sensor-modeling 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. sensor_modeling/__init__.py +45 -0
  2. sensor_modeling/alerts/__init__.py +26 -0
  3. sensor_modeling/alerts/alert.py +532 -0
  4. sensor_modeling/analysis/__init__.py +43 -0
  5. sensor_modeling/analysis/_frame.py +19 -0
  6. sensor_modeling/analysis/behavioral_analysis.py +57 -0
  7. sensor_modeling/analysis/behavioral_metrics.py +66 -0
  8. sensor_modeling/analysis/comparison.py +164 -0
  9. sensor_modeling/analysis/dependency_network.py +408 -0
  10. sensor_modeling/analysis/granger_causality.py +314 -0
  11. sensor_modeling/analysis/pipeline.py +168 -0
  12. sensor_modeling/analysis/reporting.py +109 -0
  13. sensor_modeling/baseline/__init__.py +30 -0
  14. sensor_modeling/baseline/adaptive.py +520 -0
  15. sensor_modeling/baseline/features.py +224 -0
  16. sensor_modeling/change_point/__init__.py +13 -0
  17. sensor_modeling/change_point/_validation.py +31 -0
  18. sensor_modeling/change_point/adaptive_normalization.py +55 -0
  19. sensor_modeling/change_point/embedding_cpd.py +60 -0
  20. sensor_modeling/change_point/energy_efficient.py +57 -0
  21. sensor_modeling/change_point/genetic_optimization.py +65 -0
  22. sensor_modeling/cli.py +416 -0
  23. sensor_modeling/context/__init__.py +33 -0
  24. sensor_modeling/context/occupancy.py +529 -0
  25. sensor_modeling/data/__init__.py +5 -0
  26. sensor_modeling/data/loaders.py +146 -0
  27. sensor_modeling/data/preprocessing.py +83 -0
  28. sensor_modeling/data/synthetic.py +121 -0
  29. sensor_modeling/data/validation.py +81 -0
  30. sensor_modeling/evaluation/__init__.py +92 -0
  31. sensor_modeling/evaluation/ablation.py +303 -0
  32. sensor_modeling/evaluation/attribution.py +474 -0
  33. sensor_modeling/evaluation/detection.py +297 -0
  34. sensor_modeling/evaluation/metrics.py +541 -0
  35. sensor_modeling/evaluation/provenance.py +309 -0
  36. sensor_modeling/examples/__init__.py +1 -0
  37. sensor_modeling/examples/demos/__init__.py +1 -0
  38. sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
  39. sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
  40. sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
  41. sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
  42. sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
  43. sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
  44. sensor_modeling/examples/tutorials/__init__.py +1 -0
  45. sensor_modeling/fusion/__init__.py +46 -0
  46. sensor_modeling/fusion/defaults.py +296 -0
  47. sensor_modeling/fusion/emissions.py +339 -0
  48. sensor_modeling/fusion/estimate.py +375 -0
  49. sensor_modeling/fusion/filter.py +323 -0
  50. sensor_modeling/health/__init__.py +31 -0
  51. sensor_modeling/health/monitor.py +590 -0
  52. sensor_modeling/health/status.py +74 -0
  53. sensor_modeling/hmm/__init__.py +15 -0
  54. sensor_modeling/hmm/adaptive_hmm.py +22 -0
  55. sensor_modeling/hmm/base.py +134 -0
  56. sensor_modeling/hmm/circadian_hmm.py +22 -0
  57. sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
  58. sensor_modeling/hmm/hierarchical_hmm.py +35 -0
  59. sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
  60. sensor_modeling/interop/__init__.py +57 -0
  61. sensor_modeling/interop/fhir.py +418 -0
  62. sensor_modeling/interop/privacy.py +308 -0
  63. sensor_modeling/models/__init__.py +12 -0
  64. sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
  65. sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
  66. sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
  67. sensor_modeling/models/change_point_detection/__init__.py +10 -0
  68. sensor_modeling/models/change_point_detection/deep.py +65 -0
  69. sensor_modeling/models/change_point_detection/pelt.py +159 -0
  70. sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
  71. sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
  72. sensor_modeling/models/nhpp_pelt/cli.py +243 -0
  73. sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
  74. sensor_modeling/models/nhpp_pelt/io.py +58 -0
  75. sensor_modeling/models/nhpp_pelt/model.py +408 -0
  76. sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
  77. sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
  78. sensor_modeling/models/nhpp_pelt/quad.py +72 -0
  79. sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
  80. sensor_modeling/models/nhpp_pelt/utils.py +174 -0
  81. sensor_modeling/observations/__init__.py +59 -0
  82. sensor_modeling/observations/adapters.py +195 -0
  83. sensor_modeling/observations/ingest.py +269 -0
  84. sensor_modeling/observations/observation.py +270 -0
  85. sensor_modeling/observations/registry.py +262 -0
  86. sensor_modeling/observations/stream.py +342 -0
  87. sensor_modeling/observations/types.py +107 -0
  88. sensor_modeling/observations/units.py +117 -0
  89. sensor_modeling/online/__init__.py +36 -0
  90. sensor_modeling/online/benchmarks.py +242 -0
  91. sensor_modeling/online/pipeline.py +485 -0
  92. sensor_modeling/simulation/__init__.py +54 -0
  93. sensor_modeling/simulation/faults.py +191 -0
  94. sensor_modeling/simulation/household.py +862 -0
  95. sensor_modeling/states/__init__.py +23 -0
  96. sensor_modeling/states/markov.py +105 -0
  97. sensor_modeling/states/ontology.py +238 -0
  98. sensor_modeling/utils/__init__.py +41 -0
  99. sensor_modeling/utils/data_io.py +199 -0
  100. sensor_modeling/utils/logging_config.py +10 -0
  101. sensor_modeling/utils/missing.py +188 -0
  102. sensor_modeling/utils/plotting.py +98 -0
  103. sensor_modeling/utils/validation.py +117 -0
  104. sensor_modeling/visualization/__init__.py +3 -0
  105. sensor_modeling/visualization/clinical.py +67 -0
  106. sensor_modeling/visualization/interactive.py +208 -0
  107. sensor_modeling/visualization/research.py +60 -0
  108. sensor_modeling/visualization/web_app.py +137 -0
  109. sensor_modeling-0.2.0.dist-info/METADATA +683 -0
  110. sensor_modeling-0.2.0.dist-info/RECORD +114 -0
  111. sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
  112. sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
  113. sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
  114. sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,309 @@
1
+ """Making an experimental result interpretable without the code that made it.
2
+
3
+ A JSON file containing ``{"balanced_accuracy": 0.814}`` is nearly worthless six
4
+ months later. Balanced accuracy of what, over which sensors, with which seeds,
5
+ under which version, and computed how? Every one of those has to be recoverable
6
+ from the artefact itself, because the code will have moved on and the person
7
+ reading it may not be the person who ran it.
8
+
9
+ An :class:`ExperimentRecord` therefore carries the results *and* everything
10
+ needed to interpret and reproduce them: the configuration, the seeds, the
11
+ software and library versions, the sensor subset, and a written definition of
12
+ every metric reported.
13
+
14
+ Artefacts are written to a results directory that is deliberately excluded
15
+ from version control. Results are regenerated from their seed rather than
16
+ committed, so the repository does not accumulate large generated files.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import dataclasses
22
+ import json
23
+ import logging
24
+ import platform
25
+ import subprocess
26
+ import sys
27
+ from collections.abc import Mapping, Sequence
28
+ from dataclasses import dataclass, field
29
+ from datetime import datetime, timezone
30
+ from pathlib import Path
31
+ from typing import Any
32
+
33
+ logger = logging.getLogger(__name__)
34
+
35
+ #: Default location for generated artefacts. Excluded from version control.
36
+ RESULTS_DIR = Path("results")
37
+
38
+ #: Version of the artefact layout itself.
39
+ #:
40
+ #: A reader that understands this schema knows which fields to expect. Bump it
41
+ #: when a field changes meaning, not merely when one is added.
42
+ SCHEMA_VERSION = "1.0"
43
+
44
+ #: Repository root, used to ask git which commit the code came from.
45
+ _REPO_ROOT = Path(__file__).resolve().parents[2]
46
+
47
+ #: Written definitions of every metric this package reports.
48
+ #:
49
+ #: These travel with the artefact so a stored number can be interpreted
50
+ #: without the source. Where a metric has a convention that could reasonably
51
+ #: go the other way -- notably how abstentions are scored -- the convention is
52
+ #: stated rather than left implicit.
53
+ METRIC_DEFINITIONS: dict[str, str] = {
54
+ "accuracy": (
55
+ "Fraction of scored estimates whose reported state equals the true "
56
+ "state. A reported UNKNOWN is never correct, so abstentions count as "
57
+ "errors."
58
+ ),
59
+ "selective_accuracy": (
60
+ "Accuracy among only the estimates that committed to a state. Read "
61
+ "together with abstention_rate."
62
+ ),
63
+ "abstention_rate": (
64
+ "Fraction of estimates that declined to name a state, because "
65
+ "confidence or sensor coverage fell below threshold."
66
+ ),
67
+ "balanced_accuracy": (
68
+ "Unweighted mean of per-class recall. Used in preference to accuracy "
69
+ "because the states are heavily imbalanced and a model that always "
70
+ "predicts the majority state scores highly on raw accuracy."
71
+ ),
72
+ "macro_f1": "Unweighted mean F1 across the classes present in the truth.",
73
+ "log_loss": (
74
+ "Mean negative log probability assigned to the true state. Punishes "
75
+ "confident errors far more than hedged ones."
76
+ ),
77
+ "brier": (
78
+ "Multiclass Brier score over the full posterior, in [0, 2]. A proper "
79
+ "scoring rule; lower is better."
80
+ ),
81
+ "calibration_error": (
82
+ "Expected calibration error: the bin-weighted gap between stated "
83
+ "confidence and observed accuracy. Answers whether a confidence of "
84
+ "0.9 means anything."
85
+ ),
86
+ "per_class_recall": "Recall for each state present in the ground truth.",
87
+ "precision": "True positives divided by predicted positives.",
88
+ "recall": "True positives divided by actual positives.",
89
+ "f1": "Harmonic mean of precision and recall.",
90
+ "median_delay_days": (
91
+ "Median days between a true change occurring and an alert being "
92
+ "delivered for it. A detection counts only if it falls at or after "
93
+ "the change and within max_delay_days. Where an arm aggregates several "
94
+ "seeds, the individual delays are pooled before taking the median, so "
95
+ "this is a median over detections and not a mean of per-seed medians."
96
+ ),
97
+ "mean_seed_median_delay_days": (
98
+ "Mean across seeds of each seed's own median delay. Reported alongside "
99
+ "median_delay_days because it weights every seed equally regardless of "
100
+ "how many changes it detected; it is not a median."
101
+ ),
102
+ "detected_changes": (
103
+ "Number of true changes detected across the seeds in an arm, which is "
104
+ "the sample size behind median_delay_days."
105
+ ),
106
+ "false_positives_per_person_day": (
107
+ "Delivered alerts not matched to a true change, divided by monitored "
108
+ "person-days. The alert burden a recipient experiences."
109
+ ),
110
+ "mean_difference": (
111
+ "Mean paired difference between two configurations evaluated on "
112
+ "identical simulated trajectories."
113
+ ),
114
+ "ci_low, ci_high": (
115
+ "Bootstrap confidence interval on the mean paired difference. "
116
+ "Reported instead of a p-value: with simulations, significance is a "
117
+ "statement about how long the computer ran."
118
+ ),
119
+ "effect_size": (
120
+ "Cohen's dz for the paired difference. Note that pairing removes "
121
+ "between-household variance, so dz is larger than an unpaired field "
122
+ "study of the same size would produce."
123
+ ),
124
+ "contaminated_fraction": (
125
+ "Fraction of simulated time during which someone other than the "
126
+ "monitored resident was present."
127
+ ),
128
+ }
129
+
130
+
131
+ def git_state() -> dict[str, str]:
132
+ """Return the commit the code came from, and whether it was modified.
133
+
134
+ The package version is not enough to identify code. A version stays fixed
135
+ across many commits during development, so two materially different
136
+ implementations can produce records that claim the same provenance. The
137
+ commit resolves that; ``git_dirty`` records whether the working tree had
138
+ uncommitted changes, because a dirty run is not reproducible from the
139
+ commit alone.
140
+
141
+ Returns ``"unknown"`` when git is unavailable or the package was installed
142
+ from a distribution rather than a checkout, which is not an error.
143
+ """
144
+
145
+ def _run(*args: str) -> str | None:
146
+ try:
147
+ completed = subprocess.run(
148
+ ["git", *args],
149
+ cwd=_REPO_ROOT,
150
+ capture_output=True,
151
+ text=True,
152
+ timeout=10,
153
+ check=False,
154
+ )
155
+ except (OSError, subprocess.SubprocessError):
156
+ return None
157
+ if completed.returncode != 0:
158
+ return None
159
+ return completed.stdout.strip()
160
+
161
+ commit = _run("rev-parse", "HEAD")
162
+ if commit is None:
163
+ return {"git_commit": "unknown", "git_dirty": "unknown"}
164
+ status = _run("status", "--porcelain")
165
+ return {
166
+ "git_commit": commit,
167
+ "git_dirty": "unknown" if status is None else str(bool(status)).lower(),
168
+ }
169
+
170
+
171
+ def resolved_defaults() -> dict[str, Any]:
172
+ """Snapshot the algorithm defaults an experiment actually ran under.
173
+
174
+ Recording only the options a user typed is not enough to interpret a result
175
+ later. Most of what determines the outcome lives in configuration defaults,
176
+ so a record that omits them cannot be distinguished from one produced after
177
+ those defaults changed. Capturing the resolved values freezes the model
178
+ specification alongside the numbers.
179
+ """
180
+ from ..baseline.adaptive import BaselineConfig
181
+ from ..context.occupancy import ContextConfig
182
+ from ..health.monitor import HealthConfig
183
+ from ..online.pipeline import PipelineConfig
184
+ from ..simulation.household import HouseholdConfig
185
+
186
+ snapshot: dict[str, Any] = {}
187
+ for name, factory in (
188
+ ("pipeline", PipelineConfig),
189
+ ("baseline", BaselineConfig),
190
+ ("context", ContextConfig),
191
+ ("health", HealthConfig),
192
+ ("household", HouseholdConfig),
193
+ ):
194
+ try:
195
+ snapshot[name] = dataclasses.asdict(factory())
196
+ except Exception: # pragma: no cover - a config needing arguments
197
+ logger.debug("could not snapshot %s defaults", name, exc_info=True)
198
+ snapshot[name] = "unavailable"
199
+ return snapshot
200
+
201
+
202
+ def environment() -> dict[str, str]:
203
+ """Capture the software environment a result was produced in."""
204
+ versions: dict[str, str] = {
205
+ "python": sys.version.split()[0],
206
+ "platform": platform.platform(),
207
+ }
208
+ for name in ("numpy", "scipy", "pandas", "sensor_modeling"):
209
+ try:
210
+ module = __import__(name)
211
+ versions[name] = str(getattr(module, "__version__", "unknown"))
212
+ except ImportError: # pragma: no cover - all are hard dependencies
213
+ versions[name] = "not installed"
214
+ versions.update(git_state())
215
+ return versions
216
+
217
+
218
+ @dataclass
219
+ class ExperimentRecord:
220
+ """A result together with everything needed to interpret and repeat it.
221
+
222
+ Parameters
223
+ ----------
224
+ experiment
225
+ Name of the experiment, matching the command that produces it.
226
+ configuration
227
+ Every parameter that affects the outcome. A reader must be able to
228
+ reconstruct the run from this alone.
229
+ seeds
230
+ Random seeds used. Listed separately from the configuration because
231
+ they are the first thing anyone reproducing a result reaches for.
232
+ results
233
+ The findings themselves.
234
+ sensor_subset
235
+ Sensors the experiment ran over, when it varies them.
236
+ notes
237
+ Anything a reader needs in order not to over-read the result.
238
+ """
239
+
240
+ experiment: str
241
+ configuration: Mapping[str, Any]
242
+ seeds: Sequence[int] = field(default_factory=list)
243
+ results: Mapping[str, Any] = field(default_factory=dict)
244
+ sensor_subset: Sequence[str] | None = None
245
+ notes: Sequence[str] = field(default_factory=list)
246
+
247
+ def __post_init__(self) -> None:
248
+ """Validate that the record is self-describing."""
249
+ if not str(self.experiment).strip():
250
+ raise ValueError("an experiment record needs a name")
251
+
252
+ def to_dict(self) -> dict[str, Any]:
253
+ """Return the full artefact, results and provenance together."""
254
+ return {
255
+ "experiment": self.experiment,
256
+ "schema_version": SCHEMA_VERSION,
257
+ "recorded_at": datetime.now(timezone.utc).isoformat(),
258
+ "environment": environment(),
259
+ "configuration": dict(self.configuration),
260
+ "resolved_defaults": resolved_defaults(),
261
+ "seeds": list(self.seeds),
262
+ "sensor_subset": (
263
+ list(self.sensor_subset) if self.sensor_subset is not None else None
264
+ ),
265
+ "metric_definitions": METRIC_DEFINITIONS,
266
+ "results": dict(self.results),
267
+ "notes": [
268
+ *self.notes,
269
+ "Generated from the bundled simulator. Not validated against "
270
+ "real sensor data; see docs/limitations.md.",
271
+ ],
272
+ }
273
+
274
+ def write(self, path: Path | None = None) -> Path:
275
+ """Write the artefact as JSON and return where it went.
276
+
277
+ A ``recorded_at`` stamp is added at write time, so two writes of the
278
+ same record are not byte-identical. The *results* they contain are,
279
+ which is the property reproducibility actually needs.
280
+ """
281
+ target = path if path is not None else RESULTS_DIR / f"{self.experiment}.json"
282
+ target.parent.mkdir(parents=True, exist_ok=True)
283
+ target.write_text(
284
+ json.dumps(self.to_dict(), indent=2, default=str), encoding="utf-8"
285
+ )
286
+ logger.info("Wrote experiment record to %s", target)
287
+ return target
288
+
289
+
290
+ def load_record(path: Path) -> dict[str, Any]:
291
+ """Read an artefact back, checking it carries its provenance."""
292
+ payload = json.loads(Path(path).read_text(encoding="utf-8"))
293
+ missing = [
294
+ key
295
+ for key in (
296
+ "experiment",
297
+ "schema_version",
298
+ "environment",
299
+ "configuration",
300
+ "metric_definitions",
301
+ )
302
+ if key not in payload
303
+ ]
304
+ if missing:
305
+ raise ValueError(
306
+ f"artefact at {path} is missing provenance fields: {sorted(missing)}"
307
+ )
308
+ result: dict[str, Any] = payload
309
+ return result
@@ -0,0 +1 @@
1
+ """Example scripts demonstrating package usage."""
@@ -0,0 +1 @@
1
+ """Demonstration scripts for sensor modeling."""