sensor-modeling 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. sensor_modeling/__init__.py +45 -0
  2. sensor_modeling/alerts/__init__.py +26 -0
  3. sensor_modeling/alerts/alert.py +532 -0
  4. sensor_modeling/analysis/__init__.py +43 -0
  5. sensor_modeling/analysis/_frame.py +19 -0
  6. sensor_modeling/analysis/behavioral_analysis.py +57 -0
  7. sensor_modeling/analysis/behavioral_metrics.py +66 -0
  8. sensor_modeling/analysis/comparison.py +164 -0
  9. sensor_modeling/analysis/dependency_network.py +408 -0
  10. sensor_modeling/analysis/granger_causality.py +314 -0
  11. sensor_modeling/analysis/pipeline.py +168 -0
  12. sensor_modeling/analysis/reporting.py +109 -0
  13. sensor_modeling/baseline/__init__.py +30 -0
  14. sensor_modeling/baseline/adaptive.py +520 -0
  15. sensor_modeling/baseline/features.py +224 -0
  16. sensor_modeling/change_point/__init__.py +13 -0
  17. sensor_modeling/change_point/_validation.py +31 -0
  18. sensor_modeling/change_point/adaptive_normalization.py +55 -0
  19. sensor_modeling/change_point/embedding_cpd.py +60 -0
  20. sensor_modeling/change_point/energy_efficient.py +57 -0
  21. sensor_modeling/change_point/genetic_optimization.py +65 -0
  22. sensor_modeling/cli.py +416 -0
  23. sensor_modeling/context/__init__.py +33 -0
  24. sensor_modeling/context/occupancy.py +529 -0
  25. sensor_modeling/data/__init__.py +5 -0
  26. sensor_modeling/data/loaders.py +146 -0
  27. sensor_modeling/data/preprocessing.py +83 -0
  28. sensor_modeling/data/synthetic.py +121 -0
  29. sensor_modeling/data/validation.py +81 -0
  30. sensor_modeling/evaluation/__init__.py +92 -0
  31. sensor_modeling/evaluation/ablation.py +303 -0
  32. sensor_modeling/evaluation/attribution.py +474 -0
  33. sensor_modeling/evaluation/detection.py +297 -0
  34. sensor_modeling/evaluation/metrics.py +541 -0
  35. sensor_modeling/evaluation/provenance.py +309 -0
  36. sensor_modeling/examples/__init__.py +1 -0
  37. sensor_modeling/examples/demos/__init__.py +1 -0
  38. sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
  39. sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
  40. sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
  41. sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
  42. sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
  43. sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
  44. sensor_modeling/examples/tutorials/__init__.py +1 -0
  45. sensor_modeling/fusion/__init__.py +46 -0
  46. sensor_modeling/fusion/defaults.py +296 -0
  47. sensor_modeling/fusion/emissions.py +339 -0
  48. sensor_modeling/fusion/estimate.py +375 -0
  49. sensor_modeling/fusion/filter.py +323 -0
  50. sensor_modeling/health/__init__.py +31 -0
  51. sensor_modeling/health/monitor.py +590 -0
  52. sensor_modeling/health/status.py +74 -0
  53. sensor_modeling/hmm/__init__.py +15 -0
  54. sensor_modeling/hmm/adaptive_hmm.py +22 -0
  55. sensor_modeling/hmm/base.py +134 -0
  56. sensor_modeling/hmm/circadian_hmm.py +22 -0
  57. sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
  58. sensor_modeling/hmm/hierarchical_hmm.py +35 -0
  59. sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
  60. sensor_modeling/interop/__init__.py +57 -0
  61. sensor_modeling/interop/fhir.py +418 -0
  62. sensor_modeling/interop/privacy.py +308 -0
  63. sensor_modeling/models/__init__.py +12 -0
  64. sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
  65. sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
  66. sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
  67. sensor_modeling/models/change_point_detection/__init__.py +10 -0
  68. sensor_modeling/models/change_point_detection/deep.py +65 -0
  69. sensor_modeling/models/change_point_detection/pelt.py +159 -0
  70. sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
  71. sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
  72. sensor_modeling/models/nhpp_pelt/cli.py +243 -0
  73. sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
  74. sensor_modeling/models/nhpp_pelt/io.py +58 -0
  75. sensor_modeling/models/nhpp_pelt/model.py +408 -0
  76. sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
  77. sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
  78. sensor_modeling/models/nhpp_pelt/quad.py +72 -0
  79. sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
  80. sensor_modeling/models/nhpp_pelt/utils.py +174 -0
  81. sensor_modeling/observations/__init__.py +59 -0
  82. sensor_modeling/observations/adapters.py +195 -0
  83. sensor_modeling/observations/ingest.py +269 -0
  84. sensor_modeling/observations/observation.py +270 -0
  85. sensor_modeling/observations/registry.py +262 -0
  86. sensor_modeling/observations/stream.py +342 -0
  87. sensor_modeling/observations/types.py +107 -0
  88. sensor_modeling/observations/units.py +117 -0
  89. sensor_modeling/online/__init__.py +36 -0
  90. sensor_modeling/online/benchmarks.py +242 -0
  91. sensor_modeling/online/pipeline.py +485 -0
  92. sensor_modeling/simulation/__init__.py +54 -0
  93. sensor_modeling/simulation/faults.py +191 -0
  94. sensor_modeling/simulation/household.py +862 -0
  95. sensor_modeling/states/__init__.py +23 -0
  96. sensor_modeling/states/markov.py +105 -0
  97. sensor_modeling/states/ontology.py +238 -0
  98. sensor_modeling/utils/__init__.py +41 -0
  99. sensor_modeling/utils/data_io.py +199 -0
  100. sensor_modeling/utils/logging_config.py +10 -0
  101. sensor_modeling/utils/missing.py +188 -0
  102. sensor_modeling/utils/plotting.py +98 -0
  103. sensor_modeling/utils/validation.py +117 -0
  104. sensor_modeling/visualization/__init__.py +3 -0
  105. sensor_modeling/visualization/clinical.py +67 -0
  106. sensor_modeling/visualization/interactive.py +208 -0
  107. sensor_modeling/visualization/research.py +60 -0
  108. sensor_modeling/visualization/web_app.py +137 -0
  109. sensor_modeling-0.2.0.dist-info/METADATA +683 -0
  110. sensor_modeling-0.2.0.dist-info/RECORD +114 -0
  111. sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
  112. sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
  113. sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
  114. sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,308 @@
1
+ """Pseudonymisation and redaction for exported research data.
2
+
3
+ Two claims this module does **not** make, stated first because getting them
4
+ wrong is how re-identification happens:
5
+
6
+ *Pseudonymisation is not anonymisation.* Replacing an identifier with a
7
+ pseudonym removes the name, not the person. A behavioural record is a detailed
8
+ account of when somebody sleeps, eats and leaves the house; anyone with a
9
+ little side information can often re-identify it. Pseudonymised exports remain
10
+ personal data and must be handled as such.
11
+
12
+ *A hash is not a pseudonym.* Hashing a short identifier such as ``patient_7``
13
+ protects nothing: an attacker hashes every plausible identifier and matches.
14
+ Pseudonyms here are therefore keyed with a secret salt, so the mapping cannot
15
+ be reconstructed without it. The salt must be supplied, never defaulted, and
16
+ must be kept separately from the data it protects.
17
+
18
+ What the module does provide: deterministic pseudonyms that are stable across
19
+ runs and machines, so a longitudinal study can link a subject's records
20
+ without holding their identity; and a redaction pass that strips the
21
+ free-form metadata an export does not need.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import hashlib
27
+ import hmac
28
+ import logging
29
+ import re
30
+ from collections.abc import Iterable, Mapping, MutableMapping
31
+ from dataclasses import dataclass, field
32
+ from typing import Any
33
+
34
+ logger = logging.getLogger(__name__)
35
+
36
+ #: Fields whose values are free-form and most likely to carry incidental
37
+ #: identifiers -- a room named after a person, an installer's note, a device
38
+ #: label containing a street address.
39
+ DEFAULT_REDACTED_KEYS: frozenset[str] = frozenset(
40
+ {"context", "note", "notes", "detail", "description", "display", "text"}
41
+ )
42
+
43
+ #: Patterns that look like contact details or locations wherever they appear.
44
+ _EMAIL = re.compile(r"[\w.+-]+@[\w-]+\.[\w.]+")
45
+ _PHONE = re.compile(r"(?<!\d)(?:\+\d{1,3}[\s-]?)?(?:\d[\s-]?){9,14}\d(?!\d)")
46
+ _POSTCODE = re.compile(r"\b[A-Z]{1,2}\d[A-Z\d]?\s*\d[A-Z]{2}\b", re.IGNORECASE)
47
+
48
+ REDACTED = "[redacted]"
49
+
50
+
51
+ class SaltError(ValueError):
52
+ """Raised when a pseudonymisation salt is missing or too weak."""
53
+
54
+
55
+ @dataclass
56
+ class Pseudonymiser:
57
+ """Deterministic, keyed pseudonyms for identifiers.
58
+
59
+ Parameters
60
+ ----------
61
+ salt
62
+ Secret key. Must be at least 16 characters. Keep it separately from
63
+ the exported data: together, they reverse the pseudonymisation for
64
+ anyone who can enumerate candidate identifiers.
65
+ prefix
66
+ Prepended to each pseudonym so its kind stays legible. The prefix is
67
+ cosmetic and provides no separation on its own.
68
+ domain
69
+ Optional separation label. Two pseudonymisers sharing a salt but
70
+ differing in *domain* produce unrelated digests for the same
71
+ identifier, because the domain is mixed into the key rather than into
72
+ the visible text. Leave empty to keep a single global namespace.
73
+ length
74
+ Hexadecimal characters retained. The default gives a collision
75
+ probability far below one in a billion for study-sized cohorts while
76
+ staying short enough to read.
77
+
78
+ Notes
79
+ -----
80
+ The same identifier and salt always produce the same pseudonym, on any
81
+ machine and in any process, which is what lets a longitudinal study link
82
+ records without holding identities.
83
+ """
84
+
85
+ salt: str
86
+ prefix: str = "subj"
87
+ length: int = 16
88
+ domain: str = ""
89
+
90
+ def __post_init__(self) -> None:
91
+ """Reject a salt too weak to be worth having."""
92
+ if not isinstance(self.salt, str) or len(self.salt) < 16:
93
+ raise SaltError(
94
+ "salt must be at least 16 characters; a short or guessable "
95
+ "salt lets an attacker reconstruct the mapping by enumerating "
96
+ "candidate identifiers"
97
+ )
98
+ if not 8 <= self.length <= 64:
99
+ raise ValueError("length must lie between 8 and 64 characters")
100
+
101
+ def _key(self) -> bytes:
102
+ """Return the HMAC key, separated by domain when one is set.
103
+
104
+ Deriving a subkey is what makes the domain meaningful. Mixing it into
105
+ the pseudonym text instead would leave the digest unchanged, so records
106
+ for one subject would stay joinable across domains by comparing the
107
+ part after the prefix.
108
+ """
109
+ key = self.salt.encode("utf-8")
110
+ if self.domain:
111
+ key = hmac.new(
112
+ key, b"domain:" + self.domain.encode("utf-8"), hashlib.sha256
113
+ ).digest()
114
+ return key
115
+
116
+ def pseudonym(self, identifier: str) -> str:
117
+ """Return the stable pseudonym for *identifier*."""
118
+ if not isinstance(identifier, str) or not identifier.strip():
119
+ raise ValueError("identifier must be a non-empty string")
120
+ digest = hmac.new(
121
+ self._key(), identifier.encode("utf-8"), hashlib.sha256
122
+ ).hexdigest()
123
+ return f"{self.prefix}-{digest[: self.length]}"
124
+
125
+ def mapping(self, identifiers: Iterable[str]) -> dict[str, str]:
126
+ """Return the pseudonym for each identifier.
127
+
128
+ The result is the re-identification key. Store it separately from the
129
+ exported data, or not at all.
130
+ """
131
+ return {identifier: self.pseudonym(identifier) for identifier in identifiers}
132
+
133
+
134
+ def research_identifier(study: str, subject: str, *, salt: str) -> str:
135
+ """Return a reproducible identifier for a subject within a study.
136
+
137
+ Scoping by study means the same person carries different identifiers in
138
+ different studies, so records cannot be joined across them by identifier
139
+ alone. The study is used to derive a study-specific key, not merely as a
140
+ label: sharing a salt across studies would otherwise leave every digest
141
+ identical and the records trivially linkable.
142
+ """
143
+ if not isinstance(study, str) or not study.strip():
144
+ raise ValueError("study must be a non-empty string")
145
+ return Pseudonymiser(salt=salt, prefix=study, domain=study).pseudonym(subject)
146
+
147
+
148
+ @dataclass
149
+ class RedactionPolicy:
150
+ """What an export should strip before leaving the research environment.
151
+
152
+ Parameters
153
+ ----------
154
+ drop_keys
155
+ Keys removed entirely wherever they appear.
156
+ scrub_patterns
157
+ Whether to replace anything resembling an email address, telephone
158
+ number or postcode in remaining free text.
159
+ keep_keys
160
+ Keys preserved even if they appear in *drop_keys*. Use sparingly and
161
+ deliberately.
162
+ """
163
+
164
+ drop_keys: frozenset[str] = field(default=DEFAULT_REDACTED_KEYS)
165
+ scrub_patterns: bool = True
166
+ keep_keys: frozenset[str] = field(default_factory=frozenset)
167
+
168
+ def should_drop(self, key: str) -> bool:
169
+ """Whether a key is removed under this policy."""
170
+ return key in self.drop_keys and key not in self.keep_keys
171
+
172
+
173
+ def _scrub(text: str) -> str:
174
+ """Replace contact details and locations in free text."""
175
+ scrubbed = _EMAIL.sub(REDACTED, text)
176
+ scrubbed = _PHONE.sub(REDACTED, scrubbed)
177
+ return _POSTCODE.sub(REDACTED, scrubbed)
178
+
179
+
180
+ def redact(
181
+ payload: Any,
182
+ policy: RedactionPolicy | None = None,
183
+ *,
184
+ pseudonyms: Mapping[str, str] | None = None,
185
+ ) -> Any:
186
+ """Return a copy of *payload* with identifying material removed.
187
+
188
+ Walks any nested structure of mappings and sequences, so it applies
189
+ equally to a single resource, a bundle, or a whole experiment record.
190
+
191
+ Parameters
192
+ ----------
193
+ payload
194
+ The structure to redact. Not modified.
195
+ policy
196
+ What to strip. Defaults to removing free-form metadata and scrubbing
197
+ contact details from remaining text.
198
+ pseudonyms
199
+ Replacements applied to any string that exactly matches a key. Use
200
+ :meth:`Pseudonymiser.mapping` to build it.
201
+ """
202
+ rules = policy or RedactionPolicy()
203
+ replacements = pseudonyms or {}
204
+
205
+ if isinstance(payload, Mapping):
206
+ result: MutableMapping[str, Any] = {}
207
+ for key, value in payload.items():
208
+ name = str(key)
209
+ if rules.should_drop(name):
210
+ continue
211
+ result[name] = redact(value, rules, pseudonyms=replacements)
212
+ return result
213
+
214
+ if isinstance(payload, (list, tuple)):
215
+ return [redact(item, rules, pseudonyms=replacements) for item in payload]
216
+
217
+ if isinstance(payload, str):
218
+ if payload in replacements:
219
+ return replacements[payload]
220
+ return _scrub(payload) if rules.scrub_patterns else payload
221
+
222
+ return payload
223
+
224
+
225
+ def redact_bundle(
226
+ bundle: Mapping[str, Any],
227
+ *,
228
+ salt: str,
229
+ subjects: Iterable[str] = (),
230
+ sensors: Iterable[str] = (),
231
+ policy: RedactionPolicy | None = None,
232
+ ) -> dict[str, Any]:
233
+ """Pseudonymise and redact an exported bundle in one pass.
234
+
235
+ Subjects and sensors are pseudonymised under separate prefixes, so a
236
+ reader can still tell which kind of identifier they are looking at
237
+ without being able to recover either.
238
+
239
+ Where a pseudonym is supplied, the field carrying the identifier is
240
+ *kept* and its value replaced, rather than dropped. Dropping it would be
241
+ stronger, but it would also destroy the ability to link a record to its
242
+ sensor or subject, which is the whole point of pseudonymising rather
243
+ than deleting.
244
+ """
245
+ subject_names = list(subjects)
246
+ sensor_names = list(sensors)
247
+ replacements: dict[str, str] = {}
248
+ if subject_names:
249
+ replacements.update(
250
+ Pseudonymiser(salt=salt, prefix="subj").mapping(subject_names)
251
+ )
252
+ if sensor_names:
253
+ replacements.update(
254
+ Pseudonymiser(salt=salt, prefix="sens").mapping(sensor_names)
255
+ )
256
+
257
+ # Keep the fields the pseudonyms are meant to land in. Without this the
258
+ # default policy would drop `display` outright and the sensor pseudonyms
259
+ # would have nowhere to go.
260
+ rules = policy or RedactionPolicy()
261
+ if replacements:
262
+ rules = RedactionPolicy(
263
+ drop_keys=rules.drop_keys,
264
+ scrub_patterns=rules.scrub_patterns,
265
+ keep_keys=rules.keep_keys | frozenset({"display", "reference"}),
266
+ )
267
+
268
+ redacted = redact(bundle, rules, pseudonyms=replacements)
269
+ if not isinstance(redacted, dict): # pragma: no cover - bundles are mappings
270
+ raise TypeError("a bundle must be a mapping")
271
+
272
+ meta = redacted.setdefault("meta", {})
273
+ tags = meta.setdefault("tag", [])
274
+ tags.append(
275
+ {
276
+ "code": "pseudonymised",
277
+ "display": (
278
+ "Identifiers replaced with keyed pseudonyms and free-form "
279
+ "metadata removed. Pseudonymised behavioural data remains "
280
+ "personal data: it is not anonymised and can often be "
281
+ "re-identified from side information."
282
+ ),
283
+ }
284
+ )
285
+ logger.info("Redacted bundle: %d identifiers pseudonymised", len(replacements))
286
+ return redacted
287
+
288
+
289
+ def identifiers_in(bundle: Mapping[str, Any]) -> set[str]:
290
+ """Collect the sensor and subject identifiers a bundle carries.
291
+
292
+ Provided so a caller can see what would be pseudonymised before doing it,
293
+ rather than discovering an unredacted identifier after export.
294
+ """
295
+ found: set[str] = set()
296
+
297
+ def walk(node: Any) -> None:
298
+ if isinstance(node, Mapping):
299
+ for key, value in node.items():
300
+ if key in {"display", "reference"} and isinstance(value, str):
301
+ found.add(value)
302
+ walk(value)
303
+ elif isinstance(node, (list, tuple)):
304
+ for item in node:
305
+ walk(item)
306
+
307
+ walk(bundle)
308
+ return found
@@ -0,0 +1,12 @@
1
+ """Model implementations for sensor data."""
2
+
3
+ from .bernoulli_ar.base_model import BernoulliAutoregressiveModel
4
+ from .change_point_detection.pelt import PELTChangePointDetector
5
+ from .nhpp_pelt.model import NHPPPELT, NHPPConfig
6
+
7
+ __all__ = [
8
+ "BernoulliAutoregressiveModel",
9
+ "NHPPPELT",
10
+ "NHPPConfig",
11
+ "PELTChangePointDetector",
12
+ ]
@@ -0,0 +1,6 @@
1
+ """Bernoulli autoregressive models."""
2
+
3
+ from .base_model import BernoulliAutoregressiveModel
4
+ from .multivariate_model import MultivariateAutoregressiveModel
5
+
6
+ __all__ = ["BernoulliAutoregressiveModel", "MultivariateAutoregressiveModel"]