sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Granger Causality Testing for Binary Time Series
|
|
3
|
+
|
|
4
|
+
This module implements Granger causality tests specifically adapted
|
|
5
|
+
for binary sensor time series data.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pandas as pd
|
|
12
|
+
from scipy.stats import chi2
|
|
13
|
+
from sklearn.linear_model import LogisticRegression
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
CAUSALITY_RESULT_COLUMNS = [
|
|
18
|
+
"cause",
|
|
19
|
+
"effect",
|
|
20
|
+
"test_statistic",
|
|
21
|
+
"p_value",
|
|
22
|
+
"causality_detected",
|
|
23
|
+
"lags_used",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class GrangerCausalityTest:
|
|
28
|
+
"""
|
|
29
|
+
Granger causality test for binary time series sensor data.
|
|
30
|
+
Tests whether past values of sensor X help predict sensor Y beyond what
|
|
31
|
+
past values of Y already explain.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(self, max_lags: int = 5):
|
|
35
|
+
"""
|
|
36
|
+
Initialize Granger causality test.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
max_lags: Maximum number of lags to consider
|
|
40
|
+
"""
|
|
41
|
+
if max_lags < 1:
|
|
42
|
+
raise ValueError("max_lags must be at least 1")
|
|
43
|
+
self.max_lags = max_lags
|
|
44
|
+
|
|
45
|
+
def test(self, x: np.ndarray, y: np.ndarray, lags: int | None = None) -> dict:
|
|
46
|
+
"""
|
|
47
|
+
Perform Granger causality test.
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
x: Time series of sensor X (potential cause)
|
|
51
|
+
y: Time series of sensor Y (potential effect)
|
|
52
|
+
lags: Number of lags to use (if None, automatically selected)
|
|
53
|
+
|
|
54
|
+
Returns:
|
|
55
|
+
Dictionary with test results
|
|
56
|
+
"""
|
|
57
|
+
x, y = self._validate_series(x, y)
|
|
58
|
+
if lags is None:
|
|
59
|
+
lags = self._select_optimal_lags(x, y)
|
|
60
|
+
self._validate_lags(lags, len(y))
|
|
61
|
+
|
|
62
|
+
# Restricted model: Y ~ lags of Y
|
|
63
|
+
y_lagged = self._create_lag_matrix(y, lags, include_current=False)
|
|
64
|
+
y_current = y[lags:]
|
|
65
|
+
|
|
66
|
+
# Unrestricted model: Y ~ lags of Y + lags of X
|
|
67
|
+
x_lagged = self._create_lag_matrix(x, lags, include_current=False)
|
|
68
|
+
|
|
69
|
+
# Fit restricted model (Y predicted by its own lags only)
|
|
70
|
+
restricted_ll = self._fit_binary_regression(y_lagged, y_current)
|
|
71
|
+
|
|
72
|
+
# Fit unrestricted model (Y predicted by its own lags + X lags)
|
|
73
|
+
combined_features = np.column_stack([y_lagged, x_lagged])
|
|
74
|
+
unrestricted_ll = self._fit_binary_regression(combined_features, y_current)
|
|
75
|
+
|
|
76
|
+
# Likelihood ratio test
|
|
77
|
+
lr_stat = max(0.0, float(2 * (unrestricted_ll - restricted_ll)))
|
|
78
|
+
p_value = float(1 - chi2.cdf(lr_stat, df=lags))
|
|
79
|
+
|
|
80
|
+
return {
|
|
81
|
+
"test_statistic": lr_stat,
|
|
82
|
+
"p_value": p_value,
|
|
83
|
+
"lags_used": lags,
|
|
84
|
+
"causality_detected": p_value < 0.05,
|
|
85
|
+
"restricted_ll": restricted_ll,
|
|
86
|
+
"unrestricted_ll": unrestricted_ll,
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
def _validate_series(
|
|
90
|
+
self, x: np.ndarray, y: np.ndarray
|
|
91
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
92
|
+
"""Validate and normalize input series."""
|
|
93
|
+
x_array = np.asarray(x, dtype=float).ravel()
|
|
94
|
+
y_array = np.asarray(y, dtype=float).ravel()
|
|
95
|
+
if len(x_array) != len(y_array):
|
|
96
|
+
raise ValueError("x and y must have the same length")
|
|
97
|
+
if len(x_array) < 3:
|
|
98
|
+
raise ValueError("x and y must contain at least 3 observations")
|
|
99
|
+
if np.isnan(x_array).any() or np.isnan(y_array).any():
|
|
100
|
+
raise ValueError("x and y must not contain NaN values")
|
|
101
|
+
for name, values in {"x": x_array, "y": y_array}.items():
|
|
102
|
+
unique = set(np.unique(values))
|
|
103
|
+
if not unique <= {0.0, 1.0}:
|
|
104
|
+
raise ValueError(f"{name} must be binary with values 0 and 1")
|
|
105
|
+
return x_array, y_array
|
|
106
|
+
|
|
107
|
+
def _validate_lags(self, lags: int, n_obs: int) -> None:
|
|
108
|
+
"""Validate requested lag count."""
|
|
109
|
+
if lags < 1:
|
|
110
|
+
raise ValueError("lags must be at least 1")
|
|
111
|
+
if lags >= n_obs:
|
|
112
|
+
raise ValueError("lags must be smaller than the number of observations")
|
|
113
|
+
|
|
114
|
+
def _create_lag_matrix(
|
|
115
|
+
self, series: np.ndarray, lags: int, include_current: bool = False
|
|
116
|
+
) -> np.ndarray:
|
|
117
|
+
"""Create matrix of lagged variables."""
|
|
118
|
+
n = len(series)
|
|
119
|
+
lag_matrix = np.zeros((n - lags, lags + (1 if include_current else 0)))
|
|
120
|
+
|
|
121
|
+
for row, t in enumerate(range(lags, n)):
|
|
122
|
+
col = 0
|
|
123
|
+
if include_current:
|
|
124
|
+
lag_matrix[row, col] = series[t]
|
|
125
|
+
col += 1
|
|
126
|
+
for lag in range(1, lags + 1):
|
|
127
|
+
lag_matrix[row, col] = series[t - lag]
|
|
128
|
+
col += 1
|
|
129
|
+
|
|
130
|
+
return lag_matrix
|
|
131
|
+
|
|
132
|
+
def _fit_binary_regression(self, X: np.ndarray, y: np.ndarray) -> float:
|
|
133
|
+
"""Fit logistic regression and return log-likelihood."""
|
|
134
|
+
try:
|
|
135
|
+
if len(np.unique(y)) < 2:
|
|
136
|
+
return self._constant_log_likelihood(y)
|
|
137
|
+
|
|
138
|
+
# Add intercept
|
|
139
|
+
X_with_intercept = np.column_stack([np.ones(X.shape[0]), X])
|
|
140
|
+
|
|
141
|
+
# Fit using sklearn for numerical stability
|
|
142
|
+
model = LogisticRegression(fit_intercept=False, max_iter=1000)
|
|
143
|
+
model.fit(X_with_intercept, y)
|
|
144
|
+
|
|
145
|
+
# Calculate log-likelihood manually
|
|
146
|
+
linear_pred = X_with_intercept @ model.coef_.T
|
|
147
|
+
probs = 1 / (1 + np.exp(-linear_pred.flatten()))
|
|
148
|
+
|
|
149
|
+
# Avoid log(0) issues
|
|
150
|
+
probs = np.clip(probs, 1e-15, 1 - 1e-15)
|
|
151
|
+
|
|
152
|
+
log_likelihood = np.sum(y * np.log(probs) + (1 - y) * np.log(1 - probs))
|
|
153
|
+
return log_likelihood
|
|
154
|
+
|
|
155
|
+
except (FloatingPointError, ValueError) as e:
|
|
156
|
+
logger.warning("Error in binary regression: %s", e)
|
|
157
|
+
return -np.inf
|
|
158
|
+
|
|
159
|
+
def _constant_log_likelihood(self, y: np.ndarray) -> float:
|
|
160
|
+
"""Return Bernoulli log-likelihood for a constant target."""
|
|
161
|
+
p = float(np.clip(y.mean(), 1e-15, 1 - 1e-15))
|
|
162
|
+
return float(np.sum(y * np.log(p) + (1 - y) * np.log(1 - p)))
|
|
163
|
+
|
|
164
|
+
def _select_optimal_lags(self, x: np.ndarray, y: np.ndarray) -> int:
|
|
165
|
+
"""Select optimal number of lags using BIC."""
|
|
166
|
+
best_lags = 1
|
|
167
|
+
best_bic = np.inf
|
|
168
|
+
|
|
169
|
+
for lags in range(1, min(self.max_lags + 1, len(y) // 4)):
|
|
170
|
+
try:
|
|
171
|
+
# Fit model with this number of lags
|
|
172
|
+
y_lagged = self._create_lag_matrix(y, lags, include_current=False)
|
|
173
|
+
x_lagged = self._create_lag_matrix(x, lags, include_current=False)
|
|
174
|
+
y_current = y[lags:]
|
|
175
|
+
|
|
176
|
+
combined_features = np.column_stack([y_lagged, x_lagged])
|
|
177
|
+
log_likelihood = self._fit_binary_regression(
|
|
178
|
+
combined_features, y_current
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
n_params = combined_features.shape[1] + 1 # +1 for intercept
|
|
182
|
+
n_obs = len(y_current)
|
|
183
|
+
bic = -2 * log_likelihood + n_params * np.log(n_obs)
|
|
184
|
+
|
|
185
|
+
if bic < best_bic:
|
|
186
|
+
best_bic = bic
|
|
187
|
+
best_lags = lags
|
|
188
|
+
|
|
189
|
+
except (FloatingPointError, ValueError):
|
|
190
|
+
continue
|
|
191
|
+
|
|
192
|
+
return best_lags
|
|
193
|
+
|
|
194
|
+
def test_all_pairs(self, data: pd.DataFrame) -> pd.DataFrame:
|
|
195
|
+
"""
|
|
196
|
+
Test Granger causality for all sensor pairs.
|
|
197
|
+
|
|
198
|
+
Args:
|
|
199
|
+
data: DataFrame with sensor data
|
|
200
|
+
|
|
201
|
+
Returns:
|
|
202
|
+
DataFrame with causality test results for all pairs
|
|
203
|
+
"""
|
|
204
|
+
sensors = data.columns.tolist()
|
|
205
|
+
results = []
|
|
206
|
+
|
|
207
|
+
for cause in sensors:
|
|
208
|
+
for effect in sensors:
|
|
209
|
+
if cause != effect:
|
|
210
|
+
try:
|
|
211
|
+
result = self.test(data[cause].values, data[effect].values)
|
|
212
|
+
results.append(
|
|
213
|
+
{
|
|
214
|
+
"cause": cause,
|
|
215
|
+
"effect": effect,
|
|
216
|
+
"test_statistic": result["test_statistic"],
|
|
217
|
+
"p_value": result["p_value"],
|
|
218
|
+
"causality_detected": result["causality_detected"],
|
|
219
|
+
"lags_used": result["lags_used"],
|
|
220
|
+
}
|
|
221
|
+
)
|
|
222
|
+
except (FloatingPointError, ValueError) as e:
|
|
223
|
+
logger.warning("Error testing %s -> %s: %s", cause, effect, e)
|
|
224
|
+
results.append(
|
|
225
|
+
{
|
|
226
|
+
"cause": cause,
|
|
227
|
+
"effect": effect,
|
|
228
|
+
"test_statistic": np.nan,
|
|
229
|
+
"p_value": np.nan,
|
|
230
|
+
"causality_detected": False,
|
|
231
|
+
"lags_used": np.nan,
|
|
232
|
+
}
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
return pd.DataFrame(results, columns=CAUSALITY_RESULT_COLUMNS)
|
|
236
|
+
|
|
237
|
+
def create_causality_summary(self, causality_results: pd.DataFrame) -> dict:
|
|
238
|
+
"""
|
|
239
|
+
Create summary statistics from causality test results.
|
|
240
|
+
|
|
241
|
+
Args:
|
|
242
|
+
causality_results: DataFrame from test_all_pairs()
|
|
243
|
+
|
|
244
|
+
Returns:
|
|
245
|
+
Dictionary with summary statistics
|
|
246
|
+
"""
|
|
247
|
+
significant = causality_results[
|
|
248
|
+
causality_results["causality_detected"].eq(True)
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
# Count relationships
|
|
252
|
+
total_pairs = len(causality_results)
|
|
253
|
+
significant_pairs = len(significant)
|
|
254
|
+
causality_rate = significant_pairs / total_pairs if total_pairs > 0 else 0
|
|
255
|
+
|
|
256
|
+
# Identify most influential sensors
|
|
257
|
+
if len(significant) > 0:
|
|
258
|
+
cause_counts = (
|
|
259
|
+
significant.groupby("cause").size().sort_values(ascending=False)
|
|
260
|
+
)
|
|
261
|
+
effect_counts = (
|
|
262
|
+
significant.groupby("effect").size().sort_values(ascending=False)
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
top_causes = cause_counts.head(3).to_dict()
|
|
266
|
+
top_effects = effect_counts.head(3).to_dict()
|
|
267
|
+
else:
|
|
268
|
+
top_causes = {}
|
|
269
|
+
top_effects = {}
|
|
270
|
+
|
|
271
|
+
# Average test statistics
|
|
272
|
+
avg_test_stat = self._finite_mean_or_zero(causality_results["test_statistic"])
|
|
273
|
+
avg_significant_stat = self._finite_mean_or_zero(significant["test_statistic"])
|
|
274
|
+
|
|
275
|
+
return {
|
|
276
|
+
"total_pairs_tested": total_pairs,
|
|
277
|
+
"significant_relationships": significant_pairs,
|
|
278
|
+
"causality_rate": causality_rate,
|
|
279
|
+
"average_test_statistic": avg_test_stat,
|
|
280
|
+
"average_significant_statistic": avg_significant_stat,
|
|
281
|
+
"top_causes": top_causes,
|
|
282
|
+
"top_effects": top_effects,
|
|
283
|
+
"bidirectional_relationships": self._find_bidirectional(significant),
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
def _finite_mean_or_zero(self, values: pd.Series) -> float:
|
|
287
|
+
"""Return the finite mean of values, or zero when none are finite."""
|
|
288
|
+
finite_values = pd.to_numeric(values, errors="coerce").replace(
|
|
289
|
+
[np.inf, -np.inf], np.nan
|
|
290
|
+
)
|
|
291
|
+
mean = finite_values.mean()
|
|
292
|
+
if pd.isna(mean):
|
|
293
|
+
return 0.0
|
|
294
|
+
return float(mean)
|
|
295
|
+
|
|
296
|
+
def _find_bidirectional(
|
|
297
|
+
self, significant_results: pd.DataFrame
|
|
298
|
+
) -> list[tuple[str, str]]:
|
|
299
|
+
"""Find bidirectional causal relationships."""
|
|
300
|
+
bidirectional = []
|
|
301
|
+
|
|
302
|
+
for _, row in significant_results.iterrows():
|
|
303
|
+
# Check if reverse relationship also exists
|
|
304
|
+
reverse = significant_results[
|
|
305
|
+
(significant_results["cause"] == row["effect"])
|
|
306
|
+
& (significant_results["effect"] == row["cause"])
|
|
307
|
+
]
|
|
308
|
+
|
|
309
|
+
if len(reverse) > 0:
|
|
310
|
+
pair = tuple(sorted([row["cause"], row["effect"]]))
|
|
311
|
+
if pair not in bidirectional:
|
|
312
|
+
bidirectional.append(pair)
|
|
313
|
+
|
|
314
|
+
return bidirectional
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Unified analysis pipeline for sensor modeling.
|
|
2
|
+
|
|
3
|
+
This module exposes :class:`AnalysisPipeline` which orchestrates the
|
|
4
|
+
available modeling approaches (autoregressive, HMM, change-point and NHPP)
|
|
5
|
+
on a common :class:`~sensor_modeling.utils.data_io.SensorDataset` input.
|
|
6
|
+
|
|
7
|
+
The pipeline hides the complexity of individual models while presenting a
|
|
8
|
+
simple interface for end users::
|
|
9
|
+
|
|
10
|
+
from sensor_modeling.analysis import AnalysisPipeline
|
|
11
|
+
from sensor_modeling.utils.data_io import SensorDataset
|
|
12
|
+
|
|
13
|
+
pipeline = AnalysisPipeline()
|
|
14
|
+
results = pipeline.run(dataset)
|
|
15
|
+
pipeline.generate_report(results, output_dir="out")
|
|
16
|
+
|
|
17
|
+
Each model is executed in parallel and their results are collected in a
|
|
18
|
+
single dictionary that can be further analysed or exported.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import logging
|
|
25
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any, Dict
|
|
28
|
+
|
|
29
|
+
import numpy as np
|
|
30
|
+
import pandas as pd
|
|
31
|
+
|
|
32
|
+
from ..change_point.embedding_cpd import EmbeddingCPD
|
|
33
|
+
from ..hmm.base import BaseHMM
|
|
34
|
+
from ..models.bernoulli_ar.base_model import BernoulliAutoregressiveModel
|
|
35
|
+
from ..models.nhpp_pelt.model import NHPPPELT, NHPPConfig
|
|
36
|
+
from ..utils.data_io import SensorDataset
|
|
37
|
+
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
ModelRunError = (
|
|
41
|
+
AttributeError,
|
|
42
|
+
FloatingPointError,
|
|
43
|
+
RuntimeError,
|
|
44
|
+
TypeError,
|
|
45
|
+
ValueError,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AnalysisPipeline:
|
|
50
|
+
"""Master analysis class coordinating all modeling approaches."""
|
|
51
|
+
|
|
52
|
+
def __init__(self, models: Dict[str, Any] | None = None):
|
|
53
|
+
self.models = models or {}
|
|
54
|
+
self.results: Dict[str, Any] = {}
|
|
55
|
+
self.dataset: SensorDataset | None = None
|
|
56
|
+
|
|
57
|
+
# ------------------------------------------------------------------
|
|
58
|
+
def _default_models(self, sensor_names: list[str]) -> Dict[str, Any]:
|
|
59
|
+
"""Create default model instances for the given sensors."""
|
|
60
|
+
ar_model = BernoulliAutoregressiveModel(sensor_names, sensor_names[0])
|
|
61
|
+
hmm_model = BaseHMM()
|
|
62
|
+
cpd_model = EmbeddingCPD()
|
|
63
|
+
# n_basis must be at least degree+1; the cubic default needs four
|
|
64
|
+
# basis functions. Below that the model raises on every fit and the
|
|
65
|
+
# pipeline silently records an error instead of a result.
|
|
66
|
+
nhpp_model = NHPPPELT(NHPPConfig(n_basis=4, min_seg_len=1))
|
|
67
|
+
return {
|
|
68
|
+
"ar": ar_model,
|
|
69
|
+
"hmm": hmm_model,
|
|
70
|
+
"cpd": cpd_model,
|
|
71
|
+
"nhpp": nhpp_model,
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
# ------------------------------------------------------------------
|
|
75
|
+
def _prepare_dataset(self, data: pd.DataFrame | SensorDataset) -> pd.DataFrame:
|
|
76
|
+
"""Store and validate the input dataset for a pipeline run."""
|
|
77
|
+
self.dataset = data if isinstance(data, SensorDataset) else SensorDataset(data)
|
|
78
|
+
df = self.dataset.to_dataframe()
|
|
79
|
+
if len(df.columns) == 0:
|
|
80
|
+
raise ValueError("analysis pipeline requires at least one sensor column")
|
|
81
|
+
if len(df) == 0:
|
|
82
|
+
raise ValueError("analysis pipeline requires at least one row")
|
|
83
|
+
return df
|
|
84
|
+
|
|
85
|
+
# ------------------------------------------------------------------
|
|
86
|
+
def _run_model(
|
|
87
|
+
self,
|
|
88
|
+
name: str,
|
|
89
|
+
model: Any,
|
|
90
|
+
df: pd.DataFrame,
|
|
91
|
+
sensor_names: list[str],
|
|
92
|
+
) -> Dict[str, Any]:
|
|
93
|
+
"""Run one configured model and return serializable summary output."""
|
|
94
|
+
target_sensor = sensor_names[0]
|
|
95
|
+
if name == "nhpp":
|
|
96
|
+
model.fit(self.dataset, sensor=target_sensor)
|
|
97
|
+
return {"changepoints": getattr(model, "changepoints_", [])}
|
|
98
|
+
if name == "cpd":
|
|
99
|
+
model.fit(df[target_sensor].values)
|
|
100
|
+
return {"changepoints": model.predict()}
|
|
101
|
+
if name == "hmm":
|
|
102
|
+
x = df.values
|
|
103
|
+
model.fit(x)
|
|
104
|
+
return {"states": model.predict(x)}
|
|
105
|
+
if name == "ar":
|
|
106
|
+
model.fit(df)
|
|
107
|
+
probabilities = np.asarray(model.predict_probabilities(df))
|
|
108
|
+
return {"probabilities": probabilities[:5].tolist()}
|
|
109
|
+
return {"status": "skipped"}
|
|
110
|
+
|
|
111
|
+
# ------------------------------------------------------------------
|
|
112
|
+
def run(self, data: pd.DataFrame | SensorDataset) -> Dict[str, Any]:
|
|
113
|
+
"""Run all configured models on *data* in parallel."""
|
|
114
|
+
df = self._prepare_dataset(data)
|
|
115
|
+
sensor_names = list(df.columns)
|
|
116
|
+
if not self.models:
|
|
117
|
+
self.models = self._default_models(sensor_names)
|
|
118
|
+
|
|
119
|
+
def _run(name: str, model: Any) -> Dict[str, Any]:
|
|
120
|
+
try:
|
|
121
|
+
return self._run_model(name, model, df, sensor_names)
|
|
122
|
+
except ModelRunError as exc:
|
|
123
|
+
logger.error("Model %s failed: %s", name, exc)
|
|
124
|
+
return {"error": str(exc)}
|
|
125
|
+
|
|
126
|
+
with ThreadPoolExecutor() as ex:
|
|
127
|
+
futures = {n: ex.submit(_run, n, m) for n, m in self.models.items()}
|
|
128
|
+
self.results = {n: f.result() for n, f in futures.items()}
|
|
129
|
+
return self.results
|
|
130
|
+
|
|
131
|
+
# ------------------------------------------------------------------
|
|
132
|
+
def generate_report(
|
|
133
|
+
self, results: Dict[str, Any], output_dir: str
|
|
134
|
+
) -> Dict[str, Path]:
|
|
135
|
+
"""Generate LaTeX, HTML and FHIR reports for *results*."""
|
|
136
|
+
from .reporting import (
|
|
137
|
+
create_html_dashboard,
|
|
138
|
+
export_to_fhir,
|
|
139
|
+
generate_latex_report,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
out_dir = Path(output_dir)
|
|
143
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
144
|
+
return {
|
|
145
|
+
"latex": generate_latex_report(results, out_dir / "analysis.tex"),
|
|
146
|
+
"html": create_html_dashboard(results, out_dir / "dashboard.html"),
|
|
147
|
+
"fhir": export_to_fhir(results, out_dir / "analysis_fhir.json"),
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def main() -> None:
|
|
152
|
+
"""Run the analysis pipeline from the command line."""
|
|
153
|
+
parser = argparse.ArgumentParser(description="Run the sensor analysis pipeline")
|
|
154
|
+
parser.add_argument("data", help="Path to CSV sensor data")
|
|
155
|
+
parser.add_argument(
|
|
156
|
+
"output_dir",
|
|
157
|
+
help="Directory where the generated reports should be written",
|
|
158
|
+
)
|
|
159
|
+
args = parser.parse_args()
|
|
160
|
+
|
|
161
|
+
dataset = SensorDataset.from_csv(args.data)
|
|
162
|
+
pipeline = AnalysisPipeline()
|
|
163
|
+
results = pipeline.run(dataset)
|
|
164
|
+
pipeline.generate_report(results, args.output_dir)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
if __name__ == "__main__":
|
|
168
|
+
main()
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Reporting utilities for analysis results."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from os import PathLike
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
ReportPath = str | PathLike[str]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _prepare_output_path(path: ReportPath) -> Path:
|
|
18
|
+
"""Return an output path with its parent directory created."""
|
|
19
|
+
output_path = Path(path)
|
|
20
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
21
|
+
return output_path
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
def generate_latex_report(results: dict[str, Any], path: ReportPath) -> Path:
|
|
26
|
+
"""Generate a minimal LaTeX report summarizing *results*."""
|
|
27
|
+
output_path = _prepare_output_path(path)
|
|
28
|
+
content = [
|
|
29
|
+
r"\documentclass{article}",
|
|
30
|
+
r"\begin{document}",
|
|
31
|
+
r"\section*{Sensor Modeling Report}",
|
|
32
|
+
r"\begin{verbatim}",
|
|
33
|
+
json.dumps(results, indent=2, default=str),
|
|
34
|
+
r"\end{verbatim}",
|
|
35
|
+
r"\end{document}",
|
|
36
|
+
]
|
|
37
|
+
output_path.write_text("\n".join(content), encoding="utf-8")
|
|
38
|
+
logger.info("LaTeX report written to %s", output_path)
|
|
39
|
+
return output_path
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# ---------------------------------------------------------------------------
|
|
43
|
+
def create_html_dashboard(results: dict[str, Any], path: ReportPath) -> Path:
|
|
44
|
+
"""Generate a simple HTML dashboard for *results*."""
|
|
45
|
+
output_path = _prepare_output_path(path)
|
|
46
|
+
html = [
|
|
47
|
+
"<html><body><h1>Sensor Modeling Dashboard</h1><pre>",
|
|
48
|
+
json.dumps(results, indent=2, default=str),
|
|
49
|
+
"</pre></body></html>",
|
|
50
|
+
]
|
|
51
|
+
output_path.write_text("\n".join(html), encoding="utf-8")
|
|
52
|
+
logger.info("HTML dashboard written to %s", output_path)
|
|
53
|
+
return output_path
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# ---------------------------------------------------------------------------
|
|
57
|
+
def export_to_fhir(results: dict[str, Any], path: ReportPath) -> Path:
|
|
58
|
+
"""Export *results* as a minimal FHIR-like Observation resource.
|
|
59
|
+
|
|
60
|
+
The export preserves each top-level analysis result as an Observation
|
|
61
|
+
component. It is intended as an interoperable starting point, not a full
|
|
62
|
+
clinical profile implementation.
|
|
63
|
+
"""
|
|
64
|
+
fhir = {
|
|
65
|
+
"resourceType": "Observation",
|
|
66
|
+
"id": "sensor-analysis",
|
|
67
|
+
"status": "final",
|
|
68
|
+
"category": [
|
|
69
|
+
{
|
|
70
|
+
"coding": [
|
|
71
|
+
{
|
|
72
|
+
"system": (
|
|
73
|
+
"http://terminology.hl7.org/CodeSystem/"
|
|
74
|
+
"observation-category"
|
|
75
|
+
),
|
|
76
|
+
"code": "activity",
|
|
77
|
+
"display": "Activity",
|
|
78
|
+
}
|
|
79
|
+
]
|
|
80
|
+
}
|
|
81
|
+
],
|
|
82
|
+
"code": {
|
|
83
|
+
"text": "Sensor modeling analysis summary",
|
|
84
|
+
},
|
|
85
|
+
"effectiveDateTime": datetime.now(timezone.utc)
|
|
86
|
+
.isoformat()
|
|
87
|
+
.replace("+00:00", "Z"),
|
|
88
|
+
"component": [
|
|
89
|
+
{
|
|
90
|
+
"code": {"text": name},
|
|
91
|
+
"valueString": json.dumps(value, default=str),
|
|
92
|
+
}
|
|
93
|
+
for name, value in results.items()
|
|
94
|
+
],
|
|
95
|
+
}
|
|
96
|
+
output_path = _prepare_output_path(path)
|
|
97
|
+
output_path.write_text(json.dumps(fhir, indent=2), encoding="utf-8")
|
|
98
|
+
logger.info("FHIR export written to %s", output_path)
|
|
99
|
+
return output_path
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------------------
|
|
103
|
+
def render_template(template: str, context: dict[str, Any]) -> str:
|
|
104
|
+
"""Render *template* using ``str.format`` with the provided *context*."""
|
|
105
|
+
try:
|
|
106
|
+
return template.format(**context)
|
|
107
|
+
except (KeyError, IndexError, ValueError, AttributeError) as exc:
|
|
108
|
+
logger.error("Template rendering failed: %s", exc)
|
|
109
|
+
return template
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Adaptive personal baselines over non-stationary behaviour.
|
|
2
|
+
|
|
3
|
+
Normal behaviour is modelled as a slowly moving distribution rather than a
|
|
4
|
+
fixed calibration window, and the reasons a day can look unusual -- ordinary
|
|
5
|
+
variability, weekly rhythm, a temporary disturbance, a persistent change, a
|
|
6
|
+
gradual drift, or simply a day the sensors did not watch -- are distinguished
|
|
7
|
+
rather than collapsed into a single anomaly score.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .adaptive import (
|
|
11
|
+
MAD_TO_SIGMA,
|
|
12
|
+
AdaptiveBaseline,
|
|
13
|
+
BaselineConfig,
|
|
14
|
+
BaselineReference,
|
|
15
|
+
BehaviouralChange,
|
|
16
|
+
ChangeKind,
|
|
17
|
+
)
|
|
18
|
+
from .features import DailySummary, feature_series, summarise_days
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"MAD_TO_SIGMA",
|
|
22
|
+
"AdaptiveBaseline",
|
|
23
|
+
"BaselineConfig",
|
|
24
|
+
"BaselineReference",
|
|
25
|
+
"BehaviouralChange",
|
|
26
|
+
"ChangeKind",
|
|
27
|
+
"DailySummary",
|
|
28
|
+
"feature_series",
|
|
29
|
+
"summarise_days",
|
|
30
|
+
]
|