sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""High-level behavioral analysis utilities."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from ._frame import prepare_sensor_frame
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# ---------------------------------------------------------------------------
|
|
16
|
+
def recognize_activity_patterns(data: pd.DataFrame) -> dict[str, int]:
|
|
17
|
+
"""Identify peak and quiet hours of activity."""
|
|
18
|
+
sensor_data = prepare_sensor_frame(data, context="recognize_activity_patterns")
|
|
19
|
+
hourly = sensor_data.groupby(sensor_data.index.hour).sum()
|
|
20
|
+
totals = hourly.sum(axis=1)
|
|
21
|
+
return {
|
|
22
|
+
"peak_hours": int(totals.idxmax()),
|
|
23
|
+
"quiet_hours": int(totals.idxmin()),
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# ---------------------------------------------------------------------------
|
|
28
|
+
def score_anomalies(data: pd.DataFrame) -> pd.Series:
|
|
29
|
+
"""Simple z-score based anomaly metric for each timestamp."""
|
|
30
|
+
sensor_data = prepare_sensor_frame(data, context="score_anomalies")
|
|
31
|
+
std = sensor_data.std(ddof=0).replace(0, np.nan)
|
|
32
|
+
zscores = (sensor_data - sensor_data.mean()) / std
|
|
33
|
+
return zscores.fillna(0.0).abs().sum(axis=1)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
def detect_trends(data: pd.DataFrame, window: int = 24) -> pd.DataFrame:
|
|
38
|
+
"""Rolling mean trend indicator."""
|
|
39
|
+
if window < 1:
|
|
40
|
+
raise ValueError("window must be at least 1")
|
|
41
|
+
sensor_data = prepare_sensor_frame(data, context="detect_trends")
|
|
42
|
+
return sensor_data.rolling(window=window, min_periods=1).mean()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
def health_indicators(data: pd.DataFrame) -> dict[str, float]:
|
|
47
|
+
"""Basic health status indicators derived from activity levels."""
|
|
48
|
+
sensor_data = prepare_sensor_frame(data, context="health_indicators")
|
|
49
|
+
activity = sensor_data.sum(axis=1)
|
|
50
|
+
overall = float(activity.mean())
|
|
51
|
+
variability = float(activity.std(ddof=0))
|
|
52
|
+
sedentary_ratio = float((activity == 0).mean())
|
|
53
|
+
return {
|
|
54
|
+
"overall_activity": overall,
|
|
55
|
+
"activity_variability": variability,
|
|
56
|
+
"sedentary_ratio": sedentary_ratio,
|
|
57
|
+
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Behavioral metrics calculated from sensor datasets."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from ._frame import prepare_sensor_frame
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def calculate_behavioral_metrics(data: pd.DataFrame) -> dict[str, object]:
|
|
15
|
+
"""Calculate basic behavioral pattern metrics from sensor data."""
|
|
16
|
+
sensor_data = prepare_sensor_frame(data, context="calculate_behavioral_metrics")
|
|
17
|
+
metrics: dict[str, object] = {}
|
|
18
|
+
|
|
19
|
+
total_activations = sensor_data.sum().sum()
|
|
20
|
+
total_possible = len(sensor_data) * len(sensor_data.columns)
|
|
21
|
+
metrics["overall_activity_rate"] = float(total_activations / total_possible)
|
|
22
|
+
metrics["total_activations"] = int(total_activations)
|
|
23
|
+
|
|
24
|
+
hourly_activity = sensor_data.groupby(sensor_data.index.hour).sum()
|
|
25
|
+
metrics["peak_activity_hour"] = int(hourly_activity.sum(axis=1).idxmax())
|
|
26
|
+
metrics["quietest_hour"] = int(hourly_activity.sum(axis=1).idxmin())
|
|
27
|
+
|
|
28
|
+
sensor_metrics: dict[str, dict[str, float | int]] = {}
|
|
29
|
+
for sensor in sensor_data.columns:
|
|
30
|
+
series = sensor_data[sensor]
|
|
31
|
+
daily_variance = series.groupby(series.index.date).sum().var()
|
|
32
|
+
sensor_metrics[sensor] = {
|
|
33
|
+
"activation_rate": float(series.mean()),
|
|
34
|
+
"total_activations": int(series.sum()),
|
|
35
|
+
"longest_inactive_period": _find_longest_streak(series, 0),
|
|
36
|
+
"longest_active_period": _find_longest_streak(series, 1),
|
|
37
|
+
"daily_variance": 0.0 if pd.isna(daily_variance) else float(daily_variance),
|
|
38
|
+
}
|
|
39
|
+
metrics["sensor_metrics"] = sensor_metrics
|
|
40
|
+
|
|
41
|
+
dow_activity = sensor_data.groupby(sensor_data.index.dayofweek).sum()
|
|
42
|
+
weekday_names = [
|
|
43
|
+
"Monday",
|
|
44
|
+
"Tuesday",
|
|
45
|
+
"Wednesday",
|
|
46
|
+
"Thursday",
|
|
47
|
+
"Friday",
|
|
48
|
+
"Saturday",
|
|
49
|
+
"Sunday",
|
|
50
|
+
]
|
|
51
|
+
metrics["most_active_day"] = weekday_names[dow_activity.sum(axis=1).idxmax()]
|
|
52
|
+
metrics["least_active_day"] = weekday_names[dow_activity.sum(axis=1).idxmin()]
|
|
53
|
+
|
|
54
|
+
return metrics
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _find_longest_streak(series: pd.Series, value: int) -> int:
|
|
58
|
+
max_streak = 0
|
|
59
|
+
current_streak = 0
|
|
60
|
+
for val in series:
|
|
61
|
+
if val == value:
|
|
62
|
+
current_streak += 1
|
|
63
|
+
max_streak = max(max_streak, current_streak)
|
|
64
|
+
else:
|
|
65
|
+
current_streak = 0
|
|
66
|
+
return max_streak
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Utilities for comparing different sensor models."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from collections.abc import Callable
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import matplotlib.pyplot as plt
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pandas as pd
|
|
12
|
+
from scipy.stats import ttest_rel
|
|
13
|
+
from sklearn.model_selection import TimeSeriesSplit
|
|
14
|
+
|
|
15
|
+
from ..utils.data_io import SensorDataset
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
ModelScorer = Callable[[Any, pd.DataFrame, pd.DataFrame], float]
|
|
20
|
+
FoldError = (AttributeError, FloatingPointError, RuntimeError, TypeError, ValueError)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def time_series_splits(
|
|
24
|
+
data: SensorDataset | pd.DataFrame, n_splits: int = 3
|
|
25
|
+
) -> list[tuple[np.ndarray, np.ndarray]]:
|
|
26
|
+
"""Return chronological train/test splits for sensor time series."""
|
|
27
|
+
if n_splits < 1:
|
|
28
|
+
raise ValueError("n_splits must be at least 1")
|
|
29
|
+
dataset = data if isinstance(data, SensorDataset) else SensorDataset(data)
|
|
30
|
+
df = dataset.to_dataframe()
|
|
31
|
+
if len(df) < 2:
|
|
32
|
+
raise ValueError("At least two observations are required for cross-validation")
|
|
33
|
+
split_count = min(n_splits, len(df) - 1)
|
|
34
|
+
splitter = TimeSeriesSplit(n_splits=split_count)
|
|
35
|
+
return list(splitter.split(df))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _fit_and_score_model(
|
|
39
|
+
name: str,
|
|
40
|
+
model: Any,
|
|
41
|
+
train_df: pd.DataFrame,
|
|
42
|
+
test_df: pd.DataFrame,
|
|
43
|
+
scorers: dict[str, ModelScorer],
|
|
44
|
+
) -> float:
|
|
45
|
+
"""Fit one model on a fold and return its score."""
|
|
46
|
+
model.fit(train_df.values if name == "hmm" else train_df)
|
|
47
|
+
if name in scorers:
|
|
48
|
+
return float(scorers[name](model, train_df, test_df))
|
|
49
|
+
return float(model.score(test_df.values))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _mean_score(values: list[float]) -> float:
|
|
53
|
+
"""Return the mean of valid fold scores, or NaN when every fold failed."""
|
|
54
|
+
finite_scores = [score for score in values if not np.isnan(score)]
|
|
55
|
+
if not finite_scores:
|
|
56
|
+
return float("nan")
|
|
57
|
+
return float(np.mean(finite_scores))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# ---------------------------------------------------------------------------
|
|
61
|
+
def cross_validate(
|
|
62
|
+
models: dict[str, Any],
|
|
63
|
+
data: SensorDataset | pd.DataFrame,
|
|
64
|
+
n_splits: int = 3,
|
|
65
|
+
scorers: dict[str, ModelScorer] | None = None,
|
|
66
|
+
) -> dict[str, float]:
|
|
67
|
+
"""Perform chronological cross-validation across *models*.
|
|
68
|
+
|
|
69
|
+
Parameters
|
|
70
|
+
----------
|
|
71
|
+
models : dict[str, Any]
|
|
72
|
+
Mapping from model name to model instance. Each model must implement a
|
|
73
|
+
:py:meth:`fit` method and either a :py:meth:`score` method or a scorer
|
|
74
|
+
supplied via ``scorers``.
|
|
75
|
+
data : SensorDataset | pd.DataFrame
|
|
76
|
+
Input dataset.
|
|
77
|
+
n_splits : int
|
|
78
|
+
Number of chronological folds.
|
|
79
|
+
scorers : dict[str, ModelScorer] | None
|
|
80
|
+
Optional per-model scoring functions accepting
|
|
81
|
+
``(model, train_df, test_df)``.
|
|
82
|
+
"""
|
|
83
|
+
dataset = data if isinstance(data, SensorDataset) else SensorDataset(data)
|
|
84
|
+
df = dataset.to_dataframe()
|
|
85
|
+
scorers = scorers or {}
|
|
86
|
+
unsupported = [
|
|
87
|
+
name
|
|
88
|
+
for name, model in models.items()
|
|
89
|
+
if name not in scorers and not hasattr(model, "score")
|
|
90
|
+
]
|
|
91
|
+
if unsupported:
|
|
92
|
+
names = ", ".join(sorted(unsupported))
|
|
93
|
+
raise TypeError(
|
|
94
|
+
"Cross-validation requires each model to define score() or an "
|
|
95
|
+
f"explicit scorer. Missing scorer for: {names}"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
scores: dict[str, list[float]] = {name: [] for name in models}
|
|
99
|
+
for train_idx, test_idx in time_series_splits(dataset, n_splits=n_splits):
|
|
100
|
+
train_df = df.iloc[train_idx]
|
|
101
|
+
test_df = df.iloc[test_idx]
|
|
102
|
+
for name, model in models.items():
|
|
103
|
+
try:
|
|
104
|
+
score = _fit_and_score_model(name, model, train_df, test_df, scorers)
|
|
105
|
+
scores[name].append(score)
|
|
106
|
+
except FoldError as exc:
|
|
107
|
+
logger.error("Failed fold for %s: %s", name, exc)
|
|
108
|
+
scores[name].append(float("nan"))
|
|
109
|
+
return {name: _mean_score(vals) for name, vals in scores.items()}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# ---------------------------------------------------------------------------
|
|
113
|
+
def significance_test(scores_a: list[float], scores_b: list[float]) -> float:
|
|
114
|
+
"""Paired t-test returning the p-value."""
|
|
115
|
+
values_a = np.asarray(scores_a, dtype=float)
|
|
116
|
+
values_b = np.asarray(scores_b, dtype=float)
|
|
117
|
+
if values_a.shape != values_b.shape:
|
|
118
|
+
raise ValueError("scores_a and scores_b must have the same length")
|
|
119
|
+
|
|
120
|
+
valid_pairs = np.isfinite(values_a) & np.isfinite(values_b)
|
|
121
|
+
if np.count_nonzero(valid_pairs) < 2:
|
|
122
|
+
raise ValueError("at least two paired finite scores are required")
|
|
123
|
+
|
|
124
|
+
differences = values_a[valid_pairs] - values_b[valid_pairs]
|
|
125
|
+
if np.allclose(differences, 0.0):
|
|
126
|
+
return 1.0
|
|
127
|
+
|
|
128
|
+
_, pvalue = ttest_rel(values_a[valid_pairs], values_b[valid_pairs])
|
|
129
|
+
if not np.isfinite(pvalue):
|
|
130
|
+
return 1.0
|
|
131
|
+
return float(pvalue)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
def standardize_metrics(metrics: dict[str, float]) -> dict[str, float]:
|
|
136
|
+
"""Scale metric values to the [0,1] range."""
|
|
137
|
+
if not metrics:
|
|
138
|
+
return {}
|
|
139
|
+
|
|
140
|
+
vals = np.array(list(metrics.values()), dtype=float)
|
|
141
|
+
finite_vals = vals[np.isfinite(vals)]
|
|
142
|
+
if finite_vals.size == 0:
|
|
143
|
+
return {name: float("nan") for name in metrics}
|
|
144
|
+
|
|
145
|
+
vmin = np.min(finite_vals)
|
|
146
|
+
vmax = np.max(finite_vals)
|
|
147
|
+
rng = vmax - vmin if vmax != vmin else 1.0
|
|
148
|
+
return {
|
|
149
|
+
name: float("nan") if not np.isfinite(value) else float((value - vmin) / rng)
|
|
150
|
+
for name, value in zip(metrics, vals)
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# ---------------------------------------------------------------------------
|
|
155
|
+
def visualize_comparison(metrics: dict[str, float], ax=None):
|
|
156
|
+
"""Visualize model comparison scores as a bar chart."""
|
|
157
|
+
if ax is None:
|
|
158
|
+
_, ax = plt.subplots()
|
|
159
|
+
names = list(metrics.keys())
|
|
160
|
+
vals = [metrics[n] for n in names]
|
|
161
|
+
ax.bar(names, vals)
|
|
162
|
+
ax.set_ylabel("Score")
|
|
163
|
+
ax.set_title("Model Comparison")
|
|
164
|
+
return ax
|
|
@@ -0,0 +1,408 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Sensor Dependency Network Analysis
|
|
3
|
+
|
|
4
|
+
This module builds and analyzes cross-sensor dependency networks
|
|
5
|
+
using Granger causality and network analysis techniques.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import matplotlib.pyplot as plt
|
|
13
|
+
import networkx as nx
|
|
14
|
+
import numpy as np
|
|
15
|
+
import pandas as pd
|
|
16
|
+
import seaborn as sns
|
|
17
|
+
from sklearn.metrics import mutual_info_score
|
|
18
|
+
|
|
19
|
+
from .granger_causality import GrangerCausalityTest
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SensorDependencyNetwork:
|
|
25
|
+
"""
|
|
26
|
+
Build and analyze cross-sensor dependency networks.
|
|
27
|
+
Creates directed graphs showing causal relationships between sensors.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, significance_level: float = 0.05):
|
|
31
|
+
"""
|
|
32
|
+
Initialize dependency network builder.
|
|
33
|
+
|
|
34
|
+
Args:
|
|
35
|
+
significance_level: P-value threshold for including edges
|
|
36
|
+
"""
|
|
37
|
+
if not 0 < significance_level < 1:
|
|
38
|
+
raise ValueError("significance_level must be between 0 and 1")
|
|
39
|
+
self.significance_level = significance_level
|
|
40
|
+
self.granger_test = GrangerCausalityTest()
|
|
41
|
+
self.network: nx.DiGraph | None = None
|
|
42
|
+
self.causality_results: pd.DataFrame | None = None
|
|
43
|
+
|
|
44
|
+
def build_network(self, data: pd.DataFrame) -> nx.DiGraph:
|
|
45
|
+
"""
|
|
46
|
+
Build sensor dependency network using Granger causality.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
data: DataFrame with sensor data
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
NetworkX directed graph
|
|
53
|
+
"""
|
|
54
|
+
logger.info("Building sensor dependency network...")
|
|
55
|
+
sensor_data = self._prepare_sensor_data(data)
|
|
56
|
+
|
|
57
|
+
# Test all pairs for Granger causality
|
|
58
|
+
self.causality_results = self.granger_test.test_all_pairs(sensor_data)
|
|
59
|
+
|
|
60
|
+
# Create directed graph
|
|
61
|
+
self.network = nx.DiGraph()
|
|
62
|
+
|
|
63
|
+
# Add all sensors as nodes
|
|
64
|
+
sensors = sensor_data.columns.tolist()
|
|
65
|
+
self.network.add_nodes_from(sensors)
|
|
66
|
+
|
|
67
|
+
# Add edges for significant causal relationships
|
|
68
|
+
significant_results = self.causality_results[
|
|
69
|
+
self.causality_results["causality_detected"]
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
for _, row in significant_results.iterrows():
|
|
73
|
+
self.network.add_edge(
|
|
74
|
+
row["cause"],
|
|
75
|
+
row["effect"],
|
|
76
|
+
weight=row["test_statistic"],
|
|
77
|
+
p_value=row["p_value"],
|
|
78
|
+
lags=row["lags_used"],
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
logger.info(
|
|
82
|
+
"Network created with %d nodes and %d edges",
|
|
83
|
+
len(self.network.nodes),
|
|
84
|
+
len(self.network.edges),
|
|
85
|
+
)
|
|
86
|
+
return self.network
|
|
87
|
+
|
|
88
|
+
def _prepare_sensor_data(self, data: pd.DataFrame) -> pd.DataFrame:
|
|
89
|
+
"""Return a numeric binary sensor frame for network analysis."""
|
|
90
|
+
if data.empty:
|
|
91
|
+
raise ValueError("dependency network analysis requires at least one row")
|
|
92
|
+
|
|
93
|
+
sensor_data = data.select_dtypes(include="number")
|
|
94
|
+
if sensor_data.empty:
|
|
95
|
+
raise ValueError(
|
|
96
|
+
"dependency network analysis requires at least one numeric sensor column"
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
if sensor_data.isna().any().any():
|
|
100
|
+
raise ValueError("dependency network analysis does not accept NaN values")
|
|
101
|
+
|
|
102
|
+
non_binary = [
|
|
103
|
+
column
|
|
104
|
+
for column in sensor_data.columns
|
|
105
|
+
if not set(sensor_data[column].unique()) <= {0, 1}
|
|
106
|
+
]
|
|
107
|
+
if non_binary:
|
|
108
|
+
raise ValueError(
|
|
109
|
+
"dependency network analysis requires binary sensor columns: "
|
|
110
|
+
+ ", ".join(map(str, non_binary))
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
return sensor_data.copy()
|
|
114
|
+
|
|
115
|
+
def get_network_statistics(self) -> dict[str, Any]:
|
|
116
|
+
"""Get network topology statistics."""
|
|
117
|
+
if self.network is None:
|
|
118
|
+
raise ValueError("Network must be built first")
|
|
119
|
+
|
|
120
|
+
is_connected = (
|
|
121
|
+
nx.is_weakly_connected(self.network)
|
|
122
|
+
if self.network.number_of_nodes()
|
|
123
|
+
else False
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
return {
|
|
127
|
+
"num_nodes": len(self.network.nodes),
|
|
128
|
+
"num_edges": len(self.network.edges),
|
|
129
|
+
"density": nx.density(self.network),
|
|
130
|
+
"is_connected": is_connected,
|
|
131
|
+
"num_components": nx.number_weakly_connected_components(self.network),
|
|
132
|
+
"avg_clustering": nx.average_clustering(self.network.to_undirected()),
|
|
133
|
+
"in_degree_centrality": nx.in_degree_centrality(self.network),
|
|
134
|
+
"out_degree_centrality": nx.out_degree_centrality(self.network),
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
def identify_sensor_roles(self) -> dict[str, list[str]]:
|
|
138
|
+
"""
|
|
139
|
+
Identify different roles of sensors in the network.
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
Dictionary categorizing sensors by their network roles
|
|
143
|
+
"""
|
|
144
|
+
if self.network is None:
|
|
145
|
+
raise ValueError("Network must be built first")
|
|
146
|
+
|
|
147
|
+
# Calculate centrality measures
|
|
148
|
+
in_degree = dict(self.network.in_degree())
|
|
149
|
+
out_degree = dict(self.network.out_degree())
|
|
150
|
+
|
|
151
|
+
roles: dict[str, list[str]] = {
|
|
152
|
+
"triggers": [], # High out-degree, low in-degree
|
|
153
|
+
"responders": [], # High in-degree, low out-degree
|
|
154
|
+
"hubs": [], # High both in and out degree
|
|
155
|
+
"isolated": [], # Low both in and out degree
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
if self.network.number_of_nodes() == 0:
|
|
159
|
+
return roles
|
|
160
|
+
if all(
|
|
161
|
+
degree == 0
|
|
162
|
+
for degree in list(in_degree.values()) + list(out_degree.values())
|
|
163
|
+
):
|
|
164
|
+
roles["isolated"] = list(self.network.nodes())
|
|
165
|
+
return roles
|
|
166
|
+
|
|
167
|
+
# Define thresholds (can be adjusted)
|
|
168
|
+
high_threshold = np.percentile(
|
|
169
|
+
list(in_degree.values()) + list(out_degree.values()), 75
|
|
170
|
+
)
|
|
171
|
+
low_threshold = np.percentile(
|
|
172
|
+
list(in_degree.values()) + list(out_degree.values()), 25
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
for node in self.network.nodes():
|
|
176
|
+
in_deg = in_degree[node]
|
|
177
|
+
out_deg = out_degree[node]
|
|
178
|
+
|
|
179
|
+
if out_deg >= high_threshold and in_deg <= low_threshold:
|
|
180
|
+
roles["triggers"].append(node)
|
|
181
|
+
elif in_deg >= high_threshold and out_deg <= low_threshold:
|
|
182
|
+
roles["responders"].append(node)
|
|
183
|
+
elif in_deg >= high_threshold and out_deg >= high_threshold:
|
|
184
|
+
roles["hubs"].append(node)
|
|
185
|
+
else:
|
|
186
|
+
roles["isolated"].append(node)
|
|
187
|
+
|
|
188
|
+
return roles
|
|
189
|
+
|
|
190
|
+
def detect_communities(self) -> list[list[str]]:
|
|
191
|
+
"""Detect communities/clusters in the sensor network."""
|
|
192
|
+
if self.network is None:
|
|
193
|
+
raise ValueError("Network must be built first")
|
|
194
|
+
|
|
195
|
+
if self.network.number_of_nodes() == 0:
|
|
196
|
+
return []
|
|
197
|
+
if self.network.number_of_edges() == 0:
|
|
198
|
+
return [[node] for node in self.network.nodes()]
|
|
199
|
+
|
|
200
|
+
try:
|
|
201
|
+
# Convert to undirected for community detection
|
|
202
|
+
undirected_network = self.network.to_undirected()
|
|
203
|
+
|
|
204
|
+
# Use greedy modularity optimization
|
|
205
|
+
communities = nx.community.greedy_modularity_communities(undirected_network)
|
|
206
|
+
return [list(community) for community in communities]
|
|
207
|
+
|
|
208
|
+
except (nx.NetworkXException, ValueError) as exc:
|
|
209
|
+
logger.warning("Community detection failed: %s", exc)
|
|
210
|
+
return []
|
|
211
|
+
|
|
212
|
+
def calculate_mutual_information(self, data: pd.DataFrame) -> pd.DataFrame:
|
|
213
|
+
"""Calculate mutual information between all sensor pairs."""
|
|
214
|
+
sensor_data = self._prepare_sensor_data(data)
|
|
215
|
+
sensors = sensor_data.columns.tolist()
|
|
216
|
+
n_sensors = len(sensors)
|
|
217
|
+
mi_matrix = np.zeros((n_sensors, n_sensors))
|
|
218
|
+
|
|
219
|
+
for i, sensor1 in enumerate(sensors):
|
|
220
|
+
for j, sensor2 in enumerate(sensors):
|
|
221
|
+
if i != j:
|
|
222
|
+
mi_matrix[i, j] = mutual_info_score(
|
|
223
|
+
sensor_data[sensor1].values, sensor_data[sensor2].values
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
return pd.DataFrame(mi_matrix, index=sensors, columns=sensors)
|
|
227
|
+
|
|
228
|
+
def find_critical_sensors(self) -> dict[str, Any]:
|
|
229
|
+
"""
|
|
230
|
+
Identify critical sensors whose removal would significantly affect network connectivity.
|
|
231
|
+
|
|
232
|
+
Returns:
|
|
233
|
+
Dictionary with criticality analysis
|
|
234
|
+
"""
|
|
235
|
+
if self.network is None:
|
|
236
|
+
raise ValueError("Network must be built first")
|
|
237
|
+
|
|
238
|
+
original_components = nx.number_weakly_connected_components(self.network)
|
|
239
|
+
|
|
240
|
+
criticality_scores: dict[str, dict[str, float]] = {}
|
|
241
|
+
|
|
242
|
+
for node in self.network.nodes():
|
|
243
|
+
# Create network without this node
|
|
244
|
+
temp_network = self.network.copy()
|
|
245
|
+
temp_network.remove_node(node)
|
|
246
|
+
|
|
247
|
+
# Calculate impact on connectivity
|
|
248
|
+
new_components = nx.number_weakly_connected_components(temp_network)
|
|
249
|
+
component_change = new_components - original_components
|
|
250
|
+
|
|
251
|
+
# Calculate impact on edges
|
|
252
|
+
edges_lost = len(self.network.edges()) - len(temp_network.edges())
|
|
253
|
+
|
|
254
|
+
# Criticality score combines connectivity and edge impact
|
|
255
|
+
criticality_scores[node] = {
|
|
256
|
+
"component_change": component_change,
|
|
257
|
+
"edges_lost": edges_lost,
|
|
258
|
+
"criticality_score": component_change + 0.1 * edges_lost,
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
# Sort by criticality
|
|
262
|
+
sorted_sensors = sorted(
|
|
263
|
+
criticality_scores.items(),
|
|
264
|
+
key=lambda x: x[1]["criticality_score"],
|
|
265
|
+
reverse=True,
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
return {
|
|
269
|
+
"criticality_rankings": sorted_sensors,
|
|
270
|
+
"most_critical": sorted_sensors[0][0] if sorted_sensors else None,
|
|
271
|
+
"original_components": original_components,
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
def plot_network(
|
|
275
|
+
self,
|
|
276
|
+
figsize: tuple[int, int] = (12, 8),
|
|
277
|
+
node_size_factor: int = 500,
|
|
278
|
+
*,
|
|
279
|
+
show: bool = True,
|
|
280
|
+
):
|
|
281
|
+
"""
|
|
282
|
+
Plot the sensor dependency network.
|
|
283
|
+
|
|
284
|
+
Args:
|
|
285
|
+
figsize: Figure size
|
|
286
|
+
node_size_factor: Factor to scale node sizes
|
|
287
|
+
show: Whether to display the figure immediately
|
|
288
|
+
"""
|
|
289
|
+
if self.network is None:
|
|
290
|
+
raise ValueError("Network must be built first")
|
|
291
|
+
|
|
292
|
+
fig, ax = plt.subplots(figsize=figsize)
|
|
293
|
+
|
|
294
|
+
# Calculate node sizes based on degree centrality
|
|
295
|
+
centrality = nx.degree_centrality(self.network.to_undirected())
|
|
296
|
+
node_sizes = [
|
|
297
|
+
centrality[node] * node_size_factor + 100 for node in self.network.nodes()
|
|
298
|
+
]
|
|
299
|
+
|
|
300
|
+
# Calculate edge weights for visualization
|
|
301
|
+
edge_weights = [
|
|
302
|
+
self.network[u][v]["weight"] / 10 for u, v in self.network.edges()
|
|
303
|
+
]
|
|
304
|
+
|
|
305
|
+
# Use spring layout for positioning
|
|
306
|
+
pos = nx.spring_layout(self.network, k=1, iterations=50)
|
|
307
|
+
|
|
308
|
+
# Draw network
|
|
309
|
+
nx.draw_networkx_nodes(
|
|
310
|
+
self.network,
|
|
311
|
+
pos,
|
|
312
|
+
node_size=node_sizes,
|
|
313
|
+
node_color="lightblue",
|
|
314
|
+
alpha=0.7,
|
|
315
|
+
ax=ax,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
nx.draw_networkx_edges(
|
|
319
|
+
self.network,
|
|
320
|
+
pos,
|
|
321
|
+
width=edge_weights,
|
|
322
|
+
edge_color="gray",
|
|
323
|
+
arrows=True,
|
|
324
|
+
arrowsize=20,
|
|
325
|
+
alpha=0.6,
|
|
326
|
+
ax=ax,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
nx.draw_networkx_labels(self.network, pos, font_size=10, ax=ax)
|
|
330
|
+
|
|
331
|
+
ax.set_title("Sensor Dependency Network\n(Arrows show causal direction)")
|
|
332
|
+
ax.axis("off")
|
|
333
|
+
fig.tight_layout()
|
|
334
|
+
if show:
|
|
335
|
+
plt.show()
|
|
336
|
+
return fig
|
|
337
|
+
|
|
338
|
+
def plot_causality_matrix(
|
|
339
|
+
self, figsize: tuple[int, int] = (10, 8), *, show: bool = True
|
|
340
|
+
):
|
|
341
|
+
"""
|
|
342
|
+
Plot heatmap of causality test results.
|
|
343
|
+
|
|
344
|
+
Args:
|
|
345
|
+
figsize: Figure size
|
|
346
|
+
show: Whether to display the figure immediately
|
|
347
|
+
"""
|
|
348
|
+
if self.causality_results is None:
|
|
349
|
+
raise ValueError("Network must be built first")
|
|
350
|
+
|
|
351
|
+
# Create pivot table for heatmap
|
|
352
|
+
pivot_data = self.causality_results.pivot(
|
|
353
|
+
index="effect", columns="cause", values="test_statistic"
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
fig, ax = plt.subplots(figsize=figsize)
|
|
357
|
+
|
|
358
|
+
# Create heatmap
|
|
359
|
+
mask = pivot_data.isna()
|
|
360
|
+
sns.heatmap(
|
|
361
|
+
pivot_data,
|
|
362
|
+
annot=True,
|
|
363
|
+
fmt=".2f",
|
|
364
|
+
mask=mask,
|
|
365
|
+
cmap="Reds",
|
|
366
|
+
cbar_kws={"label": "Granger Causality Test Statistic"},
|
|
367
|
+
ax=ax,
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
ax.set_title("Sensor Causality Matrix\n(Rows: Effects, Columns: Causes)")
|
|
371
|
+
ax.set_xlabel("Potential Causes")
|
|
372
|
+
ax.set_ylabel("Effects")
|
|
373
|
+
fig.tight_layout()
|
|
374
|
+
if show:
|
|
375
|
+
plt.show()
|
|
376
|
+
return fig
|
|
377
|
+
|
|
378
|
+
def export_network_data(self, filename: str | Path | None = None) -> dict[str, Any]:
|
|
379
|
+
"""
|
|
380
|
+
Export network data for external analysis.
|
|
381
|
+
|
|
382
|
+
Args:
|
|
383
|
+
filename: Optional filename to save data
|
|
384
|
+
|
|
385
|
+
Returns:
|
|
386
|
+
Dictionary with all network data
|
|
387
|
+
"""
|
|
388
|
+
if self.network is None:
|
|
389
|
+
raise ValueError("Network must be built first")
|
|
390
|
+
|
|
391
|
+
export_data = {
|
|
392
|
+
"nodes": list(self.network.nodes()),
|
|
393
|
+
"edges": [(u, v, self.network[u][v]) for u, v in self.network.edges()],
|
|
394
|
+
"network_statistics": self.get_network_statistics(),
|
|
395
|
+
"sensor_roles": self.identify_sensor_roles(),
|
|
396
|
+
"communities": self.detect_communities(),
|
|
397
|
+
"causality_results": self.causality_results.to_dict("records"),
|
|
398
|
+
"critical_sensors": self.find_critical_sensors(),
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
if filename:
|
|
402
|
+
import json
|
|
403
|
+
|
|
404
|
+
path = Path(filename)
|
|
405
|
+
path.write_text(json.dumps(export_data, indent=2, default=str))
|
|
406
|
+
logger.info("Network data exported to %s", path)
|
|
407
|
+
|
|
408
|
+
return export_data
|