sensor-modeling 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. sensor_modeling/__init__.py +45 -0
  2. sensor_modeling/alerts/__init__.py +26 -0
  3. sensor_modeling/alerts/alert.py +532 -0
  4. sensor_modeling/analysis/__init__.py +43 -0
  5. sensor_modeling/analysis/_frame.py +19 -0
  6. sensor_modeling/analysis/behavioral_analysis.py +57 -0
  7. sensor_modeling/analysis/behavioral_metrics.py +66 -0
  8. sensor_modeling/analysis/comparison.py +164 -0
  9. sensor_modeling/analysis/dependency_network.py +408 -0
  10. sensor_modeling/analysis/granger_causality.py +314 -0
  11. sensor_modeling/analysis/pipeline.py +168 -0
  12. sensor_modeling/analysis/reporting.py +109 -0
  13. sensor_modeling/baseline/__init__.py +30 -0
  14. sensor_modeling/baseline/adaptive.py +520 -0
  15. sensor_modeling/baseline/features.py +224 -0
  16. sensor_modeling/change_point/__init__.py +13 -0
  17. sensor_modeling/change_point/_validation.py +31 -0
  18. sensor_modeling/change_point/adaptive_normalization.py +55 -0
  19. sensor_modeling/change_point/embedding_cpd.py +60 -0
  20. sensor_modeling/change_point/energy_efficient.py +57 -0
  21. sensor_modeling/change_point/genetic_optimization.py +65 -0
  22. sensor_modeling/cli.py +416 -0
  23. sensor_modeling/context/__init__.py +33 -0
  24. sensor_modeling/context/occupancy.py +529 -0
  25. sensor_modeling/data/__init__.py +5 -0
  26. sensor_modeling/data/loaders.py +146 -0
  27. sensor_modeling/data/preprocessing.py +83 -0
  28. sensor_modeling/data/synthetic.py +121 -0
  29. sensor_modeling/data/validation.py +81 -0
  30. sensor_modeling/evaluation/__init__.py +92 -0
  31. sensor_modeling/evaluation/ablation.py +303 -0
  32. sensor_modeling/evaluation/attribution.py +474 -0
  33. sensor_modeling/evaluation/detection.py +297 -0
  34. sensor_modeling/evaluation/metrics.py +541 -0
  35. sensor_modeling/evaluation/provenance.py +309 -0
  36. sensor_modeling/examples/__init__.py +1 -0
  37. sensor_modeling/examples/demos/__init__.py +1 -0
  38. sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
  39. sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
  40. sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
  41. sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
  42. sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
  43. sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
  44. sensor_modeling/examples/tutorials/__init__.py +1 -0
  45. sensor_modeling/fusion/__init__.py +46 -0
  46. sensor_modeling/fusion/defaults.py +296 -0
  47. sensor_modeling/fusion/emissions.py +339 -0
  48. sensor_modeling/fusion/estimate.py +375 -0
  49. sensor_modeling/fusion/filter.py +323 -0
  50. sensor_modeling/health/__init__.py +31 -0
  51. sensor_modeling/health/monitor.py +590 -0
  52. sensor_modeling/health/status.py +74 -0
  53. sensor_modeling/hmm/__init__.py +15 -0
  54. sensor_modeling/hmm/adaptive_hmm.py +22 -0
  55. sensor_modeling/hmm/base.py +134 -0
  56. sensor_modeling/hmm/circadian_hmm.py +22 -0
  57. sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
  58. sensor_modeling/hmm/hierarchical_hmm.py +35 -0
  59. sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
  60. sensor_modeling/interop/__init__.py +57 -0
  61. sensor_modeling/interop/fhir.py +418 -0
  62. sensor_modeling/interop/privacy.py +308 -0
  63. sensor_modeling/models/__init__.py +12 -0
  64. sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
  65. sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
  66. sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
  67. sensor_modeling/models/change_point_detection/__init__.py +10 -0
  68. sensor_modeling/models/change_point_detection/deep.py +65 -0
  69. sensor_modeling/models/change_point_detection/pelt.py +159 -0
  70. sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
  71. sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
  72. sensor_modeling/models/nhpp_pelt/cli.py +243 -0
  73. sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
  74. sensor_modeling/models/nhpp_pelt/io.py +58 -0
  75. sensor_modeling/models/nhpp_pelt/model.py +408 -0
  76. sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
  77. sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
  78. sensor_modeling/models/nhpp_pelt/quad.py +72 -0
  79. sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
  80. sensor_modeling/models/nhpp_pelt/utils.py +174 -0
  81. sensor_modeling/observations/__init__.py +59 -0
  82. sensor_modeling/observations/adapters.py +195 -0
  83. sensor_modeling/observations/ingest.py +269 -0
  84. sensor_modeling/observations/observation.py +270 -0
  85. sensor_modeling/observations/registry.py +262 -0
  86. sensor_modeling/observations/stream.py +342 -0
  87. sensor_modeling/observations/types.py +107 -0
  88. sensor_modeling/observations/units.py +117 -0
  89. sensor_modeling/online/__init__.py +36 -0
  90. sensor_modeling/online/benchmarks.py +242 -0
  91. sensor_modeling/online/pipeline.py +485 -0
  92. sensor_modeling/simulation/__init__.py +54 -0
  93. sensor_modeling/simulation/faults.py +191 -0
  94. sensor_modeling/simulation/household.py +862 -0
  95. sensor_modeling/states/__init__.py +23 -0
  96. sensor_modeling/states/markov.py +105 -0
  97. sensor_modeling/states/ontology.py +238 -0
  98. sensor_modeling/utils/__init__.py +41 -0
  99. sensor_modeling/utils/data_io.py +199 -0
  100. sensor_modeling/utils/logging_config.py +10 -0
  101. sensor_modeling/utils/missing.py +188 -0
  102. sensor_modeling/utils/plotting.py +98 -0
  103. sensor_modeling/utils/validation.py +117 -0
  104. sensor_modeling/visualization/__init__.py +3 -0
  105. sensor_modeling/visualization/clinical.py +67 -0
  106. sensor_modeling/visualization/interactive.py +208 -0
  107. sensor_modeling/visualization/research.py +60 -0
  108. sensor_modeling/visualization/web_app.py +137 -0
  109. sensor_modeling-0.2.0.dist-info/METADATA +683 -0
  110. sensor_modeling-0.2.0.dist-info/RECORD +114 -0
  111. sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
  112. sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
  113. sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
  114. sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,57 @@
1
+ """High-level behavioral analysis utilities."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+
7
+ import numpy as np
8
+ import pandas as pd
9
+
10
+ from ._frame import prepare_sensor_frame
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+
15
+ # ---------------------------------------------------------------------------
16
+ def recognize_activity_patterns(data: pd.DataFrame) -> dict[str, int]:
17
+ """Identify peak and quiet hours of activity."""
18
+ sensor_data = prepare_sensor_frame(data, context="recognize_activity_patterns")
19
+ hourly = sensor_data.groupby(sensor_data.index.hour).sum()
20
+ totals = hourly.sum(axis=1)
21
+ return {
22
+ "peak_hours": int(totals.idxmax()),
23
+ "quiet_hours": int(totals.idxmin()),
24
+ }
25
+
26
+
27
+ # ---------------------------------------------------------------------------
28
+ def score_anomalies(data: pd.DataFrame) -> pd.Series:
29
+ """Simple z-score based anomaly metric for each timestamp."""
30
+ sensor_data = prepare_sensor_frame(data, context="score_anomalies")
31
+ std = sensor_data.std(ddof=0).replace(0, np.nan)
32
+ zscores = (sensor_data - sensor_data.mean()) / std
33
+ return zscores.fillna(0.0).abs().sum(axis=1)
34
+
35
+
36
+ # ---------------------------------------------------------------------------
37
+ def detect_trends(data: pd.DataFrame, window: int = 24) -> pd.DataFrame:
38
+ """Rolling mean trend indicator."""
39
+ if window < 1:
40
+ raise ValueError("window must be at least 1")
41
+ sensor_data = prepare_sensor_frame(data, context="detect_trends")
42
+ return sensor_data.rolling(window=window, min_periods=1).mean()
43
+
44
+
45
+ # ---------------------------------------------------------------------------
46
+ def health_indicators(data: pd.DataFrame) -> dict[str, float]:
47
+ """Basic health status indicators derived from activity levels."""
48
+ sensor_data = prepare_sensor_frame(data, context="health_indicators")
49
+ activity = sensor_data.sum(axis=1)
50
+ overall = float(activity.mean())
51
+ variability = float(activity.std(ddof=0))
52
+ sedentary_ratio = float((activity == 0).mean())
53
+ return {
54
+ "overall_activity": overall,
55
+ "activity_variability": variability,
56
+ "sedentary_ratio": sedentary_ratio,
57
+ }
@@ -0,0 +1,66 @@
1
+ """Behavioral metrics calculated from sensor datasets."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+
7
+ import pandas as pd
8
+
9
+ from ._frame import prepare_sensor_frame
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ def calculate_behavioral_metrics(data: pd.DataFrame) -> dict[str, object]:
15
+ """Calculate basic behavioral pattern metrics from sensor data."""
16
+ sensor_data = prepare_sensor_frame(data, context="calculate_behavioral_metrics")
17
+ metrics: dict[str, object] = {}
18
+
19
+ total_activations = sensor_data.sum().sum()
20
+ total_possible = len(sensor_data) * len(sensor_data.columns)
21
+ metrics["overall_activity_rate"] = float(total_activations / total_possible)
22
+ metrics["total_activations"] = int(total_activations)
23
+
24
+ hourly_activity = sensor_data.groupby(sensor_data.index.hour).sum()
25
+ metrics["peak_activity_hour"] = int(hourly_activity.sum(axis=1).idxmax())
26
+ metrics["quietest_hour"] = int(hourly_activity.sum(axis=1).idxmin())
27
+
28
+ sensor_metrics: dict[str, dict[str, float | int]] = {}
29
+ for sensor in sensor_data.columns:
30
+ series = sensor_data[sensor]
31
+ daily_variance = series.groupby(series.index.date).sum().var()
32
+ sensor_metrics[sensor] = {
33
+ "activation_rate": float(series.mean()),
34
+ "total_activations": int(series.sum()),
35
+ "longest_inactive_period": _find_longest_streak(series, 0),
36
+ "longest_active_period": _find_longest_streak(series, 1),
37
+ "daily_variance": 0.0 if pd.isna(daily_variance) else float(daily_variance),
38
+ }
39
+ metrics["sensor_metrics"] = sensor_metrics
40
+
41
+ dow_activity = sensor_data.groupby(sensor_data.index.dayofweek).sum()
42
+ weekday_names = [
43
+ "Monday",
44
+ "Tuesday",
45
+ "Wednesday",
46
+ "Thursday",
47
+ "Friday",
48
+ "Saturday",
49
+ "Sunday",
50
+ ]
51
+ metrics["most_active_day"] = weekday_names[dow_activity.sum(axis=1).idxmax()]
52
+ metrics["least_active_day"] = weekday_names[dow_activity.sum(axis=1).idxmin()]
53
+
54
+ return metrics
55
+
56
+
57
+ def _find_longest_streak(series: pd.Series, value: int) -> int:
58
+ max_streak = 0
59
+ current_streak = 0
60
+ for val in series:
61
+ if val == value:
62
+ current_streak += 1
63
+ max_streak = max(max_streak, current_streak)
64
+ else:
65
+ current_streak = 0
66
+ return max_streak
@@ -0,0 +1,164 @@
1
+ """Utilities for comparing different sensor models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections.abc import Callable
7
+ from typing import Any
8
+
9
+ import matplotlib.pyplot as plt
10
+ import numpy as np
11
+ import pandas as pd
12
+ from scipy.stats import ttest_rel
13
+ from sklearn.model_selection import TimeSeriesSplit
14
+
15
+ from ..utils.data_io import SensorDataset
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+ ModelScorer = Callable[[Any, pd.DataFrame, pd.DataFrame], float]
20
+ FoldError = (AttributeError, FloatingPointError, RuntimeError, TypeError, ValueError)
21
+
22
+
23
+ def time_series_splits(
24
+ data: SensorDataset | pd.DataFrame, n_splits: int = 3
25
+ ) -> list[tuple[np.ndarray, np.ndarray]]:
26
+ """Return chronological train/test splits for sensor time series."""
27
+ if n_splits < 1:
28
+ raise ValueError("n_splits must be at least 1")
29
+ dataset = data if isinstance(data, SensorDataset) else SensorDataset(data)
30
+ df = dataset.to_dataframe()
31
+ if len(df) < 2:
32
+ raise ValueError("At least two observations are required for cross-validation")
33
+ split_count = min(n_splits, len(df) - 1)
34
+ splitter = TimeSeriesSplit(n_splits=split_count)
35
+ return list(splitter.split(df))
36
+
37
+
38
+ def _fit_and_score_model(
39
+ name: str,
40
+ model: Any,
41
+ train_df: pd.DataFrame,
42
+ test_df: pd.DataFrame,
43
+ scorers: dict[str, ModelScorer],
44
+ ) -> float:
45
+ """Fit one model on a fold and return its score."""
46
+ model.fit(train_df.values if name == "hmm" else train_df)
47
+ if name in scorers:
48
+ return float(scorers[name](model, train_df, test_df))
49
+ return float(model.score(test_df.values))
50
+
51
+
52
+ def _mean_score(values: list[float]) -> float:
53
+ """Return the mean of valid fold scores, or NaN when every fold failed."""
54
+ finite_scores = [score for score in values if not np.isnan(score)]
55
+ if not finite_scores:
56
+ return float("nan")
57
+ return float(np.mean(finite_scores))
58
+
59
+
60
+ # ---------------------------------------------------------------------------
61
+ def cross_validate(
62
+ models: dict[str, Any],
63
+ data: SensorDataset | pd.DataFrame,
64
+ n_splits: int = 3,
65
+ scorers: dict[str, ModelScorer] | None = None,
66
+ ) -> dict[str, float]:
67
+ """Perform chronological cross-validation across *models*.
68
+
69
+ Parameters
70
+ ----------
71
+ models : dict[str, Any]
72
+ Mapping from model name to model instance. Each model must implement a
73
+ :py:meth:`fit` method and either a :py:meth:`score` method or a scorer
74
+ supplied via ``scorers``.
75
+ data : SensorDataset | pd.DataFrame
76
+ Input dataset.
77
+ n_splits : int
78
+ Number of chronological folds.
79
+ scorers : dict[str, ModelScorer] | None
80
+ Optional per-model scoring functions accepting
81
+ ``(model, train_df, test_df)``.
82
+ """
83
+ dataset = data if isinstance(data, SensorDataset) else SensorDataset(data)
84
+ df = dataset.to_dataframe()
85
+ scorers = scorers or {}
86
+ unsupported = [
87
+ name
88
+ for name, model in models.items()
89
+ if name not in scorers and not hasattr(model, "score")
90
+ ]
91
+ if unsupported:
92
+ names = ", ".join(sorted(unsupported))
93
+ raise TypeError(
94
+ "Cross-validation requires each model to define score() or an "
95
+ f"explicit scorer. Missing scorer for: {names}"
96
+ )
97
+
98
+ scores: dict[str, list[float]] = {name: [] for name in models}
99
+ for train_idx, test_idx in time_series_splits(dataset, n_splits=n_splits):
100
+ train_df = df.iloc[train_idx]
101
+ test_df = df.iloc[test_idx]
102
+ for name, model in models.items():
103
+ try:
104
+ score = _fit_and_score_model(name, model, train_df, test_df, scorers)
105
+ scores[name].append(score)
106
+ except FoldError as exc:
107
+ logger.error("Failed fold for %s: %s", name, exc)
108
+ scores[name].append(float("nan"))
109
+ return {name: _mean_score(vals) for name, vals in scores.items()}
110
+
111
+
112
+ # ---------------------------------------------------------------------------
113
+ def significance_test(scores_a: list[float], scores_b: list[float]) -> float:
114
+ """Paired t-test returning the p-value."""
115
+ values_a = np.asarray(scores_a, dtype=float)
116
+ values_b = np.asarray(scores_b, dtype=float)
117
+ if values_a.shape != values_b.shape:
118
+ raise ValueError("scores_a and scores_b must have the same length")
119
+
120
+ valid_pairs = np.isfinite(values_a) & np.isfinite(values_b)
121
+ if np.count_nonzero(valid_pairs) < 2:
122
+ raise ValueError("at least two paired finite scores are required")
123
+
124
+ differences = values_a[valid_pairs] - values_b[valid_pairs]
125
+ if np.allclose(differences, 0.0):
126
+ return 1.0
127
+
128
+ _, pvalue = ttest_rel(values_a[valid_pairs], values_b[valid_pairs])
129
+ if not np.isfinite(pvalue):
130
+ return 1.0
131
+ return float(pvalue)
132
+
133
+
134
+ # ---------------------------------------------------------------------------
135
+ def standardize_metrics(metrics: dict[str, float]) -> dict[str, float]:
136
+ """Scale metric values to the [0,1] range."""
137
+ if not metrics:
138
+ return {}
139
+
140
+ vals = np.array(list(metrics.values()), dtype=float)
141
+ finite_vals = vals[np.isfinite(vals)]
142
+ if finite_vals.size == 0:
143
+ return {name: float("nan") for name in metrics}
144
+
145
+ vmin = np.min(finite_vals)
146
+ vmax = np.max(finite_vals)
147
+ rng = vmax - vmin if vmax != vmin else 1.0
148
+ return {
149
+ name: float("nan") if not np.isfinite(value) else float((value - vmin) / rng)
150
+ for name, value in zip(metrics, vals)
151
+ }
152
+
153
+
154
+ # ---------------------------------------------------------------------------
155
+ def visualize_comparison(metrics: dict[str, float], ax=None):
156
+ """Visualize model comparison scores as a bar chart."""
157
+ if ax is None:
158
+ _, ax = plt.subplots()
159
+ names = list(metrics.keys())
160
+ vals = [metrics[n] for n in names]
161
+ ax.bar(names, vals)
162
+ ax.set_ylabel("Score")
163
+ ax.set_title("Model Comparison")
164
+ return ax
@@ -0,0 +1,408 @@
1
+ """
2
+ Sensor Dependency Network Analysis
3
+
4
+ This module builds and analyzes cross-sensor dependency networks
5
+ using Granger causality and network analysis techniques.
6
+ """
7
+
8
+ import logging
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ import matplotlib.pyplot as plt
13
+ import networkx as nx
14
+ import numpy as np
15
+ import pandas as pd
16
+ import seaborn as sns
17
+ from sklearn.metrics import mutual_info_score
18
+
19
+ from .granger_causality import GrangerCausalityTest
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+
24
+ class SensorDependencyNetwork:
25
+ """
26
+ Build and analyze cross-sensor dependency networks.
27
+ Creates directed graphs showing causal relationships between sensors.
28
+ """
29
+
30
+ def __init__(self, significance_level: float = 0.05):
31
+ """
32
+ Initialize dependency network builder.
33
+
34
+ Args:
35
+ significance_level: P-value threshold for including edges
36
+ """
37
+ if not 0 < significance_level < 1:
38
+ raise ValueError("significance_level must be between 0 and 1")
39
+ self.significance_level = significance_level
40
+ self.granger_test = GrangerCausalityTest()
41
+ self.network: nx.DiGraph | None = None
42
+ self.causality_results: pd.DataFrame | None = None
43
+
44
+ def build_network(self, data: pd.DataFrame) -> nx.DiGraph:
45
+ """
46
+ Build sensor dependency network using Granger causality.
47
+
48
+ Args:
49
+ data: DataFrame with sensor data
50
+
51
+ Returns:
52
+ NetworkX directed graph
53
+ """
54
+ logger.info("Building sensor dependency network...")
55
+ sensor_data = self._prepare_sensor_data(data)
56
+
57
+ # Test all pairs for Granger causality
58
+ self.causality_results = self.granger_test.test_all_pairs(sensor_data)
59
+
60
+ # Create directed graph
61
+ self.network = nx.DiGraph()
62
+
63
+ # Add all sensors as nodes
64
+ sensors = sensor_data.columns.tolist()
65
+ self.network.add_nodes_from(sensors)
66
+
67
+ # Add edges for significant causal relationships
68
+ significant_results = self.causality_results[
69
+ self.causality_results["causality_detected"]
70
+ ]
71
+
72
+ for _, row in significant_results.iterrows():
73
+ self.network.add_edge(
74
+ row["cause"],
75
+ row["effect"],
76
+ weight=row["test_statistic"],
77
+ p_value=row["p_value"],
78
+ lags=row["lags_used"],
79
+ )
80
+
81
+ logger.info(
82
+ "Network created with %d nodes and %d edges",
83
+ len(self.network.nodes),
84
+ len(self.network.edges),
85
+ )
86
+ return self.network
87
+
88
+ def _prepare_sensor_data(self, data: pd.DataFrame) -> pd.DataFrame:
89
+ """Return a numeric binary sensor frame for network analysis."""
90
+ if data.empty:
91
+ raise ValueError("dependency network analysis requires at least one row")
92
+
93
+ sensor_data = data.select_dtypes(include="number")
94
+ if sensor_data.empty:
95
+ raise ValueError(
96
+ "dependency network analysis requires at least one numeric sensor column"
97
+ )
98
+
99
+ if sensor_data.isna().any().any():
100
+ raise ValueError("dependency network analysis does not accept NaN values")
101
+
102
+ non_binary = [
103
+ column
104
+ for column in sensor_data.columns
105
+ if not set(sensor_data[column].unique()) <= {0, 1}
106
+ ]
107
+ if non_binary:
108
+ raise ValueError(
109
+ "dependency network analysis requires binary sensor columns: "
110
+ + ", ".join(map(str, non_binary))
111
+ )
112
+
113
+ return sensor_data.copy()
114
+
115
+ def get_network_statistics(self) -> dict[str, Any]:
116
+ """Get network topology statistics."""
117
+ if self.network is None:
118
+ raise ValueError("Network must be built first")
119
+
120
+ is_connected = (
121
+ nx.is_weakly_connected(self.network)
122
+ if self.network.number_of_nodes()
123
+ else False
124
+ )
125
+
126
+ return {
127
+ "num_nodes": len(self.network.nodes),
128
+ "num_edges": len(self.network.edges),
129
+ "density": nx.density(self.network),
130
+ "is_connected": is_connected,
131
+ "num_components": nx.number_weakly_connected_components(self.network),
132
+ "avg_clustering": nx.average_clustering(self.network.to_undirected()),
133
+ "in_degree_centrality": nx.in_degree_centrality(self.network),
134
+ "out_degree_centrality": nx.out_degree_centrality(self.network),
135
+ }
136
+
137
+ def identify_sensor_roles(self) -> dict[str, list[str]]:
138
+ """
139
+ Identify different roles of sensors in the network.
140
+
141
+ Returns:
142
+ Dictionary categorizing sensors by their network roles
143
+ """
144
+ if self.network is None:
145
+ raise ValueError("Network must be built first")
146
+
147
+ # Calculate centrality measures
148
+ in_degree = dict(self.network.in_degree())
149
+ out_degree = dict(self.network.out_degree())
150
+
151
+ roles: dict[str, list[str]] = {
152
+ "triggers": [], # High out-degree, low in-degree
153
+ "responders": [], # High in-degree, low out-degree
154
+ "hubs": [], # High both in and out degree
155
+ "isolated": [], # Low both in and out degree
156
+ }
157
+
158
+ if self.network.number_of_nodes() == 0:
159
+ return roles
160
+ if all(
161
+ degree == 0
162
+ for degree in list(in_degree.values()) + list(out_degree.values())
163
+ ):
164
+ roles["isolated"] = list(self.network.nodes())
165
+ return roles
166
+
167
+ # Define thresholds (can be adjusted)
168
+ high_threshold = np.percentile(
169
+ list(in_degree.values()) + list(out_degree.values()), 75
170
+ )
171
+ low_threshold = np.percentile(
172
+ list(in_degree.values()) + list(out_degree.values()), 25
173
+ )
174
+
175
+ for node in self.network.nodes():
176
+ in_deg = in_degree[node]
177
+ out_deg = out_degree[node]
178
+
179
+ if out_deg >= high_threshold and in_deg <= low_threshold:
180
+ roles["triggers"].append(node)
181
+ elif in_deg >= high_threshold and out_deg <= low_threshold:
182
+ roles["responders"].append(node)
183
+ elif in_deg >= high_threshold and out_deg >= high_threshold:
184
+ roles["hubs"].append(node)
185
+ else:
186
+ roles["isolated"].append(node)
187
+
188
+ return roles
189
+
190
+ def detect_communities(self) -> list[list[str]]:
191
+ """Detect communities/clusters in the sensor network."""
192
+ if self.network is None:
193
+ raise ValueError("Network must be built first")
194
+
195
+ if self.network.number_of_nodes() == 0:
196
+ return []
197
+ if self.network.number_of_edges() == 0:
198
+ return [[node] for node in self.network.nodes()]
199
+
200
+ try:
201
+ # Convert to undirected for community detection
202
+ undirected_network = self.network.to_undirected()
203
+
204
+ # Use greedy modularity optimization
205
+ communities = nx.community.greedy_modularity_communities(undirected_network)
206
+ return [list(community) for community in communities]
207
+
208
+ except (nx.NetworkXException, ValueError) as exc:
209
+ logger.warning("Community detection failed: %s", exc)
210
+ return []
211
+
212
+ def calculate_mutual_information(self, data: pd.DataFrame) -> pd.DataFrame:
213
+ """Calculate mutual information between all sensor pairs."""
214
+ sensor_data = self._prepare_sensor_data(data)
215
+ sensors = sensor_data.columns.tolist()
216
+ n_sensors = len(sensors)
217
+ mi_matrix = np.zeros((n_sensors, n_sensors))
218
+
219
+ for i, sensor1 in enumerate(sensors):
220
+ for j, sensor2 in enumerate(sensors):
221
+ if i != j:
222
+ mi_matrix[i, j] = mutual_info_score(
223
+ sensor_data[sensor1].values, sensor_data[sensor2].values
224
+ )
225
+
226
+ return pd.DataFrame(mi_matrix, index=sensors, columns=sensors)
227
+
228
+ def find_critical_sensors(self) -> dict[str, Any]:
229
+ """
230
+ Identify critical sensors whose removal would significantly affect network connectivity.
231
+
232
+ Returns:
233
+ Dictionary with criticality analysis
234
+ """
235
+ if self.network is None:
236
+ raise ValueError("Network must be built first")
237
+
238
+ original_components = nx.number_weakly_connected_components(self.network)
239
+
240
+ criticality_scores: dict[str, dict[str, float]] = {}
241
+
242
+ for node in self.network.nodes():
243
+ # Create network without this node
244
+ temp_network = self.network.copy()
245
+ temp_network.remove_node(node)
246
+
247
+ # Calculate impact on connectivity
248
+ new_components = nx.number_weakly_connected_components(temp_network)
249
+ component_change = new_components - original_components
250
+
251
+ # Calculate impact on edges
252
+ edges_lost = len(self.network.edges()) - len(temp_network.edges())
253
+
254
+ # Criticality score combines connectivity and edge impact
255
+ criticality_scores[node] = {
256
+ "component_change": component_change,
257
+ "edges_lost": edges_lost,
258
+ "criticality_score": component_change + 0.1 * edges_lost,
259
+ }
260
+
261
+ # Sort by criticality
262
+ sorted_sensors = sorted(
263
+ criticality_scores.items(),
264
+ key=lambda x: x[1]["criticality_score"],
265
+ reverse=True,
266
+ )
267
+
268
+ return {
269
+ "criticality_rankings": sorted_sensors,
270
+ "most_critical": sorted_sensors[0][0] if sorted_sensors else None,
271
+ "original_components": original_components,
272
+ }
273
+
274
+ def plot_network(
275
+ self,
276
+ figsize: tuple[int, int] = (12, 8),
277
+ node_size_factor: int = 500,
278
+ *,
279
+ show: bool = True,
280
+ ):
281
+ """
282
+ Plot the sensor dependency network.
283
+
284
+ Args:
285
+ figsize: Figure size
286
+ node_size_factor: Factor to scale node sizes
287
+ show: Whether to display the figure immediately
288
+ """
289
+ if self.network is None:
290
+ raise ValueError("Network must be built first")
291
+
292
+ fig, ax = plt.subplots(figsize=figsize)
293
+
294
+ # Calculate node sizes based on degree centrality
295
+ centrality = nx.degree_centrality(self.network.to_undirected())
296
+ node_sizes = [
297
+ centrality[node] * node_size_factor + 100 for node in self.network.nodes()
298
+ ]
299
+
300
+ # Calculate edge weights for visualization
301
+ edge_weights = [
302
+ self.network[u][v]["weight"] / 10 for u, v in self.network.edges()
303
+ ]
304
+
305
+ # Use spring layout for positioning
306
+ pos = nx.spring_layout(self.network, k=1, iterations=50)
307
+
308
+ # Draw network
309
+ nx.draw_networkx_nodes(
310
+ self.network,
311
+ pos,
312
+ node_size=node_sizes,
313
+ node_color="lightblue",
314
+ alpha=0.7,
315
+ ax=ax,
316
+ )
317
+
318
+ nx.draw_networkx_edges(
319
+ self.network,
320
+ pos,
321
+ width=edge_weights,
322
+ edge_color="gray",
323
+ arrows=True,
324
+ arrowsize=20,
325
+ alpha=0.6,
326
+ ax=ax,
327
+ )
328
+
329
+ nx.draw_networkx_labels(self.network, pos, font_size=10, ax=ax)
330
+
331
+ ax.set_title("Sensor Dependency Network\n(Arrows show causal direction)")
332
+ ax.axis("off")
333
+ fig.tight_layout()
334
+ if show:
335
+ plt.show()
336
+ return fig
337
+
338
+ def plot_causality_matrix(
339
+ self, figsize: tuple[int, int] = (10, 8), *, show: bool = True
340
+ ):
341
+ """
342
+ Plot heatmap of causality test results.
343
+
344
+ Args:
345
+ figsize: Figure size
346
+ show: Whether to display the figure immediately
347
+ """
348
+ if self.causality_results is None:
349
+ raise ValueError("Network must be built first")
350
+
351
+ # Create pivot table for heatmap
352
+ pivot_data = self.causality_results.pivot(
353
+ index="effect", columns="cause", values="test_statistic"
354
+ )
355
+
356
+ fig, ax = plt.subplots(figsize=figsize)
357
+
358
+ # Create heatmap
359
+ mask = pivot_data.isna()
360
+ sns.heatmap(
361
+ pivot_data,
362
+ annot=True,
363
+ fmt=".2f",
364
+ mask=mask,
365
+ cmap="Reds",
366
+ cbar_kws={"label": "Granger Causality Test Statistic"},
367
+ ax=ax,
368
+ )
369
+
370
+ ax.set_title("Sensor Causality Matrix\n(Rows: Effects, Columns: Causes)")
371
+ ax.set_xlabel("Potential Causes")
372
+ ax.set_ylabel("Effects")
373
+ fig.tight_layout()
374
+ if show:
375
+ plt.show()
376
+ return fig
377
+
378
+ def export_network_data(self, filename: str | Path | None = None) -> dict[str, Any]:
379
+ """
380
+ Export network data for external analysis.
381
+
382
+ Args:
383
+ filename: Optional filename to save data
384
+
385
+ Returns:
386
+ Dictionary with all network data
387
+ """
388
+ if self.network is None:
389
+ raise ValueError("Network must be built first")
390
+
391
+ export_data = {
392
+ "nodes": list(self.network.nodes()),
393
+ "edges": [(u, v, self.network[u][v]) for u, v in self.network.edges()],
394
+ "network_statistics": self.get_network_statistics(),
395
+ "sensor_roles": self.identify_sensor_roles(),
396
+ "communities": self.detect_communities(),
397
+ "causality_results": self.causality_results.to_dict("records"),
398
+ "critical_sensors": self.find_critical_sensors(),
399
+ }
400
+
401
+ if filename:
402
+ import json
403
+
404
+ path = Path(filename)
405
+ path.write_text(json.dumps(export_data, indent=2, default=str))
406
+ logger.info("Network data exported to %s", path)
407
+
408
+ return export_data