sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Data cleaning and preparation routines for sensor data."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from sensor_modeling.utils.data_io import SensorDataset
|
|
11
|
+
from sensor_modeling.utils.missing import handle_missing_data
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def detect_missing(dataset: SensorDataset) -> pd.Series:
|
|
17
|
+
"""Return the fraction of missing values per sensor."""
|
|
18
|
+
df = dataset.to_dataframe()
|
|
19
|
+
miss = df.isna().mean()
|
|
20
|
+
logger.debug("Missing value ratios: %s", miss.to_dict())
|
|
21
|
+
return miss
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def impute_missing(dataset: SensorDataset, strategy: str = "ffill") -> SensorDataset:
|
|
25
|
+
"""Impute missing values using the specified strategy."""
|
|
26
|
+
df = dataset.to_dataframe().copy()
|
|
27
|
+
if strategy == "ffill":
|
|
28
|
+
df = handle_missing_data(df, strategy="gap_aware").data
|
|
29
|
+
elif strategy == "interpolate":
|
|
30
|
+
df = handle_missing_data(df, strategy="interpolate").data
|
|
31
|
+
elif strategy == "mean":
|
|
32
|
+
numeric_means = df.select_dtypes(include="number").mean()
|
|
33
|
+
df = df.fillna(numeric_means)
|
|
34
|
+
else:
|
|
35
|
+
raise ValueError(f"Unsupported imputation strategy: {strategy}")
|
|
36
|
+
logger.info("Imputed missing values using %s strategy", strategy)
|
|
37
|
+
return SensorDataset(df)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def detect_outliers(dataset: SensorDataset, z_thresh: float = 3.0) -> pd.DataFrame:
|
|
41
|
+
"""Identify outlier readings using a z-score threshold."""
|
|
42
|
+
if z_thresh <= 0:
|
|
43
|
+
raise ValueError("z_thresh must be positive")
|
|
44
|
+
|
|
45
|
+
df = dataset.to_dataframe()
|
|
46
|
+
outliers = pd.DataFrame(False, index=df.index, columns=df.columns)
|
|
47
|
+
numeric = df.select_dtypes(include="number")
|
|
48
|
+
if not numeric.empty:
|
|
49
|
+
std = numeric.std(ddof=0).replace(0, np.nan)
|
|
50
|
+
z = (numeric - numeric.mean()) / std
|
|
51
|
+
outliers.loc[:, numeric.columns] = (np.abs(z) > z_thresh).fillna(False)
|
|
52
|
+
logger.debug("Outlier counts per sensor: %s", outliers.sum().to_dict())
|
|
53
|
+
return outliers
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def align_sensors(
|
|
57
|
+
datasets: list[SensorDataset], freq: str = "1min"
|
|
58
|
+
) -> list[SensorDataset]:
|
|
59
|
+
"""Temporal alignment across multiple sensors/datasets."""
|
|
60
|
+
if not datasets:
|
|
61
|
+
raise ValueError("No datasets provided for alignment")
|
|
62
|
+
target_index = pd.date_range(
|
|
63
|
+
start=min(ds.to_dataframe().index.min() for ds in datasets),
|
|
64
|
+
end=max(ds.to_dataframe().index.max() for ds in datasets),
|
|
65
|
+
freq=freq,
|
|
66
|
+
)
|
|
67
|
+
aligned = []
|
|
68
|
+
for ds in datasets:
|
|
69
|
+
df = ds.to_dataframe().reindex(target_index).interpolate()
|
|
70
|
+
aligned.append(SensorDataset(df))
|
|
71
|
+
logger.info("Aligned %d datasets to frequency %s", len(datasets), freq)
|
|
72
|
+
return aligned
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def data_quality_report(dataset: SensorDataset) -> dict[str, float]:
|
|
76
|
+
"""Compute simple data quality metrics."""
|
|
77
|
+
df = dataset.to_dataframe()
|
|
78
|
+
report = {
|
|
79
|
+
"missing_ratio": 0.0 if df.empty else float(df.isna().mean().mean()),
|
|
80
|
+
"outlier_ratio": float(detect_outliers(dataset).mean().mean()),
|
|
81
|
+
}
|
|
82
|
+
logger.info("Data quality report: %s", report)
|
|
83
|
+
return report
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Synthetic data generation utilities for benchmarking."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from numbers import Integral
|
|
10
|
+
from os import PathLike
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
from sensor_modeling.utils.data_io import SensorDataset
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class SyntheticConfig:
|
|
23
|
+
n_steps: int = 1000
|
|
24
|
+
n_sensors: int = 3
|
|
25
|
+
change_points: list[int] | None = None
|
|
26
|
+
failure_rate: float = 0.0
|
|
27
|
+
seed: int = 0
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _validate_config(config: SyntheticConfig) -> list[int]:
|
|
31
|
+
"""Validate a synthetic generation config and return change points."""
|
|
32
|
+
if config.n_steps < 1:
|
|
33
|
+
raise ValueError("n_steps must be at least 1")
|
|
34
|
+
if config.n_sensors < 1:
|
|
35
|
+
raise ValueError("n_sensors must be at least 1")
|
|
36
|
+
if not 0 <= config.failure_rate <= 1:
|
|
37
|
+
raise ValueError("failure_rate must be between 0 and 1")
|
|
38
|
+
|
|
39
|
+
change_points = config.change_points or [config.n_steps // 2]
|
|
40
|
+
invalid = [
|
|
41
|
+
cp
|
|
42
|
+
for cp in change_points
|
|
43
|
+
if not isinstance(cp, Integral)
|
|
44
|
+
or isinstance(cp, bool)
|
|
45
|
+
or cp < 0
|
|
46
|
+
or cp >= config.n_steps
|
|
47
|
+
]
|
|
48
|
+
if invalid:
|
|
49
|
+
raise ValueError(
|
|
50
|
+
"change_points must be integer offsets between 0 and n_steps - 1"
|
|
51
|
+
)
|
|
52
|
+
return change_points
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def generate(config: SyntheticConfig) -> tuple[SensorDataset, dict[str, list[int]]]:
|
|
56
|
+
"""Generate synthetic sensor data with ground truth change points."""
|
|
57
|
+
cps = _validate_config(config)
|
|
58
|
+
rng = np.random.default_rng(config.seed)
|
|
59
|
+
probs = np.zeros((config.n_steps, config.n_sensors)) + 0.1
|
|
60
|
+
for cp in cps:
|
|
61
|
+
probs[cp:] += 0.5 # behavioral change after change point
|
|
62
|
+
probs = np.clip(probs, 0.0, 1.0)
|
|
63
|
+
data = rng.binomial(1, probs)
|
|
64
|
+
# introduce sensor failures
|
|
65
|
+
for s in range(config.n_sensors):
|
|
66
|
+
if rng.random() < config.failure_rate:
|
|
67
|
+
fail_start = rng.integers(0, max(config.n_steps // 2, 1))
|
|
68
|
+
data[fail_start:, s] = 0
|
|
69
|
+
logger.warning(
|
|
70
|
+
"Injected failure in sensor %s starting at %s", s, fail_start
|
|
71
|
+
)
|
|
72
|
+
index = pd.date_range("2024-01-01", periods=config.n_steps, freq="1min")
|
|
73
|
+
df = pd.DataFrame(
|
|
74
|
+
data, index=index, columns=[f"sensor_{i}" for i in range(config.n_sensors)]
|
|
75
|
+
)
|
|
76
|
+
return SensorDataset(df), {"change_points": cps}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def export(
|
|
80
|
+
dataset: SensorDataset,
|
|
81
|
+
metadata: Mapping[str, object],
|
|
82
|
+
path: str | PathLike[str],
|
|
83
|
+
fmt: str = "csv",
|
|
84
|
+
) -> dict[str, Path]:
|
|
85
|
+
"""Export synthetic dataset and metadata in multiple formats."""
|
|
86
|
+
if fmt not in {"csv", "json", "hdf5"}:
|
|
87
|
+
raise ValueError(f"Unsupported export format: {fmt}")
|
|
88
|
+
|
|
89
|
+
df = dataset.to_dataframe()
|
|
90
|
+
output_path = Path(path)
|
|
91
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
92
|
+
|
|
93
|
+
if fmt == "csv":
|
|
94
|
+
df.to_csv(output_path)
|
|
95
|
+
metadata_path = Path(f"{output_path}.meta.json")
|
|
96
|
+
with metadata_path.open("w", encoding="utf-8") as f:
|
|
97
|
+
json.dump(metadata, f)
|
|
98
|
+
output_paths = {"data": output_path, "metadata": metadata_path}
|
|
99
|
+
elif fmt == "json":
|
|
100
|
+
records_df = df.reset_index().rename(
|
|
101
|
+
columns={df.index.name or "index": "timestamp"}
|
|
102
|
+
)
|
|
103
|
+
records_df["timestamp"] = records_df["timestamp"].astype(str)
|
|
104
|
+
records = records_df.to_dict(orient="records")
|
|
105
|
+
with output_path.open("w", encoding="utf-8") as f:
|
|
106
|
+
json.dump({"data": records, "meta": metadata}, f)
|
|
107
|
+
output_paths = {"data": output_path}
|
|
108
|
+
else:
|
|
109
|
+
try:
|
|
110
|
+
import h5py
|
|
111
|
+
except ImportError as exc: # pragma: no cover - dependency is installed in CI
|
|
112
|
+
raise ImportError("h5py is required for HDF5 export") from exc
|
|
113
|
+
|
|
114
|
+
with h5py.File(output_path, "w") as h5:
|
|
115
|
+
dset = h5.create_dataset("data", data=df.values)
|
|
116
|
+
dset.attrs["timestamp"] = df.index.astype(str).to_list()
|
|
117
|
+
h5.create_dataset("meta", data=json.dumps(metadata).encode("utf-8"))
|
|
118
|
+
output_paths = {"data": output_path}
|
|
119
|
+
|
|
120
|
+
logger.info("Exported synthetic dataset to %s (format=%s)", output_path, fmt)
|
|
121
|
+
return output_paths
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Data validation utilities."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from sensor_modeling.utils.data_io import SensorDataset
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def check_temporal_consistency(dataset: SensorDataset) -> bool:
|
|
15
|
+
"""Verify that timestamps are monotonic and evenly spaced."""
|
|
16
|
+
idx = dataset.to_dataframe().index
|
|
17
|
+
if not isinstance(idx, pd.DatetimeIndex):
|
|
18
|
+
logger.error("Index must be a DatetimeIndex")
|
|
19
|
+
return False
|
|
20
|
+
if not idx.is_monotonic_increasing:
|
|
21
|
+
logger.error("Timestamps are not sorted")
|
|
22
|
+
return False
|
|
23
|
+
if idx.has_duplicates:
|
|
24
|
+
logger.error("Duplicate timestamps detected")
|
|
25
|
+
return False
|
|
26
|
+
diffs = idx.to_series().diff().dropna().value_counts()
|
|
27
|
+
if len(diffs) > 1:
|
|
28
|
+
logger.warning("Irregular sampling detected: %s", diffs.to_dict())
|
|
29
|
+
return False
|
|
30
|
+
return True
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def validate_sensor_ranges(
|
|
34
|
+
dataset: SensorDataset, min_val: float = 0.0, max_val: float = 1.0
|
|
35
|
+
) -> bool:
|
|
36
|
+
"""Ensure sensor readings fall within expected bounds."""
|
|
37
|
+
if min_val > max_val:
|
|
38
|
+
raise ValueError("min_val must be less than or equal to max_val")
|
|
39
|
+
|
|
40
|
+
df = dataset.to_dataframe()
|
|
41
|
+
if df.empty or len(df.columns) == 0:
|
|
42
|
+
logger.error("Sensor range validation requires data")
|
|
43
|
+
return False
|
|
44
|
+
|
|
45
|
+
numeric = df.apply(pd.to_numeric, errors="coerce")
|
|
46
|
+
valid = (
|
|
47
|
+
numeric.notna().all().all()
|
|
48
|
+
and numeric.apply(lambda c: c.between(min_val, max_val).all()).all()
|
|
49
|
+
)
|
|
50
|
+
if not valid:
|
|
51
|
+
logger.error("Sensor readings outside [%s, %s]", min_val, max_val)
|
|
52
|
+
return bool(valid)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def detect_sensor_failures(
|
|
56
|
+
dataset: SensorDataset, window: int = 100
|
|
57
|
+
) -> dict[str, bool]:
|
|
58
|
+
"""Detect potential sensor failures using long constant stretches."""
|
|
59
|
+
if window < 1:
|
|
60
|
+
raise ValueError("window must be at least 1")
|
|
61
|
+
|
|
62
|
+
df = dataset.to_dataframe()
|
|
63
|
+
failures: dict[str, bool] = {}
|
|
64
|
+
for col in df.columns:
|
|
65
|
+
series = df[col]
|
|
66
|
+
rolling = series.rolling(window=window, min_periods=window)
|
|
67
|
+
constant_windows = rolling.apply(
|
|
68
|
+
lambda x: x.nunique(dropna=False) <= 1,
|
|
69
|
+
raw=False,
|
|
70
|
+
)
|
|
71
|
+
failures[col] = bool(constant_windows.fillna(0).astype(bool).any())
|
|
72
|
+
if failures[col]:
|
|
73
|
+
logger.warning("Possible failure detected in sensor '%s'", col)
|
|
74
|
+
return failures
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def cross_sensor_correlation(dataset: SensorDataset) -> pd.DataFrame:
|
|
78
|
+
"""Compute correlation matrix across sensors."""
|
|
79
|
+
corr = dataset.to_dataframe().corr()
|
|
80
|
+
logger.debug("Sensor correlation matrix:\n%s", corr)
|
|
81
|
+
return corr
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Evaluation metrics and sensor-ablation experiments.
|
|
2
|
+
|
|
3
|
+
Metrics are chosen per problem rather than defaulting to accuracy, which is
|
|
4
|
+
close to meaningless on states this imbalanced. Ablation studies are paired by
|
|
5
|
+
construction: every configuration sees identical simulated trajectories, so a
|
|
6
|
+
difference between configurations is a difference in sensing rather than in
|
|
7
|
+
the person being sensed.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .ablation import (
|
|
11
|
+
AblationReport,
|
|
12
|
+
AblationRun,
|
|
13
|
+
SensorConfiguration,
|
|
14
|
+
evaluate_configuration,
|
|
15
|
+
leave_one_out,
|
|
16
|
+
named_subsets,
|
|
17
|
+
run_ablation,
|
|
18
|
+
)
|
|
19
|
+
from .attribution import (
|
|
20
|
+
ArmResult,
|
|
21
|
+
AttributionStudy,
|
|
22
|
+
Scenario,
|
|
23
|
+
ScenarioComparison,
|
|
24
|
+
compare_scenario,
|
|
25
|
+
run_attribution_study,
|
|
26
|
+
standard_scenarios,
|
|
27
|
+
)
|
|
28
|
+
from .detection import (
|
|
29
|
+
ArmOutcome,
|
|
30
|
+
ChangeArm,
|
|
31
|
+
DetectionStudy,
|
|
32
|
+
run_detection_study,
|
|
33
|
+
standard_arms,
|
|
34
|
+
)
|
|
35
|
+
from .metrics import (
|
|
36
|
+
BinaryMetrics,
|
|
37
|
+
DetectionMetrics,
|
|
38
|
+
PairedDifference,
|
|
39
|
+
StateMetrics,
|
|
40
|
+
TimingMetrics,
|
|
41
|
+
binary_metrics,
|
|
42
|
+
detection_metrics,
|
|
43
|
+
paired_difference,
|
|
44
|
+
state_metrics,
|
|
45
|
+
summarise,
|
|
46
|
+
transition_timing,
|
|
47
|
+
)
|
|
48
|
+
from .provenance import (
|
|
49
|
+
METRIC_DEFINITIONS,
|
|
50
|
+
RESULTS_DIR,
|
|
51
|
+
ExperimentRecord,
|
|
52
|
+
environment,
|
|
53
|
+
load_record,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
__all__ = [
|
|
57
|
+
"METRIC_DEFINITIONS",
|
|
58
|
+
"RESULTS_DIR",
|
|
59
|
+
"AblationReport",
|
|
60
|
+
"ArmOutcome",
|
|
61
|
+
"ArmResult",
|
|
62
|
+
"AttributionStudy",
|
|
63
|
+
"AblationRun",
|
|
64
|
+
"BinaryMetrics",
|
|
65
|
+
"ChangeArm",
|
|
66
|
+
"DetectionMetrics",
|
|
67
|
+
"DetectionStudy",
|
|
68
|
+
"ExperimentRecord",
|
|
69
|
+
"PairedDifference",
|
|
70
|
+
"Scenario",
|
|
71
|
+
"ScenarioComparison",
|
|
72
|
+
"SensorConfiguration",
|
|
73
|
+
"StateMetrics",
|
|
74
|
+
"TimingMetrics",
|
|
75
|
+
"binary_metrics",
|
|
76
|
+
"compare_scenario",
|
|
77
|
+
"detection_metrics",
|
|
78
|
+
"environment",
|
|
79
|
+
"evaluate_configuration",
|
|
80
|
+
"leave_one_out",
|
|
81
|
+
"load_record",
|
|
82
|
+
"named_subsets",
|
|
83
|
+
"paired_difference",
|
|
84
|
+
"run_ablation",
|
|
85
|
+
"run_attribution_study",
|
|
86
|
+
"run_detection_study",
|
|
87
|
+
"standard_arms",
|
|
88
|
+
"standard_scenarios",
|
|
89
|
+
"state_metrics",
|
|
90
|
+
"summarise",
|
|
91
|
+
"transition_timing",
|
|
92
|
+
]
|
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
"""Sensor-ablation experiments: what is each modality actually worth?
|
|
2
|
+
|
|
3
|
+
The research question this package exists to answer is whether useful
|
|
4
|
+
behavioural inference survives with fewer physical sensors. Answering it needs
|
|
5
|
+
more discipline than running a few configurations and comparing numbers.
|
|
6
|
+
|
|
7
|
+
Three things are built in rather than left to the user to remember.
|
|
8
|
+
|
|
9
|
+
*The design is paired.* Every configuration is evaluated on identical
|
|
10
|
+
simulated trajectories. Simulated households differ from each other far more
|
|
11
|
+
than two sensor configurations differ on one household, so an unpaired
|
|
12
|
+
comparison buries a real effect under between-household variance.
|
|
13
|
+
|
|
14
|
+
*Ablation removes sensors, not code.* A configuration is a subset of the
|
|
15
|
+
registry. The pipeline is constructed from that subset exactly as it would be
|
|
16
|
+
from a full deployment, so an ablated run exercises the same inference path a
|
|
17
|
+
real sparse deployment would.
|
|
18
|
+
|
|
19
|
+
*Marginal value is not assumed additive.* Two sensors that each look
|
|
20
|
+
worthless alone can be jointly essential -- a door tells you little without
|
|
21
|
+
something to say who walked through it. :func:`interaction` reports how far
|
|
22
|
+
the joint contribution departs from the sum of the individual ones, so that
|
|
23
|
+
departure is measured rather than assumed away.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
30
|
+
from dataclasses import dataclass, field
|
|
31
|
+
from datetime import timedelta
|
|
32
|
+
|
|
33
|
+
import numpy as np
|
|
34
|
+
|
|
35
|
+
from ..observations.registry import SensorRegistry
|
|
36
|
+
from ..online.pipeline import BehaviouralSensingPipeline, PipelineConfig
|
|
37
|
+
from ..simulation.faults import DegradationConfig, degrade
|
|
38
|
+
from ..simulation.household import HouseholdConfig, simulate
|
|
39
|
+
from .metrics import PairedDifference, StateMetrics, paired_difference, state_metrics
|
|
40
|
+
|
|
41
|
+
logger = logging.getLogger(__name__)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class SensorConfiguration:
|
|
46
|
+
"""A named subset of a deployment to evaluate."""
|
|
47
|
+
|
|
48
|
+
name: str
|
|
49
|
+
sensors: tuple[str, ...]
|
|
50
|
+
|
|
51
|
+
def __post_init__(self) -> None:
|
|
52
|
+
"""Validate the configuration."""
|
|
53
|
+
if not self.name.strip():
|
|
54
|
+
raise ValueError("a configuration needs a name")
|
|
55
|
+
if not self.sensors:
|
|
56
|
+
raise ValueError(f"configuration '{self.name}' has no sensors")
|
|
57
|
+
|
|
58
|
+
def restrict(self, registry: SensorRegistry) -> SensorRegistry:
|
|
59
|
+
"""Return the registry restricted to this configuration."""
|
|
60
|
+
return registry.subset(self.sensors)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen=True)
|
|
64
|
+
class AblationRun:
|
|
65
|
+
"""The outcome of one configuration on one simulated trajectory."""
|
|
66
|
+
|
|
67
|
+
configuration: str
|
|
68
|
+
seed: int
|
|
69
|
+
metrics: StateMetrics
|
|
70
|
+
sensors: tuple[str, ...]
|
|
71
|
+
|
|
72
|
+
def to_dict(self) -> dict[str, object]:
|
|
73
|
+
"""Return a serialisable form of the run."""
|
|
74
|
+
return {
|
|
75
|
+
"configuration": self.configuration,
|
|
76
|
+
"seed": self.seed,
|
|
77
|
+
"sensors": list(self.sensors),
|
|
78
|
+
"metrics": self.metrics.to_dict(),
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class AblationReport:
|
|
84
|
+
"""Results of a paired ablation sweep."""
|
|
85
|
+
|
|
86
|
+
runs: list[AblationRun] = field(default_factory=list)
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def configurations(self) -> list[str]:
|
|
90
|
+
"""Names of the configurations evaluated, in first-seen order."""
|
|
91
|
+
seen: dict[str, None] = {}
|
|
92
|
+
for run in self.runs:
|
|
93
|
+
seen.setdefault(run.configuration, None)
|
|
94
|
+
return list(seen)
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def seeds(self) -> list[int]:
|
|
98
|
+
"""Seeds evaluated, sorted."""
|
|
99
|
+
return sorted({run.seed for run in self.runs})
|
|
100
|
+
|
|
101
|
+
def series(
|
|
102
|
+
self, configuration: str, metric: str = "balanced_accuracy"
|
|
103
|
+
) -> list[float]:
|
|
104
|
+
"""Return one configuration's scores, ordered by seed.
|
|
105
|
+
|
|
106
|
+
Ordering by seed is what keeps the pairing intact: position ``i`` in
|
|
107
|
+
two configurations' series refers to the same simulated household.
|
|
108
|
+
"""
|
|
109
|
+
by_seed = {
|
|
110
|
+
run.seed: getattr(run.metrics, metric)
|
|
111
|
+
for run in self.runs
|
|
112
|
+
if run.configuration == configuration
|
|
113
|
+
}
|
|
114
|
+
missing = set(self.seeds) - set(by_seed)
|
|
115
|
+
if missing:
|
|
116
|
+
raise ValueError(
|
|
117
|
+
f"configuration '{configuration}' is missing seeds {sorted(missing)}; "
|
|
118
|
+
"the paired design requires every configuration on every seed"
|
|
119
|
+
)
|
|
120
|
+
return [by_seed[seed] for seed in self.seeds]
|
|
121
|
+
|
|
122
|
+
def compare(
|
|
123
|
+
self,
|
|
124
|
+
treatment: str,
|
|
125
|
+
control: str,
|
|
126
|
+
*,
|
|
127
|
+
metric: str = "balanced_accuracy",
|
|
128
|
+
seed: int = 0,
|
|
129
|
+
) -> PairedDifference:
|
|
130
|
+
"""Compare two configurations on the trajectories they share."""
|
|
131
|
+
return paired_difference(
|
|
132
|
+
self.series(treatment, metric),
|
|
133
|
+
self.series(control, metric),
|
|
134
|
+
seed=seed,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def marginal(
|
|
138
|
+
self,
|
|
139
|
+
full: str,
|
|
140
|
+
without: str,
|
|
141
|
+
*,
|
|
142
|
+
metric: str = "balanced_accuracy",
|
|
143
|
+
) -> PairedDifference:
|
|
144
|
+
"""Return the value lost by removing a sensor from *full*.
|
|
145
|
+
|
|
146
|
+
This is ``performance(S) - performance(S without j)``, evaluated on
|
|
147
|
+
paired trajectories.
|
|
148
|
+
"""
|
|
149
|
+
return self.compare(full, without, metric=metric)
|
|
150
|
+
|
|
151
|
+
def interaction(
|
|
152
|
+
self,
|
|
153
|
+
full: str,
|
|
154
|
+
without_first: str,
|
|
155
|
+
without_second: str,
|
|
156
|
+
without_both: str,
|
|
157
|
+
*,
|
|
158
|
+
metric: str = "balanced_accuracy",
|
|
159
|
+
) -> float:
|
|
160
|
+
"""Return how far two sensors' joint value departs from additivity.
|
|
161
|
+
|
|
162
|
+
Positive means the pair is worth more together than separately --
|
|
163
|
+
they are complementary. Negative means they are redundant, each
|
|
164
|
+
covering for the other's absence.
|
|
165
|
+
"""
|
|
166
|
+
scores = {
|
|
167
|
+
name: float(np.mean(self.series(name, metric)))
|
|
168
|
+
for name in (full, without_first, without_second, without_both)
|
|
169
|
+
}
|
|
170
|
+
first = scores[full] - scores[without_first]
|
|
171
|
+
second = scores[full] - scores[without_second]
|
|
172
|
+
joint = scores[full] - scores[without_both]
|
|
173
|
+
return joint - (first + second)
|
|
174
|
+
|
|
175
|
+
def summary(self, metric: str = "balanced_accuracy") -> dict[str, dict[str, float]]:
|
|
176
|
+
"""Return mean and spread of *metric* for each configuration."""
|
|
177
|
+
result: dict[str, dict[str, float]] = {}
|
|
178
|
+
for name in self.configurations:
|
|
179
|
+
series = self.series(name, metric)
|
|
180
|
+
result[name] = {
|
|
181
|
+
"mean": float(np.mean(series)),
|
|
182
|
+
"sd": float(np.std(series, ddof=1)) if len(series) > 1 else 0.0,
|
|
183
|
+
"min": float(np.min(series)),
|
|
184
|
+
"max": float(np.max(series)),
|
|
185
|
+
"n_sensors": float(
|
|
186
|
+
len(
|
|
187
|
+
next(
|
|
188
|
+
run.sensors
|
|
189
|
+
for run in self.runs
|
|
190
|
+
if run.configuration == name
|
|
191
|
+
)
|
|
192
|
+
)
|
|
193
|
+
),
|
|
194
|
+
}
|
|
195
|
+
return result
|
|
196
|
+
|
|
197
|
+
def to_dict(self, metric: str = "balanced_accuracy") -> dict[str, object]:
|
|
198
|
+
"""Return a serialisable form of the report."""
|
|
199
|
+
return {
|
|
200
|
+
"metric": metric,
|
|
201
|
+
"seeds": self.seeds,
|
|
202
|
+
"summary": self.summary(metric),
|
|
203
|
+
"runs": [run.to_dict() for run in self.runs],
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def evaluate_configuration(
|
|
208
|
+
configuration: SensorConfiguration,
|
|
209
|
+
result: object,
|
|
210
|
+
*,
|
|
211
|
+
step: timedelta = timedelta(minutes=5),
|
|
212
|
+
degradation: DegradationConfig | None = None,
|
|
213
|
+
) -> StateMetrics:
|
|
214
|
+
"""Run one sensor configuration over one simulated trajectory."""
|
|
215
|
+
registry = configuration.restrict(result.registry) # type: ignore[attr-defined]
|
|
216
|
+
observations = result.observations_for(configuration.sensors) # type: ignore[attr-defined]
|
|
217
|
+
if degradation is not None:
|
|
218
|
+
observations, _ = degrade(observations, degradation)
|
|
219
|
+
|
|
220
|
+
pipeline = BehaviouralSensingPipeline(
|
|
221
|
+
registry,
|
|
222
|
+
config=PipelineConfig(tz=result.config.tz, step=step), # type: ignore[attr-defined]
|
|
223
|
+
)
|
|
224
|
+
steps = pipeline.run(observations)
|
|
225
|
+
steps.extend(pipeline.close(result.end)) # type: ignore[attr-defined]
|
|
226
|
+
if not steps:
|
|
227
|
+
raise ValueError(
|
|
228
|
+
f"configuration '{configuration.name}' produced no pipeline steps"
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
truth = result.truth.states_at([step_.at for step_ in steps]) # type: ignore[attr-defined]
|
|
232
|
+
return state_metrics(truth, [step_.state for step_ in steps])
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def run_ablation(
|
|
236
|
+
configurations: Sequence[SensorConfiguration],
|
|
237
|
+
*,
|
|
238
|
+
seeds: Iterable[int],
|
|
239
|
+
household: HouseholdConfig | None = None,
|
|
240
|
+
step: timedelta = timedelta(minutes=5),
|
|
241
|
+
degradation: DegradationConfig | None = None,
|
|
242
|
+
) -> AblationReport:
|
|
243
|
+
"""Evaluate every configuration on every seeded trajectory.
|
|
244
|
+
|
|
245
|
+
One household is simulated per seed and shared by all configurations, so
|
|
246
|
+
differences between configurations are differences in sensing rather than
|
|
247
|
+
differences in the person being sensed.
|
|
248
|
+
"""
|
|
249
|
+
if not configurations:
|
|
250
|
+
raise ValueError("at least one configuration is required")
|
|
251
|
+
names = [configuration.name for configuration in configurations]
|
|
252
|
+
if len(set(names)) != len(names):
|
|
253
|
+
raise ValueError("configuration names must be unique")
|
|
254
|
+
|
|
255
|
+
base = household or HouseholdConfig()
|
|
256
|
+
report = AblationReport()
|
|
257
|
+
for seed in seeds:
|
|
258
|
+
from dataclasses import replace as _replace
|
|
259
|
+
|
|
260
|
+
result = simulate(_replace(base, seed=seed))
|
|
261
|
+
for configuration in configurations:
|
|
262
|
+
metrics = evaluate_configuration(
|
|
263
|
+
configuration, result, step=step, degradation=degradation
|
|
264
|
+
)
|
|
265
|
+
report.runs.append(
|
|
266
|
+
AblationRun(
|
|
267
|
+
configuration=configuration.name,
|
|
268
|
+
seed=seed,
|
|
269
|
+
metrics=metrics,
|
|
270
|
+
sensors=configuration.sensors,
|
|
271
|
+
)
|
|
272
|
+
)
|
|
273
|
+
logger.info(
|
|
274
|
+
"seed %s | %-22s | balanced accuracy %.3f | abstention %.3f",
|
|
275
|
+
seed,
|
|
276
|
+
configuration.name,
|
|
277
|
+
metrics.balanced_accuracy,
|
|
278
|
+
metrics.abstention_rate,
|
|
279
|
+
)
|
|
280
|
+
return report
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def leave_one_out(
|
|
284
|
+
registry: SensorRegistry, sensors: Iterable[str] | None = None
|
|
285
|
+
) -> list[SensorConfiguration]:
|
|
286
|
+
"""Build the full configuration plus one with each sensor removed."""
|
|
287
|
+
everything = tuple(registry.sensor_ids())
|
|
288
|
+
candidates = tuple(sensors) if sensors is not None else everything
|
|
289
|
+
configurations = [SensorConfiguration("all", everything)]
|
|
290
|
+
for sensor_id in candidates:
|
|
291
|
+
remaining = tuple(s for s in everything if s != sensor_id)
|
|
292
|
+
if remaining:
|
|
293
|
+
configurations.append(
|
|
294
|
+
SensorConfiguration(f"without_{sensor_id}", remaining)
|
|
295
|
+
)
|
|
296
|
+
return configurations
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def named_subsets(subsets: Mapping[str, Sequence[str]]) -> list[SensorConfiguration]:
|
|
300
|
+
"""Build configurations from a mapping of name to sensor identifiers."""
|
|
301
|
+
return [
|
|
302
|
+
SensorConfiguration(name, tuple(sensors)) for name, sensors in subsets.items()
|
|
303
|
+
]
|