drift-or-shift 0.1.0a1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- caliblab/__init__.py +159 -0
- caliblab/py.typed +0 -0
- drift_or_shift/__init__.py +113 -0
- drift_or_shift/calibration.py +92 -0
- drift_or_shift/data_synth.py +170 -0
- drift_or_shift/drift_monitor.py +124 -0
- drift_or_shift/drift_variants.py +66 -0
- drift_or_shift/ess.py +26 -0
- drift_or_shift/experiments/__init__.py +10 -0
- drift_or_shift/experiments/_common.py +106 -0
- drift_or_shift/experiments/exp10_credit_card_fraud.py +179 -0
- drift_or_shift/experiments/exp11_high_variance_medical.py +221 -0
- drift_or_shift/experiments/exp1_label_shift_synth.py +157 -0
- drift_or_shift/experiments/exp2_auc_pr_invariance.py +118 -0
- drift_or_shift/experiments/exp3_ess_vs_weight.py +80 -0
- drift_or_shift/experiments/exp4_concept_drift.py +170 -0
- drift_or_shift/experiments/exp5_realdata_breast_cancer.py +163 -0
- drift_or_shift/experiments/exp6_calibration_label_shift.py +184 -0
- drift_or_shift/experiments/exp7_drift_types.py +199 -0
- drift_or_shift/experiments/exp8_multimodal_label_shift.py +176 -0
- drift_or_shift/experiments/exp9_covtype_label_shift.py +171 -0
- drift_or_shift/experiments/experiments/__init__.py +31 -0
- drift_or_shift/io_utils.py +42 -0
- drift_or_shift/metrics.py +56 -0
- drift_or_shift/models.py +43 -0
- drift_or_shift/plotting.py +75 -0
- drift_or_shift/py.typed +0 -0
- drift_or_shift/reporting.py +230 -0
- drift_or_shift/shift.py +53 -0
- drift_or_shift-0.1.0a1.dist-info/METADATA +218 -0
- drift_or_shift-0.1.0a1.dist-info/RECORD +37 -0
- drift_or_shift-0.1.0a1.dist-info/WHEEL +5 -0
- drift_or_shift-0.1.0a1.dist-info/entry_points.txt +12 -0
- drift_or_shift-0.1.0a1.dist-info/licenses/LICENSE +21 -0
- drift_or_shift-0.1.0a1.dist-info/top_level.txt +3 -0
- drift_shift_pipeline/__init__.py +30 -0
- drift_shift_pipeline/py.typed +0 -0
caliblab/__init__.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Simple benchmarking utilities for calibrated classifiers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
from sklearn.calibration import CalibratedClassifierCV
|
|
10
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
11
|
+
from sklearn.linear_model import LogisticRegression
|
|
12
|
+
from sklearn.model_selection import StratifiedKFold, train_test_split
|
|
13
|
+
|
|
14
|
+
try: # scikit-learn >= 1.6
|
|
15
|
+
from sklearn.frozen import FrozenEstimator
|
|
16
|
+
except ImportError: # pragma: no cover - scikit-learn < 1.6
|
|
17
|
+
FrozenEstimator = None
|
|
18
|
+
|
|
19
|
+
_ESTIMATOR_REGISTRY = {
|
|
20
|
+
"logistic": LogisticRegression,
|
|
21
|
+
"random_forest": RandomForestClassifier,
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _make_estimator(spec: dict, random_state: int | None):
|
|
26
|
+
"""Instantiate a base estimator from spec."""
|
|
27
|
+
estimator_type = spec.get("type")
|
|
28
|
+
if estimator_type not in _ESTIMATOR_REGISTRY:
|
|
29
|
+
raise ValueError(f"Unsupported estimator type: {estimator_type!r}")
|
|
30
|
+
estimator_cls = _ESTIMATOR_REGISTRY[estimator_type]
|
|
31
|
+
params = dict(spec.get("params", {}))
|
|
32
|
+
if random_state is not None:
|
|
33
|
+
params.setdefault("random_state", random_state)
|
|
34
|
+
if estimator_type == "logistic":
|
|
35
|
+
params.setdefault("max_iter", 1000)
|
|
36
|
+
return estimator_cls(**params)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _calibrate_prefit(estimator, method: str) -> CalibratedClassifierCV:
|
|
40
|
+
"""Wrap an already-fitted estimator in a calibrator.
|
|
41
|
+
|
|
42
|
+
``cv="prefit"`` is deprecated in scikit-learn 1.6 and removed in 1.8, so we
|
|
43
|
+
prefer ``FrozenEstimator`` where it exists and fall back only for older
|
|
44
|
+
releases still covered by our dependency floor.
|
|
45
|
+
"""
|
|
46
|
+
if FrozenEstimator is not None:
|
|
47
|
+
return CalibratedClassifierCV(FrozenEstimator(estimator), method=method)
|
|
48
|
+
return CalibratedClassifierCV( # pragma: no cover - scikit-learn < 1.6
|
|
49
|
+
estimator=estimator, method=method, cv="prefit"
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_MIN_CALIBRATION_PER_CLASS = 5
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _validate_calibration_set(y_calib: np.ndarray, holdout: float) -> None:
|
|
57
|
+
"""Fail early and clearly when the held-out split is too small to calibrate.
|
|
58
|
+
|
|
59
|
+
``CalibratedClassifierCV`` cross-validates internally even over a frozen
|
|
60
|
+
estimator, so it needs at least as many members of each class as its
|
|
61
|
+
default number of folds. Left to sklearn this surfaces as "n_splits=5
|
|
62
|
+
cannot be greater than the number of members in each class", which says
|
|
63
|
+
nothing about the knob the caller actually turned.
|
|
64
|
+
"""
|
|
65
|
+
counts = np.bincount(np.asarray(y_calib, dtype=int), minlength=2)
|
|
66
|
+
smallest = int(counts.min())
|
|
67
|
+
if smallest < _MIN_CALIBRATION_PER_CLASS:
|
|
68
|
+
raise ValueError(
|
|
69
|
+
f"calibration_holdout={holdout} leaves only {smallest} sample(s) of "
|
|
70
|
+
f"the rarest class to calibrate on; at least "
|
|
71
|
+
f"{_MIN_CALIBRATION_PER_CLASS} are needed. Raise the holdout "
|
|
72
|
+
f"fraction, use fewer CV folds, or pass more data."
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def benchmark(
|
|
77
|
+
X: np.ndarray,
|
|
78
|
+
y: np.ndarray,
|
|
79
|
+
estimators: Iterable[dict],
|
|
80
|
+
calibrations: Iterable[str],
|
|
81
|
+
*,
|
|
82
|
+
cv_params: dict,
|
|
83
|
+
calibration_holdout: float | None = None,
|
|
84
|
+
) -> pd.DataFrame:
|
|
85
|
+
"""Run calibration benchmarking over a dataset.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
calibration_holdout: Fraction of each fold's training data to hold out
|
|
89
|
+
for fitting the calibration map. ``None`` (the default) fits the
|
|
90
|
+
calibrator on the same data the base estimator was trained on.
|
|
91
|
+
|
|
92
|
+
On fitting the calibrator on the estimator's own training data:
|
|
93
|
+
|
|
94
|
+
Textbook practice is to calibrate on held-out data, because a model's
|
|
95
|
+
predictions on data it has already seen are over-confident and the map
|
|
96
|
+
learned from them need not transfer. We measured it on the shipped
|
|
97
|
+
breast-cancer benchmark, holding the base model and the calibration-set
|
|
98
|
+
size fixed so that only the overlap differed, and the effect on test Brier
|
|
99
|
+
score was within noise::
|
|
100
|
+
|
|
101
|
+
logistic sigmoid -0.0014 (SE 0.0003)
|
|
102
|
+
logistic isotonic +0.0006 (SE 0.0009)
|
|
103
|
+
random_forest sigmoid +0.0001 (SE 0.0003)
|
|
104
|
+
random_forest isotonic -0.0002 (SE 0.0010)
|
|
105
|
+
|
|
106
|
+
The default therefore stays as it was, so published numbers do not move.
|
|
107
|
+
Pass ``calibration_holdout`` to opt into a held-out split; on a harder
|
|
108
|
+
dataset or a higher-capacity estimator the difference may well matter.
|
|
109
|
+
|
|
110
|
+
``tests/test_caliblab_leakage.py`` re-runs that comparison, so the claim
|
|
111
|
+
above is checked rather than merely asserted.
|
|
112
|
+
"""
|
|
113
|
+
if calibration_holdout is not None and not 0.0 < calibration_holdout < 1.0:
|
|
114
|
+
raise ValueError("calibration_holdout must lie in the open interval (0, 1)")
|
|
115
|
+
X = np.asarray(X)
|
|
116
|
+
y = np.asarray(y)
|
|
117
|
+
cv = StratifiedKFold(
|
|
118
|
+
n_splits=cv_params.get("n_splits", 5),
|
|
119
|
+
shuffle=True,
|
|
120
|
+
random_state=cv_params.get("random_state"),
|
|
121
|
+
)
|
|
122
|
+
results = []
|
|
123
|
+
for fold, (train_idx, test_idx) in enumerate(cv.split(X, y)):
|
|
124
|
+
X_train, X_test = X[train_idx], X[test_idx]
|
|
125
|
+
y_train, y_test = y[train_idx], y[test_idx]
|
|
126
|
+
X_fit, y_fit = X_train, y_train
|
|
127
|
+
X_calib, y_calib = X_train, y_train
|
|
128
|
+
if calibration_holdout is not None:
|
|
129
|
+
X_fit, X_calib, y_fit, y_calib = train_test_split(
|
|
130
|
+
X_train,
|
|
131
|
+
y_train,
|
|
132
|
+
test_size=calibration_holdout,
|
|
133
|
+
stratify=y_train,
|
|
134
|
+
random_state=cv_params.get("random_state"),
|
|
135
|
+
)
|
|
136
|
+
_validate_calibration_set(y_calib, calibration_holdout)
|
|
137
|
+
for spec in estimators:
|
|
138
|
+
for calibration in calibrations:
|
|
139
|
+
estimator = _make_estimator(spec, cv_params.get("random_state"))
|
|
140
|
+
estimator.fit(X_fit, y_fit)
|
|
141
|
+
predictor = estimator
|
|
142
|
+
if calibration != "none":
|
|
143
|
+
calibrator = _calibrate_prefit(estimator, calibration)
|
|
144
|
+
calibrator.fit(X_calib, y_calib)
|
|
145
|
+
predictor = calibrator
|
|
146
|
+
probs = predictor.predict_proba(X_test)[:, 1]
|
|
147
|
+
for local_idx, prob in enumerate(probs):
|
|
148
|
+
results.append(
|
|
149
|
+
{
|
|
150
|
+
"fold": fold,
|
|
151
|
+
"estimator": spec.get("name", spec.get("type")),
|
|
152
|
+
"estimator_type": spec.get("type"),
|
|
153
|
+
"calibration": calibration,
|
|
154
|
+
"sample_index": int(test_idx[local_idx]),
|
|
155
|
+
"y_true": int(y_test[local_idx]),
|
|
156
|
+
"y_prob": float(prob),
|
|
157
|
+
}
|
|
158
|
+
)
|
|
159
|
+
return pd.DataFrame(results)
|
caliblab/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""drift-or-shift: label shift and offset correction utilities."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0a1"
|
|
4
|
+
|
|
5
|
+
from .calibration import (
|
|
6
|
+
apply_calibrator,
|
|
7
|
+
find_best_temperature,
|
|
8
|
+
fit_isotonic_calibrator,
|
|
9
|
+
temperature_scale,
|
|
10
|
+
)
|
|
11
|
+
from .data_synth import (
|
|
12
|
+
concept_drift_mu1,
|
|
13
|
+
default_mu0,
|
|
14
|
+
default_mu1,
|
|
15
|
+
make_gaussian_binary,
|
|
16
|
+
make_multimodal_binary,
|
|
17
|
+
resample_to_prevalence,
|
|
18
|
+
)
|
|
19
|
+
from .drift_monitor import feature_drift_summary, univariate_feature_stats
|
|
20
|
+
from .drift_variants import (
|
|
21
|
+
apply_covariance_shift,
|
|
22
|
+
density_ratio_shift,
|
|
23
|
+
drift_score_from_ratio,
|
|
24
|
+
inject_label_noise,
|
|
25
|
+
shift_mean_vector,
|
|
26
|
+
)
|
|
27
|
+
from .ess import effective_sample_size, ess_fraction
|
|
28
|
+
from .experiments._common import (
|
|
29
|
+
DEFAULT_COSTS,
|
|
30
|
+
DRIFT_FEATURE_METRICS,
|
|
31
|
+
PI_TEST_GRID,
|
|
32
|
+
PI_TRAIN,
|
|
33
|
+
SEEDS,
|
|
34
|
+
ExperimentConfig,
|
|
35
|
+
aggregate_mean_std,
|
|
36
|
+
feature_drift_metrics,
|
|
37
|
+
)
|
|
38
|
+
from .io_utils import ensure_dir, save_json, save_table, timestamped_run_dir
|
|
39
|
+
from .metrics import oracle_threshold_min_risk, pr_auc, risk_cost_sensitive, roc_auc
|
|
40
|
+
from .models import fit_logistic_regression, predict_logits
|
|
41
|
+
from .plotting import (
|
|
42
|
+
plot_auc_pr_vs_prevalence,
|
|
43
|
+
plot_ess_vs_alpha,
|
|
44
|
+
plot_risk_vs_prevalence,
|
|
45
|
+
)
|
|
46
|
+
from .reporting import (
|
|
47
|
+
collect_summary_jsons,
|
|
48
|
+
format_results_overview,
|
|
49
|
+
select_latest_summaries,
|
|
50
|
+
write_results_overview,
|
|
51
|
+
)
|
|
52
|
+
from .shift import (
|
|
53
|
+
apply_logit_offset,
|
|
54
|
+
decision_from_logits,
|
|
55
|
+
logit_offset,
|
|
56
|
+
odds,
|
|
57
|
+
threshold_from_costs,
|
|
58
|
+
validate_prevalence,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
__all__ = [
|
|
62
|
+
"DEFAULT_COSTS",
|
|
63
|
+
"DRIFT_FEATURE_METRICS",
|
|
64
|
+
"PI_TEST_GRID",
|
|
65
|
+
"PI_TRAIN",
|
|
66
|
+
"SEEDS",
|
|
67
|
+
"ExperimentConfig",
|
|
68
|
+
"__version__",
|
|
69
|
+
"aggregate_mean_std",
|
|
70
|
+
"apply_calibrator",
|
|
71
|
+
"apply_covariance_shift",
|
|
72
|
+
"apply_logit_offset",
|
|
73
|
+
"collect_summary_jsons",
|
|
74
|
+
"concept_drift_mu1",
|
|
75
|
+
"decision_from_logits",
|
|
76
|
+
"default_mu0",
|
|
77
|
+
"default_mu1",
|
|
78
|
+
"density_ratio_shift",
|
|
79
|
+
"drift_score_from_ratio",
|
|
80
|
+
"effective_sample_size",
|
|
81
|
+
"ensure_dir",
|
|
82
|
+
"ess_fraction",
|
|
83
|
+
"feature_drift_metrics",
|
|
84
|
+
"feature_drift_summary",
|
|
85
|
+
"find_best_temperature",
|
|
86
|
+
"fit_isotonic_calibrator",
|
|
87
|
+
"fit_logistic_regression",
|
|
88
|
+
"format_results_overview",
|
|
89
|
+
"inject_label_noise",
|
|
90
|
+
"logit_offset",
|
|
91
|
+
"make_gaussian_binary",
|
|
92
|
+
"make_multimodal_binary",
|
|
93
|
+
"odds",
|
|
94
|
+
"oracle_threshold_min_risk",
|
|
95
|
+
"plot_auc_pr_vs_prevalence",
|
|
96
|
+
"plot_ess_vs_alpha",
|
|
97
|
+
"plot_risk_vs_prevalence",
|
|
98
|
+
"pr_auc",
|
|
99
|
+
"predict_logits",
|
|
100
|
+
"resample_to_prevalence",
|
|
101
|
+
"risk_cost_sensitive",
|
|
102
|
+
"roc_auc",
|
|
103
|
+
"save_json",
|
|
104
|
+
"save_table",
|
|
105
|
+
"select_latest_summaries",
|
|
106
|
+
"shift_mean_vector",
|
|
107
|
+
"temperature_scale",
|
|
108
|
+
"threshold_from_costs",
|
|
109
|
+
"timestamped_run_dir",
|
|
110
|
+
"univariate_feature_stats",
|
|
111
|
+
"validate_prevalence",
|
|
112
|
+
"write_results_overview",
|
|
113
|
+
]
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Calibration helpers for logistic outputs under label shift."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
from sklearn.isotonic import IsotonicRegression
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _sigmoid(logits: np.ndarray) -> np.ndarray:
|
|
13
|
+
"""Return probability after applying the logistic sigmoid function.
|
|
14
|
+
|
|
15
|
+
Evaluated piecewise. The direct form ``1 / (1 + exp(-x))`` overflows for
|
|
16
|
+
large negative x: the result saturates to the correct 0.0, but NumPy emits
|
|
17
|
+
an ``overflow encountered in exp`` warning on the way, which is noise that
|
|
18
|
+
can mask a real one. For x < 0 we use ``exp(x) / (1 + exp(x))`` instead,
|
|
19
|
+
where the exponential underflows harmlessly rather than overflowing.
|
|
20
|
+
"""
|
|
21
|
+
x = np.asarray(logits, dtype=float)
|
|
22
|
+
out = np.empty_like(x)
|
|
23
|
+
positive = x >= 0
|
|
24
|
+
out[positive] = 1.0 / (1.0 + np.exp(-x[positive]))
|
|
25
|
+
exp_x = np.exp(x[~positive])
|
|
26
|
+
out[~positive] = exp_x / (1.0 + exp_x)
|
|
27
|
+
return out
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def temperature_scale(logits: ArrayLike, temperature: float) -> np.ndarray:
|
|
31
|
+
"""
|
|
32
|
+
Adjust logits by a temperature factor.
|
|
33
|
+
|
|
34
|
+
Higher temperatures soften the distribution, while temperatures below 1 boost confidence.
|
|
35
|
+
"""
|
|
36
|
+
if temperature <= 0:
|
|
37
|
+
raise ValueError("temperature must be positive")
|
|
38
|
+
logits_arr = np.asarray(logits, dtype=float)
|
|
39
|
+
return logits_arr / float(temperature)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def find_best_temperature(
|
|
43
|
+
logits: ArrayLike,
|
|
44
|
+
targets: ArrayLike,
|
|
45
|
+
*,
|
|
46
|
+
grid: Iterable[float] | None = None,
|
|
47
|
+
) -> float:
|
|
48
|
+
"""
|
|
49
|
+
Return the temperature that minimizes logistic cross-entropy on the provided data.
|
|
50
|
+
|
|
51
|
+
A default grid spans from 0.05 to 3.0.
|
|
52
|
+
"""
|
|
53
|
+
if grid is None:
|
|
54
|
+
grid = list(
|
|
55
|
+
np.concatenate([np.logspace(-2, -0.5, 5), np.linspace(0.1, 3.0, 50)])
|
|
56
|
+
)
|
|
57
|
+
logits_arr = np.asarray(logits, dtype=float)
|
|
58
|
+
targets_arr = np.asarray(targets, dtype=float)
|
|
59
|
+
if logits_arr.shape != targets_arr.shape:
|
|
60
|
+
raise ValueError("logits and targets must share shape")
|
|
61
|
+
if not (np.all((targets_arr == 0) | (targets_arr == 1))):
|
|
62
|
+
raise ValueError("targets must be binary (0 or 1)")
|
|
63
|
+
losses: list[tuple[float, float]] = []
|
|
64
|
+
eps = 1e-12
|
|
65
|
+
for temperature in grid:
|
|
66
|
+
if temperature <= 0:
|
|
67
|
+
continue
|
|
68
|
+
scaled = _sigmoid(logits_arr / temperature)
|
|
69
|
+
loss = -(
|
|
70
|
+
targets_arr * np.log(scaled + eps)
|
|
71
|
+
+ (1 - targets_arr) * np.log(1 - scaled + eps)
|
|
72
|
+
).mean()
|
|
73
|
+
losses.append((temperature, float(loss)))
|
|
74
|
+
if not losses:
|
|
75
|
+
raise ValueError("no valid temperatures found in the grid")
|
|
76
|
+
return min(losses, key=lambda item: item[1])[0]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def fit_isotonic_calibrator(
|
|
80
|
+
scores: ArrayLike, targets: ArrayLike, *, out_of_bounds: str = "clip"
|
|
81
|
+
) -> IsotonicRegression:
|
|
82
|
+
"""
|
|
83
|
+
Fit an isotonic regression calibrator mapping scores to calibrated probabilities.
|
|
84
|
+
"""
|
|
85
|
+
calib = IsotonicRegression(out_of_bounds=out_of_bounds)
|
|
86
|
+
calib.fit(np.asarray(scores, dtype=float), np.asarray(targets, dtype=float))
|
|
87
|
+
return calib
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def apply_calibrator(calibrator: IsotonicRegression, scores: ArrayLike) -> np.ndarray:
|
|
91
|
+
"""Apply the fitted calibrator to a new set of scores."""
|
|
92
|
+
return calibrator.predict(np.asarray(scores, dtype=float))
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Synthetic data and resampling helpers for label shift experiments."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from typing import cast
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from numpy.typing import ArrayLike
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def default_mu0(d: int) -> np.ndarray:
|
|
13
|
+
"""Return the default negative-class mean (zeros)."""
|
|
14
|
+
return np.zeros(d)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def default_mu1(d: int) -> np.ndarray:
|
|
18
|
+
"""Return the default positive-class mean (first five dims at 1)."""
|
|
19
|
+
mu = np.zeros(d)
|
|
20
|
+
mu[: min(5, d)] = 1.0
|
|
21
|
+
return mu
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def concept_drift_mu1(
|
|
25
|
+
mu1: ArrayLike, shift_dim: int = 5, delta: float = 0.5
|
|
26
|
+
) -> np.ndarray:
|
|
27
|
+
"""Return mu1 shifted along one dimension to simulate concept drift."""
|
|
28
|
+
mu_arr = np.asarray(mu1, dtype=float)
|
|
29
|
+
if shift_dim >= mu_arr.shape[0]:
|
|
30
|
+
raise IndexError("shift_dim must be within the dimensionality of mu1.")
|
|
31
|
+
mu_arr = mu_arr.copy()
|
|
32
|
+
mu_arr[shift_dim] += delta
|
|
33
|
+
return mu_arr
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _validate_prevalence(pi: float) -> None:
|
|
37
|
+
if not 0 < pi < 1:
|
|
38
|
+
raise ValueError("pi must lie in the open interval (0, 1).")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _build_covariance(sigma: float | np.ndarray, d: int) -> np.ndarray:
|
|
42
|
+
if np.isscalar(sigma):
|
|
43
|
+
scalar = float(cast(float, np.asarray(sigma, dtype=float)))
|
|
44
|
+
return np.eye(d, dtype=float) * scalar
|
|
45
|
+
cov = np.asarray(sigma, dtype=float)
|
|
46
|
+
if cov.shape != (d, d):
|
|
47
|
+
raise ValueError("Covariance must be scalar or a d-by-d array.")
|
|
48
|
+
return cov
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def make_gaussian_binary(
|
|
52
|
+
n: int,
|
|
53
|
+
d: int,
|
|
54
|
+
pi: float,
|
|
55
|
+
mu0: ArrayLike,
|
|
56
|
+
mu1: ArrayLike,
|
|
57
|
+
sigma: float | np.ndarray,
|
|
58
|
+
rng: np.random.Generator,
|
|
59
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
60
|
+
"""Generate binary Gaussian data for label shift experiments."""
|
|
61
|
+
if n <= 0:
|
|
62
|
+
raise ValueError("n must be positive.")
|
|
63
|
+
if d <= 0:
|
|
64
|
+
raise ValueError("d must be positive.")
|
|
65
|
+
_validate_prevalence(pi)
|
|
66
|
+
mu0_arr = np.asarray(mu0, dtype=float)
|
|
67
|
+
mu1_arr = np.asarray(mu1, dtype=float)
|
|
68
|
+
if mu0_arr.shape != (d,) or mu1_arr.shape != (d,):
|
|
69
|
+
raise ValueError("Mean vectors must match dimensionality d.")
|
|
70
|
+
cov = _build_covariance(sigma, d)
|
|
71
|
+
|
|
72
|
+
probabilities = [1 - pi, pi]
|
|
73
|
+
labels = rng.choice([0, 1], size=n, p=probabilities)
|
|
74
|
+
X = np.empty((n, d), dtype=float)
|
|
75
|
+
if np.any(labels == 0):
|
|
76
|
+
count0 = np.sum(labels == 0)
|
|
77
|
+
X[labels == 0] = rng.multivariate_normal(mu0_arr, cov, size=count0)
|
|
78
|
+
if np.any(labels == 1):
|
|
79
|
+
count1 = np.sum(labels == 1)
|
|
80
|
+
X[labels == 1] = rng.multivariate_normal(mu1_arr, cov, size=count1)
|
|
81
|
+
return X, labels
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def resample_to_prevalence(
|
|
85
|
+
X: np.ndarray, y: np.ndarray, pi_target: float, rng: np.random.Generator
|
|
86
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
87
|
+
"""Resample a fixed pool to a target prevalence via stratified sampling."""
|
|
88
|
+
_validate_prevalence(pi_target)
|
|
89
|
+
if X.shape[0] != y.shape[0]:
|
|
90
|
+
raise ValueError("X and y must have the same number of samples.")
|
|
91
|
+
n = len(y)
|
|
92
|
+
target_pos = round(pi_target * n)
|
|
93
|
+
target_neg = n - target_pos
|
|
94
|
+
pos_idx = np.where(y == 1)[0]
|
|
95
|
+
neg_idx = np.where(y == 0)[0]
|
|
96
|
+
replace_pos = target_pos > len(pos_idx)
|
|
97
|
+
replace_neg = target_neg > len(neg_idx)
|
|
98
|
+
chosen_pos = rng.choice(pos_idx, size=target_pos, replace=replace_pos)
|
|
99
|
+
chosen_neg = rng.choice(neg_idx, size=target_neg, replace=replace_neg)
|
|
100
|
+
indices = np.concatenate([chosen_pos, chosen_neg])
|
|
101
|
+
rng.shuffle(indices)
|
|
102
|
+
return X[indices], y[indices]
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def make_multimodal_binary(
|
|
106
|
+
n: int,
|
|
107
|
+
d: int,
|
|
108
|
+
pi: float,
|
|
109
|
+
components_neg: Sequence[tuple[ArrayLike, float]],
|
|
110
|
+
components_pos: Sequence[tuple[ArrayLike, float]],
|
|
111
|
+
sigma: float | np.ndarray,
|
|
112
|
+
rng: np.random.Generator,
|
|
113
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
114
|
+
"""Generate a binary dataset where each class is a mixture of Gaussians."""
|
|
115
|
+
if n <= 0:
|
|
116
|
+
raise ValueError("n must be positive.")
|
|
117
|
+
if d <= 0:
|
|
118
|
+
raise ValueError("d must be positive.")
|
|
119
|
+
_validate_prevalence(pi)
|
|
120
|
+
neg_means, neg_weights = _prep_mixture(components_neg, d)
|
|
121
|
+
pos_means, pos_weights = _prep_mixture(components_pos, d)
|
|
122
|
+
cov = _build_covariance(sigma, d)
|
|
123
|
+
labels = rng.choice([0, 1], size=n, p=[1 - pi, pi])
|
|
124
|
+
X = np.empty((n, d), dtype=float)
|
|
125
|
+
if np.any(labels == 0):
|
|
126
|
+
X[labels == 0] = _sample_mixture(
|
|
127
|
+
cov, neg_means, neg_weights, rng, size=np.sum(labels == 0)
|
|
128
|
+
)
|
|
129
|
+
if np.any(labels == 1):
|
|
130
|
+
X[labels == 1] = _sample_mixture(
|
|
131
|
+
cov, pos_means, pos_weights, rng, size=np.sum(labels == 1)
|
|
132
|
+
)
|
|
133
|
+
return X, labels
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _prep_mixture(
|
|
137
|
+
components: Sequence[tuple[ArrayLike, float]], d: int
|
|
138
|
+
) -> tuple[list[np.ndarray], np.ndarray]:
|
|
139
|
+
if not components:
|
|
140
|
+
raise ValueError("At least one mixture component is required.")
|
|
141
|
+
means: list[np.ndarray] = []
|
|
142
|
+
weights = []
|
|
143
|
+
for mean, weight in components:
|
|
144
|
+
if weight < 0:
|
|
145
|
+
raise ValueError("Mixture weights must be non-negative.")
|
|
146
|
+
arr = np.asarray(mean, dtype=float)
|
|
147
|
+
if arr.shape != (d,):
|
|
148
|
+
raise ValueError("Component means must match dimensionality d.")
|
|
149
|
+
means.append(arr)
|
|
150
|
+
weights.append(weight)
|
|
151
|
+
weights_arr = np.asarray(weights, dtype=float)
|
|
152
|
+
total = float(weights_arr.sum())
|
|
153
|
+
if total <= 0:
|
|
154
|
+
raise ValueError("Mixture weights must sum to a positive value.")
|
|
155
|
+
return means, weights_arr / total
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _sample_mixture(
|
|
159
|
+
cov: np.ndarray,
|
|
160
|
+
means: Sequence[np.ndarray],
|
|
161
|
+
weights: np.ndarray,
|
|
162
|
+
rng: np.random.Generator,
|
|
163
|
+
*,
|
|
164
|
+
size: int,
|
|
165
|
+
) -> np.ndarray:
|
|
166
|
+
if size == 0:
|
|
167
|
+
return np.empty((0, cov.shape[0]))
|
|
168
|
+
indices = rng.choice(len(means), size=size, p=weights)
|
|
169
|
+
samples = [rng.multivariate_normal(means[idx], cov, size=1)[0] for idx in indices]
|
|
170
|
+
return np.vstack(samples)
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Lightweight helpers for monitoring feature drift."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _ks_statistic(reference: np.ndarray, target: np.ndarray) -> float:
|
|
12
|
+
if reference.size == 0 or target.size == 0:
|
|
13
|
+
return 0.0
|
|
14
|
+
reference_sorted = np.sort(reference)
|
|
15
|
+
target_sorted = np.sort(target)
|
|
16
|
+
combined = np.sort(np.concatenate([reference_sorted, target_sorted]))
|
|
17
|
+
cdf_ref = (
|
|
18
|
+
np.searchsorted(reference_sorted, combined, side="right")
|
|
19
|
+
/ reference_sorted.size
|
|
20
|
+
)
|
|
21
|
+
cdf_target = (
|
|
22
|
+
np.searchsorted(target_sorted, combined, side="right") / target_sorted.size
|
|
23
|
+
)
|
|
24
|
+
return float(np.max(np.abs(cdf_ref - cdf_target)))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _covariance_matrix(data: np.ndarray) -> np.ndarray:
|
|
28
|
+
return np.atleast_2d(np.cov(data, rowvar=False))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _correlation_matrix(data: np.ndarray) -> np.ndarray:
|
|
32
|
+
if data.shape[1] == 0:
|
|
33
|
+
return np.zeros((0, 0))
|
|
34
|
+
corr = np.corrcoef(data, rowvar=False)
|
|
35
|
+
if corr.ndim == 0:
|
|
36
|
+
corr = np.array([[corr]])
|
|
37
|
+
return np.nan_to_num(corr, nan=0.0, posinf=0.0, neginf=0.0)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def covariance_frobenius_diff(reference: np.ndarray, target: np.ndarray) -> float:
|
|
41
|
+
"""Return the Frobenius difference between training and target covariance matrices."""
|
|
42
|
+
cov_ref = _covariance_matrix(reference)
|
|
43
|
+
cov_target = _covariance_matrix(target)
|
|
44
|
+
return float(np.linalg.norm(cov_ref - cov_target, ord="fro"))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def correlation_mean_diff(reference: np.ndarray, target: np.ndarray) -> float:
|
|
48
|
+
"""Compute average absolute deviation between correlation matrices (off-diagonal)."""
|
|
49
|
+
d = reference.shape[1]
|
|
50
|
+
if d <= 1:
|
|
51
|
+
return 0.0
|
|
52
|
+
corr_ref = _correlation_matrix(reference)
|
|
53
|
+
corr_target = _correlation_matrix(target)
|
|
54
|
+
mask = ~np.eye(d, dtype=bool)
|
|
55
|
+
diffs = np.abs(corr_ref - corr_target)
|
|
56
|
+
return float(np.mean(diffs[mask]))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def multivariate_projection_ks(
|
|
60
|
+
reference: np.ndarray,
|
|
61
|
+
target: np.ndarray,
|
|
62
|
+
*,
|
|
63
|
+
projections: int = 12,
|
|
64
|
+
rng: np.random.Generator | None = None,
|
|
65
|
+
) -> list[float]:
|
|
66
|
+
"""Approximate multivariate drift via KS on random projections."""
|
|
67
|
+
rng = rng or np.random.default_rng(0)
|
|
68
|
+
d = reference.shape[1]
|
|
69
|
+
if d == 0 or projections <= 0:
|
|
70
|
+
return [0.0]
|
|
71
|
+
stats: list[float] = []
|
|
72
|
+
for _ in range(projections):
|
|
73
|
+
direction = np.asarray(rng.normal(size=d), dtype=float)
|
|
74
|
+
norm = np.linalg.norm(direction)
|
|
75
|
+
if norm == 0:
|
|
76
|
+
continue
|
|
77
|
+
direction /= norm
|
|
78
|
+
ref_proj = reference @ direction
|
|
79
|
+
target_proj = target @ direction
|
|
80
|
+
stats.append(_ks_statistic(ref_proj, target_proj))
|
|
81
|
+
return stats or [0.0]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def univariate_feature_stats(
|
|
85
|
+
X_ref: ArrayLike,
|
|
86
|
+
X_target: ArrayLike,
|
|
87
|
+
) -> list[dict]:
|
|
88
|
+
"""Return per-feature drift statistics between reference and target arrays."""
|
|
89
|
+
ref_arr = np.asarray(X_ref, dtype=float)
|
|
90
|
+
target_arr = np.asarray(X_target, dtype=float)
|
|
91
|
+
if ref_arr.ndim != 2 or target_arr.ndim != 2:
|
|
92
|
+
raise ValueError("input arrays must be two-dimensional")
|
|
93
|
+
if ref_arr.shape[1] != target_arr.shape[1]:
|
|
94
|
+
raise ValueError("feature dimensionality must match")
|
|
95
|
+
if ref_arr.shape[0] == 0 or target_arr.shape[0] == 0:
|
|
96
|
+
raise ValueError("input datasets must contain samples")
|
|
97
|
+
stats: list[dict] = []
|
|
98
|
+
for idx in range(ref_arr.shape[1]):
|
|
99
|
+
ref_col = ref_arr[:, idx]
|
|
100
|
+
target_col = target_arr[:, idx]
|
|
101
|
+
stats.append(
|
|
102
|
+
{
|
|
103
|
+
"feature": idx,
|
|
104
|
+
"mean_diff": float(np.abs(ref_col.mean() - target_col.mean())),
|
|
105
|
+
"std_diff": float(np.abs(ref_col.std(ddof=0) - target_col.std(ddof=0))),
|
|
106
|
+
"ks_stat": _ks_statistic(ref_col, target_col),
|
|
107
|
+
}
|
|
108
|
+
)
|
|
109
|
+
return stats
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def feature_drift_summary(stats: Sequence[dict]) -> dict[str, float]:
|
|
113
|
+
"""Aggregate univariate stats into summary metrics suitable for tracking."""
|
|
114
|
+
if not stats:
|
|
115
|
+
raise ValueError("stats cannot be empty")
|
|
116
|
+
mean_diffs = [entry["mean_diff"] for entry in stats]
|
|
117
|
+
std_diffs = [entry["std_diff"] for entry in stats]
|
|
118
|
+
ks_stats = [entry["ks_stat"] for entry in stats]
|
|
119
|
+
return {
|
|
120
|
+
"feature_max_mean_diff": float(max(mean_diffs)),
|
|
121
|
+
"feature_max_std_diff": float(max(std_diffs)),
|
|
122
|
+
"feature_max_ks": float(max(ks_stats)),
|
|
123
|
+
"feature_mean_ks": float(np.mean(ks_stats)),
|
|
124
|
+
}
|