evalsuite-python 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalsuite/__init__.py +159 -0
- evalsuite/__main__.py +5 -0
- evalsuite/api.py +264 -0
- evalsuite/benchmarks.py +272 -0
- evalsuite/classification/__init__.py +51 -0
- evalsuite/classification/_common.py +111 -0
- evalsuite/classification/metrics.py +943 -0
- evalsuite/cli/__init__.py +5 -0
- evalsuite/cli/main.py +389 -0
- evalsuite/core/__init__.py +1 -0
- evalsuite/core/context.py +181 -0
- evalsuite/core/exceptions.py +52 -0
- evalsuite/core/export.py +100 -0
- evalsuite/core/registry.py +90 -0
- evalsuite/core/result.py +379 -0
- evalsuite/core/types.py +23 -0
- evalsuite/core/validation.py +202 -0
- evalsuite/plot.py +296 -0
- evalsuite/py.typed +0 -0
- evalsuite/regression/__init__.py +41 -0
- evalsuite/regression/metrics.py +604 -0
- evalsuite/reporting.py +241 -0
- evalsuite/stats/__init__.py +25 -0
- evalsuite/stats/_resolve.py +90 -0
- evalsuite/stats/compare.py +414 -0
- evalsuite/stats/effect.py +113 -0
- evalsuite/stats/intervals.py +320 -0
- evalsuite/stats/paired.py +207 -0
- evalsuite/stats/results.py +129 -0
- evalsuite/version.py +3 -0
- evalsuite_python-0.1.0.dist-info/METADATA +247 -0
- evalsuite_python-0.1.0.dist-info/RECORD +35 -0
- evalsuite_python-0.1.0.dist-info/WHEEL +4 -0
- evalsuite_python-0.1.0.dist-info/entry_points.txt +2 -0
- evalsuite_python-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
"""Confidence intervals: bootstrap (percentile, basic, BCa), Wilson and Clopper-Pearson, DeLong."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import warnings
|
|
6
|
+
from typing import Any, Literal, Optional
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from numpy.typing import NDArray
|
|
10
|
+
from scipy import special, stats
|
|
11
|
+
|
|
12
|
+
from ..core.exceptions import EvalSuiteError, InputValidationError, StatisticalTestError
|
|
13
|
+
from ..core.types import ArrayLike
|
|
14
|
+
from ..core.validation import to_numpy
|
|
15
|
+
from ._resolve import MetricCall, resolve_metric
|
|
16
|
+
from .results import ConfidenceInterval
|
|
17
|
+
|
|
18
|
+
__all__ = ["accuracy_ci", "bootstrap_ci", "proportion_ci", "roc_auc_ci"]
|
|
19
|
+
|
|
20
|
+
BootstrapMethod = Literal["percentile", "basic", "bca"]
|
|
21
|
+
MAX_FAILED_FRACTION = 0.1
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _check_level(level: float) -> float:
|
|
25
|
+
if not (isinstance(level, (int, float)) and 0 < level < 1):
|
|
26
|
+
raise InputValidationError("level must be between 0 and 1, for example 0.95.")
|
|
27
|
+
return float(level)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _check_resamples(n_resamples: int) -> int:
|
|
31
|
+
if not isinstance(n_resamples, (int, np.integer)) or n_resamples < 100:
|
|
32
|
+
raise InputValidationError("n_resamples must be an integer of at least 100 (1000 or more recommended).")
|
|
33
|
+
return int(n_resamples)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def resample_indices(
|
|
37
|
+
n: int, n_resamples: int, rng: np.random.Generator, strata: Optional[NDArray[Any]] = None
|
|
38
|
+
) -> NDArray[np.int64]:
|
|
39
|
+
"""(n_resamples, n) indices drawn with replacement, optionally within strata (keeps class balance)."""
|
|
40
|
+
if strata is None:
|
|
41
|
+
return rng.integers(0, n, size=(n_resamples, n))
|
|
42
|
+
out = np.empty((n_resamples, n), dtype=np.int64)
|
|
43
|
+
pos = 0
|
|
44
|
+
for value in np.unique(strata, axis=0) if strata.ndim > 1 else np.unique(strata):
|
|
45
|
+
members = np.flatnonzero(np.all(strata == value, axis=1) if strata.ndim > 1 else strata == value)
|
|
46
|
+
k = members.shape[0]
|
|
47
|
+
out[:, pos : pos + k] = members[rng.integers(0, k, size=(n_resamples, k))]
|
|
48
|
+
pos += k
|
|
49
|
+
return out
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def bootstrap_distribution(call: MetricCall, indices: NDArray[np.int64]) -> tuple[NDArray[np.float64], int]:
|
|
53
|
+
"""Metric value on each resample; resamples where the metric is undefined become NaN."""
|
|
54
|
+
values = np.empty(indices.shape[0])
|
|
55
|
+
failed = 0
|
|
56
|
+
with warnings.catch_warnings():
|
|
57
|
+
warnings.simplefilter("ignore") # zero-division warnings inside resamples are expected noise
|
|
58
|
+
for b, idx in enumerate(indices):
|
|
59
|
+
try:
|
|
60
|
+
values[b] = call(idx)
|
|
61
|
+
except EvalSuiteError:
|
|
62
|
+
values[b] = np.nan
|
|
63
|
+
failed += 1
|
|
64
|
+
return values, failed
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _jackknife(call: MetricCall, n: int, rng: np.random.Generator, groups: int) -> NDArray[np.float64]:
|
|
68
|
+
"""Leave-one-out (or leave-one-group-out for large n) estimates for the BCa acceleration."""
|
|
69
|
+
blocks = [np.array([i]) for i in range(n)] if n <= groups else np.array_split(rng.permutation(n), groups)
|
|
70
|
+
everything = np.arange(n)
|
|
71
|
+
out = np.empty(len(blocks))
|
|
72
|
+
with warnings.catch_warnings():
|
|
73
|
+
warnings.simplefilter("ignore")
|
|
74
|
+
for j, block in enumerate(blocks):
|
|
75
|
+
keep = np.setdiff1d(everything, block, assume_unique=True)
|
|
76
|
+
try:
|
|
77
|
+
out[j] = call(keep)
|
|
78
|
+
except EvalSuiteError:
|
|
79
|
+
out[j] = np.nan
|
|
80
|
+
return out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def bootstrap_ci(
|
|
84
|
+
metric: Any,
|
|
85
|
+
y_true: ArrayLike,
|
|
86
|
+
y_pred: Optional[ArrayLike] = None,
|
|
87
|
+
*,
|
|
88
|
+
y_prob: Optional[ArrayLike] = None,
|
|
89
|
+
sample_weight: Optional[ArrayLike] = None,
|
|
90
|
+
level: float = 0.95,
|
|
91
|
+
method: BootstrapMethod = "bca",
|
|
92
|
+
n_resamples: int = 2000,
|
|
93
|
+
random_state: Optional[int] = None,
|
|
94
|
+
stratify: Optional[bool] = None,
|
|
95
|
+
jackknife_groups: int = 1000,
|
|
96
|
+
**metric_kwargs: Any,
|
|
97
|
+
) -> ConfidenceInterval:
|
|
98
|
+
"""Bootstrap confidence interval for any metric.
|
|
99
|
+
|
|
100
|
+
``metric`` is a metric function (``es.f1``) or its name (``"f1"``). Pass ``y_pred`` for label/value
|
|
101
|
+
metrics or ``y_prob`` for probability metrics; extra keyword arguments go to the metric (for example
|
|
102
|
+
``average="macro"``). Resampling is stratified by class for classification targets unless
|
|
103
|
+
``stratify=False``, so every resample contains every class. For classification metrics that accept
|
|
104
|
+
``labels``, the full label set is fixed across resamples.
|
|
105
|
+
|
|
106
|
+
``method``: ``"bca"`` (bias-corrected and accelerated; default, Efron 1987), ``"percentile"`` or
|
|
107
|
+
``"basic"``. Set ``random_state`` for reproducible intervals. Resamples on which the metric is undefined
|
|
108
|
+
are counted in ``params["failed_resamples"]``; more than 10% raises :class:`StatisticalTestError`.
|
|
109
|
+
|
|
110
|
+
References: Efron B. Better bootstrap confidence intervals. JASA. 1987;82(397):171-185.
|
|
111
|
+
Efron B, Tibshirani RJ. An Introduction to the Bootstrap. Chapman & Hall; 1993.
|
|
112
|
+
"""
|
|
113
|
+
level = _check_level(level)
|
|
114
|
+
n_resamples = _check_resamples(n_resamples)
|
|
115
|
+
if method not in ("percentile", "basic", "bca"):
|
|
116
|
+
raise InputValidationError("method must be 'bca', 'percentile' or 'basic'.")
|
|
117
|
+
fn, name = resolve_metric(metric)
|
|
118
|
+
call = MetricCall(fn, y_true, y_pred, y_prob, sample_weight, metric_kwargs)
|
|
119
|
+
estimate = call()
|
|
120
|
+
rng = np.random.default_rng(random_state)
|
|
121
|
+
use_strata = call.categorical if stratify is None else bool(stratify)
|
|
122
|
+
idx = resample_indices(call.n, n_resamples, rng, call.y_true if use_strata else None)
|
|
123
|
+
boot, failed = bootstrap_distribution(call, idx)
|
|
124
|
+
if failed > MAX_FAILED_FRACTION * n_resamples:
|
|
125
|
+
raise StatisticalTestError(
|
|
126
|
+
f"The metric was undefined on {failed} of {n_resamples} bootstrap resamples. The sample is probably "
|
|
127
|
+
"too small or too imbalanced for a reliable bootstrap interval."
|
|
128
|
+
)
|
|
129
|
+
alpha = 1 - level
|
|
130
|
+
valid = boot[~np.isnan(boot)]
|
|
131
|
+
params: dict[str, Any] = {
|
|
132
|
+
"n_resamples": n_resamples,
|
|
133
|
+
"random_state": random_state,
|
|
134
|
+
"stratified": use_strata,
|
|
135
|
+
"failed_resamples": failed,
|
|
136
|
+
**{k: v for k, v in metric_kwargs.items()},
|
|
137
|
+
}
|
|
138
|
+
if method == "percentile":
|
|
139
|
+
low, high = np.quantile(valid, [alpha / 2, 1 - alpha / 2])
|
|
140
|
+
elif method == "basic":
|
|
141
|
+
q_low, q_high = np.quantile(valid, [alpha / 2, 1 - alpha / 2])
|
|
142
|
+
low, high = 2 * estimate - q_high, 2 * estimate - q_low
|
|
143
|
+
else:
|
|
144
|
+
low, high, extra = _bca(call, valid, estimate, alpha, rng, jackknife_groups)
|
|
145
|
+
params.update(extra)
|
|
146
|
+
return ConfidenceInterval(estimate, float(low), float(high), level, f"bootstrap-{method}", name, params)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _bca(
|
|
150
|
+
call: MetricCall,
|
|
151
|
+
boot: NDArray[np.float64],
|
|
152
|
+
estimate: float,
|
|
153
|
+
alpha: float,
|
|
154
|
+
rng: np.random.Generator,
|
|
155
|
+
groups: int,
|
|
156
|
+
) -> tuple[float, float, dict[str, Any]]:
|
|
157
|
+
# bias correction: share of the bootstrap distribution below the estimate (ties count half)
|
|
158
|
+
share = (np.sum(boot < estimate) + 0.5 * np.sum(boot == estimate)) / boot.shape[0]
|
|
159
|
+
if share <= 0 or share >= 1:
|
|
160
|
+
warnings.warn(
|
|
161
|
+
"BCa is undefined because the estimate lies outside the bootstrap distribution; "
|
|
162
|
+
"returning the percentile interval.",
|
|
163
|
+
RuntimeWarning,
|
|
164
|
+
stacklevel=3,
|
|
165
|
+
)
|
|
166
|
+
low, high = np.quantile(boot, [alpha / 2, 1 - alpha / 2])
|
|
167
|
+
return float(low), float(high), {"bca_fallback": "percentile"}
|
|
168
|
+
z0 = special.ndtri(share)
|
|
169
|
+
jack = _jackknife(call, call.n, rng, groups)
|
|
170
|
+
jack = jack[~np.isnan(jack)]
|
|
171
|
+
diffs = jack.mean() - jack
|
|
172
|
+
denom = 6.0 * np.sum(diffs**2) ** 1.5
|
|
173
|
+
accel = float(np.sum(diffs**3) / denom) if denom > 0 else 0.0
|
|
174
|
+
probs = []
|
|
175
|
+
for z_alpha in (special.ndtri(alpha / 2), special.ndtri(1 - alpha / 2)):
|
|
176
|
+
num = z0 + z_alpha
|
|
177
|
+
probs.append(special.ndtr(z0 + num / (1 - accel * num)))
|
|
178
|
+
low, high = np.quantile(boot, probs)
|
|
179
|
+
return float(low), float(high), {"bias_correction": float(z0), "acceleration": accel}
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def proportion_ci(
|
|
183
|
+
successes: int,
|
|
184
|
+
n: int,
|
|
185
|
+
*,
|
|
186
|
+
level: float = 0.95,
|
|
187
|
+
method: Literal["wilson", "clopper-pearson", "normal"] = "wilson",
|
|
188
|
+
) -> ConfidenceInterval:
|
|
189
|
+
"""Confidence interval for a proportion ``successes / n``.
|
|
190
|
+
|
|
191
|
+
``"wilson"`` (default; good coverage even for small n or extreme proportions), ``"clopper-pearson"``
|
|
192
|
+
(exact, conservative) or ``"normal"`` (Wald; shown for comparison, not recommended).
|
|
193
|
+
|
|
194
|
+
References: Wilson EB. JASA. 1927;22(158):209-212. Clopper CJ, Pearson ES. Biometrika. 1934;26(4):404-413.
|
|
195
|
+
Brown LD, Cai TT, DasGupta A. Interval estimation for a binomial proportion. Stat Sci. 2001;16(2):101-133.
|
|
196
|
+
"""
|
|
197
|
+
level = _check_level(level)
|
|
198
|
+
if not (isinstance(n, (int, np.integer)) and n > 0):
|
|
199
|
+
raise InputValidationError("n must be a positive integer.")
|
|
200
|
+
if not (0 <= successes <= n):
|
|
201
|
+
raise InputValidationError("successes must be between 0 and n.")
|
|
202
|
+
p = successes / n
|
|
203
|
+
alpha = 1 - level
|
|
204
|
+
z = special.ndtri(1 - alpha / 2)
|
|
205
|
+
if method == "wilson":
|
|
206
|
+
denom = 1 + z**2 / n
|
|
207
|
+
centre = (p + z**2 / (2 * n)) / denom
|
|
208
|
+
half = z * np.sqrt(p * (1 - p) / n + z**2 / (4 * n**2)) / denom
|
|
209
|
+
low, high = centre - half, centre + half
|
|
210
|
+
elif method == "clopper-pearson":
|
|
211
|
+
low = 0.0 if successes == 0 else stats.beta.ppf(alpha / 2, successes, n - successes + 1)
|
|
212
|
+
high = 1.0 if successes == n else stats.beta.ppf(1 - alpha / 2, successes + 1, n - successes)
|
|
213
|
+
elif method == "normal":
|
|
214
|
+
half = z * np.sqrt(p * (1 - p) / n)
|
|
215
|
+
low, high = max(0.0, p - half), min(1.0, p + half)
|
|
216
|
+
else:
|
|
217
|
+
raise InputValidationError("method must be 'wilson', 'clopper-pearson' or 'normal'.")
|
|
218
|
+
return ConfidenceInterval(
|
|
219
|
+
p, float(low), float(high), level, method, "proportion", {"successes": int(successes), "n": int(n)}
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def accuracy_ci(
|
|
224
|
+
y_true: ArrayLike,
|
|
225
|
+
y_pred: ArrayLike,
|
|
226
|
+
*,
|
|
227
|
+
level: float = 0.95,
|
|
228
|
+
method: Literal["wilson", "clopper-pearson", "normal"] = "wilson",
|
|
229
|
+
) -> ConfidenceInterval:
|
|
230
|
+
"""Analytic confidence interval for accuracy (a proportion of correct predictions)."""
|
|
231
|
+
yt = to_numpy(y_true, "y_true", allow_2d=True)
|
|
232
|
+
yp = to_numpy(y_pred, "y_pred", allow_2d=True)
|
|
233
|
+
if yt.shape != yp.shape:
|
|
234
|
+
raise InputValidationError(f"y_true has shape {yt.shape} but y_pred has shape {yp.shape}.")
|
|
235
|
+
correct = (yt == yp).all(axis=1) if yt.ndim == 2 else yt == yp
|
|
236
|
+
ci = proportion_ci(int(correct.sum()), int(correct.shape[0]), level=level, method=method)
|
|
237
|
+
return ConfidenceInterval(ci.estimate, ci.low, ci.high, level, method, "accuracy", ci.params)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
# ---- DeLong ---------------------------------------------------------------------------------------
|
|
241
|
+
def delong_placements(
|
|
242
|
+
y: NDArray[np.float64], scores: NDArray[np.float64]
|
|
243
|
+
) -> tuple[NDArray[np.float64], NDArray[np.float64], NDArray[np.float64]]:
|
|
244
|
+
"""AUCs and structural components for k score columns (Sun & Xu fast DeLong).
|
|
245
|
+
|
|
246
|
+
Returns (auc[k], v10[k, m], v01[k, n]) where m/n are the numbers of positives/negatives.
|
|
247
|
+
"""
|
|
248
|
+
pos = scores[:, y == 1]
|
|
249
|
+
neg = scores[:, y == 0]
|
|
250
|
+
m, n = pos.shape[1], neg.shape[1]
|
|
251
|
+
if m == 0 or n == 0:
|
|
252
|
+
from ..core.exceptions import MetricInputError
|
|
253
|
+
|
|
254
|
+
raise MetricInputError("DeLong's method needs at least one positive and one negative observation.")
|
|
255
|
+
tx = np.apply_along_axis(stats.rankdata, 1, pos)
|
|
256
|
+
ty = np.apply_along_axis(stats.rankdata, 1, neg)
|
|
257
|
+
tz = np.apply_along_axis(stats.rankdata, 1, np.concatenate([pos, neg], axis=1))
|
|
258
|
+
auc = (tz[:, :m].sum(axis=1) - m * (m + 1) / 2) / (m * n)
|
|
259
|
+
v10 = (tz[:, :m] - tx) / n
|
|
260
|
+
v01 = 1.0 - (tz[:, m:] - ty) / m
|
|
261
|
+
return auc, v10, v01
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def delong_covariance(v10: NDArray[np.float64], v01: NDArray[np.float64]) -> NDArray[np.float64]:
|
|
265
|
+
m, n = v10.shape[1], v01.shape[1]
|
|
266
|
+
s10 = np.atleast_2d(np.cov(v10)) if m > 1 else np.zeros((v10.shape[0],) * 2)
|
|
267
|
+
s01 = np.atleast_2d(np.cov(v01)) if n > 1 else np.zeros((v01.shape[0],) * 2)
|
|
268
|
+
cov: NDArray[np.float64] = s10 / m + s01 / n
|
|
269
|
+
return cov
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def binary_scores(
|
|
273
|
+
y_true: ArrayLike, y_prob: ArrayLike, pos_label: Any
|
|
274
|
+
) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
|
|
275
|
+
yt = to_numpy(y_true, "y_true")
|
|
276
|
+
s = to_numpy(y_prob, "y_prob").astype(np.float64)
|
|
277
|
+
if s.shape[0] != yt.shape[0]:
|
|
278
|
+
raise InputValidationError(
|
|
279
|
+
"y_true and y_prob must contain the same number of observations. "
|
|
280
|
+
f"Received {yt.shape[0]} and {s.shape[0]}."
|
|
281
|
+
)
|
|
282
|
+
labels = np.unique(yt)
|
|
283
|
+
if labels.shape[0] > 2:
|
|
284
|
+
raise InputValidationError("DeLong's method is for binary targets.")
|
|
285
|
+
if pos_label is None:
|
|
286
|
+
if not set(labels.tolist()) <= {0, 1}:
|
|
287
|
+
raise InputValidationError(f"Labels are {labels.tolist()}; specify pos_label=...")
|
|
288
|
+
pos_label = 1
|
|
289
|
+
return (yt == pos_label).astype(np.float64), s
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def roc_auc_ci(
|
|
293
|
+
y_true: ArrayLike,
|
|
294
|
+
y_prob: ArrayLike,
|
|
295
|
+
*,
|
|
296
|
+
level: float = 0.95,
|
|
297
|
+
pos_label: Any = None,
|
|
298
|
+
) -> ConfidenceInterval:
|
|
299
|
+
"""DeLong confidence interval for a binary ROC AUC (normal approximation, clipped to [0, 1]).
|
|
300
|
+
|
|
301
|
+
Reference: DeLong ER, DeLong DM, Clarke-Pearson DL. Comparing the areas under two or more correlated
|
|
302
|
+
receiver operating characteristic curves: a nonparametric approach. Biometrics. 1988;44(3):837-845.
|
|
303
|
+
Sun X, Xu W. Fast implementation of DeLong's algorithm. IEEE Signal Process Lett. 2014;21(11):1389-1393.
|
|
304
|
+
"""
|
|
305
|
+
level = _check_level(level)
|
|
306
|
+
y, s = binary_scores(y_true, y_prob, pos_label)
|
|
307
|
+
auc, v10, v01 = delong_placements(y, s[None, :])
|
|
308
|
+
var = float(delong_covariance(v10, v01)[0, 0])
|
|
309
|
+
se = np.sqrt(var)
|
|
310
|
+
z = special.ndtri(1 - (1 - level) / 2)
|
|
311
|
+
a = float(auc[0])
|
|
312
|
+
return ConfidenceInterval(
|
|
313
|
+
a,
|
|
314
|
+
max(0.0, a - z * se),
|
|
315
|
+
min(1.0, a + z * se),
|
|
316
|
+
level,
|
|
317
|
+
"delong",
|
|
318
|
+
"roc_auc",
|
|
319
|
+
{"standard_error": float(se), "n_positive": int(y.sum()), "n_negative": int((1 - y).sum())},
|
|
320
|
+
)
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""Paired tests for comparing two models evaluated on the same observations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Literal, Optional
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from scipy import special, stats
|
|
9
|
+
|
|
10
|
+
from ..core.exceptions import InputValidationError
|
|
11
|
+
from ..core.types import ArrayLike
|
|
12
|
+
from ..core.validation import to_numpy
|
|
13
|
+
from ._resolve import MetricCall, resolve_metric
|
|
14
|
+
from .intervals import (
|
|
15
|
+
_check_level,
|
|
16
|
+
_check_resamples,
|
|
17
|
+
binary_scores,
|
|
18
|
+
bootstrap_distribution,
|
|
19
|
+
delong_covariance,
|
|
20
|
+
delong_placements,
|
|
21
|
+
resample_indices,
|
|
22
|
+
)
|
|
23
|
+
from .results import ConfidenceInterval, TestResult
|
|
24
|
+
|
|
25
|
+
__all__ = ["delong_test", "mcnemar_test", "paired_bootstrap_test"]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def mcnemar_test(
|
|
29
|
+
y_true: ArrayLike,
|
|
30
|
+
y_pred_a: ArrayLike,
|
|
31
|
+
y_pred_b: ArrayLike,
|
|
32
|
+
*,
|
|
33
|
+
exact: Optional[bool] = None,
|
|
34
|
+
correction: bool = True,
|
|
35
|
+
) -> TestResult:
|
|
36
|
+
"""McNemar's test: do two classifiers have the same error rate on the same observations?
|
|
37
|
+
|
|
38
|
+
Uses only discordant pairs: b = A right & B wrong, c = A wrong & B right. ``exact=None`` (default) uses
|
|
39
|
+
the exact binomial test when b + c < 25 and the chi-squared test otherwise (with Edwards' continuity
|
|
40
|
+
correction unless ``correction=False``). ``estimate`` is the accuracy difference A − B.
|
|
41
|
+
|
|
42
|
+
References: McNemar Q. Psychometrika. 1947;12(2):153-157. Edwards AL. Psychometrika. 1948;13:185-187.
|
|
43
|
+
Dietterich TG. Approximate statistical tests for comparing supervised classification learning
|
|
44
|
+
algorithms. Neural Comput. 1998;10(7):1895-1923.
|
|
45
|
+
"""
|
|
46
|
+
yt = to_numpy(y_true, "y_true", allow_2d=True)
|
|
47
|
+
a = to_numpy(y_pred_a, "y_pred_a", allow_2d=True)
|
|
48
|
+
b_ = to_numpy(y_pred_b, "y_pred_b", allow_2d=True)
|
|
49
|
+
if not (yt.shape == a.shape == b_.shape):
|
|
50
|
+
raise InputValidationError(
|
|
51
|
+
f"y_true, y_pred_a and y_pred_b must have the same shape; got {yt.shape}, {a.shape}, {b_.shape}."
|
|
52
|
+
)
|
|
53
|
+
ok_a = (yt == a).all(axis=1) if yt.ndim == 2 else yt == a
|
|
54
|
+
ok_b = (yt == b_).all(axis=1) if yt.ndim == 2 else yt == b_
|
|
55
|
+
b = int(np.sum(ok_a & ~ok_b))
|
|
56
|
+
c = int(np.sum(~ok_a & ok_b))
|
|
57
|
+
n_disc = b + c
|
|
58
|
+
use_exact = n_disc < 25 if exact is None else bool(exact)
|
|
59
|
+
estimate = (b - c) / yt.shape[0]
|
|
60
|
+
params = {"b": b, "c": c, "n": int(yt.shape[0])}
|
|
61
|
+
if n_disc == 0:
|
|
62
|
+
return TestResult(
|
|
63
|
+
"mcnemar-exact" if use_exact else "mcnemar",
|
|
64
|
+
0.0,
|
|
65
|
+
1.0,
|
|
66
|
+
estimate=estimate,
|
|
67
|
+
params={**params, "note": "no discordant pairs: the classifiers agree on every observation"},
|
|
68
|
+
)
|
|
69
|
+
if use_exact:
|
|
70
|
+
p = min(1.0, 2 * stats.binom.cdf(min(b, c), n_disc, 0.5))
|
|
71
|
+
return TestResult("mcnemar-exact", float(min(b, c)), float(p), estimate=estimate, params=params)
|
|
72
|
+
stat = (abs(b - c) - (1 if correction else 0)) ** 2 / n_disc
|
|
73
|
+
p = stats.chi2.sf(stat, 1)
|
|
74
|
+
return TestResult(
|
|
75
|
+
"mcnemar", float(stat), float(p), estimate=estimate, params={**params, "continuity_correction": correction}
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def delong_test(
|
|
80
|
+
y_true: ArrayLike,
|
|
81
|
+
y_prob_a: ArrayLike,
|
|
82
|
+
y_prob_b: ArrayLike,
|
|
83
|
+
*,
|
|
84
|
+
pos_label: Any = None,
|
|
85
|
+
level: float = 0.95,
|
|
86
|
+
) -> TestResult:
|
|
87
|
+
"""DeLong's test for two correlated ROC AUCs (same observations, two models). ``estimate`` is AUC_A − AUC_B
|
|
88
|
+
and ``ci`` its confidence interval.
|
|
89
|
+
|
|
90
|
+
Reference: DeLong ER, DeLong DM, Clarke-Pearson DL. Biometrics. 1988;44(3):837-845.
|
|
91
|
+
"""
|
|
92
|
+
level = _check_level(level)
|
|
93
|
+
y, sa = binary_scores(y_true, y_prob_a, pos_label)
|
|
94
|
+
_, sb = binary_scores(y_true, y_prob_b, pos_label)
|
|
95
|
+
if sb.shape != sa.shape:
|
|
96
|
+
raise InputValidationError("y_prob_a and y_prob_b must have the same length.")
|
|
97
|
+
auc, v10, v01 = delong_placements(y, np.vstack([sa, sb]))
|
|
98
|
+
cov = delong_covariance(v10, v01)
|
|
99
|
+
diff = float(auc[0] - auc[1])
|
|
100
|
+
var = float(cov[0, 0] + cov[1, 1] - 2 * cov[0, 1])
|
|
101
|
+
if var <= 0:
|
|
102
|
+
z, p = 0.0, 1.0
|
|
103
|
+
se = 0.0
|
|
104
|
+
else:
|
|
105
|
+
se = float(np.sqrt(var))
|
|
106
|
+
z = diff / se
|
|
107
|
+
p = float(2 * special.ndtr(-abs(z)))
|
|
108
|
+
zc = special.ndtri(1 - (1 - level) / 2)
|
|
109
|
+
ci = ConfidenceInterval(diff, diff - zc * se, diff + zc * se, level, "delong", "roc_auc_difference")
|
|
110
|
+
return TestResult(
|
|
111
|
+
"delong",
|
|
112
|
+
z,
|
|
113
|
+
p,
|
|
114
|
+
estimate=diff,
|
|
115
|
+
ci=ci,
|
|
116
|
+
params={"auc_a": float(auc[0]), "auc_b": float(auc[1]), "standard_error": se},
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def paired_bootstrap_test(
|
|
121
|
+
metric: Any,
|
|
122
|
+
y_true: ArrayLike,
|
|
123
|
+
y_pred_a: Optional[ArrayLike] = None,
|
|
124
|
+
y_pred_b: Optional[ArrayLike] = None,
|
|
125
|
+
*,
|
|
126
|
+
y_prob_a: Optional[ArrayLike] = None,
|
|
127
|
+
y_prob_b: Optional[ArrayLike] = None,
|
|
128
|
+
sample_weight: Optional[ArrayLike] = None,
|
|
129
|
+
n_resamples: int = 2000,
|
|
130
|
+
level: float = 0.95,
|
|
131
|
+
random_state: Optional[int] = None,
|
|
132
|
+
stratify: Optional[bool] = None,
|
|
133
|
+
alternative: Literal["two-sided", "greater", "less"] = "two-sided",
|
|
134
|
+
**metric_kwargs: Any,
|
|
135
|
+
) -> TestResult:
|
|
136
|
+
"""Paired bootstrap test for any metric: is metric(A) − metric(B) different from 0?
|
|
137
|
+
|
|
138
|
+
Both models are evaluated on the same resampled observations. The p-value uses the shifted bootstrap
|
|
139
|
+
distribution (difference re-centred on 0 under the null), with the +1 correction so it is never 0:
|
|
140
|
+
p = (1 + #{|δ* − δ̂| ≥ |δ̂|}) / (B + 1) for a two-sided test. ``ci`` is the percentile interval of δ*.
|
|
141
|
+
|
|
142
|
+
Reference: Efron B, Tibshirani RJ. An Introduction to the Bootstrap. Chapman & Hall; 1993, ch. 16.
|
|
143
|
+
"""
|
|
144
|
+
level = _check_level(level)
|
|
145
|
+
n_resamples = _check_resamples(n_resamples)
|
|
146
|
+
if alternative not in ("two-sided", "greater", "less"):
|
|
147
|
+
raise InputValidationError("alternative must be 'two-sided', 'greater' or 'less'.")
|
|
148
|
+
fn, name = resolve_metric(metric)
|
|
149
|
+
by_prob = y_prob_a is not None or y_prob_b is not None
|
|
150
|
+
if by_prob and (y_prob_a is None or y_prob_b is None or y_pred_a is not None or y_pred_b is not None):
|
|
151
|
+
raise InputValidationError("Give either y_pred_a and y_pred_b, or y_prob_a and y_prob_b.")
|
|
152
|
+
if not by_prob and (y_pred_a is None or y_pred_b is None):
|
|
153
|
+
raise InputValidationError("Give either y_pred_a and y_pred_b, or y_prob_a and y_prob_b.")
|
|
154
|
+
if by_prob:
|
|
155
|
+
call_a = MetricCall(fn, y_true, None, y_prob_a, sample_weight, metric_kwargs)
|
|
156
|
+
call_b = MetricCall(fn, y_true, None, y_prob_b, sample_weight, metric_kwargs)
|
|
157
|
+
else:
|
|
158
|
+
call_a = MetricCall(fn, y_true, y_pred_a, None, sample_weight, metric_kwargs)
|
|
159
|
+
call_b = MetricCall(fn, y_true, y_pred_b, None, sample_weight, metric_kwargs)
|
|
160
|
+
if call_a.kwargs.get("labels") is not None and call_b.kwargs.get("labels") is not None:
|
|
161
|
+
shared = np.union1d(call_a.kwargs["labels"], call_b.kwargs["labels"])
|
|
162
|
+
call_a.kwargs["labels"] = call_b.kwargs["labels"] = shared
|
|
163
|
+
observed = call_a() - call_b()
|
|
164
|
+
rng = np.random.default_rng(random_state)
|
|
165
|
+
use_strata = call_a.categorical if stratify is None else bool(stratify)
|
|
166
|
+
idx = resample_indices(call_a.n, n_resamples, rng, call_a.y_true if use_strata else None)
|
|
167
|
+
da, fa = bootstrap_distribution(call_a, idx)
|
|
168
|
+
db, fb = bootstrap_distribution(call_b, idx)
|
|
169
|
+
diff = da - db
|
|
170
|
+
diff = diff[~np.isnan(diff)]
|
|
171
|
+
centred = diff - observed
|
|
172
|
+
if alternative == "two-sided":
|
|
173
|
+
extreme = np.sum(np.abs(centred) >= abs(observed))
|
|
174
|
+
elif alternative == "greater":
|
|
175
|
+
extreme = np.sum(centred >= observed)
|
|
176
|
+
else:
|
|
177
|
+
extreme = np.sum(centred <= observed)
|
|
178
|
+
p = (1 + extreme) / (diff.shape[0] + 1)
|
|
179
|
+
alpha = 1 - level
|
|
180
|
+
low, high = np.quantile(diff, [alpha / 2, 1 - alpha / 2])
|
|
181
|
+
ci = ConfidenceInterval(
|
|
182
|
+
observed,
|
|
183
|
+
float(low),
|
|
184
|
+
float(high),
|
|
185
|
+
level,
|
|
186
|
+
"bootstrap-percentile",
|
|
187
|
+
f"{name}_difference",
|
|
188
|
+
{"n_resamples": n_resamples, "random_state": random_state},
|
|
189
|
+
)
|
|
190
|
+
return TestResult(
|
|
191
|
+
"paired-bootstrap",
|
|
192
|
+
observed,
|
|
193
|
+
float(p),
|
|
194
|
+
alternative=alternative,
|
|
195
|
+
estimate=observed,
|
|
196
|
+
ci=ci,
|
|
197
|
+
params={
|
|
198
|
+
"metric": name,
|
|
199
|
+
"n_resamples": n_resamples,
|
|
200
|
+
"random_state": random_state,
|
|
201
|
+
"stratified": use_strata,
|
|
202
|
+
"failed_resamples": int(max(fa, fb)),
|
|
203
|
+
"metric_a": call_a(),
|
|
204
|
+
"metric_b": call_b(),
|
|
205
|
+
**metric_kwargs,
|
|
206
|
+
},
|
|
207
|
+
)
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Result types for intervals and statistical tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from types import MappingProxyType
|
|
10
|
+
from typing import Any, Optional, cast
|
|
11
|
+
|
|
12
|
+
from ..core.result import _json_safe
|
|
13
|
+
|
|
14
|
+
__all__ = ["ConfidenceInterval", "TestResult"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _f(x: float, digits: int = 4) -> str:
|
|
18
|
+
return "NaN" if math.isnan(x) else f"{x:.{digits}f}"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, eq=False)
|
|
22
|
+
class ConfidenceInterval:
|
|
23
|
+
"""A point estimate with a confidence interval.
|
|
24
|
+
|
|
25
|
+
``method`` names the procedure (for example ``"bootstrap-bca"``, ``"wilson"``, ``"delong"``) and
|
|
26
|
+
``params`` records everything needed to reproduce it (resamples, random state, failed resamples...).
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
estimate: float
|
|
30
|
+
low: float
|
|
31
|
+
high: float
|
|
32
|
+
level: float
|
|
33
|
+
method: str
|
|
34
|
+
metric: str = ""
|
|
35
|
+
params: Mapping[str, Any] = field(default_factory=dict)
|
|
36
|
+
|
|
37
|
+
def __post_init__(self) -> None:
|
|
38
|
+
for name in ("estimate", "low", "high", "level"):
|
|
39
|
+
object.__setattr__(self, name, float(getattr(self, name)))
|
|
40
|
+
object.__setattr__(self, "params", MappingProxyType(dict(self.params)))
|
|
41
|
+
|
|
42
|
+
def __float__(self) -> float:
|
|
43
|
+
return self.estimate
|
|
44
|
+
|
|
45
|
+
def __iter__(self): # type: ignore[no-untyped-def]
|
|
46
|
+
"""Unpack as ``low, high``."""
|
|
47
|
+
return iter((self.low, self.high))
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def width(self) -> float:
|
|
51
|
+
return self.high - self.low
|
|
52
|
+
|
|
53
|
+
def contains(self, value: float) -> bool:
|
|
54
|
+
return self.low <= value <= self.high
|
|
55
|
+
|
|
56
|
+
def __repr__(self) -> str:
|
|
57
|
+
pct = round(self.level * 100, 6)
|
|
58
|
+
label = f"{self.metric} " if self.metric else ""
|
|
59
|
+
return f"{label}{_f(self.estimate)} [{pct:g}% CI {_f(self.low)}, {_f(self.high)}] ({self.method})"
|
|
60
|
+
|
|
61
|
+
def format(self, digits: int = 3) -> str:
|
|
62
|
+
"""``0.812 (95% CI 0.774–0.846)``: the form most journals expect."""
|
|
63
|
+
pct = round(self.level * 100, 6)
|
|
64
|
+
return f"{self.estimate:.{digits}f} ({pct:g}% CI {self.low:.{digits}f}–{self.high:.{digits}f})"
|
|
65
|
+
|
|
66
|
+
def to_dict(self) -> dict[str, Any]:
|
|
67
|
+
return cast(
|
|
68
|
+
"dict[str, Any]",
|
|
69
|
+
_json_safe(
|
|
70
|
+
{
|
|
71
|
+
"metric": self.metric,
|
|
72
|
+
"estimate": self.estimate,
|
|
73
|
+
"low": self.low,
|
|
74
|
+
"high": self.high,
|
|
75
|
+
"level": self.level,
|
|
76
|
+
"method": self.method,
|
|
77
|
+
"params": dict(self.params),
|
|
78
|
+
}
|
|
79
|
+
),
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
def to_json(self, *, indent: Optional[int] = 2) -> str:
|
|
83
|
+
return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True, eq=False)
|
|
87
|
+
class TestResult:
|
|
88
|
+
"""Outcome of a statistical test: statistic, p-value and what was tested."""
|
|
89
|
+
|
|
90
|
+
__test__ = False # not a pytest test class
|
|
91
|
+
|
|
92
|
+
test: str
|
|
93
|
+
statistic: float
|
|
94
|
+
p_value: float
|
|
95
|
+
alternative: str = "two-sided"
|
|
96
|
+
estimate: Optional[float] = None
|
|
97
|
+
ci: Optional[ConfidenceInterval] = None
|
|
98
|
+
params: Mapping[str, Any] = field(default_factory=dict)
|
|
99
|
+
|
|
100
|
+
def __post_init__(self) -> None:
|
|
101
|
+
object.__setattr__(self, "statistic", float(self.statistic))
|
|
102
|
+
object.__setattr__(self, "p_value", float(self.p_value))
|
|
103
|
+
if self.estimate is not None:
|
|
104
|
+
object.__setattr__(self, "estimate", float(self.estimate))
|
|
105
|
+
object.__setattr__(self, "params", MappingProxyType(dict(self.params)))
|
|
106
|
+
|
|
107
|
+
def significant(self, alpha: float = 0.05) -> bool:
|
|
108
|
+
return self.p_value < alpha
|
|
109
|
+
|
|
110
|
+
def __repr__(self) -> str:
|
|
111
|
+
est = f", estimate={_f(self.estimate)}" if self.estimate is not None else ""
|
|
112
|
+
p = "p<0.0001" if self.p_value < 1e-4 else f"p={self.p_value:.4f}"
|
|
113
|
+
return f"TestResult({self.test}: statistic={_f(self.statistic)}, {p}{est})"
|
|
114
|
+
|
|
115
|
+
def to_dict(self) -> dict[str, Any]:
|
|
116
|
+
out = {
|
|
117
|
+
"test": self.test,
|
|
118
|
+
"statistic": self.statistic,
|
|
119
|
+
"p_value": self.p_value,
|
|
120
|
+
"alternative": self.alternative,
|
|
121
|
+
"estimate": self.estimate,
|
|
122
|
+
"params": dict(self.params),
|
|
123
|
+
}
|
|
124
|
+
if self.ci is not None:
|
|
125
|
+
out["ci"] = self.ci.to_dict()
|
|
126
|
+
return cast("dict[str, Any]", _json_safe(out))
|
|
127
|
+
|
|
128
|
+
def to_json(self, *, indent: Optional[int] = 2) -> str:
|
|
129
|
+
return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
|
evalsuite/version.py
ADDED