evalsuite-python 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,320 @@
1
+ """Confidence intervals: bootstrap (percentile, basic, BCa), Wilson and Clopper-Pearson, DeLong."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import warnings
6
+ from typing import Any, Literal, Optional
7
+
8
+ import numpy as np
9
+ from numpy.typing import NDArray
10
+ from scipy import special, stats
11
+
12
+ from ..core.exceptions import EvalSuiteError, InputValidationError, StatisticalTestError
13
+ from ..core.types import ArrayLike
14
+ from ..core.validation import to_numpy
15
+ from ._resolve import MetricCall, resolve_metric
16
+ from .results import ConfidenceInterval
17
+
18
+ __all__ = ["accuracy_ci", "bootstrap_ci", "proportion_ci", "roc_auc_ci"]
19
+
20
+ BootstrapMethod = Literal["percentile", "basic", "bca"]
21
+ MAX_FAILED_FRACTION = 0.1
22
+
23
+
24
+ def _check_level(level: float) -> float:
25
+ if not (isinstance(level, (int, float)) and 0 < level < 1):
26
+ raise InputValidationError("level must be between 0 and 1, for example 0.95.")
27
+ return float(level)
28
+
29
+
30
+ def _check_resamples(n_resamples: int) -> int:
31
+ if not isinstance(n_resamples, (int, np.integer)) or n_resamples < 100:
32
+ raise InputValidationError("n_resamples must be an integer of at least 100 (1000 or more recommended).")
33
+ return int(n_resamples)
34
+
35
+
36
+ def resample_indices(
37
+ n: int, n_resamples: int, rng: np.random.Generator, strata: Optional[NDArray[Any]] = None
38
+ ) -> NDArray[np.int64]:
39
+ """(n_resamples, n) indices drawn with replacement, optionally within strata (keeps class balance)."""
40
+ if strata is None:
41
+ return rng.integers(0, n, size=(n_resamples, n))
42
+ out = np.empty((n_resamples, n), dtype=np.int64)
43
+ pos = 0
44
+ for value in np.unique(strata, axis=0) if strata.ndim > 1 else np.unique(strata):
45
+ members = np.flatnonzero(np.all(strata == value, axis=1) if strata.ndim > 1 else strata == value)
46
+ k = members.shape[0]
47
+ out[:, pos : pos + k] = members[rng.integers(0, k, size=(n_resamples, k))]
48
+ pos += k
49
+ return out
50
+
51
+
52
+ def bootstrap_distribution(call: MetricCall, indices: NDArray[np.int64]) -> tuple[NDArray[np.float64], int]:
53
+ """Metric value on each resample; resamples where the metric is undefined become NaN."""
54
+ values = np.empty(indices.shape[0])
55
+ failed = 0
56
+ with warnings.catch_warnings():
57
+ warnings.simplefilter("ignore") # zero-division warnings inside resamples are expected noise
58
+ for b, idx in enumerate(indices):
59
+ try:
60
+ values[b] = call(idx)
61
+ except EvalSuiteError:
62
+ values[b] = np.nan
63
+ failed += 1
64
+ return values, failed
65
+
66
+
67
+ def _jackknife(call: MetricCall, n: int, rng: np.random.Generator, groups: int) -> NDArray[np.float64]:
68
+ """Leave-one-out (or leave-one-group-out for large n) estimates for the BCa acceleration."""
69
+ blocks = [np.array([i]) for i in range(n)] if n <= groups else np.array_split(rng.permutation(n), groups)
70
+ everything = np.arange(n)
71
+ out = np.empty(len(blocks))
72
+ with warnings.catch_warnings():
73
+ warnings.simplefilter("ignore")
74
+ for j, block in enumerate(blocks):
75
+ keep = np.setdiff1d(everything, block, assume_unique=True)
76
+ try:
77
+ out[j] = call(keep)
78
+ except EvalSuiteError:
79
+ out[j] = np.nan
80
+ return out
81
+
82
+
83
+ def bootstrap_ci(
84
+ metric: Any,
85
+ y_true: ArrayLike,
86
+ y_pred: Optional[ArrayLike] = None,
87
+ *,
88
+ y_prob: Optional[ArrayLike] = None,
89
+ sample_weight: Optional[ArrayLike] = None,
90
+ level: float = 0.95,
91
+ method: BootstrapMethod = "bca",
92
+ n_resamples: int = 2000,
93
+ random_state: Optional[int] = None,
94
+ stratify: Optional[bool] = None,
95
+ jackknife_groups: int = 1000,
96
+ **metric_kwargs: Any,
97
+ ) -> ConfidenceInterval:
98
+ """Bootstrap confidence interval for any metric.
99
+
100
+ ``metric`` is a metric function (``es.f1``) or its name (``"f1"``). Pass ``y_pred`` for label/value
101
+ metrics or ``y_prob`` for probability metrics; extra keyword arguments go to the metric (for example
102
+ ``average="macro"``). Resampling is stratified by class for classification targets unless
103
+ ``stratify=False``, so every resample contains every class. For classification metrics that accept
104
+ ``labels``, the full label set is fixed across resamples.
105
+
106
+ ``method``: ``"bca"`` (bias-corrected and accelerated; default, Efron 1987), ``"percentile"`` or
107
+ ``"basic"``. Set ``random_state`` for reproducible intervals. Resamples on which the metric is undefined
108
+ are counted in ``params["failed_resamples"]``; more than 10% raises :class:`StatisticalTestError`.
109
+
110
+ References: Efron B. Better bootstrap confidence intervals. JASA. 1987;82(397):171-185.
111
+ Efron B, Tibshirani RJ. An Introduction to the Bootstrap. Chapman & Hall; 1993.
112
+ """
113
+ level = _check_level(level)
114
+ n_resamples = _check_resamples(n_resamples)
115
+ if method not in ("percentile", "basic", "bca"):
116
+ raise InputValidationError("method must be 'bca', 'percentile' or 'basic'.")
117
+ fn, name = resolve_metric(metric)
118
+ call = MetricCall(fn, y_true, y_pred, y_prob, sample_weight, metric_kwargs)
119
+ estimate = call()
120
+ rng = np.random.default_rng(random_state)
121
+ use_strata = call.categorical if stratify is None else bool(stratify)
122
+ idx = resample_indices(call.n, n_resamples, rng, call.y_true if use_strata else None)
123
+ boot, failed = bootstrap_distribution(call, idx)
124
+ if failed > MAX_FAILED_FRACTION * n_resamples:
125
+ raise StatisticalTestError(
126
+ f"The metric was undefined on {failed} of {n_resamples} bootstrap resamples. The sample is probably "
127
+ "too small or too imbalanced for a reliable bootstrap interval."
128
+ )
129
+ alpha = 1 - level
130
+ valid = boot[~np.isnan(boot)]
131
+ params: dict[str, Any] = {
132
+ "n_resamples": n_resamples,
133
+ "random_state": random_state,
134
+ "stratified": use_strata,
135
+ "failed_resamples": failed,
136
+ **{k: v for k, v in metric_kwargs.items()},
137
+ }
138
+ if method == "percentile":
139
+ low, high = np.quantile(valid, [alpha / 2, 1 - alpha / 2])
140
+ elif method == "basic":
141
+ q_low, q_high = np.quantile(valid, [alpha / 2, 1 - alpha / 2])
142
+ low, high = 2 * estimate - q_high, 2 * estimate - q_low
143
+ else:
144
+ low, high, extra = _bca(call, valid, estimate, alpha, rng, jackknife_groups)
145
+ params.update(extra)
146
+ return ConfidenceInterval(estimate, float(low), float(high), level, f"bootstrap-{method}", name, params)
147
+
148
+
149
+ def _bca(
150
+ call: MetricCall,
151
+ boot: NDArray[np.float64],
152
+ estimate: float,
153
+ alpha: float,
154
+ rng: np.random.Generator,
155
+ groups: int,
156
+ ) -> tuple[float, float, dict[str, Any]]:
157
+ # bias correction: share of the bootstrap distribution below the estimate (ties count half)
158
+ share = (np.sum(boot < estimate) + 0.5 * np.sum(boot == estimate)) / boot.shape[0]
159
+ if share <= 0 or share >= 1:
160
+ warnings.warn(
161
+ "BCa is undefined because the estimate lies outside the bootstrap distribution; "
162
+ "returning the percentile interval.",
163
+ RuntimeWarning,
164
+ stacklevel=3,
165
+ )
166
+ low, high = np.quantile(boot, [alpha / 2, 1 - alpha / 2])
167
+ return float(low), float(high), {"bca_fallback": "percentile"}
168
+ z0 = special.ndtri(share)
169
+ jack = _jackknife(call, call.n, rng, groups)
170
+ jack = jack[~np.isnan(jack)]
171
+ diffs = jack.mean() - jack
172
+ denom = 6.0 * np.sum(diffs**2) ** 1.5
173
+ accel = float(np.sum(diffs**3) / denom) if denom > 0 else 0.0
174
+ probs = []
175
+ for z_alpha in (special.ndtri(alpha / 2), special.ndtri(1 - alpha / 2)):
176
+ num = z0 + z_alpha
177
+ probs.append(special.ndtr(z0 + num / (1 - accel * num)))
178
+ low, high = np.quantile(boot, probs)
179
+ return float(low), float(high), {"bias_correction": float(z0), "acceleration": accel}
180
+
181
+
182
+ def proportion_ci(
183
+ successes: int,
184
+ n: int,
185
+ *,
186
+ level: float = 0.95,
187
+ method: Literal["wilson", "clopper-pearson", "normal"] = "wilson",
188
+ ) -> ConfidenceInterval:
189
+ """Confidence interval for a proportion ``successes / n``.
190
+
191
+ ``"wilson"`` (default; good coverage even for small n or extreme proportions), ``"clopper-pearson"``
192
+ (exact, conservative) or ``"normal"`` (Wald; shown for comparison, not recommended).
193
+
194
+ References: Wilson EB. JASA. 1927;22(158):209-212. Clopper CJ, Pearson ES. Biometrika. 1934;26(4):404-413.
195
+ Brown LD, Cai TT, DasGupta A. Interval estimation for a binomial proportion. Stat Sci. 2001;16(2):101-133.
196
+ """
197
+ level = _check_level(level)
198
+ if not (isinstance(n, (int, np.integer)) and n > 0):
199
+ raise InputValidationError("n must be a positive integer.")
200
+ if not (0 <= successes <= n):
201
+ raise InputValidationError("successes must be between 0 and n.")
202
+ p = successes / n
203
+ alpha = 1 - level
204
+ z = special.ndtri(1 - alpha / 2)
205
+ if method == "wilson":
206
+ denom = 1 + z**2 / n
207
+ centre = (p + z**2 / (2 * n)) / denom
208
+ half = z * np.sqrt(p * (1 - p) / n + z**2 / (4 * n**2)) / denom
209
+ low, high = centre - half, centre + half
210
+ elif method == "clopper-pearson":
211
+ low = 0.0 if successes == 0 else stats.beta.ppf(alpha / 2, successes, n - successes + 1)
212
+ high = 1.0 if successes == n else stats.beta.ppf(1 - alpha / 2, successes + 1, n - successes)
213
+ elif method == "normal":
214
+ half = z * np.sqrt(p * (1 - p) / n)
215
+ low, high = max(0.0, p - half), min(1.0, p + half)
216
+ else:
217
+ raise InputValidationError("method must be 'wilson', 'clopper-pearson' or 'normal'.")
218
+ return ConfidenceInterval(
219
+ p, float(low), float(high), level, method, "proportion", {"successes": int(successes), "n": int(n)}
220
+ )
221
+
222
+
223
+ def accuracy_ci(
224
+ y_true: ArrayLike,
225
+ y_pred: ArrayLike,
226
+ *,
227
+ level: float = 0.95,
228
+ method: Literal["wilson", "clopper-pearson", "normal"] = "wilson",
229
+ ) -> ConfidenceInterval:
230
+ """Analytic confidence interval for accuracy (a proportion of correct predictions)."""
231
+ yt = to_numpy(y_true, "y_true", allow_2d=True)
232
+ yp = to_numpy(y_pred, "y_pred", allow_2d=True)
233
+ if yt.shape != yp.shape:
234
+ raise InputValidationError(f"y_true has shape {yt.shape} but y_pred has shape {yp.shape}.")
235
+ correct = (yt == yp).all(axis=1) if yt.ndim == 2 else yt == yp
236
+ ci = proportion_ci(int(correct.sum()), int(correct.shape[0]), level=level, method=method)
237
+ return ConfidenceInterval(ci.estimate, ci.low, ci.high, level, method, "accuracy", ci.params)
238
+
239
+
240
+ # ---- DeLong ---------------------------------------------------------------------------------------
241
+ def delong_placements(
242
+ y: NDArray[np.float64], scores: NDArray[np.float64]
243
+ ) -> tuple[NDArray[np.float64], NDArray[np.float64], NDArray[np.float64]]:
244
+ """AUCs and structural components for k score columns (Sun & Xu fast DeLong).
245
+
246
+ Returns (auc[k], v10[k, m], v01[k, n]) where m/n are the numbers of positives/negatives.
247
+ """
248
+ pos = scores[:, y == 1]
249
+ neg = scores[:, y == 0]
250
+ m, n = pos.shape[1], neg.shape[1]
251
+ if m == 0 or n == 0:
252
+ from ..core.exceptions import MetricInputError
253
+
254
+ raise MetricInputError("DeLong's method needs at least one positive and one negative observation.")
255
+ tx = np.apply_along_axis(stats.rankdata, 1, pos)
256
+ ty = np.apply_along_axis(stats.rankdata, 1, neg)
257
+ tz = np.apply_along_axis(stats.rankdata, 1, np.concatenate([pos, neg], axis=1))
258
+ auc = (tz[:, :m].sum(axis=1) - m * (m + 1) / 2) / (m * n)
259
+ v10 = (tz[:, :m] - tx) / n
260
+ v01 = 1.0 - (tz[:, m:] - ty) / m
261
+ return auc, v10, v01
262
+
263
+
264
+ def delong_covariance(v10: NDArray[np.float64], v01: NDArray[np.float64]) -> NDArray[np.float64]:
265
+ m, n = v10.shape[1], v01.shape[1]
266
+ s10 = np.atleast_2d(np.cov(v10)) if m > 1 else np.zeros((v10.shape[0],) * 2)
267
+ s01 = np.atleast_2d(np.cov(v01)) if n > 1 else np.zeros((v01.shape[0],) * 2)
268
+ cov: NDArray[np.float64] = s10 / m + s01 / n
269
+ return cov
270
+
271
+
272
+ def binary_scores(
273
+ y_true: ArrayLike, y_prob: ArrayLike, pos_label: Any
274
+ ) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
275
+ yt = to_numpy(y_true, "y_true")
276
+ s = to_numpy(y_prob, "y_prob").astype(np.float64)
277
+ if s.shape[0] != yt.shape[0]:
278
+ raise InputValidationError(
279
+ "y_true and y_prob must contain the same number of observations. "
280
+ f"Received {yt.shape[0]} and {s.shape[0]}."
281
+ )
282
+ labels = np.unique(yt)
283
+ if labels.shape[0] > 2:
284
+ raise InputValidationError("DeLong's method is for binary targets.")
285
+ if pos_label is None:
286
+ if not set(labels.tolist()) <= {0, 1}:
287
+ raise InputValidationError(f"Labels are {labels.tolist()}; specify pos_label=...")
288
+ pos_label = 1
289
+ return (yt == pos_label).astype(np.float64), s
290
+
291
+
292
+ def roc_auc_ci(
293
+ y_true: ArrayLike,
294
+ y_prob: ArrayLike,
295
+ *,
296
+ level: float = 0.95,
297
+ pos_label: Any = None,
298
+ ) -> ConfidenceInterval:
299
+ """DeLong confidence interval for a binary ROC AUC (normal approximation, clipped to [0, 1]).
300
+
301
+ Reference: DeLong ER, DeLong DM, Clarke-Pearson DL. Comparing the areas under two or more correlated
302
+ receiver operating characteristic curves: a nonparametric approach. Biometrics. 1988;44(3):837-845.
303
+ Sun X, Xu W. Fast implementation of DeLong's algorithm. IEEE Signal Process Lett. 2014;21(11):1389-1393.
304
+ """
305
+ level = _check_level(level)
306
+ y, s = binary_scores(y_true, y_prob, pos_label)
307
+ auc, v10, v01 = delong_placements(y, s[None, :])
308
+ var = float(delong_covariance(v10, v01)[0, 0])
309
+ se = np.sqrt(var)
310
+ z = special.ndtri(1 - (1 - level) / 2)
311
+ a = float(auc[0])
312
+ return ConfidenceInterval(
313
+ a,
314
+ max(0.0, a - z * se),
315
+ min(1.0, a + z * se),
316
+ level,
317
+ "delong",
318
+ "roc_auc",
319
+ {"standard_error": float(se), "n_positive": int(y.sum()), "n_negative": int((1 - y).sum())},
320
+ )
@@ -0,0 +1,207 @@
1
+ """Paired tests for comparing two models evaluated on the same observations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, Literal, Optional
6
+
7
+ import numpy as np
8
+ from scipy import special, stats
9
+
10
+ from ..core.exceptions import InputValidationError
11
+ from ..core.types import ArrayLike
12
+ from ..core.validation import to_numpy
13
+ from ._resolve import MetricCall, resolve_metric
14
+ from .intervals import (
15
+ _check_level,
16
+ _check_resamples,
17
+ binary_scores,
18
+ bootstrap_distribution,
19
+ delong_covariance,
20
+ delong_placements,
21
+ resample_indices,
22
+ )
23
+ from .results import ConfidenceInterval, TestResult
24
+
25
+ __all__ = ["delong_test", "mcnemar_test", "paired_bootstrap_test"]
26
+
27
+
28
+ def mcnemar_test(
29
+ y_true: ArrayLike,
30
+ y_pred_a: ArrayLike,
31
+ y_pred_b: ArrayLike,
32
+ *,
33
+ exact: Optional[bool] = None,
34
+ correction: bool = True,
35
+ ) -> TestResult:
36
+ """McNemar's test: do two classifiers have the same error rate on the same observations?
37
+
38
+ Uses only discordant pairs: b = A right & B wrong, c = A wrong & B right. ``exact=None`` (default) uses
39
+ the exact binomial test when b + c < 25 and the chi-squared test otherwise (with Edwards' continuity
40
+ correction unless ``correction=False``). ``estimate`` is the accuracy difference A − B.
41
+
42
+ References: McNemar Q. Psychometrika. 1947;12(2):153-157. Edwards AL. Psychometrika. 1948;13:185-187.
43
+ Dietterich TG. Approximate statistical tests for comparing supervised classification learning
44
+ algorithms. Neural Comput. 1998;10(7):1895-1923.
45
+ """
46
+ yt = to_numpy(y_true, "y_true", allow_2d=True)
47
+ a = to_numpy(y_pred_a, "y_pred_a", allow_2d=True)
48
+ b_ = to_numpy(y_pred_b, "y_pred_b", allow_2d=True)
49
+ if not (yt.shape == a.shape == b_.shape):
50
+ raise InputValidationError(
51
+ f"y_true, y_pred_a and y_pred_b must have the same shape; got {yt.shape}, {a.shape}, {b_.shape}."
52
+ )
53
+ ok_a = (yt == a).all(axis=1) if yt.ndim == 2 else yt == a
54
+ ok_b = (yt == b_).all(axis=1) if yt.ndim == 2 else yt == b_
55
+ b = int(np.sum(ok_a & ~ok_b))
56
+ c = int(np.sum(~ok_a & ok_b))
57
+ n_disc = b + c
58
+ use_exact = n_disc < 25 if exact is None else bool(exact)
59
+ estimate = (b - c) / yt.shape[0]
60
+ params = {"b": b, "c": c, "n": int(yt.shape[0])}
61
+ if n_disc == 0:
62
+ return TestResult(
63
+ "mcnemar-exact" if use_exact else "mcnemar",
64
+ 0.0,
65
+ 1.0,
66
+ estimate=estimate,
67
+ params={**params, "note": "no discordant pairs: the classifiers agree on every observation"},
68
+ )
69
+ if use_exact:
70
+ p = min(1.0, 2 * stats.binom.cdf(min(b, c), n_disc, 0.5))
71
+ return TestResult("mcnemar-exact", float(min(b, c)), float(p), estimate=estimate, params=params)
72
+ stat = (abs(b - c) - (1 if correction else 0)) ** 2 / n_disc
73
+ p = stats.chi2.sf(stat, 1)
74
+ return TestResult(
75
+ "mcnemar", float(stat), float(p), estimate=estimate, params={**params, "continuity_correction": correction}
76
+ )
77
+
78
+
79
+ def delong_test(
80
+ y_true: ArrayLike,
81
+ y_prob_a: ArrayLike,
82
+ y_prob_b: ArrayLike,
83
+ *,
84
+ pos_label: Any = None,
85
+ level: float = 0.95,
86
+ ) -> TestResult:
87
+ """DeLong's test for two correlated ROC AUCs (same observations, two models). ``estimate`` is AUC_A − AUC_B
88
+ and ``ci`` its confidence interval.
89
+
90
+ Reference: DeLong ER, DeLong DM, Clarke-Pearson DL. Biometrics. 1988;44(3):837-845.
91
+ """
92
+ level = _check_level(level)
93
+ y, sa = binary_scores(y_true, y_prob_a, pos_label)
94
+ _, sb = binary_scores(y_true, y_prob_b, pos_label)
95
+ if sb.shape != sa.shape:
96
+ raise InputValidationError("y_prob_a and y_prob_b must have the same length.")
97
+ auc, v10, v01 = delong_placements(y, np.vstack([sa, sb]))
98
+ cov = delong_covariance(v10, v01)
99
+ diff = float(auc[0] - auc[1])
100
+ var = float(cov[0, 0] + cov[1, 1] - 2 * cov[0, 1])
101
+ if var <= 0:
102
+ z, p = 0.0, 1.0
103
+ se = 0.0
104
+ else:
105
+ se = float(np.sqrt(var))
106
+ z = diff / se
107
+ p = float(2 * special.ndtr(-abs(z)))
108
+ zc = special.ndtri(1 - (1 - level) / 2)
109
+ ci = ConfidenceInterval(diff, diff - zc * se, diff + zc * se, level, "delong", "roc_auc_difference")
110
+ return TestResult(
111
+ "delong",
112
+ z,
113
+ p,
114
+ estimate=diff,
115
+ ci=ci,
116
+ params={"auc_a": float(auc[0]), "auc_b": float(auc[1]), "standard_error": se},
117
+ )
118
+
119
+
120
+ def paired_bootstrap_test(
121
+ metric: Any,
122
+ y_true: ArrayLike,
123
+ y_pred_a: Optional[ArrayLike] = None,
124
+ y_pred_b: Optional[ArrayLike] = None,
125
+ *,
126
+ y_prob_a: Optional[ArrayLike] = None,
127
+ y_prob_b: Optional[ArrayLike] = None,
128
+ sample_weight: Optional[ArrayLike] = None,
129
+ n_resamples: int = 2000,
130
+ level: float = 0.95,
131
+ random_state: Optional[int] = None,
132
+ stratify: Optional[bool] = None,
133
+ alternative: Literal["two-sided", "greater", "less"] = "two-sided",
134
+ **metric_kwargs: Any,
135
+ ) -> TestResult:
136
+ """Paired bootstrap test for any metric: is metric(A) − metric(B) different from 0?
137
+
138
+ Both models are evaluated on the same resampled observations. The p-value uses the shifted bootstrap
139
+ distribution (difference re-centred on 0 under the null), with the +1 correction so it is never 0:
140
+ p = (1 + #{|δ* − δ̂| ≥ |δ̂|}) / (B + 1) for a two-sided test. ``ci`` is the percentile interval of δ*.
141
+
142
+ Reference: Efron B, Tibshirani RJ. An Introduction to the Bootstrap. Chapman & Hall; 1993, ch. 16.
143
+ """
144
+ level = _check_level(level)
145
+ n_resamples = _check_resamples(n_resamples)
146
+ if alternative not in ("two-sided", "greater", "less"):
147
+ raise InputValidationError("alternative must be 'two-sided', 'greater' or 'less'.")
148
+ fn, name = resolve_metric(metric)
149
+ by_prob = y_prob_a is not None or y_prob_b is not None
150
+ if by_prob and (y_prob_a is None or y_prob_b is None or y_pred_a is not None or y_pred_b is not None):
151
+ raise InputValidationError("Give either y_pred_a and y_pred_b, or y_prob_a and y_prob_b.")
152
+ if not by_prob and (y_pred_a is None or y_pred_b is None):
153
+ raise InputValidationError("Give either y_pred_a and y_pred_b, or y_prob_a and y_prob_b.")
154
+ if by_prob:
155
+ call_a = MetricCall(fn, y_true, None, y_prob_a, sample_weight, metric_kwargs)
156
+ call_b = MetricCall(fn, y_true, None, y_prob_b, sample_weight, metric_kwargs)
157
+ else:
158
+ call_a = MetricCall(fn, y_true, y_pred_a, None, sample_weight, metric_kwargs)
159
+ call_b = MetricCall(fn, y_true, y_pred_b, None, sample_weight, metric_kwargs)
160
+ if call_a.kwargs.get("labels") is not None and call_b.kwargs.get("labels") is not None:
161
+ shared = np.union1d(call_a.kwargs["labels"], call_b.kwargs["labels"])
162
+ call_a.kwargs["labels"] = call_b.kwargs["labels"] = shared
163
+ observed = call_a() - call_b()
164
+ rng = np.random.default_rng(random_state)
165
+ use_strata = call_a.categorical if stratify is None else bool(stratify)
166
+ idx = resample_indices(call_a.n, n_resamples, rng, call_a.y_true if use_strata else None)
167
+ da, fa = bootstrap_distribution(call_a, idx)
168
+ db, fb = bootstrap_distribution(call_b, idx)
169
+ diff = da - db
170
+ diff = diff[~np.isnan(diff)]
171
+ centred = diff - observed
172
+ if alternative == "two-sided":
173
+ extreme = np.sum(np.abs(centred) >= abs(observed))
174
+ elif alternative == "greater":
175
+ extreme = np.sum(centred >= observed)
176
+ else:
177
+ extreme = np.sum(centred <= observed)
178
+ p = (1 + extreme) / (diff.shape[0] + 1)
179
+ alpha = 1 - level
180
+ low, high = np.quantile(diff, [alpha / 2, 1 - alpha / 2])
181
+ ci = ConfidenceInterval(
182
+ observed,
183
+ float(low),
184
+ float(high),
185
+ level,
186
+ "bootstrap-percentile",
187
+ f"{name}_difference",
188
+ {"n_resamples": n_resamples, "random_state": random_state},
189
+ )
190
+ return TestResult(
191
+ "paired-bootstrap",
192
+ observed,
193
+ float(p),
194
+ alternative=alternative,
195
+ estimate=observed,
196
+ ci=ci,
197
+ params={
198
+ "metric": name,
199
+ "n_resamples": n_resamples,
200
+ "random_state": random_state,
201
+ "stratified": use_strata,
202
+ "failed_resamples": int(max(fa, fb)),
203
+ "metric_a": call_a(),
204
+ "metric_b": call_b(),
205
+ **metric_kwargs,
206
+ },
207
+ )
@@ -0,0 +1,129 @@
1
+ """Result types for intervals and statistical tests."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import math
7
+ from collections.abc import Mapping
8
+ from dataclasses import dataclass, field
9
+ from types import MappingProxyType
10
+ from typing import Any, Optional, cast
11
+
12
+ from ..core.result import _json_safe
13
+
14
+ __all__ = ["ConfidenceInterval", "TestResult"]
15
+
16
+
17
+ def _f(x: float, digits: int = 4) -> str:
18
+ return "NaN" if math.isnan(x) else f"{x:.{digits}f}"
19
+
20
+
21
+ @dataclass(frozen=True, eq=False)
22
+ class ConfidenceInterval:
23
+ """A point estimate with a confidence interval.
24
+
25
+ ``method`` names the procedure (for example ``"bootstrap-bca"``, ``"wilson"``, ``"delong"``) and
26
+ ``params`` records everything needed to reproduce it (resamples, random state, failed resamples...).
27
+ """
28
+
29
+ estimate: float
30
+ low: float
31
+ high: float
32
+ level: float
33
+ method: str
34
+ metric: str = ""
35
+ params: Mapping[str, Any] = field(default_factory=dict)
36
+
37
+ def __post_init__(self) -> None:
38
+ for name in ("estimate", "low", "high", "level"):
39
+ object.__setattr__(self, name, float(getattr(self, name)))
40
+ object.__setattr__(self, "params", MappingProxyType(dict(self.params)))
41
+
42
+ def __float__(self) -> float:
43
+ return self.estimate
44
+
45
+ def __iter__(self): # type: ignore[no-untyped-def]
46
+ """Unpack as ``low, high``."""
47
+ return iter((self.low, self.high))
48
+
49
+ @property
50
+ def width(self) -> float:
51
+ return self.high - self.low
52
+
53
+ def contains(self, value: float) -> bool:
54
+ return self.low <= value <= self.high
55
+
56
+ def __repr__(self) -> str:
57
+ pct = round(self.level * 100, 6)
58
+ label = f"{self.metric} " if self.metric else ""
59
+ return f"{label}{_f(self.estimate)} [{pct:g}% CI {_f(self.low)}, {_f(self.high)}] ({self.method})"
60
+
61
+ def format(self, digits: int = 3) -> str:
62
+ """``0.812 (95% CI 0.774–0.846)``: the form most journals expect."""
63
+ pct = round(self.level * 100, 6)
64
+ return f"{self.estimate:.{digits}f} ({pct:g}% CI {self.low:.{digits}f}–{self.high:.{digits}f})"
65
+
66
+ def to_dict(self) -> dict[str, Any]:
67
+ return cast(
68
+ "dict[str, Any]",
69
+ _json_safe(
70
+ {
71
+ "metric": self.metric,
72
+ "estimate": self.estimate,
73
+ "low": self.low,
74
+ "high": self.high,
75
+ "level": self.level,
76
+ "method": self.method,
77
+ "params": dict(self.params),
78
+ }
79
+ ),
80
+ )
81
+
82
+ def to_json(self, *, indent: Optional[int] = 2) -> str:
83
+ return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
84
+
85
+
86
+ @dataclass(frozen=True, eq=False)
87
+ class TestResult:
88
+ """Outcome of a statistical test: statistic, p-value and what was tested."""
89
+
90
+ __test__ = False # not a pytest test class
91
+
92
+ test: str
93
+ statistic: float
94
+ p_value: float
95
+ alternative: str = "two-sided"
96
+ estimate: Optional[float] = None
97
+ ci: Optional[ConfidenceInterval] = None
98
+ params: Mapping[str, Any] = field(default_factory=dict)
99
+
100
+ def __post_init__(self) -> None:
101
+ object.__setattr__(self, "statistic", float(self.statistic))
102
+ object.__setattr__(self, "p_value", float(self.p_value))
103
+ if self.estimate is not None:
104
+ object.__setattr__(self, "estimate", float(self.estimate))
105
+ object.__setattr__(self, "params", MappingProxyType(dict(self.params)))
106
+
107
+ def significant(self, alpha: float = 0.05) -> bool:
108
+ return self.p_value < alpha
109
+
110
+ def __repr__(self) -> str:
111
+ est = f", estimate={_f(self.estimate)}" if self.estimate is not None else ""
112
+ p = "p<0.0001" if self.p_value < 1e-4 else f"p={self.p_value:.4f}"
113
+ return f"TestResult({self.test}: statistic={_f(self.statistic)}, {p}{est})"
114
+
115
+ def to_dict(self) -> dict[str, Any]:
116
+ out = {
117
+ "test": self.test,
118
+ "statistic": self.statistic,
119
+ "p_value": self.p_value,
120
+ "alternative": self.alternative,
121
+ "estimate": self.estimate,
122
+ "params": dict(self.params),
123
+ }
124
+ if self.ci is not None:
125
+ out["ci"] = self.ci.to_dict()
126
+ return cast("dict[str, Any]", _json_safe(out))
127
+
128
+ def to_json(self, *, indent: Optional[int] = 2) -> str:
129
+ return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
evalsuite/version.py ADDED
@@ -0,0 +1,3 @@
1
+ """Package version (single source of truth, read by the build backend)."""
2
+
3
+ __version__ = "0.1.0"