evalsuite-python 0.2.0__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalsuite/__init__.py +42 -1
- evalsuite/api.py +10 -0
- evalsuite/benchmarks.py +347 -51
- evalsuite/calibration.py +1 -1
- evalsuite/cli/main.py +109 -2
- evalsuite/clinical/metrics.py +14 -5
- evalsuite/core/validation.py +14 -2
- evalsuite/plot.py +165 -1
- evalsuite/stats/_resolve.py +26 -3
- evalsuite/stats/compare.py +6 -4
- evalsuite/stats/effect.py +1 -1
- evalsuite/version.py +1 -1
- evalsuite/vision/__init__.py +47 -0
- evalsuite/vision/detection.py +648 -0
- evalsuite/vision/segmentation.py +965 -0
- {evalsuite_python-0.2.0.dist-info → evalsuite_python-0.3.0.dist-info}/METADATA +83 -23
- {evalsuite_python-0.2.0.dist-info → evalsuite_python-0.3.0.dist-info}/RECORD +20 -17
- {evalsuite_python-0.2.0.dist-info → evalsuite_python-0.3.0.dist-info}/WHEEL +0 -0
- {evalsuite_python-0.2.0.dist-info → evalsuite_python-0.3.0.dist-info}/entry_points.txt +0 -0
- {evalsuite_python-0.2.0.dist-info → evalsuite_python-0.3.0.dist-info}/licenses/LICENSE +0 -0
evalsuite/__init__.py
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import calibration, classification, clinical, plot, regression, stats
|
|
8
|
+
from . import calibration, classification, clinical, plot, regression, stats, vision
|
|
9
9
|
from .api import evaluate
|
|
10
10
|
from .calibration import (
|
|
11
11
|
CalibrationReport,
|
|
@@ -111,8 +111,49 @@ from .stats import (
|
|
|
111
111
|
wilcoxon_test,
|
|
112
112
|
)
|
|
113
113
|
from .version import __version__
|
|
114
|
+
from .vision import (
|
|
115
|
+
DetectionReport,
|
|
116
|
+
SegmentationReport,
|
|
117
|
+
average_precision_detection,
|
|
118
|
+
average_surface_distance,
|
|
119
|
+
boundary_iou,
|
|
120
|
+
box_iou,
|
|
121
|
+
detection_pr_curve,
|
|
122
|
+
detection_report,
|
|
123
|
+
dice,
|
|
124
|
+
from_coco,
|
|
125
|
+
hausdorff_distance,
|
|
126
|
+
iou,
|
|
127
|
+
mean_average_precision,
|
|
128
|
+
mean_pixel_accuracy,
|
|
129
|
+
miou,
|
|
130
|
+
per_image_scores,
|
|
131
|
+
pixel_accuracy,
|
|
132
|
+
segmentation_confusion,
|
|
133
|
+
segmentation_report,
|
|
134
|
+
)
|
|
114
135
|
|
|
115
136
|
__all__ = [
|
|
137
|
+
"SegmentationReport",
|
|
138
|
+
"segmentation_report",
|
|
139
|
+
"detection_pr_curve",
|
|
140
|
+
"vision",
|
|
141
|
+
"DetectionReport",
|
|
142
|
+
"average_precision_detection",
|
|
143
|
+
"average_surface_distance",
|
|
144
|
+
"boundary_iou",
|
|
145
|
+
"box_iou",
|
|
146
|
+
"detection_report",
|
|
147
|
+
"dice",
|
|
148
|
+
"from_coco",
|
|
149
|
+
"hausdorff_distance",
|
|
150
|
+
"iou",
|
|
151
|
+
"mean_average_precision",
|
|
152
|
+
"mean_pixel_accuracy",
|
|
153
|
+
"miou",
|
|
154
|
+
"per_image_scores",
|
|
155
|
+
"pixel_accuracy",
|
|
156
|
+
"segmentation_confusion",
|
|
116
157
|
"CalibrationReport",
|
|
117
158
|
"calibration_report",
|
|
118
159
|
"calibration",
|
evalsuite/api.py
CHANGED
|
@@ -149,6 +149,16 @@ def evaluate(
|
|
|
149
149
|
>>> round(r["accuracy"], 2)
|
|
150
150
|
0.75
|
|
151
151
|
"""
|
|
152
|
+
if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
|
|
153
|
+
raise UnsupportedTaskError(
|
|
154
|
+
"y_true looks like object detection annotations (one dict per image); use "
|
|
155
|
+
"evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
|
|
156
|
+
)
|
|
157
|
+
if np.ndim(y_true) >= 3:
|
|
158
|
+
raise UnsupportedTaskError(
|
|
159
|
+
"y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
|
|
160
|
+
"evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
|
|
161
|
+
)
|
|
152
162
|
task = task or _infer_task(y_true, y_prob)
|
|
153
163
|
if task == "classification":
|
|
154
164
|
return _evaluate_classification(
|
evalsuite/benchmarks.py
CHANGED
|
@@ -1,9 +1,14 @@
|
|
|
1
|
-
"""Speed and memory benchmarks
|
|
1
|
+
"""Speed and memory benchmarks against reference implementations.
|
|
2
2
|
|
|
3
3
|
Each case times the fastest of ``repeat`` runs (after one warm-up) and measures peak traced memory with
|
|
4
|
-
``tracemalloc`` (NumPy reports its allocations to it).
|
|
5
|
-
data, and the largest absolute difference between their results is reported, so
|
|
6
|
-
numbers that disagree.
|
|
4
|
+
``tracemalloc`` (NumPy reports its allocations to it). EvalSuite and the reference compute the same
|
|
5
|
+
quantities on the same data, and the largest absolute difference between their results is reported, so
|
|
6
|
+
speed is never shown for numbers that disagree.
|
|
7
|
+
|
|
8
|
+
References: scikit-learn for classification and regression (``suite="core"``); scikit-learn, statsmodels
|
|
9
|
+
and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``);
|
|
10
|
+
scikit-learn, SciPy and pycocotools for segmentation and detection (``suite="vision"``). A case
|
|
11
|
+
whose reference library is not installed is timed for EvalSuite only.
|
|
7
12
|
|
|
8
13
|
>>> from evalsuite.benchmarks import run_benchmarks
|
|
9
14
|
>>> print(run_benchmarks(sizes=(10_000,), repeat=3)) # doctest: +SKIP
|
|
@@ -11,6 +16,7 @@ numbers that disagree.
|
|
|
11
16
|
|
|
12
17
|
from __future__ import annotations
|
|
13
18
|
|
|
19
|
+
import contextlib
|
|
14
20
|
import json
|
|
15
21
|
import platform
|
|
16
22
|
import time
|
|
@@ -33,14 +39,17 @@ __all__ = ["BenchmarkResult", "run_benchmarks"]
|
|
|
33
39
|
_HEADER = (
|
|
34
40
|
"case",
|
|
35
41
|
"n",
|
|
42
|
+
"reference",
|
|
36
43
|
"evalsuite_ms",
|
|
37
|
-
"
|
|
44
|
+
"reference_ms",
|
|
38
45
|
"speedup",
|
|
39
46
|
"evalsuite_peak_mb",
|
|
40
|
-
"
|
|
47
|
+
"reference_peak_mb",
|
|
41
48
|
"max_abs_diff",
|
|
42
49
|
)
|
|
43
50
|
|
|
51
|
+
Case = tuple[str, str, Callable[[], Any], Optional[Callable[[], Any]]]
|
|
52
|
+
|
|
44
53
|
|
|
45
54
|
def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
|
|
46
55
|
"""(fastest seconds, peak MiB, result)."""
|
|
@@ -61,7 +70,7 @@ def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
|
|
|
61
70
|
return best, peak / 2**20, result
|
|
62
71
|
|
|
63
72
|
|
|
64
|
-
def
|
|
73
|
+
def _core_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
65
74
|
import evalsuite as es
|
|
66
75
|
|
|
67
76
|
y = rng.integers(0, 2, n)
|
|
@@ -81,11 +90,11 @@ def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any
|
|
|
81
90
|
r = es.evaluate(yr, pr, metrics=["mae", "mse", "rmse", "r2"])
|
|
82
91
|
return [float(r[m]) for m in ("mae", "mse", "rmse", "r2")]
|
|
83
92
|
|
|
84
|
-
cases: list[
|
|
85
|
-
("binary: 8 label metrics via evaluate()", es_binary, None),
|
|
86
|
-
("10 classes: macro F1", lambda: [float(es.f1(yk, pk, average="macro"))], None),
|
|
87
|
-
("binary: ROC AUC", lambda: [float(es.roc_auc(y, prob))], None),
|
|
88
|
-
("regression: MAE, MSE, RMSE, R² via evaluate()", es_reg, None),
|
|
93
|
+
cases: list[Case] = [
|
|
94
|
+
("binary: 8 label metrics via evaluate()", "scikit-learn", es_binary, None),
|
|
95
|
+
("10 classes: macro F1", "scikit-learn", lambda: [float(es.f1(yk, pk, average="macro"))], None),
|
|
96
|
+
("binary: ROC AUC", "scikit-learn", lambda: [float(es.roc_auc(y, prob))], None),
|
|
97
|
+
("regression: MAE, MSE, RMSE, R² via evaluate()", "scikit-learn", es_reg, None),
|
|
89
98
|
]
|
|
90
99
|
try:
|
|
91
100
|
import sklearn.metrics as skm # type: ignore[import-untyped]
|
|
@@ -109,7 +118,257 @@ def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any
|
|
|
109
118
|
return [skm.mean_absolute_error(yr, pr), mse, float(np.sqrt(mse)), skm.r2_score(yr, pr)]
|
|
110
119
|
|
|
111
120
|
sk = [sk_binary, lambda: [skm.f1_score(yk, pk, average="macro")], lambda: [skm.roc_auc_score(y, prob)], sk_reg]
|
|
112
|
-
return [(name, es_fn, sk_fn) for (name, es_fn, _), sk_fn in zip(cases, sk)]
|
|
121
|
+
return [(name, ref, es_fn, sk_fn) for (name, ref, es_fn, _), sk_fn in zip(cases, sk)]
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _vision_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
125
|
+
"""v0.3.0: segmentation overlap and surface distance (n = pixels) and COCO detection (n / 1000 images)."""
|
|
126
|
+
import evalsuite as es
|
|
127
|
+
|
|
128
|
+
side = 64
|
|
129
|
+
n_img = max(1, n // (side * side))
|
|
130
|
+
k = 5
|
|
131
|
+
yy, xx = np.ogrid[:side, :side]
|
|
132
|
+
true = np.zeros((n_img, side, side), dtype=np.int64)
|
|
133
|
+
for i in range(n_img):
|
|
134
|
+
for c in range(1, k):
|
|
135
|
+
cy, cx, r = rng.integers(8, side - 8), rng.integers(8, side - 8), rng.integers(4, 14)
|
|
136
|
+
true[i][(yy - cy) ** 2 + (xx - cx) ** 2 <= r * r] = c
|
|
137
|
+
pred = np.roll(true, 1, axis=2)
|
|
138
|
+
noise = rng.random(pred.shape) < 0.02
|
|
139
|
+
pred[noise] = rng.integers(0, k, int(noise.sum()))
|
|
140
|
+
labels = list(range(k))
|
|
141
|
+
|
|
142
|
+
def es_overlap() -> list[float]:
|
|
143
|
+
d = np.asarray(es.dice(true, pred, average=None).value)
|
|
144
|
+
j = np.asarray(es.iou(true, pred, average=None).value)
|
|
145
|
+
return [*d, *j]
|
|
146
|
+
|
|
147
|
+
n_hd = min(n_img, 50)
|
|
148
|
+
|
|
149
|
+
def es_hd() -> list[float]:
|
|
150
|
+
return [float(es.hausdorff_distance(true[i], pred[i], labels=[k - 1])) for i in range(n_hd)]
|
|
151
|
+
|
|
152
|
+
n_det = max(10, n // 1000)
|
|
153
|
+
y_true, y_pred = [], []
|
|
154
|
+
for _ in range(n_det):
|
|
155
|
+
m = int(rng.integers(1, 8))
|
|
156
|
+
xy = rng.uniform(0, 500, (m, 2))
|
|
157
|
+
wh = rng.uniform(8, 160, (m, 2))
|
|
158
|
+
boxes = np.column_stack([xy, xy + wh])
|
|
159
|
+
lab = rng.integers(1, 6, m)
|
|
160
|
+
y_true.append({"boxes": boxes, "labels": lab, "iscrowd": np.zeros(m, int)})
|
|
161
|
+
jitter = boxes + rng.normal(0, 4, boxes.shape)
|
|
162
|
+
jitter[:, 2:] = np.maximum(jitter[:, 2:], jitter[:, :2] + 1)
|
|
163
|
+
extra = rng.uniform(0, 500, (3, 2))
|
|
164
|
+
fp = np.column_stack([extra, extra + rng.uniform(10, 80, (3, 2))])
|
|
165
|
+
y_pred.append(
|
|
166
|
+
{
|
|
167
|
+
"boxes": np.vstack([jitter, fp]),
|
|
168
|
+
"labels": np.r_[lab, rng.integers(1, 6, 3)],
|
|
169
|
+
"scores": rng.random(m + 3),
|
|
170
|
+
}
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def es_map() -> list[float]:
|
|
174
|
+
r = es.detection_report(y_true, y_pred)
|
|
175
|
+
return [r["map"], r["map_50"], r["map_75"], r["mar_100"]]
|
|
176
|
+
|
|
177
|
+
cases: list[Case] = [
|
|
178
|
+
("segmentation: Dice and IoU per class (n = pixels)", "scikit-learn", es_overlap, None),
|
|
179
|
+
(f"segmentation: Hausdorff distance ({n_hd} image{'s' if n_hd != 1 else ''})", "SciPy", es_hd, None),
|
|
180
|
+
(f"detection: COCO evaluation ({n_det} images)", "pycocotools", es_map, None),
|
|
181
|
+
]
|
|
182
|
+
refs: dict[str, Callable[[], Any]] = {}
|
|
183
|
+
try:
|
|
184
|
+
import sklearn.metrics as skm
|
|
185
|
+
|
|
186
|
+
def sk_overlap() -> list[float]:
|
|
187
|
+
ft, fp_ = true.ravel(), pred.ravel()
|
|
188
|
+
return [
|
|
189
|
+
*skm.f1_score(ft, fp_, labels=labels, average=None),
|
|
190
|
+
*skm.jaccard_score(ft, fp_, labels=labels, average=None),
|
|
191
|
+
]
|
|
192
|
+
|
|
193
|
+
refs[cases[0][0]] = sk_overlap
|
|
194
|
+
except ImportError:
|
|
195
|
+
pass
|
|
196
|
+
from scipy import ndimage
|
|
197
|
+
from scipy.spatial.distance import directed_hausdorff
|
|
198
|
+
|
|
199
|
+
def scipy_hd() -> list[float]:
|
|
200
|
+
out = []
|
|
201
|
+
for i in range(n_hd):
|
|
202
|
+
pts = []
|
|
203
|
+
for m_ in (true[i] == k - 1, pred[i] == k - 1):
|
|
204
|
+
er = ndimage.binary_erosion(m_, structure=ndimage.generate_binary_structure(2, 1), border_value=0)
|
|
205
|
+
pts.append(np.argwhere(m_ & ~er).astype(float))
|
|
206
|
+
out.append(max(directed_hausdorff(pts[0], pts[1])[0], directed_hausdorff(pts[1], pts[0])[0]))
|
|
207
|
+
return out
|
|
208
|
+
|
|
209
|
+
refs[cases[1][0]] = scipy_hd
|
|
210
|
+
try:
|
|
211
|
+
import contextlib as _ctx
|
|
212
|
+
import io
|
|
213
|
+
|
|
214
|
+
from pycocotools.coco import COCO # type: ignore[import-untyped]
|
|
215
|
+
from pycocotools.cocoeval import COCOeval # type: ignore[import-untyped]
|
|
216
|
+
|
|
217
|
+
def coco_map() -> list[float]:
|
|
218
|
+
images, anns, dets, aid = [], [], [], 1
|
|
219
|
+
for i, (t, p) in enumerate(zip(y_true, y_pred)):
|
|
220
|
+
images.append({"id": i + 1})
|
|
221
|
+
for b, c in zip(t["boxes"], t["labels"]):
|
|
222
|
+
w, h = b[2] - b[0], b[3] - b[1]
|
|
223
|
+
anns.append(
|
|
224
|
+
{
|
|
225
|
+
"id": aid,
|
|
226
|
+
"image_id": i + 1,
|
|
227
|
+
"category_id": int(c),
|
|
228
|
+
"bbox": [b[0], b[1], w, h],
|
|
229
|
+
"area": w * h,
|
|
230
|
+
"iscrowd": 0,
|
|
231
|
+
}
|
|
232
|
+
)
|
|
233
|
+
aid += 1
|
|
234
|
+
for b, c, sc in zip(p["boxes"], p["labels"], p["scores"]):
|
|
235
|
+
dets.append(
|
|
236
|
+
{
|
|
237
|
+
"image_id": i + 1,
|
|
238
|
+
"category_id": int(c),
|
|
239
|
+
"bbox": [b[0], b[1], b[2] - b[0], b[3] - b[1]],
|
|
240
|
+
"score": float(sc),
|
|
241
|
+
}
|
|
242
|
+
)
|
|
243
|
+
with _ctx.redirect_stdout(io.StringIO()):
|
|
244
|
+
gt = COCO()
|
|
245
|
+
gt.dataset = {
|
|
246
|
+
"images": images,
|
|
247
|
+
"annotations": anns,
|
|
248
|
+
"categories": [{"id": c} for c in range(1, 6)],
|
|
249
|
+
}
|
|
250
|
+
gt.createIndex()
|
|
251
|
+
ev = COCOeval(gt, gt.loadRes(dets), "bbox")
|
|
252
|
+
ev.evaluate()
|
|
253
|
+
ev.accumulate()
|
|
254
|
+
ev.summarize()
|
|
255
|
+
return [ev.stats[0], ev.stats[1], ev.stats[2], ev.stats[8]]
|
|
256
|
+
|
|
257
|
+
refs[cases[2][0]] = coco_map
|
|
258
|
+
except ImportError:
|
|
259
|
+
pass
|
|
260
|
+
return [(name, ref, es_fn, refs.get(name)) for name, ref, es_fn, _ in cases]
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _clinical_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
264
|
+
"""v0.2.0: diagnostic accuracy, calibration, decision curves and statistical tests."""
|
|
265
|
+
import evalsuite as es
|
|
266
|
+
|
|
267
|
+
y = rng.integers(0, 2, n)
|
|
268
|
+
p = np.where(rng.random(n) < 0.8, y, 1 - y)
|
|
269
|
+
x = rng.normal(size=n)
|
|
270
|
+
yc = (rng.random(n) < 1 / (1 + np.exp(-(0.4 + 1.3 * x)))).astype(int)
|
|
271
|
+
risk = 1 / (1 + np.exp(-(0.1 + 2.0 * x)))
|
|
272
|
+
a, b = rng.normal(0, 1, n), rng.normal(0.05, 1.2, n)
|
|
273
|
+
pvals = rng.random(n) ** 2
|
|
274
|
+
ga, gb = rng.integers(0, 5, n), rng.integers(0, 5, n)
|
|
275
|
+
table = np.zeros((5, 5), dtype=np.int64)
|
|
276
|
+
np.add.at(table, (ga, gb), 1)
|
|
277
|
+
thresholds = np.arange(1, 100) / 100
|
|
278
|
+
|
|
279
|
+
def es_diag() -> list[float]:
|
|
280
|
+
return [
|
|
281
|
+
float(es.sensitivity(y, p)),
|
|
282
|
+
float(es.specificity(y, p)),
|
|
283
|
+
float(es.lr_positive(y, p)),
|
|
284
|
+
float(es.lr_negative(y, p)),
|
|
285
|
+
]
|
|
286
|
+
|
|
287
|
+
report_keys = ("sensitivity", "specificity", "ppv", "npv", "accuracy", "prevalence", "diagnostic_odds_ratio")
|
|
288
|
+
|
|
289
|
+
def es_report() -> list[float]:
|
|
290
|
+
r = es.diagnostic_report(y, p)
|
|
291
|
+
return [v for k in report_keys for v in (r[k].low, r[k].high)]
|
|
292
|
+
|
|
293
|
+
def es_cal() -> list[float]:
|
|
294
|
+
return [float(es.calibration_slope(yc, risk)), float(es.calibration_intercept(yc, risk))]
|
|
295
|
+
|
|
296
|
+
def es_dca() -> list[float]:
|
|
297
|
+
curve: list[float] = es.decision_curve(yc, risk, thresholds=thresholds).net_benefit["model"].tolist()
|
|
298
|
+
return curve
|
|
299
|
+
|
|
300
|
+
def numpy_dca() -> list[float]: # the textbook loop, one threshold at a time
|
|
301
|
+
out = []
|
|
302
|
+
for t in thresholds:
|
|
303
|
+
treat = risk >= t
|
|
304
|
+
tp = np.sum(treat & (yc == 1))
|
|
305
|
+
fp = np.sum(treat & (yc == 0))
|
|
306
|
+
out.append(tp / n - fp / n * t / (1 - t))
|
|
307
|
+
return out
|
|
308
|
+
|
|
309
|
+
cases: list[Case] = [
|
|
310
|
+
("clinical: sensitivity, specificity, LR+, LR−", "scikit-learn", es_diag, None),
|
|
311
|
+
("clinical: diagnostic report (7 CIs)", "statsmodels", es_report, None),
|
|
312
|
+
("calibration: slope and intercept", "statsmodels", es_cal, None),
|
|
313
|
+
("decision curve: 99 thresholds", "NumPy loop", es_dca, numpy_dca),
|
|
314
|
+
("statistics: Welch t-test", "SciPy", lambda: [es.t_test(a, b).p_value], None),
|
|
315
|
+
("statistics: Mann–Whitney U", "SciPy", lambda: [es.mann_whitney_test(a, b).p_value], None),
|
|
316
|
+
("statistics: Cramér's V (5×5 table)", "SciPy", lambda: [es.cramers_v(table)], None),
|
|
317
|
+
(
|
|
318
|
+
"multiple testing: Hochberg (n p-values)",
|
|
319
|
+
"statsmodels",
|
|
320
|
+
lambda: es.adjust_pvalues(pvals, method="hochberg").tolist(),
|
|
321
|
+
None,
|
|
322
|
+
),
|
|
323
|
+
]
|
|
324
|
+
refs: dict[str, Callable[[], Any]] = {}
|
|
325
|
+
try:
|
|
326
|
+
import sklearn.metrics as skm
|
|
327
|
+
|
|
328
|
+
def sk_diag() -> list[float]:
|
|
329
|
+
lr_pos, lr_neg = skm.class_likelihood_ratios(y, p)
|
|
330
|
+
return [skm.recall_score(y, p), skm.recall_score(y, p, pos_label=0), lr_pos, lr_neg]
|
|
331
|
+
|
|
332
|
+
refs["clinical: sensitivity, specificity, LR+, LR−"] = sk_diag
|
|
333
|
+
except (ImportError, AttributeError):
|
|
334
|
+
pass
|
|
335
|
+
from scipy import stats
|
|
336
|
+
|
|
337
|
+
refs["statistics: Welch t-test"] = lambda: [stats.ttest_ind(a, b, equal_var=False).pvalue]
|
|
338
|
+
refs["statistics: Mann–Whitney U"] = lambda: [stats.mannwhitneyu(a, b).pvalue]
|
|
339
|
+
refs["statistics: Cramér's V (5×5 table)"] = lambda: [stats.contingency.association(table, method="cramer")]
|
|
340
|
+
try:
|
|
341
|
+
import statsmodels.api as sm # type: ignore[import-untyped]
|
|
342
|
+
from statsmodels.stats.contingency_tables import Table2x2 # type: ignore[import-untyped]
|
|
343
|
+
from statsmodels.stats.multitest import multipletests # type: ignore[import-untyped]
|
|
344
|
+
from statsmodels.stats.proportion import proportion_confint # type: ignore[import-untyped]
|
|
345
|
+
|
|
346
|
+
def sm_report() -> list[float]:
|
|
347
|
+
tp = int(np.sum((y == 1) & (p == 1)))
|
|
348
|
+
fn = int(np.sum((y == 1) & (p == 0)))
|
|
349
|
+
fp = int(np.sum((y == 0) & (p == 1)))
|
|
350
|
+
tn = int(np.sum((y == 0) & (p == 0)))
|
|
351
|
+
out: list[float] = []
|
|
352
|
+
for k, m in ((tp, tp + fn), (tn, tn + fp), (tp, tp + fp), (tn, tn + fn), (tp + tn, n), (tp + fn, n)):
|
|
353
|
+
out.extend(proportion_confint(k, m, method="wilson"))
|
|
354
|
+
out.extend(Table2x2(np.array([[tp, fn], [fp, tn]])).oddsratio_confint())
|
|
355
|
+
return out
|
|
356
|
+
|
|
357
|
+
def sm_cal() -> list[float]:
|
|
358
|
+
lp = np.log(risk / (1 - risk))
|
|
359
|
+
fam = sm.families.Binomial()
|
|
360
|
+
slope = sm.GLM(yc, sm.add_constant(lp), family=fam).fit().params[1]
|
|
361
|
+
intercept = sm.GLM(yc, np.ones((n, 1)), family=fam, offset=lp).fit().params[0]
|
|
362
|
+
return [slope, intercept]
|
|
363
|
+
|
|
364
|
+
refs["clinical: diagnostic report (7 CIs)"] = sm_report
|
|
365
|
+
refs["calibration: slope and intercept"] = sm_cal
|
|
366
|
+
refs["multiple testing: Hochberg (n p-values)"] = lambda: multipletests(pvals, method="simes-hochberg")[
|
|
367
|
+
1
|
|
368
|
+
].tolist()
|
|
369
|
+
except ImportError:
|
|
370
|
+
pass
|
|
371
|
+
return [(name, ref, es_fn, own if own is not None else refs.get(name)) for name, ref, es_fn, own in cases]
|
|
113
372
|
|
|
114
373
|
|
|
115
374
|
@dataclass(frozen=True, eq=False)
|
|
@@ -131,11 +390,12 @@ class BenchmarkResult:
|
|
|
131
390
|
[
|
|
132
391
|
r["case"],
|
|
133
392
|
f"{r['n']:,}",
|
|
393
|
+
r.get("reference") or "–",
|
|
134
394
|
f(r["evalsuite_ms"]),
|
|
135
|
-
f(r
|
|
395
|
+
f(r.get("reference_ms")),
|
|
136
396
|
"–" if r["speedup"] is None else f"{r['speedup']:.2f}×",
|
|
137
397
|
f(r["evalsuite_peak_mb"], 2),
|
|
138
|
-
f(r
|
|
398
|
+
f(r.get("reference_peak_mb"), 2),
|
|
139
399
|
"–" if r["max_abs_diff"] is None else f"{r['max_abs_diff']:.1e}",
|
|
140
400
|
]
|
|
141
401
|
for r in self.rows
|
|
@@ -144,11 +404,12 @@ class BenchmarkResult:
|
|
|
144
404
|
_TITLES = (
|
|
145
405
|
"Case",
|
|
146
406
|
"n",
|
|
407
|
+
"Reference",
|
|
147
408
|
"EvalSuite (ms)",
|
|
148
|
-
"
|
|
409
|
+
"Reference (ms)",
|
|
149
410
|
"Speed-up",
|
|
150
411
|
"EvalSuite peak (MiB)",
|
|
151
|
-
"
|
|
412
|
+
"Reference peak (MiB)",
|
|
152
413
|
"Max |difference|",
|
|
153
414
|
)
|
|
154
415
|
|
|
@@ -157,15 +418,18 @@ class BenchmarkResult:
|
|
|
157
418
|
widths = [max(len(t), *(len(c[i]) for c in cells)) for i, t in enumerate(self._TITLES)]
|
|
158
419
|
|
|
159
420
|
def line(c: Sequence[str]) -> str:
|
|
160
|
-
return " ".join(x.ljust(widths[i]) if i
|
|
421
|
+
return " ".join(x.ljust(widths[i]) if i in (0, 2) else x.rjust(widths[i]) for i, x in enumerate(c))
|
|
161
422
|
|
|
162
423
|
env = self.environment
|
|
163
424
|
head = (
|
|
164
425
|
f"EvalSuite {env['evalsuite']} benchmarks | Python {env['python']} | NumPy {env['numpy']}"
|
|
165
426
|
+ (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
|
|
427
|
+
+ (f" | statsmodels {env['statsmodels']}" if env.get("statsmodels") else "")
|
|
428
|
+
+ (f" | SciPy {env['scipy']}" if env.get("scipy") else "")
|
|
429
|
+
+ (f" | pycocotools {env['pycocotools']}" if env.get("pycocotools") else "")
|
|
166
430
|
+ f" | {env['machine']} | fastest of {env['repeat']} runs"
|
|
167
431
|
)
|
|
168
|
-
note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares
|
|
432
|
+
note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares EvalSuite with the reference."
|
|
169
433
|
return "\n".join([head, "", line(self._TITLES), *(line(c) for c in cells), "", note])
|
|
170
434
|
|
|
171
435
|
def __repr__(self) -> str:
|
|
@@ -189,7 +453,7 @@ class BenchmarkResult:
|
|
|
189
453
|
def to_csv(self, path: Optional[str] = None) -> str:
|
|
190
454
|
from .core.export import csv_text
|
|
191
455
|
|
|
192
|
-
rows = [[r
|
|
456
|
+
rows = [[r.get(h) if r.get(h) is not None else float("nan") for h in _HEADER] for r in self.rows]
|
|
193
457
|
text = csv_text(list(_HEADER), rows)
|
|
194
458
|
if path is not None:
|
|
195
459
|
with open(path, "w", encoding="utf-8", newline="") as fh:
|
|
@@ -197,7 +461,10 @@ class BenchmarkResult:
|
|
|
197
461
|
return text
|
|
198
462
|
|
|
199
463
|
def to_markdown(self, *, digits: int = 3) -> str:
|
|
200
|
-
lines = [
|
|
464
|
+
lines = [
|
|
465
|
+
"| " + " | ".join(self._TITLES) + " |",
|
|
466
|
+
"| --- | ---: | --- |" + " ---: |" * (len(self._TITLES) - 3),
|
|
467
|
+
]
|
|
201
468
|
return "\n".join(lines + ["| " + " | ".join(c) + " |" for c in self._cells(digits)])
|
|
202
469
|
|
|
203
470
|
def to_latex(self, *, digits: int = 3, caption: Optional[str] = None, label: Optional[str] = None) -> str:
|
|
@@ -223,50 +490,79 @@ def run_benchmarks(
|
|
|
223
490
|
repeat: int = 5,
|
|
224
491
|
compare_sklearn: bool = True,
|
|
225
492
|
random_state: Optional[int] = 0,
|
|
493
|
+
suite: str = "all",
|
|
226
494
|
) -> BenchmarkResult:
|
|
227
|
-
"""Time and memory for
|
|
495
|
+
"""Time and memory for evaluation workloads at each size, against a reference implementation.
|
|
496
|
+
|
|
497
|
+
``suite``: ``"core"`` (classification and regression vs scikit-learn), ``"clinical"`` (v0.2.0 clinical,
|
|
498
|
+
calibration and statistics vs scikit-learn, statsmodels, SciPy), ``"vision"`` (segmentation and COCO
|
|
499
|
+
detection vs scikit-learn, SciPy, pycocotools) or ``"all"`` (default).
|
|
500
|
+
``compare_sklearn=False`` times EvalSuite alone. Rows keep ``sklearn_ms``/``sklearn_peak_mb`` for rows
|
|
501
|
+
whose reference is scikit-learn, for compatibility with 0.1.x.
|
|
502
|
+
"""
|
|
503
|
+
import scipy
|
|
504
|
+
|
|
228
505
|
import evalsuite as es
|
|
229
506
|
|
|
230
507
|
if repeat < 1:
|
|
231
508
|
raise ValueError("repeat must be at least 1.")
|
|
509
|
+
if suite not in ("all", "core", "clinical", "vision"):
|
|
510
|
+
raise ValueError("suite must be 'all', 'core', 'clinical' or 'vision'.")
|
|
232
511
|
rng = np.random.default_rng(random_state)
|
|
233
512
|
rows: list[dict[str, Any]] = []
|
|
234
|
-
|
|
513
|
+
versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None, "pycocotools": None}
|
|
235
514
|
if compare_sklearn:
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
515
|
+
for mod in versions:
|
|
516
|
+
with contextlib.suppress(ImportError):
|
|
517
|
+
__import__(mod)
|
|
518
|
+
from importlib.metadata import version as _dist_version
|
|
519
|
+
|
|
520
|
+
versions[mod] = _dist_version("scikit-learn" if mod == "sklearn" else mod)
|
|
521
|
+
builders = {
|
|
522
|
+
"core": [_core_cases],
|
|
523
|
+
"clinical": [_clinical_cases],
|
|
524
|
+
"vision": [_vision_cases],
|
|
525
|
+
"all": [_core_cases, _clinical_cases, _vision_cases],
|
|
526
|
+
}[suite]
|
|
242
527
|
for n in sizes:
|
|
243
|
-
for
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
528
|
+
for build in builders:
|
|
529
|
+
for name, ref_name, es_fn, ref_fn in build(int(n), rng):
|
|
530
|
+
es_t, es_mem, es_val = _measure(es_fn, repeat)
|
|
531
|
+
row: dict[str, Any] = {
|
|
532
|
+
"case": name,
|
|
533
|
+
"n": int(n),
|
|
534
|
+
"reference": None,
|
|
535
|
+
"evalsuite_ms": es_t * 1000,
|
|
536
|
+
"evalsuite_peak_mb": es_mem,
|
|
537
|
+
"reference_ms": None,
|
|
538
|
+
"reference_peak_mb": None,
|
|
539
|
+
"sklearn_ms": None,
|
|
540
|
+
"sklearn_peak_mb": None,
|
|
541
|
+
"speedup": None,
|
|
542
|
+
"max_abs_diff": None,
|
|
543
|
+
}
|
|
544
|
+
if compare_sklearn and ref_fn is not None:
|
|
545
|
+
ref_t, ref_mem, ref_val = _measure(ref_fn, repeat)
|
|
546
|
+
row.update(
|
|
547
|
+
reference=ref_name,
|
|
548
|
+
reference_ms=ref_t * 1000,
|
|
549
|
+
reference_peak_mb=ref_mem,
|
|
550
|
+
speedup=ref_t / es_t if es_t else None,
|
|
551
|
+
max_abs_diff=float(np.max(np.abs(np.asarray(es_val, float) - np.asarray(ref_val, float)))),
|
|
552
|
+
)
|
|
553
|
+
if ref_name == "scikit-learn":
|
|
554
|
+
row.update(sklearn_ms=row["reference_ms"], sklearn_peak_mb=ref_mem)
|
|
555
|
+
rows.append(row)
|
|
264
556
|
env = {
|
|
265
557
|
"evalsuite": es.__version__,
|
|
266
558
|
"python": platform.python_version(),
|
|
267
559
|
"numpy": np.__version__,
|
|
268
|
-
"
|
|
560
|
+
"scipy": scipy.__version__ if suite != "core" else None,
|
|
561
|
+
"sklearn": versions["sklearn"],
|
|
562
|
+
"statsmodels": versions["statsmodels"] if suite != "core" else None,
|
|
563
|
+
"pycocotools": versions["pycocotools"] if suite in ("all", "vision") else None,
|
|
269
564
|
"machine": f"{platform.system()} {platform.machine()}",
|
|
270
565
|
"repeat": repeat,
|
|
566
|
+
"suite": suite,
|
|
271
567
|
}
|
|
272
568
|
return BenchmarkResult(tuple(rows), env)
|
evalsuite/calibration.py
CHANGED
|
@@ -128,7 +128,7 @@ def logistic_fit(
|
|
|
128
128
|
beta = np.zeros(design.shape[1])
|
|
129
129
|
for _ in range(max_iter):
|
|
130
130
|
eta = design @ beta + off
|
|
131
|
-
mu = 1 / (1 + np.exp(-eta))
|
|
131
|
+
mu = 1 / (1 + np.exp(-np.clip(eta, -700, 700)))
|
|
132
132
|
grad = design.T @ (w * (y - mu))
|
|
133
133
|
hess = (design * (w * mu * (1 - mu))[:, None]).T @ design
|
|
134
134
|
try:
|