evalsuite-python 0.2.0__py3-none-any.whl → 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evalsuite/__init__.py CHANGED
@@ -5,7 +5,7 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import calibration, classification, clinical, plot, regression, stats
8
+ from . import calibration, classification, clinical, plot, regression, stats, vision
9
9
  from .api import evaluate
10
10
  from .calibration import (
11
11
  CalibrationReport,
@@ -111,8 +111,49 @@ from .stats import (
111
111
  wilcoxon_test,
112
112
  )
113
113
  from .version import __version__
114
+ from .vision import (
115
+ DetectionReport,
116
+ SegmentationReport,
117
+ average_precision_detection,
118
+ average_surface_distance,
119
+ boundary_iou,
120
+ box_iou,
121
+ detection_pr_curve,
122
+ detection_report,
123
+ dice,
124
+ from_coco,
125
+ hausdorff_distance,
126
+ iou,
127
+ mean_average_precision,
128
+ mean_pixel_accuracy,
129
+ miou,
130
+ per_image_scores,
131
+ pixel_accuracy,
132
+ segmentation_confusion,
133
+ segmentation_report,
134
+ )
114
135
 
115
136
  __all__ = [
137
+ "SegmentationReport",
138
+ "segmentation_report",
139
+ "detection_pr_curve",
140
+ "vision",
141
+ "DetectionReport",
142
+ "average_precision_detection",
143
+ "average_surface_distance",
144
+ "boundary_iou",
145
+ "box_iou",
146
+ "detection_report",
147
+ "dice",
148
+ "from_coco",
149
+ "hausdorff_distance",
150
+ "iou",
151
+ "mean_average_precision",
152
+ "mean_pixel_accuracy",
153
+ "miou",
154
+ "per_image_scores",
155
+ "pixel_accuracy",
156
+ "segmentation_confusion",
116
157
  "CalibrationReport",
117
158
  "calibration_report",
118
159
  "calibration",
evalsuite/api.py CHANGED
@@ -149,6 +149,16 @@ def evaluate(
149
149
  >>> round(r["accuracy"], 2)
150
150
  0.75
151
151
  """
152
+ if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
153
+ raise UnsupportedTaskError(
154
+ "y_true looks like object detection annotations (one dict per image); use "
155
+ "evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
156
+ )
157
+ if np.ndim(y_true) >= 3:
158
+ raise UnsupportedTaskError(
159
+ "y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
160
+ "evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
161
+ )
152
162
  task = task or _infer_task(y_true, y_prob)
153
163
  if task == "classification":
154
164
  return _evaluate_classification(
evalsuite/benchmarks.py CHANGED
@@ -1,9 +1,14 @@
1
- """Speed and memory benchmarks, against scikit-learn when it is installed.
1
+ """Speed and memory benchmarks against reference implementations.
2
2
 
3
3
  Each case times the fastest of ``repeat`` runs (after one warm-up) and measures peak traced memory with
4
- ``tracemalloc`` (NumPy reports its allocations to it). Both libraries compute the same metrics on the same
5
- data, and the largest absolute difference between their results is reported, so speed is never shown for
6
- numbers that disagree.
4
+ ``tracemalloc`` (NumPy reports its allocations to it). EvalSuite and the reference compute the same
5
+ quantities on the same data, and the largest absolute difference between their results is reported, so
6
+ speed is never shown for numbers that disagree.
7
+
8
+ References: scikit-learn for classification and regression (``suite="core"``); scikit-learn, statsmodels
9
+ and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``);
10
+ scikit-learn, SciPy and pycocotools for segmentation and detection (``suite="vision"``). A case
11
+ whose reference library is not installed is timed for EvalSuite only.
7
12
 
8
13
  >>> from evalsuite.benchmarks import run_benchmarks
9
14
  >>> print(run_benchmarks(sizes=(10_000,), repeat=3)) # doctest: +SKIP
@@ -11,6 +16,7 @@ numbers that disagree.
11
16
 
12
17
  from __future__ import annotations
13
18
 
19
+ import contextlib
14
20
  import json
15
21
  import platform
16
22
  import time
@@ -33,14 +39,17 @@ __all__ = ["BenchmarkResult", "run_benchmarks"]
33
39
  _HEADER = (
34
40
  "case",
35
41
  "n",
42
+ "reference",
36
43
  "evalsuite_ms",
37
- "sklearn_ms",
44
+ "reference_ms",
38
45
  "speedup",
39
46
  "evalsuite_peak_mb",
40
- "sklearn_peak_mb",
47
+ "reference_peak_mb",
41
48
  "max_abs_diff",
42
49
  )
43
50
 
51
+ Case = tuple[str, str, Callable[[], Any], Optional[Callable[[], Any]]]
52
+
44
53
 
45
54
  def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
46
55
  """(fastest seconds, peak MiB, result)."""
@@ -61,7 +70,7 @@ def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
61
70
  return best, peak / 2**20, result
62
71
 
63
72
 
64
- def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]]:
73
+ def _core_cases(n: int, rng: np.random.Generator) -> list[Case]:
65
74
  import evalsuite as es
66
75
 
67
76
  y = rng.integers(0, 2, n)
@@ -81,11 +90,11 @@ def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any
81
90
  r = es.evaluate(yr, pr, metrics=["mae", "mse", "rmse", "r2"])
82
91
  return [float(r[m]) for m in ("mae", "mse", "rmse", "r2")]
83
92
 
84
- cases: list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]] = [
85
- ("binary: 8 label metrics via evaluate()", es_binary, None),
86
- ("10 classes: macro F1", lambda: [float(es.f1(yk, pk, average="macro"))], None),
87
- ("binary: ROC AUC", lambda: [float(es.roc_auc(y, prob))], None),
88
- ("regression: MAE, MSE, RMSE, R² via evaluate()", es_reg, None),
93
+ cases: list[Case] = [
94
+ ("binary: 8 label metrics via evaluate()", "scikit-learn", es_binary, None),
95
+ ("10 classes: macro F1", "scikit-learn", lambda: [float(es.f1(yk, pk, average="macro"))], None),
96
+ ("binary: ROC AUC", "scikit-learn", lambda: [float(es.roc_auc(y, prob))], None),
97
+ ("regression: MAE, MSE, RMSE, R² via evaluate()", "scikit-learn", es_reg, None),
89
98
  ]
90
99
  try:
91
100
  import sklearn.metrics as skm # type: ignore[import-untyped]
@@ -109,7 +118,257 @@ def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any
109
118
  return [skm.mean_absolute_error(yr, pr), mse, float(np.sqrt(mse)), skm.r2_score(yr, pr)]
110
119
 
111
120
  sk = [sk_binary, lambda: [skm.f1_score(yk, pk, average="macro")], lambda: [skm.roc_auc_score(y, prob)], sk_reg]
112
- return [(name, es_fn, sk_fn) for (name, es_fn, _), sk_fn in zip(cases, sk)]
121
+ return [(name, ref, es_fn, sk_fn) for (name, ref, es_fn, _), sk_fn in zip(cases, sk)]
122
+
123
+
124
+ def _vision_cases(n: int, rng: np.random.Generator) -> list[Case]:
125
+ """v0.3.0: segmentation overlap and surface distance (n = pixels) and COCO detection (n / 1000 images)."""
126
+ import evalsuite as es
127
+
128
+ side = 64
129
+ n_img = max(1, n // (side * side))
130
+ k = 5
131
+ yy, xx = np.ogrid[:side, :side]
132
+ true = np.zeros((n_img, side, side), dtype=np.int64)
133
+ for i in range(n_img):
134
+ for c in range(1, k):
135
+ cy, cx, r = rng.integers(8, side - 8), rng.integers(8, side - 8), rng.integers(4, 14)
136
+ true[i][(yy - cy) ** 2 + (xx - cx) ** 2 <= r * r] = c
137
+ pred = np.roll(true, 1, axis=2)
138
+ noise = rng.random(pred.shape) < 0.02
139
+ pred[noise] = rng.integers(0, k, int(noise.sum()))
140
+ labels = list(range(k))
141
+
142
+ def es_overlap() -> list[float]:
143
+ d = np.asarray(es.dice(true, pred, average=None).value)
144
+ j = np.asarray(es.iou(true, pred, average=None).value)
145
+ return [*d, *j]
146
+
147
+ n_hd = min(n_img, 50)
148
+
149
+ def es_hd() -> list[float]:
150
+ return [float(es.hausdorff_distance(true[i], pred[i], labels=[k - 1])) for i in range(n_hd)]
151
+
152
+ n_det = max(10, n // 1000)
153
+ y_true, y_pred = [], []
154
+ for _ in range(n_det):
155
+ m = int(rng.integers(1, 8))
156
+ xy = rng.uniform(0, 500, (m, 2))
157
+ wh = rng.uniform(8, 160, (m, 2))
158
+ boxes = np.column_stack([xy, xy + wh])
159
+ lab = rng.integers(1, 6, m)
160
+ y_true.append({"boxes": boxes, "labels": lab, "iscrowd": np.zeros(m, int)})
161
+ jitter = boxes + rng.normal(0, 4, boxes.shape)
162
+ jitter[:, 2:] = np.maximum(jitter[:, 2:], jitter[:, :2] + 1)
163
+ extra = rng.uniform(0, 500, (3, 2))
164
+ fp = np.column_stack([extra, extra + rng.uniform(10, 80, (3, 2))])
165
+ y_pred.append(
166
+ {
167
+ "boxes": np.vstack([jitter, fp]),
168
+ "labels": np.r_[lab, rng.integers(1, 6, 3)],
169
+ "scores": rng.random(m + 3),
170
+ }
171
+ )
172
+
173
+ def es_map() -> list[float]:
174
+ r = es.detection_report(y_true, y_pred)
175
+ return [r["map"], r["map_50"], r["map_75"], r["mar_100"]]
176
+
177
+ cases: list[Case] = [
178
+ ("segmentation: Dice and IoU per class (n = pixels)", "scikit-learn", es_overlap, None),
179
+ (f"segmentation: Hausdorff distance ({n_hd} image{'s' if n_hd != 1 else ''})", "SciPy", es_hd, None),
180
+ (f"detection: COCO evaluation ({n_det} images)", "pycocotools", es_map, None),
181
+ ]
182
+ refs: dict[str, Callable[[], Any]] = {}
183
+ try:
184
+ import sklearn.metrics as skm
185
+
186
+ def sk_overlap() -> list[float]:
187
+ ft, fp_ = true.ravel(), pred.ravel()
188
+ return [
189
+ *skm.f1_score(ft, fp_, labels=labels, average=None),
190
+ *skm.jaccard_score(ft, fp_, labels=labels, average=None),
191
+ ]
192
+
193
+ refs[cases[0][0]] = sk_overlap
194
+ except ImportError:
195
+ pass
196
+ from scipy import ndimage
197
+ from scipy.spatial.distance import directed_hausdorff
198
+
199
+ def scipy_hd() -> list[float]:
200
+ out = []
201
+ for i in range(n_hd):
202
+ pts = []
203
+ for m_ in (true[i] == k - 1, pred[i] == k - 1):
204
+ er = ndimage.binary_erosion(m_, structure=ndimage.generate_binary_structure(2, 1), border_value=0)
205
+ pts.append(np.argwhere(m_ & ~er).astype(float))
206
+ out.append(max(directed_hausdorff(pts[0], pts[1])[0], directed_hausdorff(pts[1], pts[0])[0]))
207
+ return out
208
+
209
+ refs[cases[1][0]] = scipy_hd
210
+ try:
211
+ import contextlib as _ctx
212
+ import io
213
+
214
+ from pycocotools.coco import COCO # type: ignore[import-untyped]
215
+ from pycocotools.cocoeval import COCOeval # type: ignore[import-untyped]
216
+
217
+ def coco_map() -> list[float]:
218
+ images, anns, dets, aid = [], [], [], 1
219
+ for i, (t, p) in enumerate(zip(y_true, y_pred)):
220
+ images.append({"id": i + 1})
221
+ for b, c in zip(t["boxes"], t["labels"]):
222
+ w, h = b[2] - b[0], b[3] - b[1]
223
+ anns.append(
224
+ {
225
+ "id": aid,
226
+ "image_id": i + 1,
227
+ "category_id": int(c),
228
+ "bbox": [b[0], b[1], w, h],
229
+ "area": w * h,
230
+ "iscrowd": 0,
231
+ }
232
+ )
233
+ aid += 1
234
+ for b, c, sc in zip(p["boxes"], p["labels"], p["scores"]):
235
+ dets.append(
236
+ {
237
+ "image_id": i + 1,
238
+ "category_id": int(c),
239
+ "bbox": [b[0], b[1], b[2] - b[0], b[3] - b[1]],
240
+ "score": float(sc),
241
+ }
242
+ )
243
+ with _ctx.redirect_stdout(io.StringIO()):
244
+ gt = COCO()
245
+ gt.dataset = {
246
+ "images": images,
247
+ "annotations": anns,
248
+ "categories": [{"id": c} for c in range(1, 6)],
249
+ }
250
+ gt.createIndex()
251
+ ev = COCOeval(gt, gt.loadRes(dets), "bbox")
252
+ ev.evaluate()
253
+ ev.accumulate()
254
+ ev.summarize()
255
+ return [ev.stats[0], ev.stats[1], ev.stats[2], ev.stats[8]]
256
+
257
+ refs[cases[2][0]] = coco_map
258
+ except ImportError:
259
+ pass
260
+ return [(name, ref, es_fn, refs.get(name)) for name, ref, es_fn, _ in cases]
261
+
262
+
263
+ def _clinical_cases(n: int, rng: np.random.Generator) -> list[Case]:
264
+ """v0.2.0: diagnostic accuracy, calibration, decision curves and statistical tests."""
265
+ import evalsuite as es
266
+
267
+ y = rng.integers(0, 2, n)
268
+ p = np.where(rng.random(n) < 0.8, y, 1 - y)
269
+ x = rng.normal(size=n)
270
+ yc = (rng.random(n) < 1 / (1 + np.exp(-(0.4 + 1.3 * x)))).astype(int)
271
+ risk = 1 / (1 + np.exp(-(0.1 + 2.0 * x)))
272
+ a, b = rng.normal(0, 1, n), rng.normal(0.05, 1.2, n)
273
+ pvals = rng.random(n) ** 2
274
+ ga, gb = rng.integers(0, 5, n), rng.integers(0, 5, n)
275
+ table = np.zeros((5, 5), dtype=np.int64)
276
+ np.add.at(table, (ga, gb), 1)
277
+ thresholds = np.arange(1, 100) / 100
278
+
279
+ def es_diag() -> list[float]:
280
+ return [
281
+ float(es.sensitivity(y, p)),
282
+ float(es.specificity(y, p)),
283
+ float(es.lr_positive(y, p)),
284
+ float(es.lr_negative(y, p)),
285
+ ]
286
+
287
+ report_keys = ("sensitivity", "specificity", "ppv", "npv", "accuracy", "prevalence", "diagnostic_odds_ratio")
288
+
289
+ def es_report() -> list[float]:
290
+ r = es.diagnostic_report(y, p)
291
+ return [v for k in report_keys for v in (r[k].low, r[k].high)]
292
+
293
+ def es_cal() -> list[float]:
294
+ return [float(es.calibration_slope(yc, risk)), float(es.calibration_intercept(yc, risk))]
295
+
296
+ def es_dca() -> list[float]:
297
+ curve: list[float] = es.decision_curve(yc, risk, thresholds=thresholds).net_benefit["model"].tolist()
298
+ return curve
299
+
300
+ def numpy_dca() -> list[float]: # the textbook loop, one threshold at a time
301
+ out = []
302
+ for t in thresholds:
303
+ treat = risk >= t
304
+ tp = np.sum(treat & (yc == 1))
305
+ fp = np.sum(treat & (yc == 0))
306
+ out.append(tp / n - fp / n * t / (1 - t))
307
+ return out
308
+
309
+ cases: list[Case] = [
310
+ ("clinical: sensitivity, specificity, LR+, LR−", "scikit-learn", es_diag, None),
311
+ ("clinical: diagnostic report (7 CIs)", "statsmodels", es_report, None),
312
+ ("calibration: slope and intercept", "statsmodels", es_cal, None),
313
+ ("decision curve: 99 thresholds", "NumPy loop", es_dca, numpy_dca),
314
+ ("statistics: Welch t-test", "SciPy", lambda: [es.t_test(a, b).p_value], None),
315
+ ("statistics: Mann–Whitney U", "SciPy", lambda: [es.mann_whitney_test(a, b).p_value], None),
316
+ ("statistics: Cramér's V (5×5 table)", "SciPy", lambda: [es.cramers_v(table)], None),
317
+ (
318
+ "multiple testing: Hochberg (n p-values)",
319
+ "statsmodels",
320
+ lambda: es.adjust_pvalues(pvals, method="hochberg").tolist(),
321
+ None,
322
+ ),
323
+ ]
324
+ refs: dict[str, Callable[[], Any]] = {}
325
+ try:
326
+ import sklearn.metrics as skm
327
+
328
+ def sk_diag() -> list[float]:
329
+ lr_pos, lr_neg = skm.class_likelihood_ratios(y, p)
330
+ return [skm.recall_score(y, p), skm.recall_score(y, p, pos_label=0), lr_pos, lr_neg]
331
+
332
+ refs["clinical: sensitivity, specificity, LR+, LR−"] = sk_diag
333
+ except (ImportError, AttributeError):
334
+ pass
335
+ from scipy import stats
336
+
337
+ refs["statistics: Welch t-test"] = lambda: [stats.ttest_ind(a, b, equal_var=False).pvalue]
338
+ refs["statistics: Mann–Whitney U"] = lambda: [stats.mannwhitneyu(a, b).pvalue]
339
+ refs["statistics: Cramér's V (5×5 table)"] = lambda: [stats.contingency.association(table, method="cramer")]
340
+ try:
341
+ import statsmodels.api as sm # type: ignore[import-untyped]
342
+ from statsmodels.stats.contingency_tables import Table2x2 # type: ignore[import-untyped]
343
+ from statsmodels.stats.multitest import multipletests # type: ignore[import-untyped]
344
+ from statsmodels.stats.proportion import proportion_confint # type: ignore[import-untyped]
345
+
346
+ def sm_report() -> list[float]:
347
+ tp = int(np.sum((y == 1) & (p == 1)))
348
+ fn = int(np.sum((y == 1) & (p == 0)))
349
+ fp = int(np.sum((y == 0) & (p == 1)))
350
+ tn = int(np.sum((y == 0) & (p == 0)))
351
+ out: list[float] = []
352
+ for k, m in ((tp, tp + fn), (tn, tn + fp), (tp, tp + fp), (tn, tn + fn), (tp + tn, n), (tp + fn, n)):
353
+ out.extend(proportion_confint(k, m, method="wilson"))
354
+ out.extend(Table2x2(np.array([[tp, fn], [fp, tn]])).oddsratio_confint())
355
+ return out
356
+
357
+ def sm_cal() -> list[float]:
358
+ lp = np.log(risk / (1 - risk))
359
+ fam = sm.families.Binomial()
360
+ slope = sm.GLM(yc, sm.add_constant(lp), family=fam).fit().params[1]
361
+ intercept = sm.GLM(yc, np.ones((n, 1)), family=fam, offset=lp).fit().params[0]
362
+ return [slope, intercept]
363
+
364
+ refs["clinical: diagnostic report (7 CIs)"] = sm_report
365
+ refs["calibration: slope and intercept"] = sm_cal
366
+ refs["multiple testing: Hochberg (n p-values)"] = lambda: multipletests(pvals, method="simes-hochberg")[
367
+ 1
368
+ ].tolist()
369
+ except ImportError:
370
+ pass
371
+ return [(name, ref, es_fn, own if own is not None else refs.get(name)) for name, ref, es_fn, own in cases]
113
372
 
114
373
 
115
374
  @dataclass(frozen=True, eq=False)
@@ -131,11 +390,12 @@ class BenchmarkResult:
131
390
  [
132
391
  r["case"],
133
392
  f"{r['n']:,}",
393
+ r.get("reference") or "–",
134
394
  f(r["evalsuite_ms"]),
135
- f(r["sklearn_ms"]),
395
+ f(r.get("reference_ms")),
136
396
  "–" if r["speedup"] is None else f"{r['speedup']:.2f}×",
137
397
  f(r["evalsuite_peak_mb"], 2),
138
- f(r["sklearn_peak_mb"], 2),
398
+ f(r.get("reference_peak_mb"), 2),
139
399
  "–" if r["max_abs_diff"] is None else f"{r['max_abs_diff']:.1e}",
140
400
  ]
141
401
  for r in self.rows
@@ -144,11 +404,12 @@ class BenchmarkResult:
144
404
  _TITLES = (
145
405
  "Case",
146
406
  "n",
407
+ "Reference",
147
408
  "EvalSuite (ms)",
148
- "scikit-learn (ms)",
409
+ "Reference (ms)",
149
410
  "Speed-up",
150
411
  "EvalSuite peak (MiB)",
151
- "scikit-learn peak (MiB)",
412
+ "Reference peak (MiB)",
152
413
  "Max |difference|",
153
414
  )
154
415
 
@@ -157,15 +418,18 @@ class BenchmarkResult:
157
418
  widths = [max(len(t), *(len(c[i]) for c in cells)) for i, t in enumerate(self._TITLES)]
158
419
 
159
420
  def line(c: Sequence[str]) -> str:
160
- return " ".join(x.ljust(widths[i]) if i == 0 else x.rjust(widths[i]) for i, x in enumerate(c))
421
+ return " ".join(x.ljust(widths[i]) if i in (0, 2) else x.rjust(widths[i]) for i, x in enumerate(c))
161
422
 
162
423
  env = self.environment
163
424
  head = (
164
425
  f"EvalSuite {env['evalsuite']} benchmarks | Python {env['python']} | NumPy {env['numpy']}"
165
426
  + (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
427
+ + (f" | statsmodels {env['statsmodels']}" if env.get("statsmodels") else "")
428
+ + (f" | SciPy {env['scipy']}" if env.get("scipy") else "")
429
+ + (f" | pycocotools {env['pycocotools']}" if env.get("pycocotools") else "")
166
430
  + f" | {env['machine']} | fastest of {env['repeat']} runs"
167
431
  )
168
- note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares the two libraries' results."
432
+ note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares EvalSuite with the reference."
169
433
  return "\n".join([head, "", line(self._TITLES), *(line(c) for c in cells), "", note])
170
434
 
171
435
  def __repr__(self) -> str:
@@ -189,7 +453,7 @@ class BenchmarkResult:
189
453
  def to_csv(self, path: Optional[str] = None) -> str:
190
454
  from .core.export import csv_text
191
455
 
192
- rows = [[r[h] if r[h] is not None else float("nan") for h in _HEADER] for r in self.rows]
456
+ rows = [[r.get(h) if r.get(h) is not None else float("nan") for h in _HEADER] for r in self.rows]
193
457
  text = csv_text(list(_HEADER), rows)
194
458
  if path is not None:
195
459
  with open(path, "w", encoding="utf-8", newline="") as fh:
@@ -197,7 +461,10 @@ class BenchmarkResult:
197
461
  return text
198
462
 
199
463
  def to_markdown(self, *, digits: int = 3) -> str:
200
- lines = ["| " + " | ".join(self._TITLES) + " |", "| --- |" + " ---: |" * (len(self._TITLES) - 1)]
464
+ lines = [
465
+ "| " + " | ".join(self._TITLES) + " |",
466
+ "| --- | ---: | --- |" + " ---: |" * (len(self._TITLES) - 3),
467
+ ]
201
468
  return "\n".join(lines + ["| " + " | ".join(c) + " |" for c in self._cells(digits)])
202
469
 
203
470
  def to_latex(self, *, digits: int = 3, caption: Optional[str] = None, label: Optional[str] = None) -> str:
@@ -223,50 +490,79 @@ def run_benchmarks(
223
490
  repeat: int = 5,
224
491
  compare_sklearn: bool = True,
225
492
  random_state: Optional[int] = 0,
493
+ suite: str = "all",
226
494
  ) -> BenchmarkResult:
227
- """Time and memory for common evaluation workloads at each size, against scikit-learn if installed."""
495
+ """Time and memory for evaluation workloads at each size, against a reference implementation.
496
+
497
+ ``suite``: ``"core"`` (classification and regression vs scikit-learn), ``"clinical"`` (v0.2.0 clinical,
498
+ calibration and statistics vs scikit-learn, statsmodels, SciPy), ``"vision"`` (segmentation and COCO
499
+ detection vs scikit-learn, SciPy, pycocotools) or ``"all"`` (default).
500
+ ``compare_sklearn=False`` times EvalSuite alone. Rows keep ``sklearn_ms``/``sklearn_peak_mb`` for rows
501
+ whose reference is scikit-learn, for compatibility with 0.1.x.
502
+ """
503
+ import scipy
504
+
228
505
  import evalsuite as es
229
506
 
230
507
  if repeat < 1:
231
508
  raise ValueError("repeat must be at least 1.")
509
+ if suite not in ("all", "core", "clinical", "vision"):
510
+ raise ValueError("suite must be 'all', 'core', 'clinical' or 'vision'.")
232
511
  rng = np.random.default_rng(random_state)
233
512
  rows: list[dict[str, Any]] = []
234
- sk_version: Optional[str] = None
513
+ versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None, "pycocotools": None}
235
514
  if compare_sklearn:
236
- try:
237
- import sklearn
238
-
239
- sk_version = sklearn.__version__
240
- except ImportError:
241
- compare_sklearn = False
515
+ for mod in versions:
516
+ with contextlib.suppress(ImportError):
517
+ __import__(mod)
518
+ from importlib.metadata import version as _dist_version
519
+
520
+ versions[mod] = _dist_version("scikit-learn" if mod == "sklearn" else mod)
521
+ builders = {
522
+ "core": [_core_cases],
523
+ "clinical": [_clinical_cases],
524
+ "vision": [_vision_cases],
525
+ "all": [_core_cases, _clinical_cases, _vision_cases],
526
+ }[suite]
242
527
  for n in sizes:
243
- for name, es_fn, sk_fn in _cases(int(n), rng):
244
- es_t, es_mem, es_val = _measure(es_fn, repeat)
245
- row: dict[str, Any] = {
246
- "case": name,
247
- "n": int(n),
248
- "evalsuite_ms": es_t * 1000,
249
- "evalsuite_peak_mb": es_mem,
250
- "sklearn_ms": None,
251
- "sklearn_peak_mb": None,
252
- "speedup": None,
253
- "max_abs_diff": None,
254
- }
255
- if compare_sklearn and sk_fn is not None:
256
- sk_t, sk_mem, sk_val = _measure(sk_fn, repeat)
257
- row.update(
258
- sklearn_ms=sk_t * 1000,
259
- sklearn_peak_mb=sk_mem,
260
- speedup=sk_t / es_t if es_t else None,
261
- max_abs_diff=float(np.max(np.abs(np.asarray(es_val, float) - np.asarray(sk_val, float)))),
262
- )
263
- rows.append(row)
528
+ for build in builders:
529
+ for name, ref_name, es_fn, ref_fn in build(int(n), rng):
530
+ es_t, es_mem, es_val = _measure(es_fn, repeat)
531
+ row: dict[str, Any] = {
532
+ "case": name,
533
+ "n": int(n),
534
+ "reference": None,
535
+ "evalsuite_ms": es_t * 1000,
536
+ "evalsuite_peak_mb": es_mem,
537
+ "reference_ms": None,
538
+ "reference_peak_mb": None,
539
+ "sklearn_ms": None,
540
+ "sklearn_peak_mb": None,
541
+ "speedup": None,
542
+ "max_abs_diff": None,
543
+ }
544
+ if compare_sklearn and ref_fn is not None:
545
+ ref_t, ref_mem, ref_val = _measure(ref_fn, repeat)
546
+ row.update(
547
+ reference=ref_name,
548
+ reference_ms=ref_t * 1000,
549
+ reference_peak_mb=ref_mem,
550
+ speedup=ref_t / es_t if es_t else None,
551
+ max_abs_diff=float(np.max(np.abs(np.asarray(es_val, float) - np.asarray(ref_val, float)))),
552
+ )
553
+ if ref_name == "scikit-learn":
554
+ row.update(sklearn_ms=row["reference_ms"], sklearn_peak_mb=ref_mem)
555
+ rows.append(row)
264
556
  env = {
265
557
  "evalsuite": es.__version__,
266
558
  "python": platform.python_version(),
267
559
  "numpy": np.__version__,
268
- "sklearn": sk_version,
560
+ "scipy": scipy.__version__ if suite != "core" else None,
561
+ "sklearn": versions["sklearn"],
562
+ "statsmodels": versions["statsmodels"] if suite != "core" else None,
563
+ "pycocotools": versions["pycocotools"] if suite in ("all", "vision") else None,
269
564
  "machine": f"{platform.system()} {platform.machine()}",
270
565
  "repeat": repeat,
566
+ "suite": suite,
271
567
  }
272
568
  return BenchmarkResult(tuple(rows), env)
evalsuite/calibration.py CHANGED
@@ -128,7 +128,7 @@ def logistic_fit(
128
128
  beta = np.zeros(design.shape[1])
129
129
  for _ in range(max_iter):
130
130
  eta = design @ beta + off
131
- mu = 1 / (1 + np.exp(-eta))
131
+ mu = 1 / (1 + np.exp(-np.clip(eta, -700, 700)))
132
132
  grad = design.T @ (w * (y - mu))
133
133
  hess = (design * (w * mu * (1 - mu))[:, None]).T @ design
134
134
  try: