evalsuite-python 0.2.1__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/CHANGELOG.md +30 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/PKG-INFO +62 -5
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/README.md +54 -3
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/pyproject.toml +6 -2
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/__init__.py +42 -1
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/api.py +10 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/benchmarks.py +158 -7
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/calibration.py +1 -1
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/cli/main.py +99 -2
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/plot.py +165 -1
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/_resolve.py +26 -3
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/compare.py +6 -4
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.3.0/src/evalsuite/vision/__init__.py +47 -0
- evalsuite_python-0.3.0/src/evalsuite/vision/detection.py +648 -0
- evalsuite_python-0.3.0/src/evalsuite/vision/segmentation.py +965 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/test_v020.py +9 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_benchmarks.py +2 -2
- evalsuite_python-0.3.0/tests/vision/__init__.py +0 -0
- evalsuite_python-0.3.0/tests/vision/test_detection.py +200 -0
- evalsuite_python-0.3.0/tests/vision/test_edges.py +126 -0
- evalsuite_python-0.3.0/tests/vision/test_outputs.py +170 -0
- evalsuite_python-0.3.0/tests/vision/test_segmentation.py +156 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/.gitignore +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/LICENSE +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/__main__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/metrics.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/cli/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/metrics.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/report.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/export.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/result.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/validation.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/regression/metrics.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/reporting.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/effect.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/hypothesis.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/test_outputs.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/conftest.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/integration/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_cli.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_plot.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_reporting.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/regression/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/test_core.py +0 -0
- {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/test_edges.py +0 -0
|
@@ -6,6 +6,36 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.3.0] - 2026-10-09
|
|
10
|
+
|
|
11
|
+
Computer vision (the v0.3.0 roadmap), with comparison, plots, reporting and benchmarks extended to it.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
- Segmentation: `dice`, `iou`, `miou`, `pixel_accuracy`, `mean_pixel_accuracy`, `boundary_iou`,
|
|
15
|
+
`hausdorff_distance` (HD and HD95, anisotropic `spacing`), `average_surface_distance`,
|
|
16
|
+
`segmentation_confusion`, `per_image_scores` and `segmentation_report`. 2-D and 3-D masks, lists of
|
|
17
|
+
differently sized masks, `ignore_index`, `aggregate="dataset"|"image"`, undefined classes reported as NaN.
|
|
18
|
+
- Object detection: `box_iou`, `detection_report` (the 12 COCO numbers plus AP per class),
|
|
19
|
+
`mean_average_precision`, `average_precision_detection` (COCO or VOC interpolation), `detection_pr_curve`,
|
|
20
|
+
`from_coco`. Matches pycocotools exactly, including crowd regions, area ranges and detection limits.
|
|
21
|
+
- Comparison for vision: `compare`, `bootstrap_ci` and `paired_bootstrap_test` resample images for
|
|
22
|
+
segmentation masks and per-image detections.
|
|
23
|
+
- Plots: `es.plot.segmentation` (prediction fill vs truth outline), `es.plot.per_class` (any per-class
|
|
24
|
+
result, a segmentation report or a detection report), `es.plot.detection_pr`.
|
|
25
|
+
- CLI: `evalsuite segmentation` (.npy/.npz or a folder of mask images) and `evalsuite detection` (COCO JSON).
|
|
26
|
+
- Benchmarks: `--suite vision` against scikit-learn, SciPy and pycocotools.
|
|
27
|
+
- `vision` extra (Pillow, for reading mask image folders); `CITATION.cff`.
|
|
28
|
+
|
|
29
|
+
### Changed
|
|
30
|
+
- `evaluate()` points segmentation masks and detection annotations to the right functions instead of
|
|
31
|
+
failing with a shape error.
|
|
32
|
+
- CI and release workflows use the Node 24 versions of the GitHub actions.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
- Calibration slope and intercept no longer emit an overflow warning before reporting a perfectly
|
|
36
|
+
separated outcome.
|
|
37
|
+
- The CLI no longer fails on consoles that cannot print Unicode (Windows code pages).
|
|
38
|
+
|
|
9
39
|
## [0.2.1] - 2026-10-09
|
|
10
40
|
|
|
11
41
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -11,7 +11,7 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
|
11
11
|
Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
|
-
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
|
|
14
|
+
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,mAP,machine learning,metrics,object detection,regression,reproducibility,segmentation,statistical tests,statistics
|
|
15
15
|
Classifier: Development Status :: 5 - Production/Stable
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: Intended Audience :: Healthcare Industry
|
|
@@ -27,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
27
27
|
Classifier: Programming Language :: Python :: 3.14
|
|
28
28
|
Classifier: Topic :: Scientific/Engineering
|
|
29
29
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
30
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
30
31
|
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
31
32
|
Classifier: Typing :: Typed
|
|
32
33
|
Requires-Python: >=3.9
|
|
@@ -35,13 +36,16 @@ Requires-Dist: pandas>=1.4
|
|
|
35
36
|
Requires-Dist: scipy>=1.8
|
|
36
37
|
Provides-Extra: all
|
|
37
38
|
Requires-Dist: matplotlib>=3.5; extra == 'all'
|
|
39
|
+
Requires-Dist: pillow>=9; extra == 'all'
|
|
38
40
|
Provides-Extra: dev
|
|
39
41
|
Requires-Dist: build; extra == 'dev'
|
|
40
42
|
Requires-Dist: hypothesis>=6.80; extra == 'dev'
|
|
41
43
|
Requires-Dist: matplotlib>=3.5; extra == 'dev'
|
|
42
44
|
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
43
45
|
Requires-Dist: pandas-stubs; extra == 'dev'
|
|
46
|
+
Requires-Dist: pillow>=9; extra == 'dev'
|
|
44
47
|
Requires-Dist: pip-audit; extra == 'dev'
|
|
48
|
+
Requires-Dist: pycocotools>=2.0.7; extra == 'dev'
|
|
45
49
|
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
46
50
|
Requires-Dist: pytest>=7; extra == 'dev'
|
|
47
51
|
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
@@ -50,6 +54,8 @@ Requires-Dist: statsmodels>=0.13; extra == 'dev'
|
|
|
50
54
|
Requires-Dist: twine; extra == 'dev'
|
|
51
55
|
Provides-Extra: plot
|
|
52
56
|
Requires-Dist: matplotlib>=3.5; extra == 'plot'
|
|
57
|
+
Provides-Extra: vision
|
|
58
|
+
Requires-Dist: pillow>=9; extra == 'vision'
|
|
53
59
|
Description-Content-Type: text/markdown
|
|
54
60
|
|
|
55
61
|
# EvalSuite
|
|
@@ -61,10 +67,10 @@ Description-Content-Type: text/markdown
|
|
|
61
67
|
|
|
62
68
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
63
69
|
|
|
64
|
-
EvalSuite brings classification, regression, clinical
|
|
65
|
-
|
|
70
|
+
EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
|
|
71
|
+
evaluation into one consistent, validated, documented framework.
|
|
66
72
|
|
|
67
|
-
> **Status: stable (0.
|
|
73
|
+
> **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
|
|
68
74
|
|
|
69
75
|
## Installation
|
|
70
76
|
|
|
@@ -181,6 +187,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
|
181
187
|
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
182
188
|
checked against SciPy and statsmodels in the test suite.
|
|
183
189
|
|
|
190
|
+
## Segmentation
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
# label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
|
|
194
|
+
es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
|
|
195
|
+
es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
|
|
196
|
+
es.miou(y_true, y_pred, ignore_index=255)
|
|
197
|
+
es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
|
|
198
|
+
es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
|
|
199
|
+
es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
|
|
200
|
+
|
|
201
|
+
report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
|
|
202
|
+
print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
|
|
203
|
+
es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
|
|
204
|
+
es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
## Object detection
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
|
|
211
|
+
y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
|
|
212
|
+
|
|
213
|
+
report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
|
|
214
|
+
report["map"], report["map_50"], report["mar_100"]
|
|
215
|
+
es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
|
|
216
|
+
es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
|
|
217
|
+
es.box_iou(boxes_a, boxes_b, box_format="xywh")
|
|
218
|
+
y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
|
|
219
|
+
es.plot.detection_pr(y_true, y_pred)
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
|
|
223
|
+
`pycocotools` to the last digit in the test suite.
|
|
224
|
+
|
|
225
|
+
Models can be compared over the same images with intervals and paired tests, as for every other task:
|
|
226
|
+
`es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
|
|
227
|
+
mAP.
|
|
228
|
+
|
|
184
229
|
## Classification report
|
|
185
230
|
|
|
186
231
|
```python
|
|
@@ -225,6 +270,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
|
225
270
|
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
226
271
|
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
227
272
|
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
273
|
+
evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
|
|
274
|
+
evalsuite detection instances_val.json detections.json --plot pr_curves.png
|
|
228
275
|
evalsuite metrics --category clinical
|
|
229
276
|
evalsuite info classification.mcc
|
|
230
277
|
evalsuite benchmark --quick
|
|
@@ -251,6 +298,9 @@ Linux x86_64). Every result agrees with the reference to floating-point rounding
|
|
|
251
298
|
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
252
299
|
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
253
300
|
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
301
|
+
| segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
|
|
302
|
+
| COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
|
|
303
|
+
| Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
|
|
254
304
|
|
|
255
305
|
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
256
306
|
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
@@ -279,6 +329,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
|
|
|
279
329
|
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
280
330
|
Benjamini–Yekutieli).
|
|
281
331
|
|
|
332
|
+
**Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
|
|
333
|
+
mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
|
|
334
|
+
surface distance (with pixel spacing), confusion matrix and a full report.
|
|
335
|
+
|
|
336
|
+
**Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
|
|
337
|
+
object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
|
|
338
|
+
|
|
282
339
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
283
340
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
284
341
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -7,10 +7,10 @@
|
|
|
7
7
|
|
|
8
8
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
9
9
|
|
|
10
|
-
EvalSuite brings classification, regression, clinical
|
|
11
|
-
|
|
10
|
+
EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
|
|
11
|
+
evaluation into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status: stable (0.
|
|
13
|
+
> **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -127,6 +127,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
|
127
127
|
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
128
128
|
checked against SciPy and statsmodels in the test suite.
|
|
129
129
|
|
|
130
|
+
## Segmentation
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
# label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
|
|
134
|
+
es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
|
|
135
|
+
es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
|
|
136
|
+
es.miou(y_true, y_pred, ignore_index=255)
|
|
137
|
+
es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
|
|
138
|
+
es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
|
|
139
|
+
es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
|
|
140
|
+
|
|
141
|
+
report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
|
|
142
|
+
print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
|
|
143
|
+
es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
|
|
144
|
+
es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
## Object detection
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
|
|
151
|
+
y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
|
|
152
|
+
|
|
153
|
+
report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
|
|
154
|
+
report["map"], report["map_50"], report["mar_100"]
|
|
155
|
+
es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
|
|
156
|
+
es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
|
|
157
|
+
es.box_iou(boxes_a, boxes_b, box_format="xywh")
|
|
158
|
+
y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
|
|
159
|
+
es.plot.detection_pr(y_true, y_pred)
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
|
|
163
|
+
`pycocotools` to the last digit in the test suite.
|
|
164
|
+
|
|
165
|
+
Models can be compared over the same images with intervals and paired tests, as for every other task:
|
|
166
|
+
`es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
|
|
167
|
+
mAP.
|
|
168
|
+
|
|
130
169
|
## Classification report
|
|
131
170
|
|
|
132
171
|
```python
|
|
@@ -171,6 +210,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
|
171
210
|
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
172
211
|
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
173
212
|
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
213
|
+
evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
|
|
214
|
+
evalsuite detection instances_val.json detections.json --plot pr_curves.png
|
|
174
215
|
evalsuite metrics --category clinical
|
|
175
216
|
evalsuite info classification.mcc
|
|
176
217
|
evalsuite benchmark --quick
|
|
@@ -197,6 +238,9 @@ Linux x86_64). Every result agrees with the reference to floating-point rounding
|
|
|
197
238
|
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
198
239
|
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
199
240
|
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
241
|
+
| segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
|
|
242
|
+
| COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
|
|
243
|
+
| Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
|
|
200
244
|
|
|
201
245
|
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
202
246
|
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
@@ -225,6 +269,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
|
|
|
225
269
|
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
226
270
|
Benjamini–Yekutieli).
|
|
227
271
|
|
|
272
|
+
**Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
|
|
273
|
+
mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
|
|
274
|
+
surface distance (with pixel spacing), confusion matrix and a full report.
|
|
275
|
+
|
|
276
|
+
**Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
|
|
277
|
+
object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
|
|
278
|
+
|
|
228
279
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
229
280
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
230
281
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -12,12 +12,13 @@ license-files = ["LICENSE"]
|
|
|
12
12
|
requires-python = ">=3.9"
|
|
13
13
|
authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
15
|
-
keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
15
|
+
keywords = ["evaluation", "metrics", "segmentation", "object detection", "mAP", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
19
|
"Intended Audience :: Healthcare Industry",
|
|
20
20
|
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
21
22
|
"Intended Audience :: Developers",
|
|
22
23
|
"Operating System :: OS Independent",
|
|
23
24
|
"Programming Language :: Python :: 3",
|
|
@@ -36,7 +37,8 @@ dependencies = ["numpy>=1.22", "scipy>=1.8", "pandas>=1.4"]
|
|
|
36
37
|
|
|
37
38
|
[project.optional-dependencies]
|
|
38
39
|
plot = ["matplotlib>=3.5"]
|
|
39
|
-
|
|
40
|
+
vision = ["pillow>=9"]
|
|
41
|
+
all = ["matplotlib>=3.5", "pillow>=9"]
|
|
40
42
|
dev = [
|
|
41
43
|
"pytest>=7",
|
|
42
44
|
"pytest-cov>=4",
|
|
@@ -44,6 +46,8 @@ dev = [
|
|
|
44
46
|
"scikit-learn>=1.2",
|
|
45
47
|
"statsmodels>=0.13",
|
|
46
48
|
"matplotlib>=3.5",
|
|
49
|
+
"pillow>=9",
|
|
50
|
+
"pycocotools>=2.0.7",
|
|
47
51
|
"ruff>=0.6",
|
|
48
52
|
"mypy>=1.10",
|
|
49
53
|
"pandas-stubs",
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import calibration, classification, clinical, plot, regression, stats
|
|
8
|
+
from . import calibration, classification, clinical, plot, regression, stats, vision
|
|
9
9
|
from .api import evaluate
|
|
10
10
|
from .calibration import (
|
|
11
11
|
CalibrationReport,
|
|
@@ -111,8 +111,49 @@ from .stats import (
|
|
|
111
111
|
wilcoxon_test,
|
|
112
112
|
)
|
|
113
113
|
from .version import __version__
|
|
114
|
+
from .vision import (
|
|
115
|
+
DetectionReport,
|
|
116
|
+
SegmentationReport,
|
|
117
|
+
average_precision_detection,
|
|
118
|
+
average_surface_distance,
|
|
119
|
+
boundary_iou,
|
|
120
|
+
box_iou,
|
|
121
|
+
detection_pr_curve,
|
|
122
|
+
detection_report,
|
|
123
|
+
dice,
|
|
124
|
+
from_coco,
|
|
125
|
+
hausdorff_distance,
|
|
126
|
+
iou,
|
|
127
|
+
mean_average_precision,
|
|
128
|
+
mean_pixel_accuracy,
|
|
129
|
+
miou,
|
|
130
|
+
per_image_scores,
|
|
131
|
+
pixel_accuracy,
|
|
132
|
+
segmentation_confusion,
|
|
133
|
+
segmentation_report,
|
|
134
|
+
)
|
|
114
135
|
|
|
115
136
|
__all__ = [
|
|
137
|
+
"SegmentationReport",
|
|
138
|
+
"segmentation_report",
|
|
139
|
+
"detection_pr_curve",
|
|
140
|
+
"vision",
|
|
141
|
+
"DetectionReport",
|
|
142
|
+
"average_precision_detection",
|
|
143
|
+
"average_surface_distance",
|
|
144
|
+
"boundary_iou",
|
|
145
|
+
"box_iou",
|
|
146
|
+
"detection_report",
|
|
147
|
+
"dice",
|
|
148
|
+
"from_coco",
|
|
149
|
+
"hausdorff_distance",
|
|
150
|
+
"iou",
|
|
151
|
+
"mean_average_precision",
|
|
152
|
+
"mean_pixel_accuracy",
|
|
153
|
+
"miou",
|
|
154
|
+
"per_image_scores",
|
|
155
|
+
"pixel_accuracy",
|
|
156
|
+
"segmentation_confusion",
|
|
116
157
|
"CalibrationReport",
|
|
117
158
|
"calibration_report",
|
|
118
159
|
"calibration",
|
|
@@ -149,6 +149,16 @@ def evaluate(
|
|
|
149
149
|
>>> round(r["accuracy"], 2)
|
|
150
150
|
0.75
|
|
151
151
|
"""
|
|
152
|
+
if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
|
|
153
|
+
raise UnsupportedTaskError(
|
|
154
|
+
"y_true looks like object detection annotations (one dict per image); use "
|
|
155
|
+
"evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
|
|
156
|
+
)
|
|
157
|
+
if np.ndim(y_true) >= 3:
|
|
158
|
+
raise UnsupportedTaskError(
|
|
159
|
+
"y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
|
|
160
|
+
"evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
|
|
161
|
+
)
|
|
152
162
|
task = task or _infer_task(y_true, y_prob)
|
|
153
163
|
if task == "classification":
|
|
154
164
|
return _evaluate_classification(
|
|
@@ -6,7 +6,8 @@ quantities on the same data, and the largest absolute difference between their r
|
|
|
6
6
|
speed is never shown for numbers that disagree.
|
|
7
7
|
|
|
8
8
|
References: scikit-learn for classification and regression (``suite="core"``); scikit-learn, statsmodels
|
|
9
|
-
and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``)
|
|
9
|
+
and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``);
|
|
10
|
+
scikit-learn, SciPy and pycocotools for segmentation and detection (``suite="vision"``). A case
|
|
10
11
|
whose reference library is not installed is timed for EvalSuite only.
|
|
11
12
|
|
|
12
13
|
>>> from evalsuite.benchmarks import run_benchmarks
|
|
@@ -120,6 +121,145 @@ def _core_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
|
120
121
|
return [(name, ref, es_fn, sk_fn) for (name, ref, es_fn, _), sk_fn in zip(cases, sk)]
|
|
121
122
|
|
|
122
123
|
|
|
124
|
+
def _vision_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
125
|
+
"""v0.3.0: segmentation overlap and surface distance (n = pixels) and COCO detection (n / 1000 images)."""
|
|
126
|
+
import evalsuite as es
|
|
127
|
+
|
|
128
|
+
side = 64
|
|
129
|
+
n_img = max(1, n // (side * side))
|
|
130
|
+
k = 5
|
|
131
|
+
yy, xx = np.ogrid[:side, :side]
|
|
132
|
+
true = np.zeros((n_img, side, side), dtype=np.int64)
|
|
133
|
+
for i in range(n_img):
|
|
134
|
+
for c in range(1, k):
|
|
135
|
+
cy, cx, r = rng.integers(8, side - 8), rng.integers(8, side - 8), rng.integers(4, 14)
|
|
136
|
+
true[i][(yy - cy) ** 2 + (xx - cx) ** 2 <= r * r] = c
|
|
137
|
+
pred = np.roll(true, 1, axis=2)
|
|
138
|
+
noise = rng.random(pred.shape) < 0.02
|
|
139
|
+
pred[noise] = rng.integers(0, k, int(noise.sum()))
|
|
140
|
+
labels = list(range(k))
|
|
141
|
+
|
|
142
|
+
def es_overlap() -> list[float]:
|
|
143
|
+
d = np.asarray(es.dice(true, pred, average=None).value)
|
|
144
|
+
j = np.asarray(es.iou(true, pred, average=None).value)
|
|
145
|
+
return [*d, *j]
|
|
146
|
+
|
|
147
|
+
n_hd = min(n_img, 50)
|
|
148
|
+
|
|
149
|
+
def es_hd() -> list[float]:
|
|
150
|
+
return [float(es.hausdorff_distance(true[i], pred[i], labels=[k - 1])) for i in range(n_hd)]
|
|
151
|
+
|
|
152
|
+
n_det = max(10, n // 1000)
|
|
153
|
+
y_true, y_pred = [], []
|
|
154
|
+
for _ in range(n_det):
|
|
155
|
+
m = int(rng.integers(1, 8))
|
|
156
|
+
xy = rng.uniform(0, 500, (m, 2))
|
|
157
|
+
wh = rng.uniform(8, 160, (m, 2))
|
|
158
|
+
boxes = np.column_stack([xy, xy + wh])
|
|
159
|
+
lab = rng.integers(1, 6, m)
|
|
160
|
+
y_true.append({"boxes": boxes, "labels": lab, "iscrowd": np.zeros(m, int)})
|
|
161
|
+
jitter = boxes + rng.normal(0, 4, boxes.shape)
|
|
162
|
+
jitter[:, 2:] = np.maximum(jitter[:, 2:], jitter[:, :2] + 1)
|
|
163
|
+
extra = rng.uniform(0, 500, (3, 2))
|
|
164
|
+
fp = np.column_stack([extra, extra + rng.uniform(10, 80, (3, 2))])
|
|
165
|
+
y_pred.append(
|
|
166
|
+
{
|
|
167
|
+
"boxes": np.vstack([jitter, fp]),
|
|
168
|
+
"labels": np.r_[lab, rng.integers(1, 6, 3)],
|
|
169
|
+
"scores": rng.random(m + 3),
|
|
170
|
+
}
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def es_map() -> list[float]:
|
|
174
|
+
r = es.detection_report(y_true, y_pred)
|
|
175
|
+
return [r["map"], r["map_50"], r["map_75"], r["mar_100"]]
|
|
176
|
+
|
|
177
|
+
cases: list[Case] = [
|
|
178
|
+
("segmentation: Dice and IoU per class (n = pixels)", "scikit-learn", es_overlap, None),
|
|
179
|
+
(f"segmentation: Hausdorff distance ({n_hd} image{'s' if n_hd != 1 else ''})", "SciPy", es_hd, None),
|
|
180
|
+
(f"detection: COCO evaluation ({n_det} images)", "pycocotools", es_map, None),
|
|
181
|
+
]
|
|
182
|
+
refs: dict[str, Callable[[], Any]] = {}
|
|
183
|
+
try:
|
|
184
|
+
import sklearn.metrics as skm
|
|
185
|
+
|
|
186
|
+
def sk_overlap() -> list[float]:
|
|
187
|
+
ft, fp_ = true.ravel(), pred.ravel()
|
|
188
|
+
return [
|
|
189
|
+
*skm.f1_score(ft, fp_, labels=labels, average=None),
|
|
190
|
+
*skm.jaccard_score(ft, fp_, labels=labels, average=None),
|
|
191
|
+
]
|
|
192
|
+
|
|
193
|
+
refs[cases[0][0]] = sk_overlap
|
|
194
|
+
except ImportError:
|
|
195
|
+
pass
|
|
196
|
+
from scipy import ndimage
|
|
197
|
+
from scipy.spatial.distance import directed_hausdorff
|
|
198
|
+
|
|
199
|
+
def scipy_hd() -> list[float]:
|
|
200
|
+
out = []
|
|
201
|
+
for i in range(n_hd):
|
|
202
|
+
pts = []
|
|
203
|
+
for m_ in (true[i] == k - 1, pred[i] == k - 1):
|
|
204
|
+
er = ndimage.binary_erosion(m_, structure=ndimage.generate_binary_structure(2, 1), border_value=0)
|
|
205
|
+
pts.append(np.argwhere(m_ & ~er).astype(float))
|
|
206
|
+
out.append(max(directed_hausdorff(pts[0], pts[1])[0], directed_hausdorff(pts[1], pts[0])[0]))
|
|
207
|
+
return out
|
|
208
|
+
|
|
209
|
+
refs[cases[1][0]] = scipy_hd
|
|
210
|
+
try:
|
|
211
|
+
import contextlib as _ctx
|
|
212
|
+
import io
|
|
213
|
+
|
|
214
|
+
from pycocotools.coco import COCO # type: ignore[import-untyped]
|
|
215
|
+
from pycocotools.cocoeval import COCOeval # type: ignore[import-untyped]
|
|
216
|
+
|
|
217
|
+
def coco_map() -> list[float]:
|
|
218
|
+
images, anns, dets, aid = [], [], [], 1
|
|
219
|
+
for i, (t, p) in enumerate(zip(y_true, y_pred)):
|
|
220
|
+
images.append({"id": i + 1})
|
|
221
|
+
for b, c in zip(t["boxes"], t["labels"]):
|
|
222
|
+
w, h = b[2] - b[0], b[3] - b[1]
|
|
223
|
+
anns.append(
|
|
224
|
+
{
|
|
225
|
+
"id": aid,
|
|
226
|
+
"image_id": i + 1,
|
|
227
|
+
"category_id": int(c),
|
|
228
|
+
"bbox": [b[0], b[1], w, h],
|
|
229
|
+
"area": w * h,
|
|
230
|
+
"iscrowd": 0,
|
|
231
|
+
}
|
|
232
|
+
)
|
|
233
|
+
aid += 1
|
|
234
|
+
for b, c, sc in zip(p["boxes"], p["labels"], p["scores"]):
|
|
235
|
+
dets.append(
|
|
236
|
+
{
|
|
237
|
+
"image_id": i + 1,
|
|
238
|
+
"category_id": int(c),
|
|
239
|
+
"bbox": [b[0], b[1], b[2] - b[0], b[3] - b[1]],
|
|
240
|
+
"score": float(sc),
|
|
241
|
+
}
|
|
242
|
+
)
|
|
243
|
+
with _ctx.redirect_stdout(io.StringIO()):
|
|
244
|
+
gt = COCO()
|
|
245
|
+
gt.dataset = {
|
|
246
|
+
"images": images,
|
|
247
|
+
"annotations": anns,
|
|
248
|
+
"categories": [{"id": c} for c in range(1, 6)],
|
|
249
|
+
}
|
|
250
|
+
gt.createIndex()
|
|
251
|
+
ev = COCOeval(gt, gt.loadRes(dets), "bbox")
|
|
252
|
+
ev.evaluate()
|
|
253
|
+
ev.accumulate()
|
|
254
|
+
ev.summarize()
|
|
255
|
+
return [ev.stats[0], ev.stats[1], ev.stats[2], ev.stats[8]]
|
|
256
|
+
|
|
257
|
+
refs[cases[2][0]] = coco_map
|
|
258
|
+
except ImportError:
|
|
259
|
+
pass
|
|
260
|
+
return [(name, ref, es_fn, refs.get(name)) for name, ref, es_fn, _ in cases]
|
|
261
|
+
|
|
262
|
+
|
|
123
263
|
def _clinical_cases(n: int, rng: np.random.Generator) -> list[Case]:
|
|
124
264
|
"""v0.2.0: diagnostic accuracy, calibration, decision curves and statistical tests."""
|
|
125
265
|
import evalsuite as es
|
|
@@ -286,6 +426,7 @@ class BenchmarkResult:
|
|
|
286
426
|
+ (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
|
|
287
427
|
+ (f" | statsmodels {env['statsmodels']}" if env.get("statsmodels") else "")
|
|
288
428
|
+ (f" | SciPy {env['scipy']}" if env.get("scipy") else "")
|
|
429
|
+
+ (f" | pycocotools {env['pycocotools']}" if env.get("pycocotools") else "")
|
|
289
430
|
+ f" | {env['machine']} | fastest of {env['repeat']} runs"
|
|
290
431
|
)
|
|
291
432
|
note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares EvalSuite with the reference."
|
|
@@ -354,7 +495,8 @@ def run_benchmarks(
|
|
|
354
495
|
"""Time and memory for evaluation workloads at each size, against a reference implementation.
|
|
355
496
|
|
|
356
497
|
``suite``: ``"core"`` (classification and regression vs scikit-learn), ``"clinical"`` (v0.2.0 clinical,
|
|
357
|
-
calibration and statistics vs scikit-learn, statsmodels, SciPy)
|
|
498
|
+
calibration and statistics vs scikit-learn, statsmodels, SciPy), ``"vision"`` (segmentation and COCO
|
|
499
|
+
detection vs scikit-learn, SciPy, pycocotools) or ``"all"`` (default).
|
|
358
500
|
``compare_sklearn=False`` times EvalSuite alone. Rows keep ``sklearn_ms``/``sklearn_peak_mb`` for rows
|
|
359
501
|
whose reference is scikit-learn, for compatibility with 0.1.x.
|
|
360
502
|
"""
|
|
@@ -364,16 +506,24 @@ def run_benchmarks(
|
|
|
364
506
|
|
|
365
507
|
if repeat < 1:
|
|
366
508
|
raise ValueError("repeat must be at least 1.")
|
|
367
|
-
if suite not in ("all", "core", "clinical"):
|
|
368
|
-
raise ValueError("suite must be 'all', 'core' or '
|
|
509
|
+
if suite not in ("all", "core", "clinical", "vision"):
|
|
510
|
+
raise ValueError("suite must be 'all', 'core', 'clinical' or 'vision'.")
|
|
369
511
|
rng = np.random.default_rng(random_state)
|
|
370
512
|
rows: list[dict[str, Any]] = []
|
|
371
|
-
versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None}
|
|
513
|
+
versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None, "pycocotools": None}
|
|
372
514
|
if compare_sklearn:
|
|
373
515
|
for mod in versions:
|
|
374
516
|
with contextlib.suppress(ImportError):
|
|
375
|
-
|
|
376
|
-
|
|
517
|
+
__import__(mod)
|
|
518
|
+
from importlib.metadata import version as _dist_version
|
|
519
|
+
|
|
520
|
+
versions[mod] = _dist_version("scikit-learn" if mod == "sklearn" else mod)
|
|
521
|
+
builders = {
|
|
522
|
+
"core": [_core_cases],
|
|
523
|
+
"clinical": [_clinical_cases],
|
|
524
|
+
"vision": [_vision_cases],
|
|
525
|
+
"all": [_core_cases, _clinical_cases, _vision_cases],
|
|
526
|
+
}[suite]
|
|
377
527
|
for n in sizes:
|
|
378
528
|
for build in builders:
|
|
379
529
|
for name, ref_name, es_fn, ref_fn in build(int(n), rng):
|
|
@@ -410,6 +560,7 @@ def run_benchmarks(
|
|
|
410
560
|
"scipy": scipy.__version__ if suite != "core" else None,
|
|
411
561
|
"sklearn": versions["sklearn"],
|
|
412
562
|
"statsmodels": versions["statsmodels"] if suite != "core" else None,
|
|
563
|
+
"pycocotools": versions["pycocotools"] if suite in ("all", "vision") else None,
|
|
413
564
|
"machine": f"{platform.system()} {platform.machine()}",
|
|
414
565
|
"repeat": repeat,
|
|
415
566
|
"suite": suite,
|
|
@@ -128,7 +128,7 @@ def logistic_fit(
|
|
|
128
128
|
beta = np.zeros(design.shape[1])
|
|
129
129
|
for _ in range(max_iter):
|
|
130
130
|
eta = design @ beta + off
|
|
131
|
-
mu = 1 / (1 + np.exp(-eta))
|
|
131
|
+
mu = 1 / (1 + np.exp(-np.clip(eta, -700, 700)))
|
|
132
132
|
grad = design.T @ (w * (y - mu))
|
|
133
133
|
hess = (design * (w * mu * (1 - mu))[:, None]).T @ design
|
|
134
134
|
try:
|