evalsuite-python 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/CHANGELOG.md +30 -0
  2. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/PKG-INFO +62 -5
  3. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/README.md +54 -3
  4. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/pyproject.toml +6 -2
  5. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/__init__.py +42 -1
  6. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/api.py +10 -0
  7. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/benchmarks.py +158 -7
  8. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/calibration.py +1 -1
  9. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/cli/main.py +99 -2
  10. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/plot.py +165 -1
  11. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/_resolve.py +26 -3
  12. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/compare.py +6 -4
  13. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/version.py +1 -1
  14. evalsuite_python-0.3.0/src/evalsuite/vision/__init__.py +47 -0
  15. evalsuite_python-0.3.0/src/evalsuite/vision/detection.py +648 -0
  16. evalsuite_python-0.3.0/src/evalsuite/vision/segmentation.py +965 -0
  17. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/test_v020.py +9 -0
  18. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_benchmarks.py +2 -2
  19. evalsuite_python-0.3.0/tests/vision/__init__.py +0 -0
  20. evalsuite_python-0.3.0/tests/vision/test_detection.py +200 -0
  21. evalsuite_python-0.3.0/tests/vision/test_edges.py +126 -0
  22. evalsuite_python-0.3.0/tests/vision/test_outputs.py +170 -0
  23. evalsuite_python-0.3.0/tests/vision/test_segmentation.py +156 -0
  24. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/.gitignore +0 -0
  25. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/CONTRIBUTING.md +0 -0
  26. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/LICENSE +0 -0
  27. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/__main__.py +0 -0
  28. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/__init__.py +0 -0
  29. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/_common.py +0 -0
  30. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/classification/metrics.py +0 -0
  31. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/cli/__init__.py +0 -0
  32. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/__init__.py +0 -0
  33. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/metrics.py +0 -0
  34. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/clinical/report.py +0 -0
  35. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/__init__.py +0 -0
  36. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/context.py +0 -0
  37. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/exceptions.py +0 -0
  38. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/export.py +0 -0
  39. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/registry.py +0 -0
  40. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/result.py +0 -0
  41. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/types.py +0 -0
  42. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/core/validation.py +0 -0
  43. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/py.typed +0 -0
  44. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/regression/__init__.py +0 -0
  45. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/regression/metrics.py +0 -0
  46. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/reporting.py +0 -0
  47. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/__init__.py +0 -0
  48. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/effect.py +0 -0
  49. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/hypothesis.py +0 -0
  50. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/intervals.py +0 -0
  51. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/paired.py +0 -0
  52. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/src/evalsuite/stats/results.py +0 -0
  53. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/__init__.py +0 -0
  54. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/classification/__init__.py +0 -0
  55. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/classification/test_against_sklearn.py +0 -0
  56. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/__init__.py +0 -0
  57. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/clinical/test_outputs.py +0 -0
  58. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/conftest.py +0 -0
  59. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/integration/__init__.py +0 -0
  60. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/__init__.py +0 -0
  61. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_cli.py +0 -0
  62. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_plot.py +0 -0
  63. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/output/test_reporting.py +0 -0
  64. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/regression/__init__.py +0 -0
  65. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/regression/test_against_sklearn.py +0 -0
  66. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/__init__.py +0 -0
  67. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_branches.py +0 -0
  68. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_compare.py +0 -0
  69. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/stats/test_reference.py +0 -0
  70. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/__init__.py +0 -0
  71. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/test_core.py +0 -0
  72. {evalsuite_python-0.2.1 → evalsuite_python-0.3.0}/tests/unit/test_edges.py +0 -0
@@ -6,6 +6,36 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.3.0] - 2026-10-09
10
+
11
+ Computer vision (the v0.3.0 roadmap), with comparison, plots, reporting and benchmarks extended to it.
12
+
13
+ ### Added
14
+ - Segmentation: `dice`, `iou`, `miou`, `pixel_accuracy`, `mean_pixel_accuracy`, `boundary_iou`,
15
+ `hausdorff_distance` (HD and HD95, anisotropic `spacing`), `average_surface_distance`,
16
+ `segmentation_confusion`, `per_image_scores` and `segmentation_report`. 2-D and 3-D masks, lists of
17
+ differently sized masks, `ignore_index`, `aggregate="dataset"|"image"`, undefined classes reported as NaN.
18
+ - Object detection: `box_iou`, `detection_report` (the 12 COCO numbers plus AP per class),
19
+ `mean_average_precision`, `average_precision_detection` (COCO or VOC interpolation), `detection_pr_curve`,
20
+ `from_coco`. Matches pycocotools exactly, including crowd regions, area ranges and detection limits.
21
+ - Comparison for vision: `compare`, `bootstrap_ci` and `paired_bootstrap_test` resample images for
22
+ segmentation masks and per-image detections.
23
+ - Plots: `es.plot.segmentation` (prediction fill vs truth outline), `es.plot.per_class` (any per-class
24
+ result, a segmentation report or a detection report), `es.plot.detection_pr`.
25
+ - CLI: `evalsuite segmentation` (.npy/.npz or a folder of mask images) and `evalsuite detection` (COCO JSON).
26
+ - Benchmarks: `--suite vision` against scikit-learn, SciPy and pycocotools.
27
+ - `vision` extra (Pillow, for reading mask image folders); `CITATION.cff`.
28
+
29
+ ### Changed
30
+ - `evaluate()` points segmentation masks and detection annotations to the right functions instead of
31
+ failing with a shape error.
32
+ - CI and release workflows use the Node 24 versions of the GitHub actions.
33
+
34
+ ### Fixed
35
+ - Calibration slope and intercept no longer emit an overflow warning before reporting a perfectly
36
+ separated outcome.
37
+ - The CLI no longer fails on consoles that cannot print Unicode (Windows code pages).
38
+
9
39
  ## [0.2.1] - 2026-10-09
10
40
 
11
41
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -11,7 +11,7 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
11
11
  Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
- Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
14
+ Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,mAP,machine learning,metrics,object detection,regression,reproducibility,segmentation,statistical tests,statistics
15
15
  Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: Intended Audience :: Healthcare Industry
@@ -27,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
27
27
  Classifier: Programming Language :: Python :: 3.14
28
28
  Classifier: Topic :: Scientific/Engineering
29
29
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
30
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
30
31
  Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
31
32
  Classifier: Typing :: Typed
32
33
  Requires-Python: >=3.9
@@ -35,13 +36,16 @@ Requires-Dist: pandas>=1.4
35
36
  Requires-Dist: scipy>=1.8
36
37
  Provides-Extra: all
37
38
  Requires-Dist: matplotlib>=3.5; extra == 'all'
39
+ Requires-Dist: pillow>=9; extra == 'all'
38
40
  Provides-Extra: dev
39
41
  Requires-Dist: build; extra == 'dev'
40
42
  Requires-Dist: hypothesis>=6.80; extra == 'dev'
41
43
  Requires-Dist: matplotlib>=3.5; extra == 'dev'
42
44
  Requires-Dist: mypy>=1.10; extra == 'dev'
43
45
  Requires-Dist: pandas-stubs; extra == 'dev'
46
+ Requires-Dist: pillow>=9; extra == 'dev'
44
47
  Requires-Dist: pip-audit; extra == 'dev'
48
+ Requires-Dist: pycocotools>=2.0.7; extra == 'dev'
45
49
  Requires-Dist: pytest-cov>=4; extra == 'dev'
46
50
  Requires-Dist: pytest>=7; extra == 'dev'
47
51
  Requires-Dist: ruff>=0.6; extra == 'dev'
@@ -50,6 +54,8 @@ Requires-Dist: statsmodels>=0.13; extra == 'dev'
50
54
  Requires-Dist: twine; extra == 'dev'
51
55
  Provides-Extra: plot
52
56
  Requires-Dist: matplotlib>=3.5; extra == 'plot'
57
+ Provides-Extra: vision
58
+ Requires-Dist: pillow>=9; extra == 'vision'
53
59
  Description-Content-Type: text/markdown
54
60
 
55
61
  # EvalSuite
@@ -61,10 +67,10 @@ Description-Content-Type: text/markdown
61
67
 
62
68
  **Unified, reproducible evaluation for machine learning and research.**
63
69
 
64
- EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
65
- object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
70
+ EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
71
+ evaluation into one consistent, validated, documented framework.
66
72
 
67
- > **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
73
+ > **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
68
74
 
69
75
  ## Installation
70
76
 
@@ -181,6 +187,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
181
187
  Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
182
188
  checked against SciPy and statsmodels in the test suite.
183
189
 
190
+ ## Segmentation
191
+
192
+ ```python
193
+ # label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
194
+ es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
195
+ es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
196
+ es.miou(y_true, y_pred, ignore_index=255)
197
+ es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
198
+ es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
199
+ es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
200
+
201
+ report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
202
+ print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
203
+ es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
204
+ es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
205
+ ```
206
+
207
+ ## Object detection
208
+
209
+ ```python
210
+ y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
211
+ y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
212
+
213
+ report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
214
+ report["map"], report["map_50"], report["mar_100"]
215
+ es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
216
+ es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
217
+ es.box_iou(boxes_a, boxes_b, box_format="xywh")
218
+ y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
219
+ es.plot.detection_pr(y_true, y_pred)
220
+ ```
221
+
222
+ The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
223
+ `pycocotools` to the last digit in the test suite.
224
+
225
+ Models can be compared over the same images with intervals and paired tests, as for every other task:
226
+ `es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
227
+ mAP.
228
+
184
229
  ## Classification report
185
230
 
186
231
  ```python
@@ -225,6 +270,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
225
270
  evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
226
271
  evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
227
272
  evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
273
+ evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
274
+ evalsuite detection instances_val.json detections.json --plot pr_curves.png
228
275
  evalsuite metrics --category clinical
229
276
  evalsuite info classification.mcc
230
277
  evalsuite benchmark --quick
@@ -251,6 +298,9 @@ Linux x86_64). Every result agrees with the reference to floating-point rounding
251
298
  | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
252
299
  | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
253
300
  | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
301
+ | segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
302
+ | COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
303
+ | Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
254
304
 
255
305
  `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
256
306
  of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
@@ -279,6 +329,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
279
329
  delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
280
330
  Benjamini–Yekutieli).
281
331
 
332
+ **Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
333
+ mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
334
+ surface distance (with pixel spacing), confusion matrix and a full report.
335
+
336
+ **Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
337
+ object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
338
+
282
339
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
283
340
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
284
341
  loss, Huber loss, relative absolute error, relative squared error.
@@ -7,10 +7,10 @@
7
7
 
8
8
  **Unified, reproducible evaluation for machine learning and research.**
9
9
 
10
- EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
11
- object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
10
+ EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
11
+ evaluation into one consistent, validated, documented framework.
12
12
 
13
- > **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
13
+ > **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -127,6 +127,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
127
127
  Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
128
128
  checked against SciPy and statsmodels in the test suite.
129
129
 
130
+ ## Segmentation
131
+
132
+ ```python
133
+ # label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
134
+ es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
135
+ es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
136
+ es.miou(y_true, y_pred, ignore_index=255)
137
+ es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
138
+ es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
139
+ es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
140
+
141
+ report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
142
+ print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
143
+ es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
144
+ es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
145
+ ```
146
+
147
+ ## Object detection
148
+
149
+ ```python
150
+ y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
151
+ y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
152
+
153
+ report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
154
+ report["map"], report["map_50"], report["mar_100"]
155
+ es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
156
+ es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
157
+ es.box_iou(boxes_a, boxes_b, box_format="xywh")
158
+ y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
159
+ es.plot.detection_pr(y_true, y_pred)
160
+ ```
161
+
162
+ The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
163
+ `pycocotools` to the last digit in the test suite.
164
+
165
+ Models can be compared over the same images with intervals and paired tests, as for every other task:
166
+ `es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
167
+ mAP.
168
+
130
169
  ## Classification report
131
170
 
132
171
  ```python
@@ -171,6 +210,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
171
210
  evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
172
211
  evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
173
212
  evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
213
+ evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
214
+ evalsuite detection instances_val.json detections.json --plot pr_curves.png
174
215
  evalsuite metrics --category clinical
175
216
  evalsuite info classification.mcc
176
217
  evalsuite benchmark --quick
@@ -197,6 +238,9 @@ Linux x86_64). Every result agrees with the reference to floating-point rounding
197
238
  | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
198
239
  | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
199
240
  | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
241
+ | segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
242
+ | COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
243
+ | Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
200
244
 
201
245
  `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
202
246
  of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
@@ -225,6 +269,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
225
269
  delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
226
270
  Benjamini–Yekutieli).
227
271
 
272
+ **Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
273
+ mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
274
+ surface distance (with pixel spacing), confusion matrix and a full report.
275
+
276
+ **Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
277
+ object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
278
+
228
279
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
229
280
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
230
281
  loss, Huber loss, relative absolute error, relative squared error.
@@ -12,12 +12,13 @@ license-files = ["LICENSE"]
12
12
  requires-python = ">=3.9"
13
13
  authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
15
- keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
15
+ keywords = ["evaluation", "metrics", "segmentation", "object detection", "mAP", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
17
  "Development Status :: 5 - Production/Stable",
18
18
  "Intended Audience :: Science/Research",
19
19
  "Intended Audience :: Healthcare Industry",
20
20
  "Topic :: Scientific/Engineering :: Medical Science Apps.",
21
+ "Topic :: Scientific/Engineering :: Image Recognition",
21
22
  "Intended Audience :: Developers",
22
23
  "Operating System :: OS Independent",
23
24
  "Programming Language :: Python :: 3",
@@ -36,7 +37,8 @@ dependencies = ["numpy>=1.22", "scipy>=1.8", "pandas>=1.4"]
36
37
 
37
38
  [project.optional-dependencies]
38
39
  plot = ["matplotlib>=3.5"]
39
- all = ["matplotlib>=3.5"]
40
+ vision = ["pillow>=9"]
41
+ all = ["matplotlib>=3.5", "pillow>=9"]
40
42
  dev = [
41
43
  "pytest>=7",
42
44
  "pytest-cov>=4",
@@ -44,6 +46,8 @@ dev = [
44
46
  "scikit-learn>=1.2",
45
47
  "statsmodels>=0.13",
46
48
  "matplotlib>=3.5",
49
+ "pillow>=9",
50
+ "pycocotools>=2.0.7",
47
51
  "ruff>=0.6",
48
52
  "mypy>=1.10",
49
53
  "pandas-stubs",
@@ -5,7 +5,7 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import calibration, classification, clinical, plot, regression, stats
8
+ from . import calibration, classification, clinical, plot, regression, stats, vision
9
9
  from .api import evaluate
10
10
  from .calibration import (
11
11
  CalibrationReport,
@@ -111,8 +111,49 @@ from .stats import (
111
111
  wilcoxon_test,
112
112
  )
113
113
  from .version import __version__
114
+ from .vision import (
115
+ DetectionReport,
116
+ SegmentationReport,
117
+ average_precision_detection,
118
+ average_surface_distance,
119
+ boundary_iou,
120
+ box_iou,
121
+ detection_pr_curve,
122
+ detection_report,
123
+ dice,
124
+ from_coco,
125
+ hausdorff_distance,
126
+ iou,
127
+ mean_average_precision,
128
+ mean_pixel_accuracy,
129
+ miou,
130
+ per_image_scores,
131
+ pixel_accuracy,
132
+ segmentation_confusion,
133
+ segmentation_report,
134
+ )
114
135
 
115
136
  __all__ = [
137
+ "SegmentationReport",
138
+ "segmentation_report",
139
+ "detection_pr_curve",
140
+ "vision",
141
+ "DetectionReport",
142
+ "average_precision_detection",
143
+ "average_surface_distance",
144
+ "boundary_iou",
145
+ "box_iou",
146
+ "detection_report",
147
+ "dice",
148
+ "from_coco",
149
+ "hausdorff_distance",
150
+ "iou",
151
+ "mean_average_precision",
152
+ "mean_pixel_accuracy",
153
+ "miou",
154
+ "per_image_scores",
155
+ "pixel_accuracy",
156
+ "segmentation_confusion",
116
157
  "CalibrationReport",
117
158
  "calibration_report",
118
159
  "calibration",
@@ -149,6 +149,16 @@ def evaluate(
149
149
  >>> round(r["accuracy"], 2)
150
150
  0.75
151
151
  """
152
+ if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
153
+ raise UnsupportedTaskError(
154
+ "y_true looks like object detection annotations (one dict per image); use "
155
+ "evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
156
+ )
157
+ if np.ndim(y_true) >= 3:
158
+ raise UnsupportedTaskError(
159
+ "y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
160
+ "evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
161
+ )
152
162
  task = task or _infer_task(y_true, y_prob)
153
163
  if task == "classification":
154
164
  return _evaluate_classification(
@@ -6,7 +6,8 @@ quantities on the same data, and the largest absolute difference between their r
6
6
  speed is never shown for numbers that disagree.
7
7
 
8
8
  References: scikit-learn for classification and regression (``suite="core"``); scikit-learn, statsmodels
9
- and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``). A case
9
+ and SciPy for the v0.2.0 clinical, calibration and statistics functions (``suite="clinical"``);
10
+ scikit-learn, SciPy and pycocotools for segmentation and detection (``suite="vision"``). A case
10
11
  whose reference library is not installed is timed for EvalSuite only.
11
12
 
12
13
  >>> from evalsuite.benchmarks import run_benchmarks
@@ -120,6 +121,145 @@ def _core_cases(n: int, rng: np.random.Generator) -> list[Case]:
120
121
  return [(name, ref, es_fn, sk_fn) for (name, ref, es_fn, _), sk_fn in zip(cases, sk)]
121
122
 
122
123
 
124
+ def _vision_cases(n: int, rng: np.random.Generator) -> list[Case]:
125
+ """v0.3.0: segmentation overlap and surface distance (n = pixels) and COCO detection (n / 1000 images)."""
126
+ import evalsuite as es
127
+
128
+ side = 64
129
+ n_img = max(1, n // (side * side))
130
+ k = 5
131
+ yy, xx = np.ogrid[:side, :side]
132
+ true = np.zeros((n_img, side, side), dtype=np.int64)
133
+ for i in range(n_img):
134
+ for c in range(1, k):
135
+ cy, cx, r = rng.integers(8, side - 8), rng.integers(8, side - 8), rng.integers(4, 14)
136
+ true[i][(yy - cy) ** 2 + (xx - cx) ** 2 <= r * r] = c
137
+ pred = np.roll(true, 1, axis=2)
138
+ noise = rng.random(pred.shape) < 0.02
139
+ pred[noise] = rng.integers(0, k, int(noise.sum()))
140
+ labels = list(range(k))
141
+
142
+ def es_overlap() -> list[float]:
143
+ d = np.asarray(es.dice(true, pred, average=None).value)
144
+ j = np.asarray(es.iou(true, pred, average=None).value)
145
+ return [*d, *j]
146
+
147
+ n_hd = min(n_img, 50)
148
+
149
+ def es_hd() -> list[float]:
150
+ return [float(es.hausdorff_distance(true[i], pred[i], labels=[k - 1])) for i in range(n_hd)]
151
+
152
+ n_det = max(10, n // 1000)
153
+ y_true, y_pred = [], []
154
+ for _ in range(n_det):
155
+ m = int(rng.integers(1, 8))
156
+ xy = rng.uniform(0, 500, (m, 2))
157
+ wh = rng.uniform(8, 160, (m, 2))
158
+ boxes = np.column_stack([xy, xy + wh])
159
+ lab = rng.integers(1, 6, m)
160
+ y_true.append({"boxes": boxes, "labels": lab, "iscrowd": np.zeros(m, int)})
161
+ jitter = boxes + rng.normal(0, 4, boxes.shape)
162
+ jitter[:, 2:] = np.maximum(jitter[:, 2:], jitter[:, :2] + 1)
163
+ extra = rng.uniform(0, 500, (3, 2))
164
+ fp = np.column_stack([extra, extra + rng.uniform(10, 80, (3, 2))])
165
+ y_pred.append(
166
+ {
167
+ "boxes": np.vstack([jitter, fp]),
168
+ "labels": np.r_[lab, rng.integers(1, 6, 3)],
169
+ "scores": rng.random(m + 3),
170
+ }
171
+ )
172
+
173
+ def es_map() -> list[float]:
174
+ r = es.detection_report(y_true, y_pred)
175
+ return [r["map"], r["map_50"], r["map_75"], r["mar_100"]]
176
+
177
+ cases: list[Case] = [
178
+ ("segmentation: Dice and IoU per class (n = pixels)", "scikit-learn", es_overlap, None),
179
+ (f"segmentation: Hausdorff distance ({n_hd} image{'s' if n_hd != 1 else ''})", "SciPy", es_hd, None),
180
+ (f"detection: COCO evaluation ({n_det} images)", "pycocotools", es_map, None),
181
+ ]
182
+ refs: dict[str, Callable[[], Any]] = {}
183
+ try:
184
+ import sklearn.metrics as skm
185
+
186
+ def sk_overlap() -> list[float]:
187
+ ft, fp_ = true.ravel(), pred.ravel()
188
+ return [
189
+ *skm.f1_score(ft, fp_, labels=labels, average=None),
190
+ *skm.jaccard_score(ft, fp_, labels=labels, average=None),
191
+ ]
192
+
193
+ refs[cases[0][0]] = sk_overlap
194
+ except ImportError:
195
+ pass
196
+ from scipy import ndimage
197
+ from scipy.spatial.distance import directed_hausdorff
198
+
199
+ def scipy_hd() -> list[float]:
200
+ out = []
201
+ for i in range(n_hd):
202
+ pts = []
203
+ for m_ in (true[i] == k - 1, pred[i] == k - 1):
204
+ er = ndimage.binary_erosion(m_, structure=ndimage.generate_binary_structure(2, 1), border_value=0)
205
+ pts.append(np.argwhere(m_ & ~er).astype(float))
206
+ out.append(max(directed_hausdorff(pts[0], pts[1])[0], directed_hausdorff(pts[1], pts[0])[0]))
207
+ return out
208
+
209
+ refs[cases[1][0]] = scipy_hd
210
+ try:
211
+ import contextlib as _ctx
212
+ import io
213
+
214
+ from pycocotools.coco import COCO # type: ignore[import-untyped]
215
+ from pycocotools.cocoeval import COCOeval # type: ignore[import-untyped]
216
+
217
+ def coco_map() -> list[float]:
218
+ images, anns, dets, aid = [], [], [], 1
219
+ for i, (t, p) in enumerate(zip(y_true, y_pred)):
220
+ images.append({"id": i + 1})
221
+ for b, c in zip(t["boxes"], t["labels"]):
222
+ w, h = b[2] - b[0], b[3] - b[1]
223
+ anns.append(
224
+ {
225
+ "id": aid,
226
+ "image_id": i + 1,
227
+ "category_id": int(c),
228
+ "bbox": [b[0], b[1], w, h],
229
+ "area": w * h,
230
+ "iscrowd": 0,
231
+ }
232
+ )
233
+ aid += 1
234
+ for b, c, sc in zip(p["boxes"], p["labels"], p["scores"]):
235
+ dets.append(
236
+ {
237
+ "image_id": i + 1,
238
+ "category_id": int(c),
239
+ "bbox": [b[0], b[1], b[2] - b[0], b[3] - b[1]],
240
+ "score": float(sc),
241
+ }
242
+ )
243
+ with _ctx.redirect_stdout(io.StringIO()):
244
+ gt = COCO()
245
+ gt.dataset = {
246
+ "images": images,
247
+ "annotations": anns,
248
+ "categories": [{"id": c} for c in range(1, 6)],
249
+ }
250
+ gt.createIndex()
251
+ ev = COCOeval(gt, gt.loadRes(dets), "bbox")
252
+ ev.evaluate()
253
+ ev.accumulate()
254
+ ev.summarize()
255
+ return [ev.stats[0], ev.stats[1], ev.stats[2], ev.stats[8]]
256
+
257
+ refs[cases[2][0]] = coco_map
258
+ except ImportError:
259
+ pass
260
+ return [(name, ref, es_fn, refs.get(name)) for name, ref, es_fn, _ in cases]
261
+
262
+
123
263
  def _clinical_cases(n: int, rng: np.random.Generator) -> list[Case]:
124
264
  """v0.2.0: diagnostic accuracy, calibration, decision curves and statistical tests."""
125
265
  import evalsuite as es
@@ -286,6 +426,7 @@ class BenchmarkResult:
286
426
  + (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
287
427
  + (f" | statsmodels {env['statsmodels']}" if env.get("statsmodels") else "")
288
428
  + (f" | SciPy {env['scipy']}" if env.get("scipy") else "")
429
+ + (f" | pycocotools {env['pycocotools']}" if env.get("pycocotools") else "")
289
430
  + f" | {env['machine']} | fastest of {env['repeat']} runs"
290
431
  )
291
432
  note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares EvalSuite with the reference."
@@ -354,7 +495,8 @@ def run_benchmarks(
354
495
  """Time and memory for evaluation workloads at each size, against a reference implementation.
355
496
 
356
497
  ``suite``: ``"core"`` (classification and regression vs scikit-learn), ``"clinical"`` (v0.2.0 clinical,
357
- calibration and statistics vs scikit-learn, statsmodels, SciPy) or ``"all"`` (default).
498
+ calibration and statistics vs scikit-learn, statsmodels, SciPy), ``"vision"`` (segmentation and COCO
499
+ detection vs scikit-learn, SciPy, pycocotools) or ``"all"`` (default).
358
500
  ``compare_sklearn=False`` times EvalSuite alone. Rows keep ``sklearn_ms``/``sklearn_peak_mb`` for rows
359
501
  whose reference is scikit-learn, for compatibility with 0.1.x.
360
502
  """
@@ -364,16 +506,24 @@ def run_benchmarks(
364
506
 
365
507
  if repeat < 1:
366
508
  raise ValueError("repeat must be at least 1.")
367
- if suite not in ("all", "core", "clinical"):
368
- raise ValueError("suite must be 'all', 'core' or 'clinical'.")
509
+ if suite not in ("all", "core", "clinical", "vision"):
510
+ raise ValueError("suite must be 'all', 'core', 'clinical' or 'vision'.")
369
511
  rng = np.random.default_rng(random_state)
370
512
  rows: list[dict[str, Any]] = []
371
- versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None}
513
+ versions: dict[str, Optional[str]] = {"sklearn": None, "statsmodels": None, "pycocotools": None}
372
514
  if compare_sklearn:
373
515
  for mod in versions:
374
516
  with contextlib.suppress(ImportError):
375
- versions[mod] = __import__(mod).__version__
376
- builders = {"core": [_core_cases], "clinical": [_clinical_cases], "all": [_core_cases, _clinical_cases]}[suite]
517
+ __import__(mod)
518
+ from importlib.metadata import version as _dist_version
519
+
520
+ versions[mod] = _dist_version("scikit-learn" if mod == "sklearn" else mod)
521
+ builders = {
522
+ "core": [_core_cases],
523
+ "clinical": [_clinical_cases],
524
+ "vision": [_vision_cases],
525
+ "all": [_core_cases, _clinical_cases, _vision_cases],
526
+ }[suite]
377
527
  for n in sizes:
378
528
  for build in builders:
379
529
  for name, ref_name, es_fn, ref_fn in build(int(n), rng):
@@ -410,6 +560,7 @@ def run_benchmarks(
410
560
  "scipy": scipy.__version__ if suite != "core" else None,
411
561
  "sklearn": versions["sklearn"],
412
562
  "statsmodels": versions["statsmodels"] if suite != "core" else None,
563
+ "pycocotools": versions["pycocotools"] if suite in ("all", "vision") else None,
413
564
  "machine": f"{platform.system()} {platform.machine()}",
414
565
  "repeat": repeat,
415
566
  "suite": suite,
@@ -128,7 +128,7 @@ def logistic_fit(
128
128
  beta = np.zeros(design.shape[1])
129
129
  for _ in range(max_iter):
130
130
  eta = design @ beta + off
131
- mu = 1 / (1 + np.exp(-eta))
131
+ mu = 1 / (1 + np.exp(-np.clip(eta, -700, 700)))
132
132
  grad = design.T @ (w * (y - mu))
133
133
  hess = (design * (w * mu * (1 - mu))[:, None]).T @ design
134
134
  try: