evalsuite-python 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/CHANGELOG.md +43 -0
  2. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/PKG-INFO +83 -23
  3. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/README.md +75 -21
  4. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/pyproject.toml +6 -2
  5. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/__init__.py +42 -1
  6. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/api.py +10 -0
  7. evalsuite_python-0.3.0/src/evalsuite/benchmarks.py +568 -0
  8. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/calibration.py +1 -1
  9. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/cli/main.py +109 -2
  10. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/metrics.py +14 -5
  11. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/validation.py +14 -2
  12. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/plot.py +165 -1
  13. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/_resolve.py +26 -3
  14. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/compare.py +6 -4
  15. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/effect.py +1 -1
  16. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/version.py +1 -1
  17. evalsuite_python-0.3.0/src/evalsuite/vision/__init__.py +47 -0
  18. evalsuite_python-0.3.0/src/evalsuite/vision/detection.py +648 -0
  19. evalsuite_python-0.3.0/src/evalsuite/vision/segmentation.py +965 -0
  20. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/test_v020.py +9 -0
  21. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_benchmarks.py +19 -3
  22. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/test_edges.py +18 -0
  23. evalsuite_python-0.3.0/tests/vision/__init__.py +0 -0
  24. evalsuite_python-0.3.0/tests/vision/test_detection.py +200 -0
  25. evalsuite_python-0.3.0/tests/vision/test_edges.py +126 -0
  26. evalsuite_python-0.3.0/tests/vision/test_outputs.py +170 -0
  27. evalsuite_python-0.3.0/tests/vision/test_segmentation.py +156 -0
  28. evalsuite_python-0.2.0/src/evalsuite/benchmarks.py +0 -272
  29. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/.gitignore +0 -0
  30. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/CONTRIBUTING.md +0 -0
  31. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/LICENSE +0 -0
  32. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/__main__.py +0 -0
  33. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/__init__.py +0 -0
  34. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/_common.py +0 -0
  35. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/metrics.py +0 -0
  36. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/cli/__init__.py +0 -0
  37. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/__init__.py +0 -0
  38. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/report.py +0 -0
  39. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/__init__.py +0 -0
  40. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/context.py +0 -0
  41. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/exceptions.py +0 -0
  42. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/export.py +0 -0
  43. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/registry.py +0 -0
  44. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/result.py +0 -0
  45. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/types.py +0 -0
  46. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/py.typed +0 -0
  47. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/regression/__init__.py +0 -0
  48. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/regression/metrics.py +0 -0
  49. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/reporting.py +0 -0
  50. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/__init__.py +0 -0
  51. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/hypothesis.py +0 -0
  52. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/intervals.py +0 -0
  53. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/paired.py +0 -0
  54. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/results.py +0 -0
  55. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/__init__.py +0 -0
  56. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/classification/__init__.py +0 -0
  57. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/classification/test_against_sklearn.py +0 -0
  58. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/__init__.py +0 -0
  59. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/test_outputs.py +0 -0
  60. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/conftest.py +0 -0
  61. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/integration/__init__.py +0 -0
  62. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/__init__.py +0 -0
  63. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_cli.py +0 -0
  64. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_plot.py +0 -0
  65. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_reporting.py +0 -0
  66. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/regression/__init__.py +0 -0
  67. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/regression/test_against_sklearn.py +0 -0
  68. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/__init__.py +0 -0
  69. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_branches.py +0 -0
  70. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_compare.py +0 -0
  71. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_reference.py +0 -0
  72. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/__init__.py +0 -0
  73. {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/test_core.py +0 -0
@@ -6,6 +6,49 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.3.0] - 2026-10-09
10
+
11
+ Computer vision (the v0.3.0 roadmap), with comparison, plots, reporting and benchmarks extended to it.
12
+
13
+ ### Added
14
+ - Segmentation: `dice`, `iou`, `miou`, `pixel_accuracy`, `mean_pixel_accuracy`, `boundary_iou`,
15
+ `hausdorff_distance` (HD and HD95, anisotropic `spacing`), `average_surface_distance`,
16
+ `segmentation_confusion`, `per_image_scores` and `segmentation_report`. 2-D and 3-D masks, lists of
17
+ differently sized masks, `ignore_index`, `aggregate="dataset"|"image"`, undefined classes reported as NaN.
18
+ - Object detection: `box_iou`, `detection_report` (the 12 COCO numbers plus AP per class),
19
+ `mean_average_precision`, `average_precision_detection` (COCO or VOC interpolation), `detection_pr_curve`,
20
+ `from_coco`. Matches pycocotools exactly, including crowd regions, area ranges and detection limits.
21
+ - Comparison for vision: `compare`, `bootstrap_ci` and `paired_bootstrap_test` resample images for
22
+ segmentation masks and per-image detections.
23
+ - Plots: `es.plot.segmentation` (prediction fill vs truth outline), `es.plot.per_class` (any per-class
24
+ result, a segmentation report or a detection report), `es.plot.detection_pr`.
25
+ - CLI: `evalsuite segmentation` (.npy/.npz or a folder of mask images) and `evalsuite detection` (COCO JSON).
26
+ - Benchmarks: `--suite vision` against scikit-learn, SciPy and pycocotools.
27
+ - `vision` extra (Pillow, for reading mask image folders); `CITATION.cff`.
28
+
29
+ ### Changed
30
+ - `evaluate()` points segmentation masks and detection annotations to the right functions instead of
31
+ failing with a shape error.
32
+ - CI and release workflows use the Node 24 versions of the GitHub actions.
33
+
34
+ ### Fixed
35
+ - Calibration slope and intercept no longer emit an overflow warning before reporting a perfectly
36
+ separated outcome.
37
+ - The CLI no longer fails on consoles that cannot print Unicode (Windows code pages).
38
+
39
+ ## [0.2.1] - 2026-10-09
40
+
41
+ ### Added
42
+ - Benchmarks for the v0.2.0 functions (`evalsuite benchmark --suite clinical`): diagnostic metrics and
43
+ report, calibration slope and intercept, decision curves, t-test, Mann–Whitney, Cramér's V and Hochberg,
44
+ each against its reference (scikit-learn, statsmodels, SciPy or the textbook NumPy loop). Benchmark rows
45
+ now name their reference library.
46
+
47
+ ### Changed
48
+ - Faster label handling: integer class labels are found with one marking pass instead of a sort
49
+ (8 label metrics at 1M samples: 34× faster than scikit-learn, up from 10×; macro F1 5.8×, up from 1.6×).
50
+ - Decision curves sort the risks once and use cumulative sums (O((n + k) log n), no n × k matrix).
51
+
9
52
  ## [0.2.0] - 2026-10-08
10
53
 
11
54
  Clinical and statistical evaluation (the v0.2.0 roadmap).
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -11,7 +11,7 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
11
11
  Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
- Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
14
+ Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,mAP,machine learning,metrics,object detection,regression,reproducibility,segmentation,statistical tests,statistics
15
15
  Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: Intended Audience :: Healthcare Industry
@@ -27,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
27
27
  Classifier: Programming Language :: Python :: 3.14
28
28
  Classifier: Topic :: Scientific/Engineering
29
29
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
30
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
30
31
  Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
31
32
  Classifier: Typing :: Typed
32
33
  Requires-Python: >=3.9
@@ -35,13 +36,16 @@ Requires-Dist: pandas>=1.4
35
36
  Requires-Dist: scipy>=1.8
36
37
  Provides-Extra: all
37
38
  Requires-Dist: matplotlib>=3.5; extra == 'all'
39
+ Requires-Dist: pillow>=9; extra == 'all'
38
40
  Provides-Extra: dev
39
41
  Requires-Dist: build; extra == 'dev'
40
42
  Requires-Dist: hypothesis>=6.80; extra == 'dev'
41
43
  Requires-Dist: matplotlib>=3.5; extra == 'dev'
42
44
  Requires-Dist: mypy>=1.10; extra == 'dev'
43
45
  Requires-Dist: pandas-stubs; extra == 'dev'
46
+ Requires-Dist: pillow>=9; extra == 'dev'
44
47
  Requires-Dist: pip-audit; extra == 'dev'
48
+ Requires-Dist: pycocotools>=2.0.7; extra == 'dev'
45
49
  Requires-Dist: pytest-cov>=4; extra == 'dev'
46
50
  Requires-Dist: pytest>=7; extra == 'dev'
47
51
  Requires-Dist: ruff>=0.6; extra == 'dev'
@@ -50,6 +54,8 @@ Requires-Dist: statsmodels>=0.13; extra == 'dev'
50
54
  Requires-Dist: twine; extra == 'dev'
51
55
  Provides-Extra: plot
52
56
  Requires-Dist: matplotlib>=3.5; extra == 'plot'
57
+ Provides-Extra: vision
58
+ Requires-Dist: pillow>=9; extra == 'vision'
53
59
  Description-Content-Type: text/markdown
54
60
 
55
61
  # EvalSuite
@@ -61,10 +67,10 @@ Description-Content-Type: text/markdown
61
67
 
62
68
  **Unified, reproducible evaluation for machine learning and research.**
63
69
 
64
- EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
65
- object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
70
+ EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
71
+ evaluation into one consistent, validated, documented framework.
66
72
 
67
- > **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
73
+ > **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
68
74
 
69
75
  ## Installation
70
76
 
@@ -181,6 +187,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
181
187
  Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
182
188
  checked against SciPy and statsmodels in the test suite.
183
189
 
190
+ ## Segmentation
191
+
192
+ ```python
193
+ # label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
194
+ es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
195
+ es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
196
+ es.miou(y_true, y_pred, ignore_index=255)
197
+ es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
198
+ es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
199
+ es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
200
+
201
+ report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
202
+ print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
203
+ es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
204
+ es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
205
+ ```
206
+
207
+ ## Object detection
208
+
209
+ ```python
210
+ y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
211
+ y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
212
+
213
+ report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
214
+ report["map"], report["map_50"], report["mar_100"]
215
+ es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
216
+ es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
217
+ es.box_iou(boxes_a, boxes_b, box_format="xywh")
218
+ y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
219
+ es.plot.detection_pr(y_true, y_pred)
220
+ ```
221
+
222
+ The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
223
+ `pycocotools` to the last digit in the test suite.
224
+
225
+ Models can be compared over the same images with intervals and paired tests, as for every other task:
226
+ `es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
227
+ mAP.
228
+
184
229
  ## Classification report
185
230
 
186
231
  ```python
@@ -225,6 +270,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
225
270
  evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
226
271
  evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
227
272
  evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
273
+ evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
274
+ evalsuite detection instances_val.json detections.json --plot pr_curves.png
228
275
  evalsuite metrics --category clinical
229
276
  evalsuite info classification.mcc
230
277
  evalsuite benchmark --quick
@@ -235,24 +282,30 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
235
282
 
236
283
  ## Performance
237
284
 
238
- Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
239
- scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
240
- (largest difference 1.1e-16).
241
-
242
- | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
243
- | --- | ---: | ---: | ---: | ---: | ---: |
244
- | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
245
- | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
246
- | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
247
- | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
248
- | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
249
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
250
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
251
-
252
- `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
253
- speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
254
- infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
255
- table and notes in
285
+ Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
286
+ Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
287
+ 1.4e-14).
288
+
289
+ | Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
290
+ | --- | ---: | --- | ---: | ---: | ---: |
291
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
292
+ | macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
293
+ | ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
294
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
295
+ | sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
296
+ | calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
297
+ | decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
298
+ | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
299
+ | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
300
+ | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
301
+ | segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
302
+ | COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
303
+ | Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
304
+
305
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
306
+ of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
307
+ below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
308
+ your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
256
309
  [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
257
310
 
258
311
  ## Metrics in this release
@@ -276,6 +329,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
276
329
  delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
277
330
  Benjamini–Yekutieli).
278
331
 
332
+ **Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
333
+ mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
334
+ surface distance (with pixel spacing), confusion matrix and a full report.
335
+
336
+ **Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
337
+ object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
338
+
279
339
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
280
340
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
281
341
  loss, Huber loss, relative absolute error, relative squared error.
@@ -7,10 +7,10 @@
7
7
 
8
8
  **Unified, reproducible evaluation for machine learning and research.**
9
9
 
10
- EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
11
- object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
10
+ EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
11
+ evaluation into one consistent, validated, documented framework.
12
12
 
13
- > **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
13
+ > **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -127,6 +127,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
127
127
  Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
128
128
  checked against SciPy and statsmodels in the test suite.
129
129
 
130
+ ## Segmentation
131
+
132
+ ```python
133
+ # label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
134
+ es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
135
+ es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
136
+ es.miou(y_true, y_pred, ignore_index=255)
137
+ es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
138
+ es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
139
+ es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
140
+
141
+ report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
142
+ print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
143
+ es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
144
+ es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
145
+ ```
146
+
147
+ ## Object detection
148
+
149
+ ```python
150
+ y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
151
+ y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
152
+
153
+ report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
154
+ report["map"], report["map_50"], report["mar_100"]
155
+ es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
156
+ es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
157
+ es.box_iou(boxes_a, boxes_b, box_format="xywh")
158
+ y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
159
+ es.plot.detection_pr(y_true, y_pred)
160
+ ```
161
+
162
+ The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
163
+ `pycocotools` to the last digit in the test suite.
164
+
165
+ Models can be compared over the same images with intervals and paired tests, as for every other task:
166
+ `es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
167
+ mAP.
168
+
130
169
  ## Classification report
131
170
 
132
171
  ```python
@@ -171,6 +210,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
171
210
  evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
172
211
  evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
173
212
  evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
213
+ evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
214
+ evalsuite detection instances_val.json detections.json --plot pr_curves.png
174
215
  evalsuite metrics --category clinical
175
216
  evalsuite info classification.mcc
176
217
  evalsuite benchmark --quick
@@ -181,24 +222,30 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
181
222
 
182
223
  ## Performance
183
224
 
184
- Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
185
- scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
186
- (largest difference 1.1e-16).
187
-
188
- | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
189
- | --- | ---: | ---: | ---: | ---: | ---: |
190
- | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
191
- | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
192
- | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
193
- | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
194
- | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
195
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
196
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
197
-
198
- `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
199
- speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
200
- infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
201
- table and notes in
225
+ Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
226
+ Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
227
+ 1.4e-14).
228
+
229
+ | Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
230
+ | --- | ---: | --- | ---: | ---: | ---: |
231
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
232
+ | macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
233
+ | ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
234
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
235
+ | sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
236
+ | calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
237
+ | decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
238
+ | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
239
+ | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
240
+ | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
241
+ | segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
242
+ | COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
243
+ | Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
244
+
245
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
246
+ of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
247
+ below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
248
+ your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
202
249
  [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
203
250
 
204
251
  ## Metrics in this release
@@ -222,6 +269,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
222
269
  delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
223
270
  Benjamini–Yekutieli).
224
271
 
272
+ **Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
273
+ mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
274
+ surface distance (with pixel spacing), confusion matrix and a full report.
275
+
276
+ **Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
277
+ object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
278
+
225
279
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
226
280
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
227
281
  loss, Huber loss, relative absolute error, relative squared error.
@@ -12,12 +12,13 @@ license-files = ["LICENSE"]
12
12
  requires-python = ">=3.9"
13
13
  authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
15
- keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
15
+ keywords = ["evaluation", "metrics", "segmentation", "object detection", "mAP", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
17
  "Development Status :: 5 - Production/Stable",
18
18
  "Intended Audience :: Science/Research",
19
19
  "Intended Audience :: Healthcare Industry",
20
20
  "Topic :: Scientific/Engineering :: Medical Science Apps.",
21
+ "Topic :: Scientific/Engineering :: Image Recognition",
21
22
  "Intended Audience :: Developers",
22
23
  "Operating System :: OS Independent",
23
24
  "Programming Language :: Python :: 3",
@@ -36,7 +37,8 @@ dependencies = ["numpy>=1.22", "scipy>=1.8", "pandas>=1.4"]
36
37
 
37
38
  [project.optional-dependencies]
38
39
  plot = ["matplotlib>=3.5"]
39
- all = ["matplotlib>=3.5"]
40
+ vision = ["pillow>=9"]
41
+ all = ["matplotlib>=3.5", "pillow>=9"]
40
42
  dev = [
41
43
  "pytest>=7",
42
44
  "pytest-cov>=4",
@@ -44,6 +46,8 @@ dev = [
44
46
  "scikit-learn>=1.2",
45
47
  "statsmodels>=0.13",
46
48
  "matplotlib>=3.5",
49
+ "pillow>=9",
50
+ "pycocotools>=2.0.7",
47
51
  "ruff>=0.6",
48
52
  "mypy>=1.10",
49
53
  "pandas-stubs",
@@ -5,7 +5,7 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import calibration, classification, clinical, plot, regression, stats
8
+ from . import calibration, classification, clinical, plot, regression, stats, vision
9
9
  from .api import evaluate
10
10
  from .calibration import (
11
11
  CalibrationReport,
@@ -111,8 +111,49 @@ from .stats import (
111
111
  wilcoxon_test,
112
112
  )
113
113
  from .version import __version__
114
+ from .vision import (
115
+ DetectionReport,
116
+ SegmentationReport,
117
+ average_precision_detection,
118
+ average_surface_distance,
119
+ boundary_iou,
120
+ box_iou,
121
+ detection_pr_curve,
122
+ detection_report,
123
+ dice,
124
+ from_coco,
125
+ hausdorff_distance,
126
+ iou,
127
+ mean_average_precision,
128
+ mean_pixel_accuracy,
129
+ miou,
130
+ per_image_scores,
131
+ pixel_accuracy,
132
+ segmentation_confusion,
133
+ segmentation_report,
134
+ )
114
135
 
115
136
  __all__ = [
137
+ "SegmentationReport",
138
+ "segmentation_report",
139
+ "detection_pr_curve",
140
+ "vision",
141
+ "DetectionReport",
142
+ "average_precision_detection",
143
+ "average_surface_distance",
144
+ "boundary_iou",
145
+ "box_iou",
146
+ "detection_report",
147
+ "dice",
148
+ "from_coco",
149
+ "hausdorff_distance",
150
+ "iou",
151
+ "mean_average_precision",
152
+ "mean_pixel_accuracy",
153
+ "miou",
154
+ "per_image_scores",
155
+ "pixel_accuracy",
156
+ "segmentation_confusion",
116
157
  "CalibrationReport",
117
158
  "calibration_report",
118
159
  "calibration",
@@ -149,6 +149,16 @@ def evaluate(
149
149
  >>> round(r["accuracy"], 2)
150
150
  0.75
151
151
  """
152
+ if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
153
+ raise UnsupportedTaskError(
154
+ "y_true looks like object detection annotations (one dict per image); use "
155
+ "evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
156
+ )
157
+ if np.ndim(y_true) >= 3:
158
+ raise UnsupportedTaskError(
159
+ "y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
160
+ "evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
161
+ )
152
162
  task = task or _infer_task(y_true, y_prob)
153
163
  if task == "classification":
154
164
  return _evaluate_classification(