evalsuite-python 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/CHANGELOG.md +43 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/PKG-INFO +83 -23
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/README.md +75 -21
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/pyproject.toml +6 -2
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/__init__.py +42 -1
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/api.py +10 -0
- evalsuite_python-0.3.0/src/evalsuite/benchmarks.py +568 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/calibration.py +1 -1
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/cli/main.py +109 -2
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/metrics.py +14 -5
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/validation.py +14 -2
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/plot.py +165 -1
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/_resolve.py +26 -3
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/compare.py +6 -4
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/effect.py +1 -1
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.3.0/src/evalsuite/vision/__init__.py +47 -0
- evalsuite_python-0.3.0/src/evalsuite/vision/detection.py +648 -0
- evalsuite_python-0.3.0/src/evalsuite/vision/segmentation.py +965 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/test_v020.py +9 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_benchmarks.py +19 -3
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/test_edges.py +18 -0
- evalsuite_python-0.3.0/tests/vision/__init__.py +0 -0
- evalsuite_python-0.3.0/tests/vision/test_detection.py +200 -0
- evalsuite_python-0.3.0/tests/vision/test_edges.py +126 -0
- evalsuite_python-0.3.0/tests/vision/test_outputs.py +170 -0
- evalsuite_python-0.3.0/tests/vision/test_segmentation.py +156 -0
- evalsuite_python-0.2.0/src/evalsuite/benchmarks.py +0 -272
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/.gitignore +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/LICENSE +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/__main__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/classification/metrics.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/cli/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/clinical/report.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/export.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/result.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/regression/metrics.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/reporting.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/hypothesis.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/clinical/test_outputs.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/conftest.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/integration/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_cli.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_plot.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/output/test_reporting.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/regression/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.2.0 → evalsuite_python-0.3.0}/tests/unit/test_core.py +0 -0
|
@@ -6,6 +6,49 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.3.0] - 2026-10-09
|
|
10
|
+
|
|
11
|
+
Computer vision (the v0.3.0 roadmap), with comparison, plots, reporting and benchmarks extended to it.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
- Segmentation: `dice`, `iou`, `miou`, `pixel_accuracy`, `mean_pixel_accuracy`, `boundary_iou`,
|
|
15
|
+
`hausdorff_distance` (HD and HD95, anisotropic `spacing`), `average_surface_distance`,
|
|
16
|
+
`segmentation_confusion`, `per_image_scores` and `segmentation_report`. 2-D and 3-D masks, lists of
|
|
17
|
+
differently sized masks, `ignore_index`, `aggregate="dataset"|"image"`, undefined classes reported as NaN.
|
|
18
|
+
- Object detection: `box_iou`, `detection_report` (the 12 COCO numbers plus AP per class),
|
|
19
|
+
`mean_average_precision`, `average_precision_detection` (COCO or VOC interpolation), `detection_pr_curve`,
|
|
20
|
+
`from_coco`. Matches pycocotools exactly, including crowd regions, area ranges and detection limits.
|
|
21
|
+
- Comparison for vision: `compare`, `bootstrap_ci` and `paired_bootstrap_test` resample images for
|
|
22
|
+
segmentation masks and per-image detections.
|
|
23
|
+
- Plots: `es.plot.segmentation` (prediction fill vs truth outline), `es.plot.per_class` (any per-class
|
|
24
|
+
result, a segmentation report or a detection report), `es.plot.detection_pr`.
|
|
25
|
+
- CLI: `evalsuite segmentation` (.npy/.npz or a folder of mask images) and `evalsuite detection` (COCO JSON).
|
|
26
|
+
- Benchmarks: `--suite vision` against scikit-learn, SciPy and pycocotools.
|
|
27
|
+
- `vision` extra (Pillow, for reading mask image folders); `CITATION.cff`.
|
|
28
|
+
|
|
29
|
+
### Changed
|
|
30
|
+
- `evaluate()` points segmentation masks and detection annotations to the right functions instead of
|
|
31
|
+
failing with a shape error.
|
|
32
|
+
- CI and release workflows use the Node 24 versions of the GitHub actions.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
- Calibration slope and intercept no longer emit an overflow warning before reporting a perfectly
|
|
36
|
+
separated outcome.
|
|
37
|
+
- The CLI no longer fails on consoles that cannot print Unicode (Windows code pages).
|
|
38
|
+
|
|
39
|
+
## [0.2.1] - 2026-10-09
|
|
40
|
+
|
|
41
|
+
### Added
|
|
42
|
+
- Benchmarks for the v0.2.0 functions (`evalsuite benchmark --suite clinical`): diagnostic metrics and
|
|
43
|
+
report, calibration slope and intercept, decision curves, t-test, Mann–Whitney, Cramér's V and Hochberg,
|
|
44
|
+
each against its reference (scikit-learn, statsmodels, SciPy or the textbook NumPy loop). Benchmark rows
|
|
45
|
+
now name their reference library.
|
|
46
|
+
|
|
47
|
+
### Changed
|
|
48
|
+
- Faster label handling: integer class labels are found with one marking pass instead of a sort
|
|
49
|
+
(8 label metrics at 1M samples: 34× faster than scikit-learn, up from 10×; macro F1 5.8×, up from 1.6×).
|
|
50
|
+
- Decision curves sort the risks once and use cumulative sums (O((n + k) log n), no n × k matrix).
|
|
51
|
+
|
|
9
52
|
## [0.2.0] - 2026-10-08
|
|
10
53
|
|
|
11
54
|
Clinical and statistical evaluation (the v0.2.0 roadmap).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -11,7 +11,7 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
|
11
11
|
Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
|
-
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
|
|
14
|
+
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,mAP,machine learning,metrics,object detection,regression,reproducibility,segmentation,statistical tests,statistics
|
|
15
15
|
Classifier: Development Status :: 5 - Production/Stable
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: Intended Audience :: Healthcare Industry
|
|
@@ -27,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
27
27
|
Classifier: Programming Language :: Python :: 3.14
|
|
28
28
|
Classifier: Topic :: Scientific/Engineering
|
|
29
29
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
30
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
30
31
|
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
31
32
|
Classifier: Typing :: Typed
|
|
32
33
|
Requires-Python: >=3.9
|
|
@@ -35,13 +36,16 @@ Requires-Dist: pandas>=1.4
|
|
|
35
36
|
Requires-Dist: scipy>=1.8
|
|
36
37
|
Provides-Extra: all
|
|
37
38
|
Requires-Dist: matplotlib>=3.5; extra == 'all'
|
|
39
|
+
Requires-Dist: pillow>=9; extra == 'all'
|
|
38
40
|
Provides-Extra: dev
|
|
39
41
|
Requires-Dist: build; extra == 'dev'
|
|
40
42
|
Requires-Dist: hypothesis>=6.80; extra == 'dev'
|
|
41
43
|
Requires-Dist: matplotlib>=3.5; extra == 'dev'
|
|
42
44
|
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
43
45
|
Requires-Dist: pandas-stubs; extra == 'dev'
|
|
46
|
+
Requires-Dist: pillow>=9; extra == 'dev'
|
|
44
47
|
Requires-Dist: pip-audit; extra == 'dev'
|
|
48
|
+
Requires-Dist: pycocotools>=2.0.7; extra == 'dev'
|
|
45
49
|
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
46
50
|
Requires-Dist: pytest>=7; extra == 'dev'
|
|
47
51
|
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
@@ -50,6 +54,8 @@ Requires-Dist: statsmodels>=0.13; extra == 'dev'
|
|
|
50
54
|
Requires-Dist: twine; extra == 'dev'
|
|
51
55
|
Provides-Extra: plot
|
|
52
56
|
Requires-Dist: matplotlib>=3.5; extra == 'plot'
|
|
57
|
+
Provides-Extra: vision
|
|
58
|
+
Requires-Dist: pillow>=9; extra == 'vision'
|
|
53
59
|
Description-Content-Type: text/markdown
|
|
54
60
|
|
|
55
61
|
# EvalSuite
|
|
@@ -61,10 +67,10 @@ Description-Content-Type: text/markdown
|
|
|
61
67
|
|
|
62
68
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
63
69
|
|
|
64
|
-
EvalSuite brings classification, regression, clinical
|
|
65
|
-
|
|
70
|
+
EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
|
|
71
|
+
evaluation into one consistent, validated, documented framework.
|
|
66
72
|
|
|
67
|
-
> **Status: stable (0.
|
|
73
|
+
> **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
|
|
68
74
|
|
|
69
75
|
## Installation
|
|
70
76
|
|
|
@@ -181,6 +187,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
|
181
187
|
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
182
188
|
checked against SciPy and statsmodels in the test suite.
|
|
183
189
|
|
|
190
|
+
## Segmentation
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
# label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
|
|
194
|
+
es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
|
|
195
|
+
es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
|
|
196
|
+
es.miou(y_true, y_pred, ignore_index=255)
|
|
197
|
+
es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
|
|
198
|
+
es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
|
|
199
|
+
es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
|
|
200
|
+
|
|
201
|
+
report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
|
|
202
|
+
print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
|
|
203
|
+
es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
|
|
204
|
+
es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
## Object detection
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
|
|
211
|
+
y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
|
|
212
|
+
|
|
213
|
+
report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
|
|
214
|
+
report["map"], report["map_50"], report["mar_100"]
|
|
215
|
+
es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
|
|
216
|
+
es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
|
|
217
|
+
es.box_iou(boxes_a, boxes_b, box_format="xywh")
|
|
218
|
+
y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
|
|
219
|
+
es.plot.detection_pr(y_true, y_pred)
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
|
|
223
|
+
`pycocotools` to the last digit in the test suite.
|
|
224
|
+
|
|
225
|
+
Models can be compared over the same images with intervals and paired tests, as for every other task:
|
|
226
|
+
`es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
|
|
227
|
+
mAP.
|
|
228
|
+
|
|
184
229
|
## Classification report
|
|
185
230
|
|
|
186
231
|
```python
|
|
@@ -225,6 +270,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
|
225
270
|
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
226
271
|
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
227
272
|
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
273
|
+
evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
|
|
274
|
+
evalsuite detection instances_val.json detections.json --plot pr_curves.png
|
|
228
275
|
evalsuite metrics --category clinical
|
|
229
276
|
evalsuite info classification.mcc
|
|
230
277
|
evalsuite benchmark --quick
|
|
@@ -235,24 +282,30 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
235
282
|
|
|
236
283
|
## Performance
|
|
237
284
|
|
|
238
|
-
Benchmarked against
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
| Case | n | EvalSuite (ms) |
|
|
243
|
-
| --- | ---: |
|
|
244
|
-
| 8 binary label metrics via `evaluate()` | 1,000 |
|
|
245
|
-
|
|
|
246
|
-
|
|
|
247
|
-
|
|
|
248
|
-
|
|
|
249
|
-
|
|
|
250
|
-
|
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
285
|
+
Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
286
|
+
Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
|
|
287
|
+
1.4e-14).
|
|
288
|
+
|
|
289
|
+
| Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
|
|
290
|
+
| --- | ---: | --- | ---: | ---: | ---: |
|
|
291
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
|
|
292
|
+
| macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
|
|
293
|
+
| ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
|
|
294
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
|
|
295
|
+
| sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
|
|
296
|
+
| calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
|
|
297
|
+
| decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
|
|
298
|
+
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
299
|
+
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
300
|
+
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
301
|
+
| segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
|
|
302
|
+
| COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
|
|
303
|
+
| Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
|
|
304
|
+
|
|
305
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
306
|
+
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
307
|
+
below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
|
|
308
|
+
your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
|
|
256
309
|
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
257
310
|
|
|
258
311
|
## Metrics in this release
|
|
@@ -276,6 +329,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
|
|
|
276
329
|
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
277
330
|
Benjamini–Yekutieli).
|
|
278
331
|
|
|
332
|
+
**Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
|
|
333
|
+
mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
|
|
334
|
+
surface distance (with pixel spacing), confusion matrix and a full report.
|
|
335
|
+
|
|
336
|
+
**Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
|
|
337
|
+
object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
|
|
338
|
+
|
|
279
339
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
280
340
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
281
341
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -7,10 +7,10 @@
|
|
|
7
7
|
|
|
8
8
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
9
9
|
|
|
10
|
-
EvalSuite brings classification, regression, clinical
|
|
11
|
-
|
|
10
|
+
EvalSuite brings classification, regression, clinical, statistical, segmentation and object-detection
|
|
11
|
+
evaluation into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status: stable (0.
|
|
13
|
+
> **Status: stable (0.3.0).** Every item on the 0.1.0, 0.2.0 and 0.3.0 roadmaps is implemented and verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -127,6 +127,45 @@ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
|
127
127
|
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
128
128
|
checked against SciPy and statsmodels in the test suite.
|
|
129
129
|
|
|
130
|
+
## Segmentation
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
# label masks: one 2-D image, or images on the first axis (N, H, W) / (N, D, H, W), or a list of masks
|
|
134
|
+
es.dice(y_true, y_pred) # macro over classes, pixel counts summed over the dataset
|
|
135
|
+
es.iou(y_true, y_pred, average=None) # per class; classes absent from both masks are NaN, not 0
|
|
136
|
+
es.miou(y_true, y_pred, ignore_index=255)
|
|
137
|
+
es.dice(y_true, y_pred, aggregate="image") # mean of per-image scores (medical imaging convention)
|
|
138
|
+
es.boundary_iou(y_true, y_pred) # Cheng et al. 2021
|
|
139
|
+
es.hausdorff_distance(y_true, y_pred, percentile=95, spacing=(0.8, 0.8)) # HD95 in mm
|
|
140
|
+
|
|
141
|
+
report = es.segmentation_report(y_true, y_pred, class_names={0: "background", 1: "liver"})
|
|
142
|
+
print(report) # mIoU, Dice, pixel accuracy, Boundary IoU, HD95, ASSD + per-class table
|
|
143
|
+
es.plot.segmentation(image, y_true[0], y_pred[0]) # prediction fill, truth outline
|
|
144
|
+
es.plot.per_class(report, metric="iou") # per-class bars (also takes a detection report)
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
## Object detection
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
y_true = [{"boxes": [[x1, y1, x2, y2], ...], "labels": [3, ...]}, ...] # one dict per image
|
|
151
|
+
y_pred = [{"boxes": [...], "labels": [...], "scores": [...]}, ...]
|
|
152
|
+
|
|
153
|
+
report = es.detection_report(y_true, y_pred) # the 12 COCO numbers + AP per class
|
|
154
|
+
report["map"], report["map_50"], report["mar_100"]
|
|
155
|
+
es.mean_average_precision(y_true, y_pred, iou_threshold=0.5) # mAP@.50
|
|
156
|
+
es.average_precision_detection(y_true, y_pred, interpolation="voc") # per class, VOC-style
|
|
157
|
+
es.box_iou(boxes_a, boxes_b, box_format="xywh")
|
|
158
|
+
y_true, y_pred = es.from_coco("instances_val.json", "detections.json")
|
|
159
|
+
es.plot.detection_pr(y_true, y_pred)
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
The COCO protocol (crowd regions, area ranges, max detections, 101-point interpolation) matches
|
|
163
|
+
`pycocotools` to the last digit in the test suite.
|
|
164
|
+
|
|
165
|
+
Models can be compared over the same images with intervals and paired tests, as for every other task:
|
|
166
|
+
`es.compare(y_true, {"unet": masks_a, "deeplab": masks_b})` resamples images; for detection it compares
|
|
167
|
+
mAP.
|
|
168
|
+
|
|
130
169
|
## Classification report
|
|
131
170
|
|
|
132
171
|
```python
|
|
@@ -171,6 +210,8 @@ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
|
171
210
|
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
172
211
|
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
173
212
|
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
213
|
+
evalsuite segmentation true_masks.npy pred_masks.npy --ignore-index 255 --plot per_class.png
|
|
214
|
+
evalsuite detection instances_val.json detections.json --plot pr_curves.png
|
|
174
215
|
evalsuite metrics --category clinical
|
|
175
216
|
evalsuite info classification.mcc
|
|
176
217
|
evalsuite benchmark --quick
|
|
@@ -181,24 +222,30 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
181
222
|
|
|
182
223
|
## Performance
|
|
183
224
|
|
|
184
|
-
Benchmarked against
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
| Case | n | EvalSuite (ms) |
|
|
189
|
-
| --- | ---: |
|
|
190
|
-
| 8 binary label metrics via `evaluate()` | 1,000 |
|
|
191
|
-
|
|
|
192
|
-
|
|
|
193
|
-
|
|
|
194
|
-
|
|
|
195
|
-
|
|
|
196
|
-
|
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
225
|
+
Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
226
|
+
Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
|
|
227
|
+
1.4e-14).
|
|
228
|
+
|
|
229
|
+
| Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
|
|
230
|
+
| --- | ---: | --- | ---: | ---: | ---: |
|
|
231
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
|
|
232
|
+
| macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
|
|
233
|
+
| ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
|
|
234
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
|
|
235
|
+
| sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
|
|
236
|
+
| calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
|
|
237
|
+
| decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
|
|
238
|
+
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
239
|
+
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
240
|
+
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
241
|
+
| segmentation Dice and IoU per class | 1,000,000 px | scikit-learn | 40.2 | 282.0 | **7.0×** |
|
|
242
|
+
| COCO detection evaluation (12 numbers) | 1,000 images | pycocotools | 967.4 | 980.6 | **1.0×** |
|
|
243
|
+
| Hausdorff distance | 50 images | SciPy | 26.2 | 20.8 | 0.79× |
|
|
244
|
+
|
|
245
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
246
|
+
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
247
|
+
below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
|
|
248
|
+
your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
|
|
202
249
|
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
203
250
|
|
|
204
251
|
## Metrics in this release
|
|
@@ -222,6 +269,13 @@ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (
|
|
|
222
269
|
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
223
270
|
Benjamini–Yekutieli).
|
|
224
271
|
|
|
272
|
+
**Segmentation** (2-D and 3-D label masks; `ignore_index`; dataset or per-image aggregation): Dice, IoU,
|
|
273
|
+
mIoU, pixel accuracy, mean pixel accuracy, Boundary IoU, Hausdorff distance and HD95, average symmetric
|
|
274
|
+
surface distance (with pixel spacing), confusion matrix and a full report.
|
|
275
|
+
|
|
276
|
+
**Object detection**: box IoU (xyxy, xywh, cxcywh), COCO mAP@[.50:.95], mAP@.50, mAP@.75, mAP and mAR by
|
|
277
|
+
object size, AP per class with COCO or VOC interpolation, precision-recall curves, COCO file import.
|
|
278
|
+
|
|
225
279
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
226
280
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
227
281
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -12,12 +12,13 @@ license-files = ["LICENSE"]
|
|
|
12
12
|
requires-python = ">=3.9"
|
|
13
13
|
authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
15
|
-
keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
15
|
+
keywords = ["evaluation", "metrics", "segmentation", "object detection", "mAP", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
19
|
"Intended Audience :: Healthcare Industry",
|
|
20
20
|
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
21
22
|
"Intended Audience :: Developers",
|
|
22
23
|
"Operating System :: OS Independent",
|
|
23
24
|
"Programming Language :: Python :: 3",
|
|
@@ -36,7 +37,8 @@ dependencies = ["numpy>=1.22", "scipy>=1.8", "pandas>=1.4"]
|
|
|
36
37
|
|
|
37
38
|
[project.optional-dependencies]
|
|
38
39
|
plot = ["matplotlib>=3.5"]
|
|
39
|
-
|
|
40
|
+
vision = ["pillow>=9"]
|
|
41
|
+
all = ["matplotlib>=3.5", "pillow>=9"]
|
|
40
42
|
dev = [
|
|
41
43
|
"pytest>=7",
|
|
42
44
|
"pytest-cov>=4",
|
|
@@ -44,6 +46,8 @@ dev = [
|
|
|
44
46
|
"scikit-learn>=1.2",
|
|
45
47
|
"statsmodels>=0.13",
|
|
46
48
|
"matplotlib>=3.5",
|
|
49
|
+
"pillow>=9",
|
|
50
|
+
"pycocotools>=2.0.7",
|
|
47
51
|
"ruff>=0.6",
|
|
48
52
|
"mypy>=1.10",
|
|
49
53
|
"pandas-stubs",
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import calibration, classification, clinical, plot, regression, stats
|
|
8
|
+
from . import calibration, classification, clinical, plot, regression, stats, vision
|
|
9
9
|
from .api import evaluate
|
|
10
10
|
from .calibration import (
|
|
11
11
|
CalibrationReport,
|
|
@@ -111,8 +111,49 @@ from .stats import (
|
|
|
111
111
|
wilcoxon_test,
|
|
112
112
|
)
|
|
113
113
|
from .version import __version__
|
|
114
|
+
from .vision import (
|
|
115
|
+
DetectionReport,
|
|
116
|
+
SegmentationReport,
|
|
117
|
+
average_precision_detection,
|
|
118
|
+
average_surface_distance,
|
|
119
|
+
boundary_iou,
|
|
120
|
+
box_iou,
|
|
121
|
+
detection_pr_curve,
|
|
122
|
+
detection_report,
|
|
123
|
+
dice,
|
|
124
|
+
from_coco,
|
|
125
|
+
hausdorff_distance,
|
|
126
|
+
iou,
|
|
127
|
+
mean_average_precision,
|
|
128
|
+
mean_pixel_accuracy,
|
|
129
|
+
miou,
|
|
130
|
+
per_image_scores,
|
|
131
|
+
pixel_accuracy,
|
|
132
|
+
segmentation_confusion,
|
|
133
|
+
segmentation_report,
|
|
134
|
+
)
|
|
114
135
|
|
|
115
136
|
__all__ = [
|
|
137
|
+
"SegmentationReport",
|
|
138
|
+
"segmentation_report",
|
|
139
|
+
"detection_pr_curve",
|
|
140
|
+
"vision",
|
|
141
|
+
"DetectionReport",
|
|
142
|
+
"average_precision_detection",
|
|
143
|
+
"average_surface_distance",
|
|
144
|
+
"boundary_iou",
|
|
145
|
+
"box_iou",
|
|
146
|
+
"detection_report",
|
|
147
|
+
"dice",
|
|
148
|
+
"from_coco",
|
|
149
|
+
"hausdorff_distance",
|
|
150
|
+
"iou",
|
|
151
|
+
"mean_average_precision",
|
|
152
|
+
"mean_pixel_accuracy",
|
|
153
|
+
"miou",
|
|
154
|
+
"per_image_scores",
|
|
155
|
+
"pixel_accuracy",
|
|
156
|
+
"segmentation_confusion",
|
|
116
157
|
"CalibrationReport",
|
|
117
158
|
"calibration_report",
|
|
118
159
|
"calibration",
|
|
@@ -149,6 +149,16 @@ def evaluate(
|
|
|
149
149
|
>>> round(r["accuracy"], 2)
|
|
150
150
|
0.75
|
|
151
151
|
"""
|
|
152
|
+
if isinstance(y_true, (list, tuple)) and y_true and isinstance(y_true[0], dict):
|
|
153
|
+
raise UnsupportedTaskError(
|
|
154
|
+
"y_true looks like object detection annotations (one dict per image); use "
|
|
155
|
+
"evalsuite.detection_report(y_true, y_pred) or evalsuite.mean_average_precision(...)."
|
|
156
|
+
)
|
|
157
|
+
if np.ndim(y_true) >= 3:
|
|
158
|
+
raise UnsupportedTaskError(
|
|
159
|
+
"y_true has 3 or more dimensions, which looks like segmentation masks (images first); use "
|
|
160
|
+
"evalsuite.segmentation_report(y_true, y_pred) or evalsuite.dice / evalsuite.iou."
|
|
161
|
+
)
|
|
152
162
|
task = task or _infer_task(y_true, y_prob)
|
|
153
163
|
if task == "classification":
|
|
154
164
|
return _evaluate_classification(
|