evalsuite-python 0.1.2__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/CHANGELOG.md +32 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/PKG-INFO +89 -23
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/README.md +85 -21
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/pyproject.toml +3 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/__init__.py +61 -1
- evalsuite_python-0.2.1/src/evalsuite/benchmarks.py +417 -0
- evalsuite_python-0.2.1/src/evalsuite/calibration.py +386 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/cli/main.py +97 -8
- evalsuite_python-0.2.1/src/evalsuite/clinical/__init__.py +18 -0
- evalsuite_python-0.2.1/src/evalsuite/clinical/metrics.py +339 -0
- evalsuite_python-0.2.1/src/evalsuite/clinical/report.py +360 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/validation.py +14 -2
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/plot.py +46 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/__init__.py +22 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/compare.py +1 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/effect.py +46 -8
- evalsuite_python-0.2.1/src/evalsuite/stats/hypothesis.py +321 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.2.1/tests/clinical/test_outputs.py +99 -0
- evalsuite_python-0.2.1/tests/clinical/test_v020.py +269 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_benchmarks.py +19 -3
- evalsuite_python-0.2.1/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/unit/test_edges.py +18 -0
- evalsuite_python-0.1.2/src/evalsuite/benchmarks.py +0 -272
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/.gitignore +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/LICENSE +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/__main__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/api.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/metrics.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/cli/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/export.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/result.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/regression/metrics.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/reporting.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/_resolve.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.2/tests/integration → evalsuite_python-0.2.1/tests/clinical}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/conftest.py +0 -0
- {evalsuite_python-0.1.2/tests/output → evalsuite_python-0.2.1/tests/integration}/__init__.py +0 -0
- {evalsuite_python-0.1.2/tests/regression → evalsuite_python-0.2.1/tests/output}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_cli.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_plot.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_reporting.py +0 -0
- {evalsuite_python-0.1.2/tests/stats → evalsuite_python-0.2.1/tests/regression}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.2/tests/unit → evalsuite_python-0.2.1/tests/stats}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/unit/test_core.py +0 -0
|
@@ -6,6 +6,38 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.2.1] - 2026-10-09
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- Benchmarks for the v0.2.0 functions (`evalsuite benchmark --suite clinical`): diagnostic metrics and
|
|
13
|
+
report, calibration slope and intercept, decision curves, t-test, Mann–Whitney, Cramér's V and Hochberg,
|
|
14
|
+
each against its reference (scikit-learn, statsmodels, SciPy or the textbook NumPy loop). Benchmark rows
|
|
15
|
+
now name their reference library.
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
- Faster label handling: integer class labels are found with one marking pass instead of a sort
|
|
19
|
+
(8 label metrics at 1M samples: 34× faster than scikit-learn, up from 10×; macro F1 5.8×, up from 1.6×).
|
|
20
|
+
- Decision curves sort the risks once and use cumulative sums (O((n + k) log n), no n × k matrix).
|
|
21
|
+
|
|
22
|
+
## [0.2.0] - 2026-10-08
|
|
23
|
+
|
|
24
|
+
Clinical and statistical evaluation (the v0.2.0 roadmap).
|
|
25
|
+
|
|
26
|
+
### Added
|
|
27
|
+
- Clinical: `sensitivity`, `ppv`, `lr_positive`, `lr_negative`, `diagnostic_odds_ratio`, `youden_j`,
|
|
28
|
+
`net_benefit`; `diagnostic_report` with confidence intervals for every measure (Wilson or Clopper–Pearson
|
|
29
|
+
for proportions, log method for likelihood ratios, Woolf for the DOR, Wald for Youden's J);
|
|
30
|
+
`decision_curve` (net benefit, treat all, treat none, useful threshold range).
|
|
31
|
+
- Calibration: `maximum_calibration_error`, `calibration_slope`, `calibration_intercept`,
|
|
32
|
+
`hosmer_lemeshow`, `calibration_report`.
|
|
33
|
+
- Statistical tests: `t_test` (Welch/Student), `paired_t_test`, `mann_whitney_test`, `wilcoxon_test`,
|
|
34
|
+
`kruskal_wallis_test`, `friedman_test`, `shapiro_wilk_test`, `chi_square_test`, `fisher_exact_test`, each
|
|
35
|
+
with an effect size; `cramers_v` (optional bias correction); Hochberg correction in `adjust_pvalues` and
|
|
36
|
+
`compare`.
|
|
37
|
+
- `es.plot.decision_curve`; CLI commands `evalsuite diagnostic`, `evalsuite calibration` and
|
|
38
|
+
`evalsuite plot decision`.
|
|
39
|
+
- Validated against statsmodels (GLM, Table2x2, proportion_confint, multipletests, CompareMeans) and SciPy.
|
|
40
|
+
|
|
9
41
|
## [0.1.2] - 2026-10-08
|
|
10
42
|
|
|
11
43
|
### Changed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -11,9 +11,10 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
|
11
11
|
Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
|
-
Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
|
|
14
|
+
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
|
|
15
15
|
Classifier: Development Status :: 5 - Production/Stable
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Intended Audience :: Healthcare Industry
|
|
17
18
|
Classifier: Intended Audience :: Science/Research
|
|
18
19
|
Classifier: Operating System :: OS Independent
|
|
19
20
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -26,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
26
27
|
Classifier: Programming Language :: Python :: 3.14
|
|
27
28
|
Classifier: Topic :: Scientific/Engineering
|
|
28
29
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
30
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
29
31
|
Classifier: Typing :: Typed
|
|
30
32
|
Requires-Python: >=3.9
|
|
31
33
|
Requires-Dist: numpy>=1.22
|
|
@@ -59,10 +61,10 @@ Description-Content-Type: text/markdown
|
|
|
59
61
|
|
|
60
62
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
61
63
|
|
|
62
|
-
EvalSuite brings classification and
|
|
64
|
+
EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
|
|
63
65
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
64
66
|
|
|
65
|
-
> **Status: stable (0.1
|
|
67
|
+
> **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
|
|
66
68
|
|
|
67
69
|
## Installation
|
|
68
70
|
|
|
@@ -134,6 +136,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
134
136
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
135
137
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
136
138
|
|
|
139
|
+
## Clinical evaluation
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
|
|
143
|
+
print(report)
|
|
144
|
+
# Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
|
|
145
|
+
# diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
|
|
146
|
+
|
|
147
|
+
es.lr_positive(y_true, y_pred)
|
|
148
|
+
es.youden_j(y_true, y_pred)
|
|
149
|
+
|
|
150
|
+
dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
|
|
151
|
+
dca.useful_range() # thresholds where the model beats both
|
|
152
|
+
es.plot.decision_curve(y_true, {"model": y_prob})
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Ratios that divide by zero are `inf` or NaN with a warning, never 0.
|
|
156
|
+
|
|
157
|
+
## Calibration
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
|
|
161
|
+
es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
|
|
162
|
+
es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
|
|
163
|
+
es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Statistical tests
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
|
|
170
|
+
es.paired_t_test(fold_scores_a, fold_scores_b)
|
|
171
|
+
es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
|
|
172
|
+
es.mann_whitney_test(a, b) # with rank-biserial r
|
|
173
|
+
es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
|
|
174
|
+
es.kruskal_wallis_test(g1, g2, g3)
|
|
175
|
+
es.shapiro_wilk_test(residuals)
|
|
176
|
+
es.chi_square_test(table) # with Cramér's V
|
|
177
|
+
es.fisher_exact_test([[8, 2], [1, 5]])
|
|
178
|
+
es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
182
|
+
checked against SciPy and statsmodels in the test suite.
|
|
183
|
+
|
|
137
184
|
## Classification report
|
|
138
185
|
|
|
139
186
|
```python
|
|
@@ -175,7 +222,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
|
175
222
|
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
176
223
|
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
177
224
|
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
178
|
-
evalsuite
|
|
225
|
+
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
226
|
+
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
227
|
+
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
228
|
+
evalsuite metrics --category clinical
|
|
179
229
|
evalsuite info classification.mcc
|
|
180
230
|
evalsuite benchmark --quick
|
|
181
231
|
```
|
|
@@ -185,24 +235,27 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
185
235
|
|
|
186
236
|
## Performance
|
|
187
237
|
|
|
188
|
-
Benchmarked against
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
| Case | n | EvalSuite (ms) |
|
|
193
|
-
| --- | ---: |
|
|
194
|
-
| 8 binary label metrics via `evaluate()` | 1,000 |
|
|
195
|
-
|
|
|
196
|
-
|
|
|
197
|
-
|
|
|
198
|
-
|
|
|
199
|
-
|
|
|
200
|
-
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
238
|
+
Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
239
|
+
Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
|
|
240
|
+
1.4e-14).
|
|
241
|
+
|
|
242
|
+
| Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
|
|
243
|
+
| --- | ---: | --- | ---: | ---: | ---: |
|
|
244
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
|
|
245
|
+
| macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
|
|
246
|
+
| ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
|
|
247
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
|
|
248
|
+
| sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
|
|
249
|
+
| calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
|
|
250
|
+
| decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
|
|
251
|
+
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
252
|
+
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
253
|
+
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
254
|
+
|
|
255
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
256
|
+
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
257
|
+
below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
|
|
258
|
+
your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
|
|
206
259
|
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
207
260
|
|
|
208
261
|
## Metrics in this release
|
|
@@ -213,6 +266,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
|
|
|
213
266
|
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
214
267
|
calibration curve and expected calibration error.
|
|
215
268
|
|
|
269
|
+
**Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
|
|
270
|
+
negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
|
|
271
|
+
diagnostic report with confidence intervals for all of them.
|
|
272
|
+
|
|
273
|
+
**Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
|
|
274
|
+
and intercept, Hosmer–Lemeshow test.
|
|
275
|
+
|
|
276
|
+
**Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
|
|
277
|
+
paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
|
|
278
|
+
Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
|
|
279
|
+
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
280
|
+
Benjamini–Yekutieli).
|
|
281
|
+
|
|
216
282
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
217
283
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
218
284
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -7,10 +7,10 @@
|
|
|
7
7
|
|
|
8
8
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
9
9
|
|
|
10
|
-
EvalSuite brings classification and
|
|
10
|
+
EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
|
|
11
11
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status: stable (0.1
|
|
13
|
+
> **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -82,6 +82,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
82
82
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
83
83
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
84
84
|
|
|
85
|
+
## Clinical evaluation
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
|
|
89
|
+
print(report)
|
|
90
|
+
# Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
|
|
91
|
+
# diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
|
|
92
|
+
|
|
93
|
+
es.lr_positive(y_true, y_pred)
|
|
94
|
+
es.youden_j(y_true, y_pred)
|
|
95
|
+
|
|
96
|
+
dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
|
|
97
|
+
dca.useful_range() # thresholds where the model beats both
|
|
98
|
+
es.plot.decision_curve(y_true, {"model": y_prob})
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Ratios that divide by zero are `inf` or NaN with a warning, never 0.
|
|
102
|
+
|
|
103
|
+
## Calibration
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
|
|
107
|
+
es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
|
|
108
|
+
es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
|
|
109
|
+
es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Statistical tests
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
|
|
116
|
+
es.paired_t_test(fold_scores_a, fold_scores_b)
|
|
117
|
+
es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
|
|
118
|
+
es.mann_whitney_test(a, b) # with rank-biserial r
|
|
119
|
+
es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
|
|
120
|
+
es.kruskal_wallis_test(g1, g2, g3)
|
|
121
|
+
es.shapiro_wilk_test(residuals)
|
|
122
|
+
es.chi_square_test(table) # with Cramér's V
|
|
123
|
+
es.fisher_exact_test([[8, 2], [1, 5]])
|
|
124
|
+
es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
128
|
+
checked against SciPy and statsmodels in the test suite.
|
|
129
|
+
|
|
85
130
|
## Classification report
|
|
86
131
|
|
|
87
132
|
```python
|
|
@@ -123,7 +168,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
|
123
168
|
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
124
169
|
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
125
170
|
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
126
|
-
evalsuite
|
|
171
|
+
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
172
|
+
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
173
|
+
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
174
|
+
evalsuite metrics --category clinical
|
|
127
175
|
evalsuite info classification.mcc
|
|
128
176
|
evalsuite benchmark --quick
|
|
129
177
|
```
|
|
@@ -133,24 +181,27 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
133
181
|
|
|
134
182
|
## Performance
|
|
135
183
|
|
|
136
|
-
Benchmarked against
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
| Case | n | EvalSuite (ms) |
|
|
141
|
-
| --- | ---: |
|
|
142
|
-
| 8 binary label metrics via `evaluate()` | 1,000 |
|
|
143
|
-
|
|
|
144
|
-
|
|
|
145
|
-
|
|
|
146
|
-
|
|
|
147
|
-
|
|
|
148
|
-
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
184
|
+
Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
185
|
+
Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
|
|
186
|
+
1.4e-14).
|
|
187
|
+
|
|
188
|
+
| Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
|
|
189
|
+
| --- | ---: | --- | ---: | ---: | ---: |
|
|
190
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
|
|
191
|
+
| macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
|
|
192
|
+
| ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
|
|
193
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
|
|
194
|
+
| sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
|
|
195
|
+
| calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
|
|
196
|
+
| decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
|
|
197
|
+
| diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
|
|
198
|
+
| Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
|
|
199
|
+
| Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
|
|
200
|
+
|
|
201
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
|
|
202
|
+
of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
|
|
203
|
+
below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
|
|
204
|
+
your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
|
|
154
205
|
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
155
206
|
|
|
156
207
|
## Metrics in this release
|
|
@@ -161,6 +212,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
|
|
|
161
212
|
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
162
213
|
calibration curve and expected calibration error.
|
|
163
214
|
|
|
215
|
+
**Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
|
|
216
|
+
negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
|
|
217
|
+
diagnostic report with confidence intervals for all of them.
|
|
218
|
+
|
|
219
|
+
**Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
|
|
220
|
+
and intercept, Hosmer–Lemeshow test.
|
|
221
|
+
|
|
222
|
+
**Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
|
|
223
|
+
paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
|
|
224
|
+
Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
|
|
225
|
+
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
226
|
+
Benjamini–Yekutieli).
|
|
227
|
+
|
|
164
228
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
165
229
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
166
230
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -12,10 +12,12 @@ license-files = ["LICENSE"]
|
|
|
12
12
|
requires-python = ">=3.9"
|
|
13
13
|
authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
15
|
-
keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
15
|
+
keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
|
+
"Intended Audience :: Healthcare Industry",
|
|
20
|
+
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
19
21
|
"Intended Audience :: Developers",
|
|
20
22
|
"Operating System :: OS Independent",
|
|
21
23
|
"Programming Language :: Python :: 3",
|
|
@@ -5,8 +5,16 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import classification, plot, regression, stats
|
|
8
|
+
from . import calibration, classification, clinical, plot, regression, stats
|
|
9
9
|
from .api import evaluate
|
|
10
|
+
from .calibration import (
|
|
11
|
+
CalibrationReport,
|
|
12
|
+
calibration_intercept,
|
|
13
|
+
calibration_report,
|
|
14
|
+
calibration_slope,
|
|
15
|
+
hosmer_lemeshow,
|
|
16
|
+
maximum_calibration_error,
|
|
17
|
+
)
|
|
10
18
|
from .classification import (
|
|
11
19
|
accuracy,
|
|
12
20
|
average_precision,
|
|
@@ -31,6 +39,19 @@ from .classification import (
|
|
|
31
39
|
specificity,
|
|
32
40
|
top_k_accuracy,
|
|
33
41
|
)
|
|
42
|
+
from .clinical import (
|
|
43
|
+
DecisionCurve,
|
|
44
|
+
DiagnosticReport,
|
|
45
|
+
decision_curve,
|
|
46
|
+
diagnostic_odds_ratio,
|
|
47
|
+
diagnostic_report,
|
|
48
|
+
lr_negative,
|
|
49
|
+
lr_positive,
|
|
50
|
+
net_benefit,
|
|
51
|
+
ppv,
|
|
52
|
+
sensitivity,
|
|
53
|
+
youden_j,
|
|
54
|
+
)
|
|
34
55
|
from .core.exceptions import (
|
|
35
56
|
EvalSuiteError,
|
|
36
57
|
InputValidationError,
|
|
@@ -69,19 +90,58 @@ from .stats import (
|
|
|
69
90
|
accuracy_ci,
|
|
70
91
|
adjust_pvalues,
|
|
71
92
|
bootstrap_ci,
|
|
93
|
+
chi_square_test,
|
|
72
94
|
cliffs_delta,
|
|
73
95
|
cohens_d,
|
|
74
96
|
compare,
|
|
97
|
+
cramers_v,
|
|
75
98
|
delong_test,
|
|
99
|
+
fisher_exact_test,
|
|
100
|
+
friedman_test,
|
|
76
101
|
hedges_g,
|
|
102
|
+
kruskal_wallis_test,
|
|
103
|
+
mann_whitney_test,
|
|
77
104
|
mcnemar_test,
|
|
78
105
|
paired_bootstrap_test,
|
|
106
|
+
paired_t_test,
|
|
79
107
|
proportion_ci,
|
|
80
108
|
roc_auc_ci,
|
|
109
|
+
shapiro_wilk_test,
|
|
110
|
+
t_test,
|
|
111
|
+
wilcoxon_test,
|
|
81
112
|
)
|
|
82
113
|
from .version import __version__
|
|
83
114
|
|
|
84
115
|
__all__ = [
|
|
116
|
+
"CalibrationReport",
|
|
117
|
+
"calibration_report",
|
|
118
|
+
"calibration",
|
|
119
|
+
"clinical",
|
|
120
|
+
"calibration_intercept",
|
|
121
|
+
"calibration_slope",
|
|
122
|
+
"hosmer_lemeshow",
|
|
123
|
+
"maximum_calibration_error",
|
|
124
|
+
"DecisionCurve",
|
|
125
|
+
"DiagnosticReport",
|
|
126
|
+
"decision_curve",
|
|
127
|
+
"diagnostic_odds_ratio",
|
|
128
|
+
"diagnostic_report",
|
|
129
|
+
"lr_negative",
|
|
130
|
+
"lr_positive",
|
|
131
|
+
"net_benefit",
|
|
132
|
+
"ppv",
|
|
133
|
+
"sensitivity",
|
|
134
|
+
"youden_j",
|
|
135
|
+
"chi_square_test",
|
|
136
|
+
"cramers_v",
|
|
137
|
+
"fisher_exact_test",
|
|
138
|
+
"friedman_test",
|
|
139
|
+
"kruskal_wallis_test",
|
|
140
|
+
"mann_whitney_test",
|
|
141
|
+
"paired_t_test",
|
|
142
|
+
"shapiro_wilk_test",
|
|
143
|
+
"t_test",
|
|
144
|
+
"wilcoxon_test",
|
|
85
145
|
"plot",
|
|
86
146
|
"expected_calibration_error",
|
|
87
147
|
"calibration_curve",
|