evalsuite-python 0.1.2__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/CHANGELOG.md +32 -0
  2. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/PKG-INFO +89 -23
  3. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/README.md +85 -21
  4. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/pyproject.toml +3 -1
  5. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/__init__.py +61 -1
  6. evalsuite_python-0.2.1/src/evalsuite/benchmarks.py +417 -0
  7. evalsuite_python-0.2.1/src/evalsuite/calibration.py +386 -0
  8. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/cli/main.py +97 -8
  9. evalsuite_python-0.2.1/src/evalsuite/clinical/__init__.py +18 -0
  10. evalsuite_python-0.2.1/src/evalsuite/clinical/metrics.py +339 -0
  11. evalsuite_python-0.2.1/src/evalsuite/clinical/report.py +360 -0
  12. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/validation.py +14 -2
  13. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/plot.py +46 -1
  14. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/__init__.py +22 -1
  15. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/compare.py +1 -1
  16. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/effect.py +46 -8
  17. evalsuite_python-0.2.1/src/evalsuite/stats/hypothesis.py +321 -0
  18. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/version.py +1 -1
  19. evalsuite_python-0.2.1/tests/clinical/test_outputs.py +99 -0
  20. evalsuite_python-0.2.1/tests/clinical/test_v020.py +269 -0
  21. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_benchmarks.py +19 -3
  22. evalsuite_python-0.2.1/tests/unit/__init__.py +0 -0
  23. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/unit/test_edges.py +18 -0
  24. evalsuite_python-0.1.2/src/evalsuite/benchmarks.py +0 -272
  25. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/.gitignore +0 -0
  26. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/CONTRIBUTING.md +0 -0
  27. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/LICENSE +0 -0
  28. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/__main__.py +0 -0
  29. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/api.py +0 -0
  30. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/__init__.py +0 -0
  31. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/_common.py +0 -0
  32. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/classification/metrics.py +0 -0
  33. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/cli/__init__.py +0 -0
  34. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/__init__.py +0 -0
  35. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/context.py +0 -0
  36. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/exceptions.py +0 -0
  37. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/export.py +0 -0
  38. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/registry.py +0 -0
  39. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/result.py +0 -0
  40. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/core/types.py +0 -0
  41. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/py.typed +0 -0
  42. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/regression/__init__.py +0 -0
  43. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/regression/metrics.py +0 -0
  44. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/reporting.py +0 -0
  45. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/_resolve.py +0 -0
  46. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/intervals.py +0 -0
  47. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/paired.py +0 -0
  48. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/src/evalsuite/stats/results.py +0 -0
  49. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/__init__.py +0 -0
  50. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/classification/__init__.py +0 -0
  51. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/classification/test_against_sklearn.py +0 -0
  52. {evalsuite_python-0.1.2/tests/integration → evalsuite_python-0.2.1/tests/clinical}/__init__.py +0 -0
  53. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/conftest.py +0 -0
  54. {evalsuite_python-0.1.2/tests/output → evalsuite_python-0.2.1/tests/integration}/__init__.py +0 -0
  55. {evalsuite_python-0.1.2/tests/regression → evalsuite_python-0.2.1/tests/output}/__init__.py +0 -0
  56. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_cli.py +0 -0
  57. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_plot.py +0 -0
  58. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/output/test_reporting.py +0 -0
  59. {evalsuite_python-0.1.2/tests/stats → evalsuite_python-0.2.1/tests/regression}/__init__.py +0 -0
  60. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/regression/test_against_sklearn.py +0 -0
  61. {evalsuite_python-0.1.2/tests/unit → evalsuite_python-0.2.1/tests/stats}/__init__.py +0 -0
  62. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_branches.py +0 -0
  63. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_compare.py +0 -0
  64. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/stats/test_reference.py +0 -0
  65. {evalsuite_python-0.1.2 → evalsuite_python-0.2.1}/tests/unit/test_core.py +0 -0
@@ -6,6 +6,38 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.2.1] - 2026-10-09
10
+
11
+ ### Added
12
+ - Benchmarks for the v0.2.0 functions (`evalsuite benchmark --suite clinical`): diagnostic metrics and
13
+ report, calibration slope and intercept, decision curves, t-test, Mann–Whitney, Cramér's V and Hochberg,
14
+ each against its reference (scikit-learn, statsmodels, SciPy or the textbook NumPy loop). Benchmark rows
15
+ now name their reference library.
16
+
17
+ ### Changed
18
+ - Faster label handling: integer class labels are found with one marking pass instead of a sort
19
+ (8 label metrics at 1M samples: 34× faster than scikit-learn, up from 10×; macro F1 5.8×, up from 1.6×).
20
+ - Decision curves sort the risks once and use cumulative sums (O((n + k) log n), no n × k matrix).
21
+
22
+ ## [0.2.0] - 2026-10-08
23
+
24
+ Clinical and statistical evaluation (the v0.2.0 roadmap).
25
+
26
+ ### Added
27
+ - Clinical: `sensitivity`, `ppv`, `lr_positive`, `lr_negative`, `diagnostic_odds_ratio`, `youden_j`,
28
+ `net_benefit`; `diagnostic_report` with confidence intervals for every measure (Wilson or Clopper–Pearson
29
+ for proportions, log method for likelihood ratios, Woolf for the DOR, Wald for Youden's J);
30
+ `decision_curve` (net benefit, treat all, treat none, useful threshold range).
31
+ - Calibration: `maximum_calibration_error`, `calibration_slope`, `calibration_intercept`,
32
+ `hosmer_lemeshow`, `calibration_report`.
33
+ - Statistical tests: `t_test` (Welch/Student), `paired_t_test`, `mann_whitney_test`, `wilcoxon_test`,
34
+ `kruskal_wallis_test`, `friedman_test`, `shapiro_wilk_test`, `chi_square_test`, `fisher_exact_test`, each
35
+ with an effect size; `cramers_v` (optional bias correction); Hochberg correction in `adjust_pvalues` and
36
+ `compare`.
37
+ - `es.plot.decision_curve`; CLI commands `evalsuite diagnostic`, `evalsuite calibration` and
38
+ `evalsuite plot decision`.
39
+ - Validated against statsmodels (GLM, Table2x2, proportion_confint, multipletests, CompareMeans) and SciPy.
40
+
9
41
  ## [0.1.2] - 2026-10-08
10
42
 
11
43
  ### Changed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.1.2
3
+ Version: 0.2.1
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -11,9 +11,10 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
11
11
  Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
- Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
14
+ Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
15
15
  Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Intended Audience :: Developers
17
+ Classifier: Intended Audience :: Healthcare Industry
17
18
  Classifier: Intended Audience :: Science/Research
18
19
  Classifier: Operating System :: OS Independent
19
20
  Classifier: Programming Language :: Python :: 3
@@ -26,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
26
27
  Classifier: Programming Language :: Python :: 3.14
27
28
  Classifier: Topic :: Scientific/Engineering
28
29
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
30
+ Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
29
31
  Classifier: Typing :: Typed
30
32
  Requires-Python: >=3.9
31
33
  Requires-Dist: numpy>=1.22
@@ -59,10 +61,10 @@ Description-Content-Type: text/markdown
59
61
 
60
62
  **Unified, reproducible evaluation for machine learning and research.**
61
63
 
62
- EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
64
+ EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
63
65
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
64
66
 
65
- > **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
67
+ > **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
66
68
 
67
69
  ## Installation
68
70
 
@@ -134,6 +136,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
134
136
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
135
137
  p-values are adjusted for multiple comparisons (Holm by default).
136
138
 
139
+ ## Clinical evaluation
140
+
141
+ ```python
142
+ report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
143
+ print(report)
144
+ # Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
145
+ # diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
146
+
147
+ es.lr_positive(y_true, y_pred)
148
+ es.youden_j(y_true, y_pred)
149
+
150
+ dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
151
+ dca.useful_range() # thresholds where the model beats both
152
+ es.plot.decision_curve(y_true, {"model": y_prob})
153
+ ```
154
+
155
+ Ratios that divide by zero are `inf` or NaN with a warning, never 0.
156
+
157
+ ## Calibration
158
+
159
+ ```python
160
+ es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
161
+ es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
162
+ es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
163
+ es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
164
+ ```
165
+
166
+ ## Statistical tests
167
+
168
+ ```python
169
+ es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
170
+ es.paired_t_test(fold_scores_a, fold_scores_b)
171
+ es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
172
+ es.mann_whitney_test(a, b) # with rank-biserial r
173
+ es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
174
+ es.kruskal_wallis_test(g1, g2, g3)
175
+ es.shapiro_wilk_test(residuals)
176
+ es.chi_square_test(table) # with Cramér's V
177
+ es.fisher_exact_test([[8, 2], [1, 5]])
178
+ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
179
+ ```
180
+
181
+ Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
182
+ checked against SciPy and statsmodels in the test suite.
183
+
137
184
  ## Classification report
138
185
 
139
186
  ```python
@@ -175,7 +222,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
175
222
  evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
176
223
  --prob lr=p_lr --prob rf=p_rf --plot comparison.png
177
224
  evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
178
- evalsuite metrics --category classification
225
+ evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
226
+ evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
227
+ evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
228
+ evalsuite metrics --category clinical
179
229
  evalsuite info classification.mcc
180
230
  evalsuite benchmark --quick
181
231
  ```
@@ -185,24 +235,27 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
185
235
 
186
236
  ## Performance
187
237
 
188
- Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
189
- scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
190
- (largest difference 1.1e-16).
191
-
192
- | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
193
- | --- | ---: | ---: | ---: | ---: | ---: |
194
- | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
195
- | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
196
- | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
197
- | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
198
- | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
199
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
200
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
201
-
202
- `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
203
- speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
204
- infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
205
- table and notes in
238
+ Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
239
+ Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
240
+ 1.4e-14).
241
+
242
+ | Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
243
+ | --- | ---: | --- | ---: | ---: | ---: |
244
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
245
+ | macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
246
+ | ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
247
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
248
+ | sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
249
+ | calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
250
+ | decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
251
+ | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
252
+ | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
253
+ | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
254
+
255
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
256
+ of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
257
+ below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
258
+ your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
206
259
  [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
207
260
 
208
261
  ## Metrics in this release
@@ -213,6 +266,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
213
266
  one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
214
267
  calibration curve and expected calibration error.
215
268
 
269
+ **Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
270
+ negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
271
+ diagnostic report with confidence intervals for all of them.
272
+
273
+ **Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
274
+ and intercept, Hosmer–Lemeshow test.
275
+
276
+ **Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
277
+ paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
278
+ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
279
+ delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
280
+ Benjamini–Yekutieli).
281
+
216
282
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
217
283
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
218
284
  loss, Huber loss, relative absolute error, relative squared error.
@@ -7,10 +7,10 @@
7
7
 
8
8
  **Unified, reproducible evaluation for machine learning and research.**
9
9
 
10
- EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
10
+ EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
11
11
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
12
 
13
- > **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
13
+ > **Status: stable (0.2.1).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -82,6 +82,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
82
82
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
83
83
  p-values are adjusted for multiple comparisons (Holm by default).
84
84
 
85
+ ## Clinical evaluation
86
+
87
+ ```python
88
+ report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
89
+ print(report)
90
+ # Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
91
+ # diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
92
+
93
+ es.lr_positive(y_true, y_pred)
94
+ es.youden_j(y_true, y_pred)
95
+
96
+ dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
97
+ dca.useful_range() # thresholds where the model beats both
98
+ es.plot.decision_curve(y_true, {"model": y_prob})
99
+ ```
100
+
101
+ Ratios that divide by zero are `inf` or NaN with a warning, never 0.
102
+
103
+ ## Calibration
104
+
105
+ ```python
106
+ es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
107
+ es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
108
+ es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
109
+ es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
110
+ ```
111
+
112
+ ## Statistical tests
113
+
114
+ ```python
115
+ es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
116
+ es.paired_t_test(fold_scores_a, fold_scores_b)
117
+ es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
118
+ es.mann_whitney_test(a, b) # with rank-biserial r
119
+ es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
120
+ es.kruskal_wallis_test(g1, g2, g3)
121
+ es.shapiro_wilk_test(residuals)
122
+ es.chi_square_test(table) # with Cramér's V
123
+ es.fisher_exact_test([[8, 2], [1, 5]])
124
+ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
125
+ ```
126
+
127
+ Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
128
+ checked against SciPy and statsmodels in the test suite.
129
+
85
130
  ## Classification report
86
131
 
87
132
  ```python
@@ -123,7 +168,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
123
168
  evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
124
169
  --prob lr=p_lr --prob rf=p_rf --plot comparison.png
125
170
  evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
126
- evalsuite metrics --category classification
171
+ evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
172
+ evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
173
+ evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
174
+ evalsuite metrics --category clinical
127
175
  evalsuite info classification.mcc
128
176
  evalsuite benchmark --quick
129
177
  ```
@@ -133,24 +181,27 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
133
181
 
134
182
  ## Performance
135
183
 
136
- Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
137
- scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
138
- (largest difference 1.1e-16).
139
-
140
- | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
141
- | --- | ---: | ---: | ---: | ---: | ---: |
142
- | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
143
- | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
144
- | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
145
- | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
146
- | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
147
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
148
- | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
149
-
150
- `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
151
- speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
152
- infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
153
- table and notes in
184
+ Benchmarked against reference implementations on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
185
+ Linux x86_64). Every result agrees with the reference to floating-point rounding (largest difference
186
+ 1.4e-14).
187
+
188
+ | Case | n | Reference | EvalSuite (ms) | Reference (ms) | Speed-up |
189
+ | --- | ---: | --- | ---: | ---: | ---: |
190
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | scikit-learn | 30.2 | 1020.2 | **33.8×** |
191
+ | macro F1, 10 classes | 1,000,000 | scikit-learn | 22.4 | 128.7 | **5.8×** |
192
+ | ROC AUC, binary | 1,000,000 | scikit-learn | 173.2 | 300.6 | **1.7×** |
193
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | scikit-learn | 19.1 | 9.7 | 0.51× |
194
+ | sensitivity, specificity, LR+, LR− | 1,000,000 | scikit-learn | 66.2 | 392.8 | **5.9×** |
195
+ | calibration slope and intercept | 1,000,000 | statsmodels | 178.0 | 1014.8 | **5.7×** |
196
+ | decision curve, 99 thresholds | 1,000,000 | NumPy loop | 155.2 | 174.3 | **1.1×** |
197
+ | diagnostic report (7 CIs) | 1,000,000 | statsmodels | 16.6 | 4.6 | 0.28× |
198
+ | Welch t-test | 1,000,000 | SciPy | 14.5 | 7.3 | 0.51× |
199
+ | Hochberg correction | 1,000,000 | statsmodels | 75.4 | 81.9 | **1.1×** |
200
+
201
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where most
202
+ of the speed-up comes from. Hypothesis tests use SciPy underneath, so they match its speed at best; rows
203
+ below 1× pay for input validation and the extra intervals and effect sizes EvalSuite reports. Reproduce on
204
+ your machine with `evalsuite benchmark`; full table (1k, 100k and 1M samples, peak memory) and notes in
154
205
  [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
155
206
 
156
207
  ## Metrics in this release
@@ -161,6 +212,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
161
212
  one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
162
213
  calibration curve and expected calibration error.
163
214
 
215
+ **Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
216
+ negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
217
+ diagnostic report with confidence intervals for all of them.
218
+
219
+ **Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
220
+ and intercept, Hosmer–Lemeshow test.
221
+
222
+ **Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
223
+ paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
224
+ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
225
+ delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
226
+ Benjamini–Yekutieli).
227
+
164
228
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
165
229
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
166
230
  loss, Huber loss, relative absolute error, relative squared error.
@@ -12,10 +12,12 @@ license-files = ["LICENSE"]
12
12
  requires-python = ">=3.9"
13
13
  authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
15
- keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
15
+ keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
17
  "Development Status :: 5 - Production/Stable",
18
18
  "Intended Audience :: Science/Research",
19
+ "Intended Audience :: Healthcare Industry",
20
+ "Topic :: Scientific/Engineering :: Medical Science Apps.",
19
21
  "Intended Audience :: Developers",
20
22
  "Operating System :: OS Independent",
21
23
  "Programming Language :: Python :: 3",
@@ -5,8 +5,16 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import classification, plot, regression, stats
8
+ from . import calibration, classification, clinical, plot, regression, stats
9
9
  from .api import evaluate
10
+ from .calibration import (
11
+ CalibrationReport,
12
+ calibration_intercept,
13
+ calibration_report,
14
+ calibration_slope,
15
+ hosmer_lemeshow,
16
+ maximum_calibration_error,
17
+ )
10
18
  from .classification import (
11
19
  accuracy,
12
20
  average_precision,
@@ -31,6 +39,19 @@ from .classification import (
31
39
  specificity,
32
40
  top_k_accuracy,
33
41
  )
42
+ from .clinical import (
43
+ DecisionCurve,
44
+ DiagnosticReport,
45
+ decision_curve,
46
+ diagnostic_odds_ratio,
47
+ diagnostic_report,
48
+ lr_negative,
49
+ lr_positive,
50
+ net_benefit,
51
+ ppv,
52
+ sensitivity,
53
+ youden_j,
54
+ )
34
55
  from .core.exceptions import (
35
56
  EvalSuiteError,
36
57
  InputValidationError,
@@ -69,19 +90,58 @@ from .stats import (
69
90
  accuracy_ci,
70
91
  adjust_pvalues,
71
92
  bootstrap_ci,
93
+ chi_square_test,
72
94
  cliffs_delta,
73
95
  cohens_d,
74
96
  compare,
97
+ cramers_v,
75
98
  delong_test,
99
+ fisher_exact_test,
100
+ friedman_test,
76
101
  hedges_g,
102
+ kruskal_wallis_test,
103
+ mann_whitney_test,
77
104
  mcnemar_test,
78
105
  paired_bootstrap_test,
106
+ paired_t_test,
79
107
  proportion_ci,
80
108
  roc_auc_ci,
109
+ shapiro_wilk_test,
110
+ t_test,
111
+ wilcoxon_test,
81
112
  )
82
113
  from .version import __version__
83
114
 
84
115
  __all__ = [
116
+ "CalibrationReport",
117
+ "calibration_report",
118
+ "calibration",
119
+ "clinical",
120
+ "calibration_intercept",
121
+ "calibration_slope",
122
+ "hosmer_lemeshow",
123
+ "maximum_calibration_error",
124
+ "DecisionCurve",
125
+ "DiagnosticReport",
126
+ "decision_curve",
127
+ "diagnostic_odds_ratio",
128
+ "diagnostic_report",
129
+ "lr_negative",
130
+ "lr_positive",
131
+ "net_benefit",
132
+ "ppv",
133
+ "sensitivity",
134
+ "youden_j",
135
+ "chi_square_test",
136
+ "cramers_v",
137
+ "fisher_exact_test",
138
+ "friedman_test",
139
+ "kruskal_wallis_test",
140
+ "mann_whitney_test",
141
+ "paired_t_test",
142
+ "shapiro_wilk_test",
143
+ "t_test",
144
+ "wilcoxon_test",
85
145
  "plot",
86
146
  "expected_calibration_error",
87
147
  "calibration_curve",