evalsuite-python 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/CHANGELOG.md +19 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/PKG-INFO +68 -5
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/README.md +64 -3
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/pyproject.toml +3 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/__init__.py +61 -1
- evalsuite_python-0.2.0/src/evalsuite/calibration.py +386 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/cli/main.py +85 -6
- evalsuite_python-0.2.0/src/evalsuite/clinical/__init__.py +18 -0
- evalsuite_python-0.2.0/src/evalsuite/clinical/metrics.py +330 -0
- evalsuite_python-0.2.0/src/evalsuite/clinical/report.py +360 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/plot.py +46 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/__init__.py +22 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/compare.py +1 -1
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/effect.py +45 -7
- evalsuite_python-0.2.0/src/evalsuite/stats/hypothesis.py +321 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.2.0/tests/clinical/test_outputs.py +99 -0
- evalsuite_python-0.2.0/tests/clinical/test_v020.py +269 -0
- evalsuite_python-0.2.0/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/.gitignore +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/LICENSE +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/__main__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/api.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/benchmarks.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/metrics.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/cli/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/export.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/result.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/validation.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/regression/metrics.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/reporting.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/_resolve.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.2/tests/integration → evalsuite_python-0.2.0/tests/clinical}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/conftest.py +0 -0
- {evalsuite_python-0.1.2/tests/output → evalsuite_python-0.2.0/tests/integration}/__init__.py +0 -0
- {evalsuite_python-0.1.2/tests/regression → evalsuite_python-0.2.0/tests/output}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_benchmarks.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_cli.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_plot.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_reporting.py +0 -0
- {evalsuite_python-0.1.2/tests/stats → evalsuite_python-0.2.0/tests/regression}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.2/tests/unit → evalsuite_python-0.2.0/tests/stats}/__init__.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/unit/test_core.py +0 -0
- {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/unit/test_edges.py +0 -0
|
@@ -6,6 +6,25 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.2.0] - 2026-10-08
|
|
10
|
+
|
|
11
|
+
Clinical and statistical evaluation (the v0.2.0 roadmap).
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
- Clinical: `sensitivity`, `ppv`, `lr_positive`, `lr_negative`, `diagnostic_odds_ratio`, `youden_j`,
|
|
15
|
+
`net_benefit`; `diagnostic_report` with confidence intervals for every measure (Wilson or Clopper–Pearson
|
|
16
|
+
for proportions, log method for likelihood ratios, Woolf for the DOR, Wald for Youden's J);
|
|
17
|
+
`decision_curve` (net benefit, treat all, treat none, useful threshold range).
|
|
18
|
+
- Calibration: `maximum_calibration_error`, `calibration_slope`, `calibration_intercept`,
|
|
19
|
+
`hosmer_lemeshow`, `calibration_report`.
|
|
20
|
+
- Statistical tests: `t_test` (Welch/Student), `paired_t_test`, `mann_whitney_test`, `wilcoxon_test`,
|
|
21
|
+
`kruskal_wallis_test`, `friedman_test`, `shapiro_wilk_test`, `chi_square_test`, `fisher_exact_test`, each
|
|
22
|
+
with an effect size; `cramers_v` (optional bias correction); Hochberg correction in `adjust_pvalues` and
|
|
23
|
+
`compare`.
|
|
24
|
+
- `es.plot.decision_curve`; CLI commands `evalsuite diagnostic`, `evalsuite calibration` and
|
|
25
|
+
`evalsuite plot decision`.
|
|
26
|
+
- Validated against statsmodels (GLM, Table2x2, proportion_confint, multipletests, CompareMeans) and SciPy.
|
|
27
|
+
|
|
9
28
|
## [0.1.2] - 2026-10-08
|
|
10
29
|
|
|
11
30
|
### Changed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -11,9 +11,10 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
|
11
11
|
Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
|
-
Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
|
|
14
|
+
Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
|
|
15
15
|
Classifier: Development Status :: 5 - Production/Stable
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Intended Audience :: Healthcare Industry
|
|
17
18
|
Classifier: Intended Audience :: Science/Research
|
|
18
19
|
Classifier: Operating System :: OS Independent
|
|
19
20
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -26,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
26
27
|
Classifier: Programming Language :: Python :: 3.14
|
|
27
28
|
Classifier: Topic :: Scientific/Engineering
|
|
28
29
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
30
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
29
31
|
Classifier: Typing :: Typed
|
|
30
32
|
Requires-Python: >=3.9
|
|
31
33
|
Requires-Dist: numpy>=1.22
|
|
@@ -59,10 +61,10 @@ Description-Content-Type: text/markdown
|
|
|
59
61
|
|
|
60
62
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
61
63
|
|
|
62
|
-
EvalSuite brings classification and
|
|
64
|
+
EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
|
|
63
65
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
64
66
|
|
|
65
|
-
> **Status: stable (0.
|
|
67
|
+
> **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
|
|
66
68
|
|
|
67
69
|
## Installation
|
|
68
70
|
|
|
@@ -134,6 +136,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
134
136
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
135
137
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
136
138
|
|
|
139
|
+
## Clinical evaluation
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
|
|
143
|
+
print(report)
|
|
144
|
+
# Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
|
|
145
|
+
# diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
|
|
146
|
+
|
|
147
|
+
es.lr_positive(y_true, y_pred)
|
|
148
|
+
es.youden_j(y_true, y_pred)
|
|
149
|
+
|
|
150
|
+
dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
|
|
151
|
+
dca.useful_range() # thresholds where the model beats both
|
|
152
|
+
es.plot.decision_curve(y_true, {"model": y_prob})
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Ratios that divide by zero are `inf` or NaN with a warning, never 0.
|
|
156
|
+
|
|
157
|
+
## Calibration
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
|
|
161
|
+
es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
|
|
162
|
+
es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
|
|
163
|
+
es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Statistical tests
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
|
|
170
|
+
es.paired_t_test(fold_scores_a, fold_scores_b)
|
|
171
|
+
es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
|
|
172
|
+
es.mann_whitney_test(a, b) # with rank-biserial r
|
|
173
|
+
es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
|
|
174
|
+
es.kruskal_wallis_test(g1, g2, g3)
|
|
175
|
+
es.shapiro_wilk_test(residuals)
|
|
176
|
+
es.chi_square_test(table) # with Cramér's V
|
|
177
|
+
es.fisher_exact_test([[8, 2], [1, 5]])
|
|
178
|
+
es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
182
|
+
checked against SciPy and statsmodels in the test suite.
|
|
183
|
+
|
|
137
184
|
## Classification report
|
|
138
185
|
|
|
139
186
|
```python
|
|
@@ -175,7 +222,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
|
175
222
|
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
176
223
|
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
177
224
|
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
178
|
-
evalsuite
|
|
225
|
+
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
226
|
+
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
227
|
+
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
228
|
+
evalsuite metrics --category clinical
|
|
179
229
|
evalsuite info classification.mcc
|
|
180
230
|
evalsuite benchmark --quick
|
|
181
231
|
```
|
|
@@ -213,6 +263,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
|
|
|
213
263
|
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
214
264
|
calibration curve and expected calibration error.
|
|
215
265
|
|
|
266
|
+
**Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
|
|
267
|
+
negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
|
|
268
|
+
diagnostic report with confidence intervals for all of them.
|
|
269
|
+
|
|
270
|
+
**Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
|
|
271
|
+
and intercept, Hosmer–Lemeshow test.
|
|
272
|
+
|
|
273
|
+
**Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
|
|
274
|
+
paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
|
|
275
|
+
Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
|
|
276
|
+
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
277
|
+
Benjamini–Yekutieli).
|
|
278
|
+
|
|
216
279
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
217
280
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
218
281
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -7,10 +7,10 @@
|
|
|
7
7
|
|
|
8
8
|
**Unified, reproducible evaluation for machine learning and research.**
|
|
9
9
|
|
|
10
|
-
EvalSuite brings classification and
|
|
10
|
+
EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
|
|
11
11
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status: stable (0.
|
|
13
|
+
> **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -82,6 +82,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
82
82
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
83
83
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
84
84
|
|
|
85
|
+
## Clinical evaluation
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
|
|
89
|
+
print(report)
|
|
90
|
+
# Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
|
|
91
|
+
# diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
|
|
92
|
+
|
|
93
|
+
es.lr_positive(y_true, y_pred)
|
|
94
|
+
es.youden_j(y_true, y_pred)
|
|
95
|
+
|
|
96
|
+
dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
|
|
97
|
+
dca.useful_range() # thresholds where the model beats both
|
|
98
|
+
es.plot.decision_curve(y_true, {"model": y_prob})
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Ratios that divide by zero are `inf` or NaN with a warning, never 0.
|
|
102
|
+
|
|
103
|
+
## Calibration
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
|
|
107
|
+
es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
|
|
108
|
+
es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
|
|
109
|
+
es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Statistical tests
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
|
|
116
|
+
es.paired_t_test(fold_scores_a, fold_scores_b)
|
|
117
|
+
es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
|
|
118
|
+
es.mann_whitney_test(a, b) # with rank-biserial r
|
|
119
|
+
es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
|
|
120
|
+
es.kruskal_wallis_test(g1, g2, g3)
|
|
121
|
+
es.shapiro_wilk_test(residuals)
|
|
122
|
+
es.chi_square_test(table) # with Cramér's V
|
|
123
|
+
es.fisher_exact_test([[8, 2], [1, 5]])
|
|
124
|
+
es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
|
|
128
|
+
checked against SciPy and statsmodels in the test suite.
|
|
129
|
+
|
|
85
130
|
## Classification report
|
|
86
131
|
|
|
87
132
|
```python
|
|
@@ -123,7 +168,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
|
123
168
|
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
124
169
|
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
125
170
|
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
126
|
-
evalsuite
|
|
171
|
+
evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
|
|
172
|
+
evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
|
|
173
|
+
evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
|
|
174
|
+
evalsuite metrics --category clinical
|
|
127
175
|
evalsuite info classification.mcc
|
|
128
176
|
evalsuite benchmark --quick
|
|
129
177
|
```
|
|
@@ -161,6 +209,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
|
|
|
161
209
|
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
162
210
|
calibration curve and expected calibration error.
|
|
163
211
|
|
|
212
|
+
**Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
|
|
213
|
+
negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
|
|
214
|
+
diagnostic report with confidence intervals for all of them.
|
|
215
|
+
|
|
216
|
+
**Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
|
|
217
|
+
and intercept, Hosmer–Lemeshow test.
|
|
218
|
+
|
|
219
|
+
**Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
|
|
220
|
+
paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
|
|
221
|
+
Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
|
|
222
|
+
delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
|
|
223
|
+
Benjamini–Yekutieli).
|
|
224
|
+
|
|
164
225
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
165
226
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
166
227
|
loss, Huber loss, relative absolute error, relative squared error.
|
|
@@ -12,10 +12,12 @@ license-files = ["LICENSE"]
|
|
|
12
12
|
requires-python = ">=3.9"
|
|
13
13
|
authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
15
|
-
keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
15
|
+
keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
|
+
"Intended Audience :: Healthcare Industry",
|
|
20
|
+
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
19
21
|
"Intended Audience :: Developers",
|
|
20
22
|
"Operating System :: OS Independent",
|
|
21
23
|
"Programming Language :: Python :: 3",
|
|
@@ -5,8 +5,16 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import classification, plot, regression, stats
|
|
8
|
+
from . import calibration, classification, clinical, plot, regression, stats
|
|
9
9
|
from .api import evaluate
|
|
10
|
+
from .calibration import (
|
|
11
|
+
CalibrationReport,
|
|
12
|
+
calibration_intercept,
|
|
13
|
+
calibration_report,
|
|
14
|
+
calibration_slope,
|
|
15
|
+
hosmer_lemeshow,
|
|
16
|
+
maximum_calibration_error,
|
|
17
|
+
)
|
|
10
18
|
from .classification import (
|
|
11
19
|
accuracy,
|
|
12
20
|
average_precision,
|
|
@@ -31,6 +39,19 @@ from .classification import (
|
|
|
31
39
|
specificity,
|
|
32
40
|
top_k_accuracy,
|
|
33
41
|
)
|
|
42
|
+
from .clinical import (
|
|
43
|
+
DecisionCurve,
|
|
44
|
+
DiagnosticReport,
|
|
45
|
+
decision_curve,
|
|
46
|
+
diagnostic_odds_ratio,
|
|
47
|
+
diagnostic_report,
|
|
48
|
+
lr_negative,
|
|
49
|
+
lr_positive,
|
|
50
|
+
net_benefit,
|
|
51
|
+
ppv,
|
|
52
|
+
sensitivity,
|
|
53
|
+
youden_j,
|
|
54
|
+
)
|
|
34
55
|
from .core.exceptions import (
|
|
35
56
|
EvalSuiteError,
|
|
36
57
|
InputValidationError,
|
|
@@ -69,19 +90,58 @@ from .stats import (
|
|
|
69
90
|
accuracy_ci,
|
|
70
91
|
adjust_pvalues,
|
|
71
92
|
bootstrap_ci,
|
|
93
|
+
chi_square_test,
|
|
72
94
|
cliffs_delta,
|
|
73
95
|
cohens_d,
|
|
74
96
|
compare,
|
|
97
|
+
cramers_v,
|
|
75
98
|
delong_test,
|
|
99
|
+
fisher_exact_test,
|
|
100
|
+
friedman_test,
|
|
76
101
|
hedges_g,
|
|
102
|
+
kruskal_wallis_test,
|
|
103
|
+
mann_whitney_test,
|
|
77
104
|
mcnemar_test,
|
|
78
105
|
paired_bootstrap_test,
|
|
106
|
+
paired_t_test,
|
|
79
107
|
proportion_ci,
|
|
80
108
|
roc_auc_ci,
|
|
109
|
+
shapiro_wilk_test,
|
|
110
|
+
t_test,
|
|
111
|
+
wilcoxon_test,
|
|
81
112
|
)
|
|
82
113
|
from .version import __version__
|
|
83
114
|
|
|
84
115
|
__all__ = [
|
|
116
|
+
"CalibrationReport",
|
|
117
|
+
"calibration_report",
|
|
118
|
+
"calibration",
|
|
119
|
+
"clinical",
|
|
120
|
+
"calibration_intercept",
|
|
121
|
+
"calibration_slope",
|
|
122
|
+
"hosmer_lemeshow",
|
|
123
|
+
"maximum_calibration_error",
|
|
124
|
+
"DecisionCurve",
|
|
125
|
+
"DiagnosticReport",
|
|
126
|
+
"decision_curve",
|
|
127
|
+
"diagnostic_odds_ratio",
|
|
128
|
+
"diagnostic_report",
|
|
129
|
+
"lr_negative",
|
|
130
|
+
"lr_positive",
|
|
131
|
+
"net_benefit",
|
|
132
|
+
"ppv",
|
|
133
|
+
"sensitivity",
|
|
134
|
+
"youden_j",
|
|
135
|
+
"chi_square_test",
|
|
136
|
+
"cramers_v",
|
|
137
|
+
"fisher_exact_test",
|
|
138
|
+
"friedman_test",
|
|
139
|
+
"kruskal_wallis_test",
|
|
140
|
+
"mann_whitney_test",
|
|
141
|
+
"paired_t_test",
|
|
142
|
+
"shapiro_wilk_test",
|
|
143
|
+
"t_test",
|
|
144
|
+
"wilcoxon_test",
|
|
85
145
|
"plot",
|
|
86
146
|
"expected_calibration_error",
|
|
87
147
|
"calibration_curve",
|