evalsuite-python 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/CHANGELOG.md +19 -0
  2. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/PKG-INFO +68 -5
  3. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/README.md +64 -3
  4. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/pyproject.toml +3 -1
  5. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/__init__.py +61 -1
  6. evalsuite_python-0.2.0/src/evalsuite/calibration.py +386 -0
  7. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/cli/main.py +85 -6
  8. evalsuite_python-0.2.0/src/evalsuite/clinical/__init__.py +18 -0
  9. evalsuite_python-0.2.0/src/evalsuite/clinical/metrics.py +330 -0
  10. evalsuite_python-0.2.0/src/evalsuite/clinical/report.py +360 -0
  11. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/plot.py +46 -1
  12. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/__init__.py +22 -1
  13. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/compare.py +1 -1
  14. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/effect.py +45 -7
  15. evalsuite_python-0.2.0/src/evalsuite/stats/hypothesis.py +321 -0
  16. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/version.py +1 -1
  17. evalsuite_python-0.2.0/tests/clinical/test_outputs.py +99 -0
  18. evalsuite_python-0.2.0/tests/clinical/test_v020.py +269 -0
  19. evalsuite_python-0.2.0/tests/unit/__init__.py +0 -0
  20. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/.gitignore +0 -0
  21. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/CONTRIBUTING.md +0 -0
  22. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/LICENSE +0 -0
  23. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/__main__.py +0 -0
  24. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/api.py +0 -0
  25. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/benchmarks.py +0 -0
  26. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/__init__.py +0 -0
  27. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/_common.py +0 -0
  28. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/classification/metrics.py +0 -0
  29. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/cli/__init__.py +0 -0
  30. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/__init__.py +0 -0
  31. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/context.py +0 -0
  32. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/exceptions.py +0 -0
  33. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/export.py +0 -0
  34. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/registry.py +0 -0
  35. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/result.py +0 -0
  36. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/types.py +0 -0
  37. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/core/validation.py +0 -0
  38. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/py.typed +0 -0
  39. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/regression/__init__.py +0 -0
  40. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/regression/metrics.py +0 -0
  41. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/reporting.py +0 -0
  42. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/_resolve.py +0 -0
  43. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/intervals.py +0 -0
  44. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/paired.py +0 -0
  45. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/src/evalsuite/stats/results.py +0 -0
  46. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/__init__.py +0 -0
  47. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/classification/__init__.py +0 -0
  48. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/classification/test_against_sklearn.py +0 -0
  49. {evalsuite_python-0.1.2/tests/integration → evalsuite_python-0.2.0/tests/clinical}/__init__.py +0 -0
  50. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/conftest.py +0 -0
  51. {evalsuite_python-0.1.2/tests/output → evalsuite_python-0.2.0/tests/integration}/__init__.py +0 -0
  52. {evalsuite_python-0.1.2/tests/regression → evalsuite_python-0.2.0/tests/output}/__init__.py +0 -0
  53. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_benchmarks.py +0 -0
  54. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_cli.py +0 -0
  55. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_plot.py +0 -0
  56. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/output/test_reporting.py +0 -0
  57. {evalsuite_python-0.1.2/tests/stats → evalsuite_python-0.2.0/tests/regression}/__init__.py +0 -0
  58. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/regression/test_against_sklearn.py +0 -0
  59. {evalsuite_python-0.1.2/tests/unit → evalsuite_python-0.2.0/tests/stats}/__init__.py +0 -0
  60. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_branches.py +0 -0
  61. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_compare.py +0 -0
  62. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/stats/test_reference.py +0 -0
  63. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/unit/test_core.py +0 -0
  64. {evalsuite_python-0.1.2 → evalsuite_python-0.2.0}/tests/unit/test_edges.py +0 -0
@@ -6,6 +6,25 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.2.0] - 2026-10-08
10
+
11
+ Clinical and statistical evaluation (the v0.2.0 roadmap).
12
+
13
+ ### Added
14
+ - Clinical: `sensitivity`, `ppv`, `lr_positive`, `lr_negative`, `diagnostic_odds_ratio`, `youden_j`,
15
+ `net_benefit`; `diagnostic_report` with confidence intervals for every measure (Wilson or Clopper–Pearson
16
+ for proportions, log method for likelihood ratios, Woolf for the DOR, Wald for Youden's J);
17
+ `decision_curve` (net benefit, treat all, treat none, useful threshold range).
18
+ - Calibration: `maximum_calibration_error`, `calibration_slope`, `calibration_intercept`,
19
+ `hosmer_lemeshow`, `calibration_report`.
20
+ - Statistical tests: `t_test` (Welch/Student), `paired_t_test`, `mann_whitney_test`, `wilcoxon_test`,
21
+ `kruskal_wallis_test`, `friedman_test`, `shapiro_wilk_test`, `chi_square_test`, `fisher_exact_test`, each
22
+ with an effect size; `cramers_v` (optional bias correction); Hochberg correction in `adjust_pvalues` and
23
+ `compare`.
24
+ - `es.plot.decision_curve`; CLI commands `evalsuite diagnostic`, `evalsuite calibration` and
25
+ `evalsuite plot decision`.
26
+ - Validated against statsmodels (GLM, Table2x2, proportion_confint, multipletests, CompareMeans) and SciPy.
27
+
9
28
  ## [0.1.2] - 2026-10-08
10
29
 
11
30
  ### Changed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -11,9 +11,10 @@ Author: Manoj Kumar C S, Nikhil D Bharadwaj
11
11
  Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
- Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
14
+ Keywords: calibration,classification,clinical,diagnostic accuracy,evaluation,machine learning,metrics,regression,reproducibility,statistical tests,statistics
15
15
  Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Intended Audience :: Developers
17
+ Classifier: Intended Audience :: Healthcare Industry
17
18
  Classifier: Intended Audience :: Science/Research
18
19
  Classifier: Operating System :: OS Independent
19
20
  Classifier: Programming Language :: Python :: 3
@@ -26,6 +27,7 @@ Classifier: Programming Language :: Python :: 3.13
26
27
  Classifier: Programming Language :: Python :: 3.14
27
28
  Classifier: Topic :: Scientific/Engineering
28
29
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
30
+ Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
29
31
  Classifier: Typing :: Typed
30
32
  Requires-Python: >=3.9
31
33
  Requires-Dist: numpy>=1.22
@@ -59,10 +61,10 @@ Description-Content-Type: text/markdown
59
61
 
60
62
  **Unified, reproducible evaluation for machine learning and research.**
61
63
 
62
- EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
64
+ EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
63
65
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
64
66
 
65
- > **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
67
+ > **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
66
68
 
67
69
  ## Installation
68
70
 
@@ -134,6 +136,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
134
136
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
135
137
  p-values are adjusted for multiple comparisons (Holm by default).
136
138
 
139
+ ## Clinical evaluation
140
+
141
+ ```python
142
+ report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
143
+ print(report)
144
+ # Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
145
+ # diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
146
+
147
+ es.lr_positive(y_true, y_pred)
148
+ es.youden_j(y_true, y_pred)
149
+
150
+ dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
151
+ dca.useful_range() # thresholds where the model beats both
152
+ es.plot.decision_curve(y_true, {"model": y_prob})
153
+ ```
154
+
155
+ Ratios that divide by zero are `inf` or NaN with a warning, never 0.
156
+
157
+ ## Calibration
158
+
159
+ ```python
160
+ es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
161
+ es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
162
+ es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
163
+ es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
164
+ ```
165
+
166
+ ## Statistical tests
167
+
168
+ ```python
169
+ es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
170
+ es.paired_t_test(fold_scores_a, fold_scores_b)
171
+ es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
172
+ es.mann_whitney_test(a, b) # with rank-biserial r
173
+ es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
174
+ es.kruskal_wallis_test(g1, g2, g3)
175
+ es.shapiro_wilk_test(residuals)
176
+ es.chi_square_test(table) # with Cramér's V
177
+ es.fisher_exact_test([[8, 2], [1, 5]])
178
+ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
179
+ ```
180
+
181
+ Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
182
+ checked against SciPy and statsmodels in the test suite.
183
+
137
184
  ## Classification report
138
185
 
139
186
  ```python
@@ -175,7 +222,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
175
222
  evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
176
223
  --prob lr=p_lr --prob rf=p_rf --plot comparison.png
177
224
  evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
178
- evalsuite metrics --category classification
225
+ evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
226
+ evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
227
+ evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
228
+ evalsuite metrics --category clinical
179
229
  evalsuite info classification.mcc
180
230
  evalsuite benchmark --quick
181
231
  ```
@@ -213,6 +263,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
213
263
  one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
214
264
  calibration curve and expected calibration error.
215
265
 
266
+ **Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
267
+ negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
268
+ diagnostic report with confidence intervals for all of them.
269
+
270
+ **Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
271
+ and intercept, Hosmer–Lemeshow test.
272
+
273
+ **Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
274
+ paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
275
+ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
276
+ delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
277
+ Benjamini–Yekutieli).
278
+
216
279
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
217
280
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
218
281
  loss, Huber loss, relative absolute error, relative squared error.
@@ -7,10 +7,10 @@
7
7
 
8
8
  **Unified, reproducible evaluation for machine learning and research.**
9
9
 
10
- EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
10
+ EvalSuite brings classification, regression, clinical and statistical evaluation (with segmentation and
11
11
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
12
 
13
- > **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
13
+ > **Status: stable (0.2.0).** Every item on the 0.1.0 and 0.2.0 roadmaps is implemented and verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -82,6 +82,51 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
82
82
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
83
83
  p-values are adjusted for multiple comparisons (Holm by default).
84
84
 
85
+ ## Clinical evaluation
86
+
87
+ ```python
88
+ report = es.diagnostic_report(y_true, y_pred) # binary test vs reference standard
89
+ print(report)
90
+ # Sensitivity, specificity, PPV, NPV (Wilson CIs), LR+ and LR− (log CIs, Simel 1991),
91
+ # diagnostic odds ratio (Woolf), Youden's J, accuracy and prevalence
92
+
93
+ es.lr_positive(y_true, y_pred)
94
+ es.youden_j(y_true, y_pred)
95
+
96
+ dca = es.decision_curve(y_true, {"model": y_prob}) # net benefit vs treat all / treat none
97
+ dca.useful_range() # thresholds where the model beats both
98
+ es.plot.decision_curve(y_true, {"model": y_prob})
99
+ ```
100
+
101
+ Ratios that divide by zero are `inf` or NaN with a warning, never 0.
102
+
103
+ ## Calibration
104
+
105
+ ```python
106
+ es.calibration_report(y_true, y_prob) # Brier, ECE, MCE, intercept, slope, Hosmer–Lemeshow
107
+ es.calibration_slope(y_true, y_prob) # ideal 1; < 1 means predictions are too extreme
108
+ es.calibration_intercept(y_true, y_prob) # ideal 0 (calibration-in-the-large)
109
+ es.hosmer_lemeshow(y_true, y_prob, n_groups=10)
110
+ ```
111
+
112
+ ## Statistical tests
113
+
114
+ ```python
115
+ es.t_test(scores_a, scores_b) # Welch by default; mean difference with CI and Cohen's d
116
+ es.paired_t_test(fold_scores_a, fold_scores_b)
117
+ es.wilcoxon_test(fold_scores_a, fold_scores_b) # with matched-pairs rank-biserial r
118
+ es.mann_whitney_test(a, b) # with rank-biserial r
119
+ es.friedman_test(scores_a, scores_b, scores_c) # with Kendall's W
120
+ es.kruskal_wallis_test(g1, g2, g3)
121
+ es.shapiro_wilk_test(residuals)
122
+ es.chi_square_test(table) # with Cramér's V
123
+ es.fisher_exact_test([[8, 2], [1, 5]])
124
+ es.adjust_pvalues(p_values, method="hochberg") # also holm, bonferroni, bh, by
125
+ ```
126
+
127
+ Every test returns a `TestResult` with the statistic, p-value and an effect size, computed with SciPy and
128
+ checked against SciPy and statsmodels in the test suite.
129
+
85
130
  ## Classification report
86
131
 
87
132
  ```python
@@ -123,7 +168,10 @@ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
123
168
  evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
124
169
  --prob lr=p_lr --prob rf=p_rf --plot comparison.png
125
170
  evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
126
- evalsuite metrics --category classification
171
+ evalsuite diagnostic predictions.csv --y-true label --y-pred pred # sensitivity, LR+, DOR... with CIs
172
+ evalsuite calibration predictions.csv --y-true label --y-prob prob # slope, intercept, ECE, HL
173
+ evalsuite plot decision predictions.csv --y-true label --y-prob prob -o dca.png
174
+ evalsuite metrics --category clinical
127
175
  evalsuite info classification.mcc
128
176
  evalsuite benchmark --quick
129
177
  ```
@@ -161,6 +209,19 @@ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matr
161
209
  one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
162
210
  calibration curve and expected calibration error.
163
211
 
212
+ **Clinical** (binary; `pos_label`; sample weights): sensitivity, specificity, PPV, NPV, positive and
213
+ negative likelihood ratios, diagnostic odds ratio, Youden's J, net benefit and decision curves, and a
214
+ diagnostic report with confidence intervals for all of them.
215
+
216
+ **Calibration**: calibration curve, Brier score, expected and maximum calibration error, calibration slope
217
+ and intercept, Hosmer–Lemeshow test.
218
+
219
+ **Statistics**: confidence intervals (bootstrap percentile/basic/BCa, Wilson, Clopper–Pearson, DeLong),
220
+ paired tests (McNemar, DeLong, paired bootstrap), t-tests (Welch, Student, paired), Mann–Whitney, Wilcoxon,
221
+ Kruskal–Wallis, Friedman, Shapiro–Wilk, χ², Fisher's exact; effect sizes (Cohen's d, Hedges' g, Cliff's
222
+ delta, Cramér's V); multiple-testing corrections (Bonferroni, Holm, Hochberg, Benjamini–Hochberg,
223
+ Benjamini–Yekutieli).
224
+
164
225
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
165
226
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
166
227
  loss, Huber loss, relative absolute error, relative squared error.
@@ -12,10 +12,12 @@ license-files = ["LICENSE"]
12
12
  requires-python = ">=3.9"
13
13
  authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
15
- keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
15
+ keywords = ["evaluation", "metrics", "clinical", "diagnostic accuracy", "calibration", "statistical tests", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
17
  "Development Status :: 5 - Production/Stable",
18
18
  "Intended Audience :: Science/Research",
19
+ "Intended Audience :: Healthcare Industry",
20
+ "Topic :: Scientific/Engineering :: Medical Science Apps.",
19
21
  "Intended Audience :: Developers",
20
22
  "Operating System :: OS Independent",
21
23
  "Programming Language :: Python :: 3",
@@ -5,8 +5,16 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import classification, plot, regression, stats
8
+ from . import calibration, classification, clinical, plot, regression, stats
9
9
  from .api import evaluate
10
+ from .calibration import (
11
+ CalibrationReport,
12
+ calibration_intercept,
13
+ calibration_report,
14
+ calibration_slope,
15
+ hosmer_lemeshow,
16
+ maximum_calibration_error,
17
+ )
10
18
  from .classification import (
11
19
  accuracy,
12
20
  average_precision,
@@ -31,6 +39,19 @@ from .classification import (
31
39
  specificity,
32
40
  top_k_accuracy,
33
41
  )
42
+ from .clinical import (
43
+ DecisionCurve,
44
+ DiagnosticReport,
45
+ decision_curve,
46
+ diagnostic_odds_ratio,
47
+ diagnostic_report,
48
+ lr_negative,
49
+ lr_positive,
50
+ net_benefit,
51
+ ppv,
52
+ sensitivity,
53
+ youden_j,
54
+ )
34
55
  from .core.exceptions import (
35
56
  EvalSuiteError,
36
57
  InputValidationError,
@@ -69,19 +90,58 @@ from .stats import (
69
90
  accuracy_ci,
70
91
  adjust_pvalues,
71
92
  bootstrap_ci,
93
+ chi_square_test,
72
94
  cliffs_delta,
73
95
  cohens_d,
74
96
  compare,
97
+ cramers_v,
75
98
  delong_test,
99
+ fisher_exact_test,
100
+ friedman_test,
76
101
  hedges_g,
102
+ kruskal_wallis_test,
103
+ mann_whitney_test,
77
104
  mcnemar_test,
78
105
  paired_bootstrap_test,
106
+ paired_t_test,
79
107
  proportion_ci,
80
108
  roc_auc_ci,
109
+ shapiro_wilk_test,
110
+ t_test,
111
+ wilcoxon_test,
81
112
  )
82
113
  from .version import __version__
83
114
 
84
115
  __all__ = [
116
+ "CalibrationReport",
117
+ "calibration_report",
118
+ "calibration",
119
+ "clinical",
120
+ "calibration_intercept",
121
+ "calibration_slope",
122
+ "hosmer_lemeshow",
123
+ "maximum_calibration_error",
124
+ "DecisionCurve",
125
+ "DiagnosticReport",
126
+ "decision_curve",
127
+ "diagnostic_odds_ratio",
128
+ "diagnostic_report",
129
+ "lr_negative",
130
+ "lr_positive",
131
+ "net_benefit",
132
+ "ppv",
133
+ "sensitivity",
134
+ "youden_j",
135
+ "chi_square_test",
136
+ "cramers_v",
137
+ "fisher_exact_test",
138
+ "friedman_test",
139
+ "kruskal_wallis_test",
140
+ "mann_whitney_test",
141
+ "paired_t_test",
142
+ "shapiro_wilk_test",
143
+ "t_test",
144
+ "wilcoxon_test",
85
145
  "plot",
86
146
  "expected_calibration_error",
87
147
  "calibration_curve",