evalsuite-python 0.1.0a1__tar.gz → 0.1.0b1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. evalsuite_python-0.1.0b1/CHANGELOG.md +66 -0
  2. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/PKG-INFO +84 -4
  3. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/README.md +79 -2
  4. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/pyproject.toml +4 -1
  5. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/__init__.py +42 -1
  6. evalsuite_python-0.1.0b1/src/evalsuite/__main__.py +5 -0
  7. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/api.py +22 -7
  8. evalsuite_python-0.1.0b1/src/evalsuite/benchmarks.py +272 -0
  9. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/__init__.py +4 -0
  10. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/metrics.py +71 -0
  11. evalsuite_python-0.1.0b1/src/evalsuite/cli/__init__.py +5 -0
  12. evalsuite_python-0.1.0b1/src/evalsuite/cli/main.py +389 -0
  13. evalsuite_python-0.1.0b1/src/evalsuite/core/export.py +100 -0
  14. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/result.py +102 -0
  15. evalsuite_python-0.1.0b1/src/evalsuite/plot.py +289 -0
  16. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/metrics.py +36 -19
  17. evalsuite_python-0.1.0b1/src/evalsuite/reporting.py +241 -0
  18. evalsuite_python-0.1.0b1/src/evalsuite/stats/__init__.py +25 -0
  19. evalsuite_python-0.1.0b1/src/evalsuite/stats/_resolve.py +90 -0
  20. evalsuite_python-0.1.0b1/src/evalsuite/stats/compare.py +414 -0
  21. evalsuite_python-0.1.0b1/src/evalsuite/stats/effect.py +113 -0
  22. evalsuite_python-0.1.0b1/src/evalsuite/stats/intervals.py +320 -0
  23. evalsuite_python-0.1.0b1/src/evalsuite/stats/paired.py +207 -0
  24. evalsuite_python-0.1.0b1/src/evalsuite/stats/results.py +129 -0
  25. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/version.py +1 -1
  26. evalsuite_python-0.1.0b1/tests/output/test_benchmarks.py +32 -0
  27. evalsuite_python-0.1.0b1/tests/output/test_cli.py +195 -0
  28. evalsuite_python-0.1.0b1/tests/output/test_plot.py +123 -0
  29. evalsuite_python-0.1.0b1/tests/output/test_reporting.py +143 -0
  30. evalsuite_python-0.1.0b1/tests/stats/__init__.py +0 -0
  31. evalsuite_python-0.1.0b1/tests/stats/test_branches.py +86 -0
  32. evalsuite_python-0.1.0b1/tests/stats/test_compare.py +152 -0
  33. evalsuite_python-0.1.0b1/tests/stats/test_reference.py +258 -0
  34. evalsuite_python-0.1.0b1/tests/unit/__init__.py +0 -0
  35. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/unit/test_core.py +21 -0
  36. evalsuite_python-0.1.0a1/CHANGELOG.md +0 -23
  37. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/.gitignore +0 -0
  38. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/CONTRIBUTING.md +0 -0
  39. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/LICENSE +0 -0
  40. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/_common.py +0 -0
  41. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/__init__.py +0 -0
  42. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/context.py +0 -0
  43. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/exceptions.py +0 -0
  44. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/registry.py +0 -0
  45. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/types.py +0 -0
  46. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/validation.py +0 -0
  47. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/py.typed +0 -0
  48. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/__init__.py +0 -0
  49. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/__init__.py +0 -0
  50. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/classification/__init__.py +0 -0
  51. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/classification/test_against_sklearn.py +0 -0
  52. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/conftest.py +0 -0
  53. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/integration/__init__.py +0 -0
  54. {evalsuite_python-0.1.0a1/tests/regression → evalsuite_python-0.1.0b1/tests/output}/__init__.py +0 -0
  55. {evalsuite_python-0.1.0a1/tests/unit → evalsuite_python-0.1.0b1/tests/regression}/__init__.py +0 -0
  56. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/regression/test_against_sklearn.py +0 -0
  57. {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/unit/test_edges.py +0 -0
@@ -0,0 +1,66 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0b1]
10
+
11
+ Feature-complete for 0.1.0.
12
+
13
+ ### Added
14
+
15
+ - Plots (`es.plot`, optional `[plot]` extra): ROC, precision-recall, calibration, confusion matrix, residuals /
16
+ predicted-vs-true, and model-comparison forest plots. Values come from EvalSuite's metrics; several models
17
+ get distinct colours and line styles. Importing `evalsuite` never imports matplotlib.
18
+ - Reporting: `classification_report` (per-class precision, recall, F1, specificity, support; accuracy, micro,
19
+ macro and weighted averages; matches scikit-learn); `to_html()` and `to_csv()` on every result; `save(path)`
20
+ choosing the format from the extension (.json .csv .md .tex .html .txt).
21
+ - `calibration_curve` and `expected_calibration_error` (matches scikit-learn's calibration curve).
22
+ - Command line: `evalsuite evaluate | report | compare | plot | metrics | info | benchmark`, and
23
+ `python -m evalsuite`. Reads CSV, TSV, Parquet and JSON; clear one-line errors with exit code 2.
24
+ - Benchmarks (`evalsuite.benchmarks.run_benchmarks`, `evalsuite benchmark`): time and peak memory against
25
+ scikit-learn, with a check that both libraries return the same numbers. Results in `BENCHMARKS.md`.
26
+
27
+ ### Changed
28
+
29
+ - Regression `evaluate()` validates inputs once for all metrics and uses a faster unweighted mean (2× faster
30
+ on large arrays).
31
+
32
+ ## [0.1.0a2]
33
+
34
+ ### Added
35
+
36
+ - Model comparison (`es.compare`): per-model confidence intervals from paired bootstrap resamples, pairwise
37
+ tests (McNemar for accuracy, DeLong for binary ROC AUC, paired bootstrap otherwise), multiple-comparison
38
+ correction, best model per metric, and summary/pandas/Markdown/LaTeX/JSON export.
39
+ - Confidence intervals: `bootstrap_ci` (percentile, basic, BCa; stratified and reproducible),
40
+ `proportion_ci` and `accuracy_ci` (Wilson, Clopper-Pearson, normal), `roc_auc_ci` (DeLong).
41
+ - Paired tests: `mcnemar_test`, `delong_test`, `paired_bootstrap_test`.
42
+ - Effect sizes: `cohens_d` (independent and paired), `hedges_g`, `cliffs_delta`; `adjust_pvalues`
43
+ (Holm, Bonferroni, Benjamini-Hochberg, Benjamini-Yekutieli).
44
+ - Reference tests against statsmodels and SciPy, brute-force DeLong checks and coverage simulations.
45
+ - Python 3.14 support and CI.
46
+
47
+ ### Fixed
48
+
49
+ - 0.1.0a1 installed an `evalsuite` console command although the CLI is not implemented yet, so the command
50
+ failed with `ModuleNotFoundError`. The entry point is removed until the CLI ships.
51
+
52
+ ## [0.1.0a1]
53
+
54
+ First alpha, published to reserve the name and test the release pipeline. Distribution name
55
+ `evalsuite-python` (`pip install --pre evalsuite-python`), imported as `evalsuite`.
56
+
57
+ ### Added
58
+
59
+ - Core: input validation with actionable errors, exception hierarchy, immutable result objects with
60
+ JSON/pandas/Markdown/LaTeX export, metric registry (`list_metrics`, `metric_info`), and an evaluation
61
+ context that computes the confusion matrix once per evaluation.
62
+ - Classification metrics for binary, multiclass and multilabel targets with all averaging modes and
63
+ sample weights.
64
+ - Regression metrics for single- and multi-output targets with sample weights.
65
+ - `evaluate()` high-level API with task inference and default metric sets.
66
+ - Reference tests against scikit-learn and property-based tests; Python 3.9 to 3.13.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.1.0a1
3
+ Version: 0.1.0b1
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -12,7 +12,7 @@ Maintainer: Manoj Kumar C S
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
14
  Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
15
- Classifier: Development Status :: 3 - Alpha
15
+ Classifier: Development Status :: 4 - Beta
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: Intended Audience :: Science/Research
18
18
  Classifier: Operating System :: OS Independent
@@ -23,6 +23,7 @@ Classifier: Programming Language :: Python :: 3.10
23
23
  Classifier: Programming Language :: Python :: 3.11
24
24
  Classifier: Programming Language :: Python :: 3.12
25
25
  Classifier: Programming Language :: Python :: 3.13
26
+ Classifier: Programming Language :: Python :: 3.14
26
27
  Classifier: Topic :: Scientific/Engineering
27
28
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
28
29
  Classifier: Typing :: Typed
@@ -35,6 +36,7 @@ Requires-Dist: matplotlib>=3.5; extra == 'all'
35
36
  Provides-Extra: dev
36
37
  Requires-Dist: build; extra == 'dev'
37
38
  Requires-Dist: hypothesis>=6.80; extra == 'dev'
39
+ Requires-Dist: matplotlib>=3.5; extra == 'dev'
38
40
  Requires-Dist: mypy>=1.10; extra == 'dev'
39
41
  Requires-Dist: pandas-stubs; extra == 'dev'
40
42
  Requires-Dist: pip-audit; extra == 'dev'
@@ -42,6 +44,7 @@ Requires-Dist: pytest-cov>=4; extra == 'dev'
42
44
  Requires-Dist: pytest>=7; extra == 'dev'
43
45
  Requires-Dist: ruff>=0.6; extra == 'dev'
44
46
  Requires-Dist: scikit-learn>=1.2; extra == 'dev'
47
+ Requires-Dist: statsmodels>=0.13; extra == 'dev'
45
48
  Requires-Dist: twine; extra == 'dev'
46
49
  Provides-Extra: plot
47
50
  Requires-Dist: matplotlib>=3.5; extra == 'plot'
@@ -59,7 +62,7 @@ Description-Content-Type: text/markdown
59
62
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
60
63
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
61
64
 
62
- > **Status: in development (v0.1.0 in progress).** The API may change before 0.1.0.
65
+ > **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
63
66
 
64
67
  ## Installation
65
68
 
@@ -109,12 +112,89 @@ es.metric_info("classification.mcc").formula # documentation
109
112
  es.list_metrics("regression")
110
113
  ```
111
114
 
115
+ ## Comparing models
116
+
117
+ ```python
118
+ result = es.compare(
119
+ y_true,
120
+ {"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
121
+ probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
122
+ random_state=0,
123
+ )
124
+ print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
125
+ result.to_latex(label="tab:models")
126
+
127
+ es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
128
+ es.accuracy_ci(y_true, y_pred) # Wilson interval
129
+ es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
130
+ es.mcnemar_test(y_true, pred_a, pred_b)
131
+ ```
132
+
133
+ Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
134
+ with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
135
+ p-values are adjusted for multiple comparisons (Holm by default).
136
+
137
+ ## Classification report
138
+
139
+ ```python
140
+ report = es.classification_report(y_true, y_pred)
141
+ print(report) # per-class precision, recall, F1, specificity, support + averages
142
+ report.save("report.html") # also .csv .md .tex .json .txt
143
+ ```
144
+
145
+ ## Plots
146
+
147
+ ```bash
148
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
149
+ ```
150
+
151
+ ```python
152
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
153
+ es.plot.pr(y_true, prob) # AP and the prevalence line
154
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
155
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
156
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
157
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
158
+ ```
159
+
160
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
161
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
162
+ *and* line styles, so figures stay readable in greyscale print.
163
+
164
+ ## Exports
165
+
166
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
167
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
168
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
169
+
170
+ ## Command line
171
+
172
+ ```bash
173
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
174
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
175
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
176
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
177
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
178
+ evalsuite metrics --category classification
179
+ evalsuite info classification.mcc
180
+ evalsuite benchmark --quick
181
+ ```
182
+
183
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
184
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
185
+
186
+ ## Performance
187
+
188
+ `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
189
+ than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
190
+
112
191
  ## Metrics in this release
113
192
 
114
193
  **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
115
194
  sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
116
195
  MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
117
- one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy.
196
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
197
+ calibration curve and expected calibration error.
118
198
 
119
199
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
120
200
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
@@ -10,7 +10,7 @@
10
10
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
11
11
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
12
 
13
- > **Status: in development (v0.1.0 in progress).** The API may change before 0.1.0.
13
+ > **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -60,12 +60,89 @@ es.metric_info("classification.mcc").formula # documentation
60
60
  es.list_metrics("regression")
61
61
  ```
62
62
 
63
+ ## Comparing models
64
+
65
+ ```python
66
+ result = es.compare(
67
+ y_true,
68
+ {"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
69
+ probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
70
+ random_state=0,
71
+ )
72
+ print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
73
+ result.to_latex(label="tab:models")
74
+
75
+ es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
76
+ es.accuracy_ci(y_true, y_pred) # Wilson interval
77
+ es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
78
+ es.mcnemar_test(y_true, pred_a, pred_b)
79
+ ```
80
+
81
+ Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
82
+ with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
83
+ p-values are adjusted for multiple comparisons (Holm by default).
84
+
85
+ ## Classification report
86
+
87
+ ```python
88
+ report = es.classification_report(y_true, y_pred)
89
+ print(report) # per-class precision, recall, F1, specificity, support + averages
90
+ report.save("report.html") # also .csv .md .tex .json .txt
91
+ ```
92
+
93
+ ## Plots
94
+
95
+ ```bash
96
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
97
+ ```
98
+
99
+ ```python
100
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
101
+ es.plot.pr(y_true, prob) # AP and the prevalence line
102
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
103
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
104
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
105
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
106
+ ```
107
+
108
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
109
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
110
+ *and* line styles, so figures stay readable in greyscale print.
111
+
112
+ ## Exports
113
+
114
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
115
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
116
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
117
+
118
+ ## Command line
119
+
120
+ ```bash
121
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
122
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
123
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
124
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
125
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
126
+ evalsuite metrics --category classification
127
+ evalsuite info classification.mcc
128
+ evalsuite benchmark --quick
129
+ ```
130
+
131
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
132
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
133
+
134
+ ## Performance
135
+
136
+ `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
137
+ than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
138
+
63
139
  ## Metrics in this release
64
140
 
65
141
  **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
66
142
  sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
67
143
  MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
68
- one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy.
144
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
145
+ calibration curve and expected calibration error.
69
146
 
70
147
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
71
148
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
@@ -14,7 +14,7 @@ authors = [{ name = "Manoj Kumar C S" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }]
15
15
  keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
- "Development Status :: 3 - Alpha",
17
+ "Development Status :: 4 - Beta",
18
18
  "Intended Audience :: Science/Research",
19
19
  "Intended Audience :: Developers",
20
20
  "Operating System :: OS Independent",
@@ -25,6 +25,7 @@ classifiers = [
25
25
  "Programming Language :: Python :: 3.11",
26
26
  "Programming Language :: Python :: 3.12",
27
27
  "Programming Language :: Python :: 3.13",
28
+ "Programming Language :: Python :: 3.14",
28
29
  "Topic :: Scientific/Engineering",
29
30
  "Topic :: Scientific/Engineering :: Artificial Intelligence",
30
31
  "Typing :: Typed",
@@ -39,6 +40,8 @@ dev = [
39
40
  "pytest-cov>=4",
40
41
  "hypothesis>=6.80",
41
42
  "scikit-learn>=1.2",
43
+ "statsmodels>=0.13",
44
+ "matplotlib>=3.5",
42
45
  "ruff>=0.6",
43
46
  "mypy>=1.10",
44
47
  "pandas-stubs",
@@ -5,15 +5,17 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import classification, regression
8
+ from . import classification, plot, regression, stats
9
9
  from .api import evaluate
10
10
  from .classification import (
11
11
  accuracy,
12
12
  average_precision,
13
13
  balanced_accuracy,
14
14
  brier_score,
15
+ calibration_curve,
15
16
  cohen_kappa,
16
17
  confusion_matrix,
18
+ expected_calibration_error,
17
19
  f1,
18
20
  fbeta,
19
21
  hamming_loss,
@@ -59,9 +61,48 @@ from .regression import (
59
61
  rse,
60
62
  smape,
61
63
  )
64
+ from .reporting import ClassificationReport, classification_report
65
+ from .stats import (
66
+ ComparisonResult,
67
+ ConfidenceInterval,
68
+ TestResult,
69
+ accuracy_ci,
70
+ adjust_pvalues,
71
+ bootstrap_ci,
72
+ cliffs_delta,
73
+ cohens_d,
74
+ compare,
75
+ delong_test,
76
+ hedges_g,
77
+ mcnemar_test,
78
+ paired_bootstrap_test,
79
+ proportion_ci,
80
+ roc_auc_ci,
81
+ )
62
82
  from .version import __version__
63
83
 
64
84
  __all__ = [
85
+ "plot",
86
+ "expected_calibration_error",
87
+ "calibration_curve",
88
+ "classification_report",
89
+ "ClassificationReport",
90
+ "stats",
91
+ "roc_auc_ci",
92
+ "proportion_ci",
93
+ "paired_bootstrap_test",
94
+ "mcnemar_test",
95
+ "hedges_g",
96
+ "delong_test",
97
+ "compare",
98
+ "cohens_d",
99
+ "cliffs_delta",
100
+ "bootstrap_ci",
101
+ "adjust_pvalues",
102
+ "accuracy_ci",
103
+ "TestResult",
104
+ "ConfidenceInterval",
105
+ "ComparisonResult",
65
106
  "EvalSuiteError",
66
107
  "EvaluationResult",
67
108
  "InputValidationError",
@@ -0,0 +1,5 @@
1
+ """``python -m evalsuite`` runs the command-line interface."""
2
+
3
+ from .cli.main import main
4
+
5
+ raise SystemExit(main())
@@ -232,6 +232,28 @@ def _evaluate_regression(
232
232
  f"Unknown regression metric(s): {', '.join(unknown)}. Available: {', '.join(_REGRESSION)}."
233
233
  )
234
234
  out: dict[str, MetricResult] = {}
235
+ shared = reg._Inputs(y_true, y_pred, sample_weight) # validate once for every metric
236
+ token = reg._SHARED.set((id(y_true), id(y_pred), id(sample_weight), shared))
237
+ try:
238
+ _regression_loop(names, out, y_true, y_pred, sample_weight)
239
+ finally:
240
+ reg._SHARED.reset(token)
241
+ return EvaluationResult(
242
+ task="regression",
243
+ metrics=out,
244
+ n_samples=yt.shape[0],
245
+ target_type="continuous",
246
+ metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
247
+ )
248
+
249
+
250
+ def _regression_loop(
251
+ names: list[str],
252
+ out: dict[str, MetricResult],
253
+ y_true: ArrayLike,
254
+ y_pred: ArrayLike,
255
+ sample_weight: Optional[ArrayLike],
256
+ ) -> None:
235
257
  for name in names:
236
258
  fn = _REGRESSION[name]
237
259
  if name in _NO_WEIGHTS:
@@ -240,10 +262,3 @@ def _evaluate_regression(
240
262
  out[name] = fn(y_true, y_pred)
241
263
  else:
242
264
  out[name] = fn(y_true, y_pred, sample_weight=sample_weight)
243
- return EvaluationResult(
244
- task="regression",
245
- metrics=out,
246
- n_samples=yt.shape[0],
247
- target_type="continuous",
248
- metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
249
- )