evalsuite-python 0.1.0a1__tar.gz → 0.1.0b1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalsuite_python-0.1.0b1/CHANGELOG.md +66 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/PKG-INFO +84 -4
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/README.md +79 -2
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/pyproject.toml +4 -1
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/__init__.py +42 -1
- evalsuite_python-0.1.0b1/src/evalsuite/__main__.py +5 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/api.py +22 -7
- evalsuite_python-0.1.0b1/src/evalsuite/benchmarks.py +272 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/__init__.py +4 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/metrics.py +71 -0
- evalsuite_python-0.1.0b1/src/evalsuite/cli/__init__.py +5 -0
- evalsuite_python-0.1.0b1/src/evalsuite/cli/main.py +389 -0
- evalsuite_python-0.1.0b1/src/evalsuite/core/export.py +100 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/result.py +102 -0
- evalsuite_python-0.1.0b1/src/evalsuite/plot.py +289 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/metrics.py +36 -19
- evalsuite_python-0.1.0b1/src/evalsuite/reporting.py +241 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/__init__.py +25 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/_resolve.py +90 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/compare.py +414 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/effect.py +113 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/intervals.py +320 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/paired.py +207 -0
- evalsuite_python-0.1.0b1/src/evalsuite/stats/results.py +129 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.1.0b1/tests/output/test_benchmarks.py +32 -0
- evalsuite_python-0.1.0b1/tests/output/test_cli.py +195 -0
- evalsuite_python-0.1.0b1/tests/output/test_plot.py +123 -0
- evalsuite_python-0.1.0b1/tests/output/test_reporting.py +143 -0
- evalsuite_python-0.1.0b1/tests/stats/__init__.py +0 -0
- evalsuite_python-0.1.0b1/tests/stats/test_branches.py +86 -0
- evalsuite_python-0.1.0b1/tests/stats/test_compare.py +152 -0
- evalsuite_python-0.1.0b1/tests/stats/test_reference.py +258 -0
- evalsuite_python-0.1.0b1/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/unit/test_core.py +21 -0
- evalsuite_python-0.1.0a1/CHANGELOG.md +0 -23
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/.gitignore +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/LICENSE +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/core/validation.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/conftest.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/integration/__init__.py +0 -0
- {evalsuite_python-0.1.0a1/tests/regression → evalsuite_python-0.1.0b1/tests/output}/__init__.py +0 -0
- {evalsuite_python-0.1.0a1/tests/unit → evalsuite_python-0.1.0b1/tests/regression}/__init__.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0a1 → evalsuite_python-0.1.0b1}/tests/unit/test_edges.py +0 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.1.0b1]
|
|
10
|
+
|
|
11
|
+
Feature-complete for 0.1.0.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Plots (`es.plot`, optional `[plot]` extra): ROC, precision-recall, calibration, confusion matrix, residuals /
|
|
16
|
+
predicted-vs-true, and model-comparison forest plots. Values come from EvalSuite's metrics; several models
|
|
17
|
+
get distinct colours and line styles. Importing `evalsuite` never imports matplotlib.
|
|
18
|
+
- Reporting: `classification_report` (per-class precision, recall, F1, specificity, support; accuracy, micro,
|
|
19
|
+
macro and weighted averages; matches scikit-learn); `to_html()` and `to_csv()` on every result; `save(path)`
|
|
20
|
+
choosing the format from the extension (.json .csv .md .tex .html .txt).
|
|
21
|
+
- `calibration_curve` and `expected_calibration_error` (matches scikit-learn's calibration curve).
|
|
22
|
+
- Command line: `evalsuite evaluate | report | compare | plot | metrics | info | benchmark`, and
|
|
23
|
+
`python -m evalsuite`. Reads CSV, TSV, Parquet and JSON; clear one-line errors with exit code 2.
|
|
24
|
+
- Benchmarks (`evalsuite.benchmarks.run_benchmarks`, `evalsuite benchmark`): time and peak memory against
|
|
25
|
+
scikit-learn, with a check that both libraries return the same numbers. Results in `BENCHMARKS.md`.
|
|
26
|
+
|
|
27
|
+
### Changed
|
|
28
|
+
|
|
29
|
+
- Regression `evaluate()` validates inputs once for all metrics and uses a faster unweighted mean (2× faster
|
|
30
|
+
on large arrays).
|
|
31
|
+
|
|
32
|
+
## [0.1.0a2]
|
|
33
|
+
|
|
34
|
+
### Added
|
|
35
|
+
|
|
36
|
+
- Model comparison (`es.compare`): per-model confidence intervals from paired bootstrap resamples, pairwise
|
|
37
|
+
tests (McNemar for accuracy, DeLong for binary ROC AUC, paired bootstrap otherwise), multiple-comparison
|
|
38
|
+
correction, best model per metric, and summary/pandas/Markdown/LaTeX/JSON export.
|
|
39
|
+
- Confidence intervals: `bootstrap_ci` (percentile, basic, BCa; stratified and reproducible),
|
|
40
|
+
`proportion_ci` and `accuracy_ci` (Wilson, Clopper-Pearson, normal), `roc_auc_ci` (DeLong).
|
|
41
|
+
- Paired tests: `mcnemar_test`, `delong_test`, `paired_bootstrap_test`.
|
|
42
|
+
- Effect sizes: `cohens_d` (independent and paired), `hedges_g`, `cliffs_delta`; `adjust_pvalues`
|
|
43
|
+
(Holm, Bonferroni, Benjamini-Hochberg, Benjamini-Yekutieli).
|
|
44
|
+
- Reference tests against statsmodels and SciPy, brute-force DeLong checks and coverage simulations.
|
|
45
|
+
- Python 3.14 support and CI.
|
|
46
|
+
|
|
47
|
+
### Fixed
|
|
48
|
+
|
|
49
|
+
- 0.1.0a1 installed an `evalsuite` console command although the CLI is not implemented yet, so the command
|
|
50
|
+
failed with `ModuleNotFoundError`. The entry point is removed until the CLI ships.
|
|
51
|
+
|
|
52
|
+
## [0.1.0a1]
|
|
53
|
+
|
|
54
|
+
First alpha, published to reserve the name and test the release pipeline. Distribution name
|
|
55
|
+
`evalsuite-python` (`pip install --pre evalsuite-python`), imported as `evalsuite`.
|
|
56
|
+
|
|
57
|
+
### Added
|
|
58
|
+
|
|
59
|
+
- Core: input validation with actionable errors, exception hierarchy, immutable result objects with
|
|
60
|
+
JSON/pandas/Markdown/LaTeX export, metric registry (`list_metrics`, `metric_info`), and an evaluation
|
|
61
|
+
context that computes the confusion matrix once per evaluation.
|
|
62
|
+
- Classification metrics for binary, multiclass and multilabel targets with all averaging modes and
|
|
63
|
+
sample weights.
|
|
64
|
+
- Regression metrics for single- and multi-output targets with sample weights.
|
|
65
|
+
- `evaluate()` high-level API with task inference and default metric sets.
|
|
66
|
+
- Reference tests against scikit-learn and property-based tests; Python 3.9 to 3.13.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.0b1
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -12,7 +12,7 @@ Maintainer: Manoj Kumar C S
|
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
14
|
Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
|
|
15
|
-
Classifier: Development Status ::
|
|
15
|
+
Classifier: Development Status :: 4 - Beta
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: Intended Audience :: Science/Research
|
|
18
18
|
Classifier: Operating System :: OS Independent
|
|
@@ -23,6 +23,7 @@ Classifier: Programming Language :: Python :: 3.10
|
|
|
23
23
|
Classifier: Programming Language :: Python :: 3.11
|
|
24
24
|
Classifier: Programming Language :: Python :: 3.12
|
|
25
25
|
Classifier: Programming Language :: Python :: 3.13
|
|
26
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
26
27
|
Classifier: Topic :: Scientific/Engineering
|
|
27
28
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
28
29
|
Classifier: Typing :: Typed
|
|
@@ -35,6 +36,7 @@ Requires-Dist: matplotlib>=3.5; extra == 'all'
|
|
|
35
36
|
Provides-Extra: dev
|
|
36
37
|
Requires-Dist: build; extra == 'dev'
|
|
37
38
|
Requires-Dist: hypothesis>=6.80; extra == 'dev'
|
|
39
|
+
Requires-Dist: matplotlib>=3.5; extra == 'dev'
|
|
38
40
|
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
39
41
|
Requires-Dist: pandas-stubs; extra == 'dev'
|
|
40
42
|
Requires-Dist: pip-audit; extra == 'dev'
|
|
@@ -42,6 +44,7 @@ Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
|
42
44
|
Requires-Dist: pytest>=7; extra == 'dev'
|
|
43
45
|
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
44
46
|
Requires-Dist: scikit-learn>=1.2; extra == 'dev'
|
|
47
|
+
Requires-Dist: statsmodels>=0.13; extra == 'dev'
|
|
45
48
|
Requires-Dist: twine; extra == 'dev'
|
|
46
49
|
Provides-Extra: plot
|
|
47
50
|
Requires-Dist: matplotlib>=3.5; extra == 'plot'
|
|
@@ -59,7 +62,7 @@ Description-Content-Type: text/markdown
|
|
|
59
62
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
60
63
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
61
64
|
|
|
62
|
-
> **Status:
|
|
65
|
+
> **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
|
|
63
66
|
|
|
64
67
|
## Installation
|
|
65
68
|
|
|
@@ -109,12 +112,89 @@ es.metric_info("classification.mcc").formula # documentation
|
|
|
109
112
|
es.list_metrics("regression")
|
|
110
113
|
```
|
|
111
114
|
|
|
115
|
+
## Comparing models
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
result = es.compare(
|
|
119
|
+
y_true,
|
|
120
|
+
{"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
|
|
121
|
+
probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
|
|
122
|
+
random_state=0,
|
|
123
|
+
)
|
|
124
|
+
print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
|
|
125
|
+
result.to_latex(label="tab:models")
|
|
126
|
+
|
|
127
|
+
es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
|
|
128
|
+
es.accuracy_ci(y_true, y_pred) # Wilson interval
|
|
129
|
+
es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
|
|
130
|
+
es.mcnemar_test(y_true, pred_a, pred_b)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
|
|
134
|
+
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
135
|
+
p-values are adjusted for multiple comparisons (Holm by default).
|
|
136
|
+
|
|
137
|
+
## Classification report
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
report = es.classification_report(y_true, y_pred)
|
|
141
|
+
print(report) # per-class precision, recall, F1, specificity, support + averages
|
|
142
|
+
report.save("report.html") # also .csv .md .tex .json .txt
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Plots
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
|
|
153
|
+
es.plot.pr(y_true, prob) # AP and the prevalence line
|
|
154
|
+
es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
|
|
155
|
+
es.plot.confusion_matrix(y_true, y_pred, normalize="true")
|
|
156
|
+
es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
|
|
157
|
+
es.plot.comparison(es.compare(...)) # forest plot with CIs
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
|
|
161
|
+
computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
|
|
162
|
+
*and* line styles, so figures stay readable in greyscale print.
|
|
163
|
+
|
|
164
|
+
## Exports
|
|
165
|
+
|
|
166
|
+
Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
|
|
167
|
+
`to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
|
|
168
|
+
the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
|
|
169
|
+
|
|
170
|
+
## Command line
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
|
|
174
|
+
evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
175
|
+
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
176
|
+
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
177
|
+
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
178
|
+
evalsuite metrics --category classification
|
|
179
|
+
evalsuite info classification.mcc
|
|
180
|
+
evalsuite benchmark --quick
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
|
|
184
|
+
(text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
|
|
185
|
+
|
|
186
|
+
## Performance
|
|
187
|
+
|
|
188
|
+
`evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
|
|
189
|
+
than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
|
|
190
|
+
|
|
112
191
|
## Metrics in this release
|
|
113
192
|
|
|
114
193
|
**Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
|
|
115
194
|
sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
|
|
116
195
|
MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
|
|
117
|
-
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy
|
|
196
|
+
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
197
|
+
calibration curve and expected calibration error.
|
|
118
198
|
|
|
119
199
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
120
200
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
11
11
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status:
|
|
13
|
+
> **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -60,12 +60,89 @@ es.metric_info("classification.mcc").formula # documentation
|
|
|
60
60
|
es.list_metrics("regression")
|
|
61
61
|
```
|
|
62
62
|
|
|
63
|
+
## Comparing models
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
result = es.compare(
|
|
67
|
+
y_true,
|
|
68
|
+
{"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
|
|
69
|
+
probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
|
|
70
|
+
random_state=0,
|
|
71
|
+
)
|
|
72
|
+
print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
|
|
73
|
+
result.to_latex(label="tab:models")
|
|
74
|
+
|
|
75
|
+
es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
|
|
76
|
+
es.accuracy_ci(y_true, y_pred) # Wilson interval
|
|
77
|
+
es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
|
|
78
|
+
es.mcnemar_test(y_true, pred_a, pred_b)
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
|
|
82
|
+
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
83
|
+
p-values are adjusted for multiple comparisons (Holm by default).
|
|
84
|
+
|
|
85
|
+
## Classification report
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
report = es.classification_report(y_true, y_pred)
|
|
89
|
+
print(report) # per-class precision, recall, F1, specificity, support + averages
|
|
90
|
+
report.save("report.html") # also .csv .md .tex .json .txt
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Plots
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
|
|
101
|
+
es.plot.pr(y_true, prob) # AP and the prevalence line
|
|
102
|
+
es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
|
|
103
|
+
es.plot.confusion_matrix(y_true, y_pred, normalize="true")
|
|
104
|
+
es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
|
|
105
|
+
es.plot.comparison(es.compare(...)) # forest plot with CIs
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
|
|
109
|
+
computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
|
|
110
|
+
*and* line styles, so figures stay readable in greyscale print.
|
|
111
|
+
|
|
112
|
+
## Exports
|
|
113
|
+
|
|
114
|
+
Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
|
|
115
|
+
`to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
|
|
116
|
+
the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
|
|
117
|
+
|
|
118
|
+
## Command line
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
|
|
122
|
+
evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
123
|
+
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
124
|
+
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
125
|
+
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
126
|
+
evalsuite metrics --category classification
|
|
127
|
+
evalsuite info classification.mcc
|
|
128
|
+
evalsuite benchmark --quick
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
|
|
132
|
+
(text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
|
|
133
|
+
|
|
134
|
+
## Performance
|
|
135
|
+
|
|
136
|
+
`evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
|
|
137
|
+
than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
|
|
138
|
+
|
|
63
139
|
## Metrics in this release
|
|
64
140
|
|
|
65
141
|
**Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
|
|
66
142
|
sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
|
|
67
143
|
MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
|
|
68
|
-
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy
|
|
144
|
+
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
145
|
+
calibration curve and expected calibration error.
|
|
69
146
|
|
|
70
147
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
71
148
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
@@ -14,7 +14,7 @@ authors = [{ name = "Manoj Kumar C S" }]
|
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }]
|
|
15
15
|
keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
|
-
"Development Status ::
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
19
|
"Intended Audience :: Developers",
|
|
20
20
|
"Operating System :: OS Independent",
|
|
@@ -25,6 +25,7 @@ classifiers = [
|
|
|
25
25
|
"Programming Language :: Python :: 3.11",
|
|
26
26
|
"Programming Language :: Python :: 3.12",
|
|
27
27
|
"Programming Language :: Python :: 3.13",
|
|
28
|
+
"Programming Language :: Python :: 3.14",
|
|
28
29
|
"Topic :: Scientific/Engineering",
|
|
29
30
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
30
31
|
"Typing :: Typed",
|
|
@@ -39,6 +40,8 @@ dev = [
|
|
|
39
40
|
"pytest-cov>=4",
|
|
40
41
|
"hypothesis>=6.80",
|
|
41
42
|
"scikit-learn>=1.2",
|
|
43
|
+
"statsmodels>=0.13",
|
|
44
|
+
"matplotlib>=3.5",
|
|
42
45
|
"ruff>=0.6",
|
|
43
46
|
"mypy>=1.10",
|
|
44
47
|
"pandas-stubs",
|
|
@@ -5,15 +5,17 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import classification, regression
|
|
8
|
+
from . import classification, plot, regression, stats
|
|
9
9
|
from .api import evaluate
|
|
10
10
|
from .classification import (
|
|
11
11
|
accuracy,
|
|
12
12
|
average_precision,
|
|
13
13
|
balanced_accuracy,
|
|
14
14
|
brier_score,
|
|
15
|
+
calibration_curve,
|
|
15
16
|
cohen_kappa,
|
|
16
17
|
confusion_matrix,
|
|
18
|
+
expected_calibration_error,
|
|
17
19
|
f1,
|
|
18
20
|
fbeta,
|
|
19
21
|
hamming_loss,
|
|
@@ -59,9 +61,48 @@ from .regression import (
|
|
|
59
61
|
rse,
|
|
60
62
|
smape,
|
|
61
63
|
)
|
|
64
|
+
from .reporting import ClassificationReport, classification_report
|
|
65
|
+
from .stats import (
|
|
66
|
+
ComparisonResult,
|
|
67
|
+
ConfidenceInterval,
|
|
68
|
+
TestResult,
|
|
69
|
+
accuracy_ci,
|
|
70
|
+
adjust_pvalues,
|
|
71
|
+
bootstrap_ci,
|
|
72
|
+
cliffs_delta,
|
|
73
|
+
cohens_d,
|
|
74
|
+
compare,
|
|
75
|
+
delong_test,
|
|
76
|
+
hedges_g,
|
|
77
|
+
mcnemar_test,
|
|
78
|
+
paired_bootstrap_test,
|
|
79
|
+
proportion_ci,
|
|
80
|
+
roc_auc_ci,
|
|
81
|
+
)
|
|
62
82
|
from .version import __version__
|
|
63
83
|
|
|
64
84
|
__all__ = [
|
|
85
|
+
"plot",
|
|
86
|
+
"expected_calibration_error",
|
|
87
|
+
"calibration_curve",
|
|
88
|
+
"classification_report",
|
|
89
|
+
"ClassificationReport",
|
|
90
|
+
"stats",
|
|
91
|
+
"roc_auc_ci",
|
|
92
|
+
"proportion_ci",
|
|
93
|
+
"paired_bootstrap_test",
|
|
94
|
+
"mcnemar_test",
|
|
95
|
+
"hedges_g",
|
|
96
|
+
"delong_test",
|
|
97
|
+
"compare",
|
|
98
|
+
"cohens_d",
|
|
99
|
+
"cliffs_delta",
|
|
100
|
+
"bootstrap_ci",
|
|
101
|
+
"adjust_pvalues",
|
|
102
|
+
"accuracy_ci",
|
|
103
|
+
"TestResult",
|
|
104
|
+
"ConfidenceInterval",
|
|
105
|
+
"ComparisonResult",
|
|
65
106
|
"EvalSuiteError",
|
|
66
107
|
"EvaluationResult",
|
|
67
108
|
"InputValidationError",
|
|
@@ -232,6 +232,28 @@ def _evaluate_regression(
|
|
|
232
232
|
f"Unknown regression metric(s): {', '.join(unknown)}. Available: {', '.join(_REGRESSION)}."
|
|
233
233
|
)
|
|
234
234
|
out: dict[str, MetricResult] = {}
|
|
235
|
+
shared = reg._Inputs(y_true, y_pred, sample_weight) # validate once for every metric
|
|
236
|
+
token = reg._SHARED.set((id(y_true), id(y_pred), id(sample_weight), shared))
|
|
237
|
+
try:
|
|
238
|
+
_regression_loop(names, out, y_true, y_pred, sample_weight)
|
|
239
|
+
finally:
|
|
240
|
+
reg._SHARED.reset(token)
|
|
241
|
+
return EvaluationResult(
|
|
242
|
+
task="regression",
|
|
243
|
+
metrics=out,
|
|
244
|
+
n_samples=yt.shape[0],
|
|
245
|
+
target_type="continuous",
|
|
246
|
+
metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _regression_loop(
|
|
251
|
+
names: list[str],
|
|
252
|
+
out: dict[str, MetricResult],
|
|
253
|
+
y_true: ArrayLike,
|
|
254
|
+
y_pred: ArrayLike,
|
|
255
|
+
sample_weight: Optional[ArrayLike],
|
|
256
|
+
) -> None:
|
|
235
257
|
for name in names:
|
|
236
258
|
fn = _REGRESSION[name]
|
|
237
259
|
if name in _NO_WEIGHTS:
|
|
@@ -240,10 +262,3 @@ def _evaluate_regression(
|
|
|
240
262
|
out[name] = fn(y_true, y_pred)
|
|
241
263
|
else:
|
|
242
264
|
out[name] = fn(y_true, y_pred, sample_weight=sample_weight)
|
|
243
|
-
return EvaluationResult(
|
|
244
|
-
task="regression",
|
|
245
|
-
metrics=out,
|
|
246
|
-
n_samples=yt.shape[0],
|
|
247
|
-
target_type="continuous",
|
|
248
|
-
metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
|
|
249
|
-
)
|