evalsuite-python 0.1.0a2__tar.gz → 0.1.0b1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/CHANGELOG.md +23 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/PKG-INFO +60 -4
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/README.md +57 -2
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/pyproject.toml +4 -2
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/__init__.py +9 -1
- evalsuite_python-0.1.0b1/src/evalsuite/__main__.py +5 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/api.py +22 -7
- evalsuite_python-0.1.0b1/src/evalsuite/benchmarks.py +272 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/__init__.py +4 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/metrics.py +71 -0
- evalsuite_python-0.1.0b1/src/evalsuite/cli/__init__.py +5 -0
- evalsuite_python-0.1.0b1/src/evalsuite/cli/main.py +389 -0
- evalsuite_python-0.1.0b1/src/evalsuite/core/export.py +100 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/result.py +102 -0
- evalsuite_python-0.1.0b1/src/evalsuite/plot.py +289 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/metrics.py +36 -19
- evalsuite_python-0.1.0b1/src/evalsuite/reporting.py +241 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/compare.py +91 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/version.py +1 -1
- evalsuite_python-0.1.0b1/tests/output/test_benchmarks.py +32 -0
- evalsuite_python-0.1.0b1/tests/output/test_cli.py +195 -0
- evalsuite_python-0.1.0b1/tests/output/test_plot.py +123 -0
- evalsuite_python-0.1.0b1/tests/output/test_reporting.py +143 -0
- evalsuite_python-0.1.0b1/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/unit/test_core.py +21 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/.gitignore +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/LICENSE +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/core/validation.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/_resolve.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/effect.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/conftest.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/integration/__init__.py +0 -0
- {evalsuite_python-0.1.0a2/tests/regression → evalsuite_python-0.1.0b1/tests/output}/__init__.py +0 -0
- {evalsuite_python-0.1.0a2/tests/stats → evalsuite_python-0.1.0b1/tests/regression}/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0a2/tests/unit → evalsuite_python-0.1.0b1/tests/stats}/__init__.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/tests/unit/test_edges.py +0 -0
|
@@ -6,6 +6,29 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.0b1]
|
|
10
|
+
|
|
11
|
+
Feature-complete for 0.1.0.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Plots (`es.plot`, optional `[plot]` extra): ROC, precision-recall, calibration, confusion matrix, residuals /
|
|
16
|
+
predicted-vs-true, and model-comparison forest plots. Values come from EvalSuite's metrics; several models
|
|
17
|
+
get distinct colours and line styles. Importing `evalsuite` never imports matplotlib.
|
|
18
|
+
- Reporting: `classification_report` (per-class precision, recall, F1, specificity, support; accuracy, micro,
|
|
19
|
+
macro and weighted averages; matches scikit-learn); `to_html()` and `to_csv()` on every result; `save(path)`
|
|
20
|
+
choosing the format from the extension (.json .csv .md .tex .html .txt).
|
|
21
|
+
- `calibration_curve` and `expected_calibration_error` (matches scikit-learn's calibration curve).
|
|
22
|
+
- Command line: `evalsuite evaluate | report | compare | plot | metrics | info | benchmark`, and
|
|
23
|
+
`python -m evalsuite`. Reads CSV, TSV, Parquet and JSON; clear one-line errors with exit code 2.
|
|
24
|
+
- Benchmarks (`evalsuite.benchmarks.run_benchmarks`, `evalsuite benchmark`): time and peak memory against
|
|
25
|
+
scikit-learn, with a check that both libraries return the same numbers. Results in `BENCHMARKS.md`.
|
|
26
|
+
|
|
27
|
+
### Changed
|
|
28
|
+
|
|
29
|
+
- Regression `evaluate()` validates inputs once for all metrics and uses a faster unweighted mean (2× faster
|
|
30
|
+
on large arrays).
|
|
31
|
+
|
|
9
32
|
## [0.1.0a2]
|
|
10
33
|
|
|
11
34
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.0b1
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
@@ -12,7 +12,7 @@ Maintainer: Manoj Kumar C S
|
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
14
|
Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
|
|
15
|
-
Classifier: Development Status ::
|
|
15
|
+
Classifier: Development Status :: 4 - Beta
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: Intended Audience :: Science/Research
|
|
18
18
|
Classifier: Operating System :: OS Independent
|
|
@@ -36,6 +36,7 @@ Requires-Dist: matplotlib>=3.5; extra == 'all'
|
|
|
36
36
|
Provides-Extra: dev
|
|
37
37
|
Requires-Dist: build; extra == 'dev'
|
|
38
38
|
Requires-Dist: hypothesis>=6.80; extra == 'dev'
|
|
39
|
+
Requires-Dist: matplotlib>=3.5; extra == 'dev'
|
|
39
40
|
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
40
41
|
Requires-Dist: pandas-stubs; extra == 'dev'
|
|
41
42
|
Requires-Dist: pip-audit; extra == 'dev'
|
|
@@ -61,7 +62,7 @@ Description-Content-Type: text/markdown
|
|
|
61
62
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
62
63
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
63
64
|
|
|
64
|
-
> **Status:
|
|
65
|
+
> **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
|
|
65
66
|
|
|
66
67
|
## Installation
|
|
67
68
|
|
|
@@ -133,12 +134,67 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
133
134
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
134
135
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
135
136
|
|
|
137
|
+
## Classification report
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
report = es.classification_report(y_true, y_pred)
|
|
141
|
+
print(report) # per-class precision, recall, F1, specificity, support + averages
|
|
142
|
+
report.save("report.html") # also .csv .md .tex .json .txt
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Plots
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
|
|
153
|
+
es.plot.pr(y_true, prob) # AP and the prevalence line
|
|
154
|
+
es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
|
|
155
|
+
es.plot.confusion_matrix(y_true, y_pred, normalize="true")
|
|
156
|
+
es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
|
|
157
|
+
es.plot.comparison(es.compare(...)) # forest plot with CIs
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
|
|
161
|
+
computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
|
|
162
|
+
*and* line styles, so figures stay readable in greyscale print.
|
|
163
|
+
|
|
164
|
+
## Exports
|
|
165
|
+
|
|
166
|
+
Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
|
|
167
|
+
`to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
|
|
168
|
+
the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
|
|
169
|
+
|
|
170
|
+
## Command line
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
|
|
174
|
+
evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
175
|
+
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
176
|
+
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
177
|
+
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
178
|
+
evalsuite metrics --category classification
|
|
179
|
+
evalsuite info classification.mcc
|
|
180
|
+
evalsuite benchmark --quick
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
|
|
184
|
+
(text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
|
|
185
|
+
|
|
186
|
+
## Performance
|
|
187
|
+
|
|
188
|
+
`evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
|
|
189
|
+
than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
|
|
190
|
+
|
|
136
191
|
## Metrics in this release
|
|
137
192
|
|
|
138
193
|
**Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
|
|
139
194
|
sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
|
|
140
195
|
MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
|
|
141
|
-
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy
|
|
196
|
+
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
197
|
+
calibration curve and expected calibration error.
|
|
142
198
|
|
|
143
199
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
144
200
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
11
11
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status:
|
|
13
|
+
> **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -82,12 +82,67 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
|
|
|
82
82
|
with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
|
|
83
83
|
p-values are adjusted for multiple comparisons (Holm by default).
|
|
84
84
|
|
|
85
|
+
## Classification report
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
report = es.classification_report(y_true, y_pred)
|
|
89
|
+
print(report) # per-class precision, recall, F1, specificity, support + averages
|
|
90
|
+
report.save("report.html") # also .csv .md .tex .json .txt
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Plots
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
|
|
101
|
+
es.plot.pr(y_true, prob) # AP and the prevalence line
|
|
102
|
+
es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
|
|
103
|
+
es.plot.confusion_matrix(y_true, y_pred, normalize="true")
|
|
104
|
+
es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
|
|
105
|
+
es.plot.comparison(es.compare(...)) # forest plot with CIs
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
|
|
109
|
+
computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
|
|
110
|
+
*and* line styles, so figures stay readable in greyscale print.
|
|
111
|
+
|
|
112
|
+
## Exports
|
|
113
|
+
|
|
114
|
+
Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
|
|
115
|
+
`to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
|
|
116
|
+
the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
|
|
117
|
+
|
|
118
|
+
## Command line
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
|
|
122
|
+
evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
|
|
123
|
+
evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
|
|
124
|
+
--prob lr=p_lr --prob rf=p_rf --plot comparison.png
|
|
125
|
+
evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
|
|
126
|
+
evalsuite metrics --category classification
|
|
127
|
+
evalsuite info classification.mcc
|
|
128
|
+
evalsuite benchmark --quick
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
|
|
132
|
+
(text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
|
|
133
|
+
|
|
134
|
+
## Performance
|
|
135
|
+
|
|
136
|
+
`evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
|
|
137
|
+
than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
|
|
138
|
+
|
|
85
139
|
## Metrics in this release
|
|
86
140
|
|
|
87
141
|
**Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
|
|
88
142
|
sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
|
|
89
143
|
MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
|
|
90
|
-
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy
|
|
144
|
+
one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
|
|
145
|
+
calibration curve and expected calibration error.
|
|
91
146
|
|
|
92
147
|
**Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
|
|
93
148
|
MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
|
|
@@ -14,7 +14,7 @@ authors = [{ name = "Manoj Kumar C S" }]
|
|
|
14
14
|
maintainers = [{ name = "Manoj Kumar C S" }]
|
|
15
15
|
keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
|
-
"Development Status ::
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
19
|
"Intended Audience :: Developers",
|
|
20
20
|
"Operating System :: OS Independent",
|
|
@@ -41,6 +41,7 @@ dev = [
|
|
|
41
41
|
"hypothesis>=6.80",
|
|
42
42
|
"scikit-learn>=1.2",
|
|
43
43
|
"statsmodels>=0.13",
|
|
44
|
+
"matplotlib>=3.5",
|
|
44
45
|
"ruff>=0.6",
|
|
45
46
|
"mypy>=1.10",
|
|
46
47
|
"pandas-stubs",
|
|
@@ -56,7 +57,8 @@ Source = "https://github.com/mkcs28/evalsuite-python"
|
|
|
56
57
|
Issues = "https://github.com/mkcs28/evalsuite-python/issues"
|
|
57
58
|
Changelog = "https://github.com/mkcs28/evalsuite-python/blob/main/CHANGELOG.md"
|
|
58
59
|
|
|
59
|
-
|
|
60
|
+
[project.scripts]
|
|
61
|
+
evalsuite = "evalsuite.cli.main:main"
|
|
60
62
|
|
|
61
63
|
[tool.hatch.version]
|
|
62
64
|
path = "src/evalsuite/version.py"
|
|
@@ -5,15 +5,17 @@
|
|
|
5
5
|
>>> print(result.summary()) # doctest: +SKIP
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
from . import classification, regression, stats
|
|
8
|
+
from . import classification, plot, regression, stats
|
|
9
9
|
from .api import evaluate
|
|
10
10
|
from .classification import (
|
|
11
11
|
accuracy,
|
|
12
12
|
average_precision,
|
|
13
13
|
balanced_accuracy,
|
|
14
14
|
brier_score,
|
|
15
|
+
calibration_curve,
|
|
15
16
|
cohen_kappa,
|
|
16
17
|
confusion_matrix,
|
|
18
|
+
expected_calibration_error,
|
|
17
19
|
f1,
|
|
18
20
|
fbeta,
|
|
19
21
|
hamming_loss,
|
|
@@ -59,6 +61,7 @@ from .regression import (
|
|
|
59
61
|
rse,
|
|
60
62
|
smape,
|
|
61
63
|
)
|
|
64
|
+
from .reporting import ClassificationReport, classification_report
|
|
62
65
|
from .stats import (
|
|
63
66
|
ComparisonResult,
|
|
64
67
|
ConfidenceInterval,
|
|
@@ -79,6 +82,11 @@ from .stats import (
|
|
|
79
82
|
from .version import __version__
|
|
80
83
|
|
|
81
84
|
__all__ = [
|
|
85
|
+
"plot",
|
|
86
|
+
"expected_calibration_error",
|
|
87
|
+
"calibration_curve",
|
|
88
|
+
"classification_report",
|
|
89
|
+
"ClassificationReport",
|
|
82
90
|
"stats",
|
|
83
91
|
"roc_auc_ci",
|
|
84
92
|
"proportion_ci",
|
|
@@ -232,6 +232,28 @@ def _evaluate_regression(
|
|
|
232
232
|
f"Unknown regression metric(s): {', '.join(unknown)}. Available: {', '.join(_REGRESSION)}."
|
|
233
233
|
)
|
|
234
234
|
out: dict[str, MetricResult] = {}
|
|
235
|
+
shared = reg._Inputs(y_true, y_pred, sample_weight) # validate once for every metric
|
|
236
|
+
token = reg._SHARED.set((id(y_true), id(y_pred), id(sample_weight), shared))
|
|
237
|
+
try:
|
|
238
|
+
_regression_loop(names, out, y_true, y_pred, sample_weight)
|
|
239
|
+
finally:
|
|
240
|
+
reg._SHARED.reset(token)
|
|
241
|
+
return EvaluationResult(
|
|
242
|
+
task="regression",
|
|
243
|
+
metrics=out,
|
|
244
|
+
n_samples=yt.shape[0],
|
|
245
|
+
target_type="continuous",
|
|
246
|
+
metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _regression_loop(
|
|
251
|
+
names: list[str],
|
|
252
|
+
out: dict[str, MetricResult],
|
|
253
|
+
y_true: ArrayLike,
|
|
254
|
+
y_pred: ArrayLike,
|
|
255
|
+
sample_weight: Optional[ArrayLike],
|
|
256
|
+
) -> None:
|
|
235
257
|
for name in names:
|
|
236
258
|
fn = _REGRESSION[name]
|
|
237
259
|
if name in _NO_WEIGHTS:
|
|
@@ -240,10 +262,3 @@ def _evaluate_regression(
|
|
|
240
262
|
out[name] = fn(y_true, y_pred)
|
|
241
263
|
else:
|
|
242
264
|
out[name] = fn(y_true, y_pred, sample_weight=sample_weight)
|
|
243
|
-
return EvaluationResult(
|
|
244
|
-
task="regression",
|
|
245
|
-
metrics=out,
|
|
246
|
-
n_samples=yt.shape[0],
|
|
247
|
-
target_type="continuous",
|
|
248
|
-
metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
|
|
249
|
-
)
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""Speed and memory benchmarks, against scikit-learn when it is installed.
|
|
2
|
+
|
|
3
|
+
Each case times the fastest of ``repeat`` runs (after one warm-up) and measures peak traced memory with
|
|
4
|
+
``tracemalloc`` (NumPy reports its allocations to it). Both libraries compute the same metrics on the same
|
|
5
|
+
data, and the largest absolute difference between their results is reported, so speed is never shown for
|
|
6
|
+
numbers that disagree.
|
|
7
|
+
|
|
8
|
+
>>> from evalsuite.benchmarks import run_benchmarks
|
|
9
|
+
>>> print(run_benchmarks(sizes=(10_000,), repeat=3)) # doctest: +SKIP
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import platform
|
|
16
|
+
import time
|
|
17
|
+
import tracemalloc
|
|
18
|
+
import warnings
|
|
19
|
+
from collections.abc import Callable, Sequence
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from types import MappingProxyType
|
|
22
|
+
from typing import TYPE_CHECKING, Any, Optional, cast
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
|
|
26
|
+
from .core.result import _json_safe, _latex_escape, _latex_table
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
import pandas as pd
|
|
30
|
+
|
|
31
|
+
__all__ = ["BenchmarkResult", "run_benchmarks"]
|
|
32
|
+
|
|
33
|
+
_HEADER = (
|
|
34
|
+
"case",
|
|
35
|
+
"n",
|
|
36
|
+
"evalsuite_ms",
|
|
37
|
+
"sklearn_ms",
|
|
38
|
+
"speedup",
|
|
39
|
+
"evalsuite_peak_mb",
|
|
40
|
+
"sklearn_peak_mb",
|
|
41
|
+
"max_abs_diff",
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
|
|
46
|
+
"""(fastest seconds, peak MiB, result)."""
|
|
47
|
+
with warnings.catch_warnings():
|
|
48
|
+
warnings.simplefilter("ignore")
|
|
49
|
+
result = fn() # warm-up
|
|
50
|
+
best = float("inf")
|
|
51
|
+
for _ in range(repeat):
|
|
52
|
+
t0 = time.perf_counter()
|
|
53
|
+
fn()
|
|
54
|
+
best = min(best, time.perf_counter() - t0)
|
|
55
|
+
tracemalloc.start()
|
|
56
|
+
try:
|
|
57
|
+
fn()
|
|
58
|
+
_, peak = tracemalloc.get_traced_memory()
|
|
59
|
+
finally:
|
|
60
|
+
tracemalloc.stop()
|
|
61
|
+
return best, peak / 2**20, result
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]]:
|
|
65
|
+
import evalsuite as es
|
|
66
|
+
|
|
67
|
+
y = rng.integers(0, 2, n)
|
|
68
|
+
p = np.where(rng.random(n) < 0.8, y, 1 - y)
|
|
69
|
+
prob = 1 / (1 + np.exp(-(2.0 * (y - 0.5) + rng.normal(0, 1, n))))
|
|
70
|
+
yk = rng.integers(0, 10, n)
|
|
71
|
+
pk = np.where(rng.random(n) < 0.7, yk, rng.integers(0, 10, n))
|
|
72
|
+
yr = rng.normal(10, 3, n)
|
|
73
|
+
pr = yr + rng.normal(0, 1, n)
|
|
74
|
+
names = ["accuracy", "balanced_accuracy", "precision", "recall", "f1", "specificity", "mcc", "cohen_kappa"]
|
|
75
|
+
|
|
76
|
+
def es_binary() -> list[float]:
|
|
77
|
+
r = es.evaluate(y, p, metrics=names)
|
|
78
|
+
return [float(r[m]) for m in names]
|
|
79
|
+
|
|
80
|
+
def es_reg() -> list[float]:
|
|
81
|
+
r = es.evaluate(yr, pr, metrics=["mae", "mse", "rmse", "r2"])
|
|
82
|
+
return [float(r[m]) for m in ("mae", "mse", "rmse", "r2")]
|
|
83
|
+
|
|
84
|
+
cases: list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]] = [
|
|
85
|
+
("binary: 8 label metrics via evaluate()", es_binary, None),
|
|
86
|
+
("10 classes: macro F1", lambda: [float(es.f1(yk, pk, average="macro"))], None),
|
|
87
|
+
("binary: ROC AUC", lambda: [float(es.roc_auc(y, prob))], None),
|
|
88
|
+
("regression: MAE, MSE, RMSE, R² via evaluate()", es_reg, None),
|
|
89
|
+
]
|
|
90
|
+
try:
|
|
91
|
+
import sklearn.metrics as skm # type: ignore[import-untyped]
|
|
92
|
+
except ImportError:
|
|
93
|
+
return cases
|
|
94
|
+
|
|
95
|
+
def sk_binary() -> list[float]:
|
|
96
|
+
return [
|
|
97
|
+
skm.accuracy_score(y, p),
|
|
98
|
+
skm.balanced_accuracy_score(y, p),
|
|
99
|
+
skm.precision_score(y, p),
|
|
100
|
+
skm.recall_score(y, p),
|
|
101
|
+
skm.f1_score(y, p),
|
|
102
|
+
skm.recall_score(y, p, pos_label=0),
|
|
103
|
+
skm.matthews_corrcoef(y, p),
|
|
104
|
+
skm.cohen_kappa_score(y, p),
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
def sk_reg() -> list[float]:
|
|
108
|
+
mse = skm.mean_squared_error(yr, pr)
|
|
109
|
+
return [skm.mean_absolute_error(yr, pr), mse, float(np.sqrt(mse)), skm.r2_score(yr, pr)]
|
|
110
|
+
|
|
111
|
+
sk = [sk_binary, lambda: [skm.f1_score(yk, pk, average="macro")], lambda: [skm.roc_auc_score(y, prob)], sk_reg]
|
|
112
|
+
return [(name, es_fn, sk_fn) for (name, es_fn, _), sk_fn in zip(cases, sk)]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@dataclass(frozen=True, eq=False)
|
|
116
|
+
class BenchmarkResult:
|
|
117
|
+
"""Rows of timings (milliseconds), peak memory (MiB) and agreement, plus the environment they ran in."""
|
|
118
|
+
|
|
119
|
+
rows: tuple[Any, ...]
|
|
120
|
+
environment: Any = field(default_factory=dict)
|
|
121
|
+
|
|
122
|
+
def __post_init__(self) -> None:
|
|
123
|
+
object.__setattr__(self, "rows", tuple(MappingProxyType(dict(r)) for r in self.rows))
|
|
124
|
+
object.__setattr__(self, "environment", MappingProxyType(dict(self.environment)))
|
|
125
|
+
|
|
126
|
+
def _cells(self, digits: int) -> list[list[str]]:
|
|
127
|
+
def f(v: Any, d: int = digits) -> str:
|
|
128
|
+
return "–" if v is None else f"{v:.{d}f}"
|
|
129
|
+
|
|
130
|
+
return [
|
|
131
|
+
[
|
|
132
|
+
r["case"],
|
|
133
|
+
f"{r['n']:,}",
|
|
134
|
+
f(r["evalsuite_ms"]),
|
|
135
|
+
f(r["sklearn_ms"]),
|
|
136
|
+
"–" if r["speedup"] is None else f"{r['speedup']:.2f}×",
|
|
137
|
+
f(r["evalsuite_peak_mb"], 2),
|
|
138
|
+
f(r["sklearn_peak_mb"], 2),
|
|
139
|
+
"–" if r["max_abs_diff"] is None else f"{r['max_abs_diff']:.1e}",
|
|
140
|
+
]
|
|
141
|
+
for r in self.rows
|
|
142
|
+
]
|
|
143
|
+
|
|
144
|
+
_TITLES = (
|
|
145
|
+
"Case",
|
|
146
|
+
"n",
|
|
147
|
+
"EvalSuite (ms)",
|
|
148
|
+
"scikit-learn (ms)",
|
|
149
|
+
"Speed-up",
|
|
150
|
+
"EvalSuite peak (MiB)",
|
|
151
|
+
"scikit-learn peak (MiB)",
|
|
152
|
+
"Max |difference|",
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
def summary(self, *, digits: int = 3) -> str:
|
|
156
|
+
cells = self._cells(digits)
|
|
157
|
+
widths = [max(len(t), *(len(c[i]) for c in cells)) for i, t in enumerate(self._TITLES)]
|
|
158
|
+
|
|
159
|
+
def line(c: Sequence[str]) -> str:
|
|
160
|
+
return " ".join(x.ljust(widths[i]) if i == 0 else x.rjust(widths[i]) for i, x in enumerate(c))
|
|
161
|
+
|
|
162
|
+
env = self.environment
|
|
163
|
+
head = (
|
|
164
|
+
f"EvalSuite {env['evalsuite']} benchmarks | Python {env['python']} | NumPy {env['numpy']}"
|
|
165
|
+
+ (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
|
|
166
|
+
+ f" | {env['machine']} | fastest of {env['repeat']} runs"
|
|
167
|
+
)
|
|
168
|
+
note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares the two libraries' results."
|
|
169
|
+
return "\n".join([head, "", line(self._TITLES), *(line(c) for c in cells), "", note])
|
|
170
|
+
|
|
171
|
+
def __repr__(self) -> str:
|
|
172
|
+
return self.summary()
|
|
173
|
+
|
|
174
|
+
def to_dict(self) -> dict[str, Any]:
|
|
175
|
+
return cast(
|
|
176
|
+
"dict[str, Any]",
|
|
177
|
+
_json_safe({"environment": dict(self.environment), "rows": [dict(r) for r in self.rows]}),
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
def to_json(self, *, indent: Optional[int] = 2) -> str:
|
|
181
|
+
return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
|
|
182
|
+
|
|
183
|
+
def to_dataframe(self) -> pd.DataFrame:
|
|
184
|
+
import pandas as pd
|
|
185
|
+
|
|
186
|
+
frame: pd.DataFrame = pd.DataFrame([dict(r) for r in self.rows])
|
|
187
|
+
return frame
|
|
188
|
+
|
|
189
|
+
def to_csv(self, path: Optional[str] = None) -> str:
|
|
190
|
+
from .core.export import csv_text
|
|
191
|
+
|
|
192
|
+
rows = [[r[h] if r[h] is not None else float("nan") for h in _HEADER] for r in self.rows]
|
|
193
|
+
text = csv_text(list(_HEADER), rows)
|
|
194
|
+
if path is not None:
|
|
195
|
+
with open(path, "w", encoding="utf-8", newline="") as fh:
|
|
196
|
+
fh.write(text)
|
|
197
|
+
return text
|
|
198
|
+
|
|
199
|
+
def to_markdown(self, *, digits: int = 3) -> str:
|
|
200
|
+
lines = ["| " + " | ".join(self._TITLES) + " |", "| --- |" + " ---: |" * (len(self._TITLES) - 1)]
|
|
201
|
+
return "\n".join(lines + ["| " + " | ".join(c) + " |" for c in self._cells(digits)])
|
|
202
|
+
|
|
203
|
+
def to_latex(self, *, digits: int = 3, caption: Optional[str] = None, label: Optional[str] = None) -> str:
|
|
204
|
+
return _latex_table(
|
|
205
|
+
[_latex_escape(t) for t in self._TITLES],
|
|
206
|
+
[[_latex_escape(x) for x in c] for c in self._cells(digits)],
|
|
207
|
+
caption=caption or "EvalSuite benchmarks.",
|
|
208
|
+
label=label,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
def to_html(self, *, digits: int = 3, full: bool = False) -> str:
|
|
212
|
+
from .core.export import html_document, html_table
|
|
213
|
+
|
|
214
|
+
table = html_table(list(self._TITLES), self._cells(digits), caption="Benchmarks")
|
|
215
|
+
env = self.environment
|
|
216
|
+
meta = f"Python {env['python']}, NumPy {env['numpy']}, {env['machine']}"
|
|
217
|
+
return html_document("EvalSuite benchmarks", table, meta) if full else table
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def run_benchmarks(
|
|
221
|
+
sizes: Sequence[int] = (1_000, 100_000, 1_000_000),
|
|
222
|
+
*,
|
|
223
|
+
repeat: int = 5,
|
|
224
|
+
compare_sklearn: bool = True,
|
|
225
|
+
random_state: Optional[int] = 0,
|
|
226
|
+
) -> BenchmarkResult:
|
|
227
|
+
"""Time and memory for common evaluation workloads at each size, against scikit-learn if installed."""
|
|
228
|
+
import evalsuite as es
|
|
229
|
+
|
|
230
|
+
if repeat < 1:
|
|
231
|
+
raise ValueError("repeat must be at least 1.")
|
|
232
|
+
rng = np.random.default_rng(random_state)
|
|
233
|
+
rows: list[dict[str, Any]] = []
|
|
234
|
+
sk_version: Optional[str] = None
|
|
235
|
+
if compare_sklearn:
|
|
236
|
+
try:
|
|
237
|
+
import sklearn
|
|
238
|
+
|
|
239
|
+
sk_version = sklearn.__version__
|
|
240
|
+
except ImportError:
|
|
241
|
+
compare_sklearn = False
|
|
242
|
+
for n in sizes:
|
|
243
|
+
for name, es_fn, sk_fn in _cases(int(n), rng):
|
|
244
|
+
es_t, es_mem, es_val = _measure(es_fn, repeat)
|
|
245
|
+
row: dict[str, Any] = {
|
|
246
|
+
"case": name,
|
|
247
|
+
"n": int(n),
|
|
248
|
+
"evalsuite_ms": es_t * 1000,
|
|
249
|
+
"evalsuite_peak_mb": es_mem,
|
|
250
|
+
"sklearn_ms": None,
|
|
251
|
+
"sklearn_peak_mb": None,
|
|
252
|
+
"speedup": None,
|
|
253
|
+
"max_abs_diff": None,
|
|
254
|
+
}
|
|
255
|
+
if compare_sklearn and sk_fn is not None:
|
|
256
|
+
sk_t, sk_mem, sk_val = _measure(sk_fn, repeat)
|
|
257
|
+
row.update(
|
|
258
|
+
sklearn_ms=sk_t * 1000,
|
|
259
|
+
sklearn_peak_mb=sk_mem,
|
|
260
|
+
speedup=sk_t / es_t if es_t else None,
|
|
261
|
+
max_abs_diff=float(np.max(np.abs(np.asarray(es_val, float) - np.asarray(sk_val, float)))),
|
|
262
|
+
)
|
|
263
|
+
rows.append(row)
|
|
264
|
+
env = {
|
|
265
|
+
"evalsuite": es.__version__,
|
|
266
|
+
"python": platform.python_version(),
|
|
267
|
+
"numpy": np.__version__,
|
|
268
|
+
"sklearn": sk_version,
|
|
269
|
+
"machine": f"{platform.system()} {platform.machine()}",
|
|
270
|
+
"repeat": repeat,
|
|
271
|
+
}
|
|
272
|
+
return BenchmarkResult(tuple(rows), env)
|
{evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b1}/src/evalsuite/classification/__init__.py
RENAMED
|
@@ -5,8 +5,10 @@ from .metrics import (
|
|
|
5
5
|
average_precision,
|
|
6
6
|
balanced_accuracy,
|
|
7
7
|
brier_score,
|
|
8
|
+
calibration_curve,
|
|
8
9
|
cohen_kappa,
|
|
9
10
|
confusion_matrix,
|
|
11
|
+
expected_calibration_error,
|
|
10
12
|
f1,
|
|
11
13
|
fbeta,
|
|
12
14
|
hamming_loss,
|
|
@@ -28,8 +30,10 @@ __all__ = [
|
|
|
28
30
|
"average_precision",
|
|
29
31
|
"balanced_accuracy",
|
|
30
32
|
"brier_score",
|
|
33
|
+
"calibration_curve",
|
|
31
34
|
"cohen_kappa",
|
|
32
35
|
"confusion_matrix",
|
|
36
|
+
"expected_calibration_error",
|
|
33
37
|
"f1",
|
|
34
38
|
"fbeta",
|
|
35
39
|
"hamming_loss",
|