evalsuite-python 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. evalsuite_python-0.1.0/.gitignore +19 -0
  2. evalsuite_python-0.1.0/CHANGELOG.md +82 -0
  3. evalsuite_python-0.1.0/CONTRIBUTING.md +26 -0
  4. evalsuite_python-0.1.0/LICENSE +21 -0
  5. evalsuite_python-0.1.0/PKG-INFO +247 -0
  6. evalsuite_python-0.1.0/README.md +195 -0
  7. evalsuite_python-0.1.0/pyproject.toml +99 -0
  8. evalsuite_python-0.1.0/src/evalsuite/__init__.py +159 -0
  9. evalsuite_python-0.1.0/src/evalsuite/__main__.py +5 -0
  10. evalsuite_python-0.1.0/src/evalsuite/api.py +264 -0
  11. evalsuite_python-0.1.0/src/evalsuite/benchmarks.py +272 -0
  12. evalsuite_python-0.1.0/src/evalsuite/classification/__init__.py +51 -0
  13. evalsuite_python-0.1.0/src/evalsuite/classification/_common.py +111 -0
  14. evalsuite_python-0.1.0/src/evalsuite/classification/metrics.py +943 -0
  15. evalsuite_python-0.1.0/src/evalsuite/cli/__init__.py +5 -0
  16. evalsuite_python-0.1.0/src/evalsuite/cli/main.py +389 -0
  17. evalsuite_python-0.1.0/src/evalsuite/core/__init__.py +1 -0
  18. evalsuite_python-0.1.0/src/evalsuite/core/context.py +181 -0
  19. evalsuite_python-0.1.0/src/evalsuite/core/exceptions.py +52 -0
  20. evalsuite_python-0.1.0/src/evalsuite/core/export.py +100 -0
  21. evalsuite_python-0.1.0/src/evalsuite/core/registry.py +90 -0
  22. evalsuite_python-0.1.0/src/evalsuite/core/result.py +379 -0
  23. evalsuite_python-0.1.0/src/evalsuite/core/types.py +23 -0
  24. evalsuite_python-0.1.0/src/evalsuite/core/validation.py +202 -0
  25. evalsuite_python-0.1.0/src/evalsuite/plot.py +296 -0
  26. evalsuite_python-0.1.0/src/evalsuite/py.typed +0 -0
  27. evalsuite_python-0.1.0/src/evalsuite/regression/__init__.py +41 -0
  28. evalsuite_python-0.1.0/src/evalsuite/regression/metrics.py +604 -0
  29. evalsuite_python-0.1.0/src/evalsuite/reporting.py +241 -0
  30. evalsuite_python-0.1.0/src/evalsuite/stats/__init__.py +25 -0
  31. evalsuite_python-0.1.0/src/evalsuite/stats/_resolve.py +90 -0
  32. evalsuite_python-0.1.0/src/evalsuite/stats/compare.py +414 -0
  33. evalsuite_python-0.1.0/src/evalsuite/stats/effect.py +113 -0
  34. evalsuite_python-0.1.0/src/evalsuite/stats/intervals.py +320 -0
  35. evalsuite_python-0.1.0/src/evalsuite/stats/paired.py +207 -0
  36. evalsuite_python-0.1.0/src/evalsuite/stats/results.py +129 -0
  37. evalsuite_python-0.1.0/src/evalsuite/version.py +3 -0
  38. evalsuite_python-0.1.0/tests/__init__.py +0 -0
  39. evalsuite_python-0.1.0/tests/classification/__init__.py +0 -0
  40. evalsuite_python-0.1.0/tests/classification/test_against_sklearn.py +239 -0
  41. evalsuite_python-0.1.0/tests/conftest.py +18 -0
  42. evalsuite_python-0.1.0/tests/integration/__init__.py +0 -0
  43. evalsuite_python-0.1.0/tests/output/__init__.py +0 -0
  44. evalsuite_python-0.1.0/tests/output/test_benchmarks.py +32 -0
  45. evalsuite_python-0.1.0/tests/output/test_cli.py +195 -0
  46. evalsuite_python-0.1.0/tests/output/test_plot.py +133 -0
  47. evalsuite_python-0.1.0/tests/output/test_reporting.py +143 -0
  48. evalsuite_python-0.1.0/tests/regression/__init__.py +0 -0
  49. evalsuite_python-0.1.0/tests/regression/test_against_sklearn.py +130 -0
  50. evalsuite_python-0.1.0/tests/stats/__init__.py +0 -0
  51. evalsuite_python-0.1.0/tests/stats/test_branches.py +86 -0
  52. evalsuite_python-0.1.0/tests/stats/test_compare.py +152 -0
  53. evalsuite_python-0.1.0/tests/stats/test_reference.py +258 -0
  54. evalsuite_python-0.1.0/tests/unit/__init__.py +0 -0
  55. evalsuite_python-0.1.0/tests/unit/test_core.py +218 -0
  56. evalsuite_python-0.1.0/tests/unit/test_edges.py +231 -0
@@ -0,0 +1,19 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .eggs/
5
+ build/
6
+ dist/
7
+ .venv/
8
+ venv/
9
+ .env
10
+ .coverage
11
+ coverage.xml
12
+ htmlcov/
13
+ .pytest_cache/
14
+ .mypy_cache/
15
+ .ruff_cache/
16
+ .hypothesis/
17
+ .DS_Store
18
+ .idea/
19
+ .vscode/
@@ -0,0 +1,82 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0] - 2026-10-08
10
+
11
+ First stable release. Everything from the 0.1.0 roadmap: classification and regression metrics, result
12
+ system, input validation, metric registry, model comparison with confidence intervals and paired tests,
13
+ plots, HTML/CSV/LaTeX/Markdown reports, classification report, command-line tool and benchmarks.
14
+
15
+ ### Changed
16
+ - Development status: Production/Stable.
17
+ - README (PyPI description) now includes the benchmark table against scikit-learn.
18
+
19
+ ## [0.1.0b2]
20
+
21
+ ### Changed
22
+ - `es.plot.calibration`: the legend now sits below the axes by default so it no longer covers the
23
+ curves; new `legend_loc` argument (`"below"` or any matplotlib location).
24
+
25
+ ## [0.1.0b1]
26
+
27
+ Feature-complete for 0.1.0.
28
+
29
+ ### Added
30
+
31
+ - Plots (`es.plot`, optional `[plot]` extra): ROC, precision-recall, calibration, confusion matrix, residuals /
32
+ predicted-vs-true, and model-comparison forest plots. Values come from EvalSuite's metrics; several models
33
+ get distinct colours and line styles. Importing `evalsuite` never imports matplotlib.
34
+ - Reporting: `classification_report` (per-class precision, recall, F1, specificity, support; accuracy, micro,
35
+ macro and weighted averages; matches scikit-learn); `to_html()` and `to_csv()` on every result; `save(path)`
36
+ choosing the format from the extension (.json .csv .md .tex .html .txt).
37
+ - `calibration_curve` and `expected_calibration_error` (matches scikit-learn's calibration curve).
38
+ - Command line: `evalsuite evaluate | report | compare | plot | metrics | info | benchmark`, and
39
+ `python -m evalsuite`. Reads CSV, TSV, Parquet and JSON; clear one-line errors with exit code 2.
40
+ - Benchmarks (`evalsuite.benchmarks.run_benchmarks`, `evalsuite benchmark`): time and peak memory against
41
+ scikit-learn, with a check that both libraries return the same numbers. Results in `BENCHMARKS.md`.
42
+
43
+ ### Changed
44
+
45
+ - Regression `evaluate()` validates inputs once for all metrics and uses a faster unweighted mean (2× faster
46
+ on large arrays).
47
+
48
+ ## [0.1.0a2]
49
+
50
+ ### Added
51
+
52
+ - Model comparison (`es.compare`): per-model confidence intervals from paired bootstrap resamples, pairwise
53
+ tests (McNemar for accuracy, DeLong for binary ROC AUC, paired bootstrap otherwise), multiple-comparison
54
+ correction, best model per metric, and summary/pandas/Markdown/LaTeX/JSON export.
55
+ - Confidence intervals: `bootstrap_ci` (percentile, basic, BCa; stratified and reproducible),
56
+ `proportion_ci` and `accuracy_ci` (Wilson, Clopper-Pearson, normal), `roc_auc_ci` (DeLong).
57
+ - Paired tests: `mcnemar_test`, `delong_test`, `paired_bootstrap_test`.
58
+ - Effect sizes: `cohens_d` (independent and paired), `hedges_g`, `cliffs_delta`; `adjust_pvalues`
59
+ (Holm, Bonferroni, Benjamini-Hochberg, Benjamini-Yekutieli).
60
+ - Reference tests against statsmodels and SciPy, brute-force DeLong checks and coverage simulations.
61
+ - Python 3.14 support and CI.
62
+
63
+ ### Fixed
64
+
65
+ - 0.1.0a1 installed an `evalsuite` console command although the CLI is not implemented yet, so the command
66
+ failed with `ModuleNotFoundError`. The entry point is removed until the CLI ships.
67
+
68
+ ## [0.1.0a1]
69
+
70
+ First alpha, published to reserve the name and test the release pipeline. Distribution name
71
+ `evalsuite-python` (`pip install --pre evalsuite-python`), imported as `evalsuite`.
72
+
73
+ ### Added
74
+
75
+ - Core: input validation with actionable errors, exception hierarchy, immutable result objects with
76
+ JSON/pandas/Markdown/LaTeX export, metric registry (`list_metrics`, `metric_info`), and an evaluation
77
+ context that computes the confusion matrix once per evaluation.
78
+ - Classification metrics for binary, multiclass and multilabel targets with all averaging modes and
79
+ sample weights.
80
+ - Regression metrics for single- and multi-output targets with sample weights.
81
+ - `evaluate()` high-level API with task inference and default metric sets.
82
+ - Reference tests against scikit-learn and property-based tests; Python 3.9 to 3.13.
@@ -0,0 +1,26 @@
1
+ # Contributing to EvalSuite
2
+
3
+ Thank you for helping. EvalSuite values correctness and clarity over breadth.
4
+
5
+ ## Adding or changing a metric
6
+
7
+ 1. Implement it with `@register(...)`: definition, formula, range, input requirements and at least one
8
+ reference.
9
+ 2. Validate inputs through `evalsuite.core.validation`; raise `InputValidationError` for bad inputs and
10
+ `MetricInputError` for values outside the metric's domain, with a message saying how to fix it.
11
+ 3. Test it: against an established implementation where definitions coincide (scikit-learn, SciPy), plus
12
+ hand-computed cases, edge cases and, where useful, property-based tests.
13
+ 4. Document conventions explicitly (averaging, label order, zero division) and update `CHANGELOG.md`.
14
+
15
+ ## Checks
16
+
17
+ ```bash
18
+ pytest --cov=evalsuite # coverage must stay at or above 95% for core modules
19
+ ruff check . && ruff format --check .
20
+ mypy # strict
21
+ ```
22
+
23
+ ## Reporting issues
24
+
25
+ Include the EvalSuite, Python and NumPy versions, a minimal example, the expected result (with a reference)
26
+ and the actual result.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Manoj Kumar C S
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,247 @@
1
+ Metadata-Version: 2.5
2
+ Name: evalsuite-python
3
+ Version: 0.1.0
4
+ Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
+ Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
+ Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
7
+ Project-URL: Source, https://github.com/mkcs28/evalsuite-python
8
+ Project-URL: Issues, https://github.com/mkcs28/evalsuite-python/issues
9
+ Project-URL: Changelog, https://github.com/mkcs28/evalsuite-python/blob/main/CHANGELOG.md
10
+ Author: Manoj Kumar C S
11
+ Maintainer: Manoj Kumar C S
12
+ License-Expression: MIT
13
+ License-File: LICENSE
14
+ Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
15
+ Classifier: Development Status :: 5 - Production/Stable
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: Intended Audience :: Science/Research
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3 :: Only
21
+ Classifier: Programming Language :: Python :: 3.9
22
+ Classifier: Programming Language :: Python :: 3.10
23
+ Classifier: Programming Language :: Python :: 3.11
24
+ Classifier: Programming Language :: Python :: 3.12
25
+ Classifier: Programming Language :: Python :: 3.13
26
+ Classifier: Programming Language :: Python :: 3.14
27
+ Classifier: Topic :: Scientific/Engineering
28
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
29
+ Classifier: Typing :: Typed
30
+ Requires-Python: >=3.9
31
+ Requires-Dist: numpy>=1.22
32
+ Requires-Dist: pandas>=1.4
33
+ Requires-Dist: scipy>=1.8
34
+ Provides-Extra: all
35
+ Requires-Dist: matplotlib>=3.5; extra == 'all'
36
+ Provides-Extra: dev
37
+ Requires-Dist: build; extra == 'dev'
38
+ Requires-Dist: hypothesis>=6.80; extra == 'dev'
39
+ Requires-Dist: matplotlib>=3.5; extra == 'dev'
40
+ Requires-Dist: mypy>=1.10; extra == 'dev'
41
+ Requires-Dist: pandas-stubs; extra == 'dev'
42
+ Requires-Dist: pip-audit; extra == 'dev'
43
+ Requires-Dist: pytest-cov>=4; extra == 'dev'
44
+ Requires-Dist: pytest>=7; extra == 'dev'
45
+ Requires-Dist: ruff>=0.6; extra == 'dev'
46
+ Requires-Dist: scikit-learn>=1.2; extra == 'dev'
47
+ Requires-Dist: statsmodels>=0.13; extra == 'dev'
48
+ Requires-Dist: twine; extra == 'dev'
49
+ Provides-Extra: plot
50
+ Requires-Dist: matplotlib>=3.5; extra == 'plot'
51
+ Description-Content-Type: text/markdown
52
+
53
+ # EvalSuite
54
+
55
+ [![CI](https://github.com/mkcs28/evalsuite-python/actions/workflows/ci.yml/badge.svg)](https://github.com/mkcs28/evalsuite-python/actions/workflows/ci.yml)
56
+ [![PyPI](https://img.shields.io/pypi/v/evalsuite-python)](https://pypi.org/project/evalsuite-python/)
57
+ [![Python](https://img.shields.io/pypi/pyversions/evalsuite-python)](https://pypi.org/project/evalsuite-python/)
58
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
59
+
60
+ **Unified, reproducible evaluation for machine learning and research.**
61
+
62
+ EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
63
+ object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
64
+
65
+ > **Status: stable (0.1.0).** Every item on the 0.1.0 roadmap is implemented and verified.
66
+
67
+ ## Installation
68
+
69
+ ```bash
70
+ pip install evalsuite-python
71
+ ```
72
+
73
+ The package is installed as `evalsuite-python` and imported as `evalsuite`:
74
+
75
+ ```python
76
+ import evalsuite as es
77
+ ```
78
+
79
+ ## Why EvalSuite
80
+
81
+ - **One consistent API.** Every metric returns a result object that behaves like a number and exports to
82
+ JSON, pandas, Markdown and LaTeX.
83
+ - **Explicit conventions.** Averaging, label order, the positive class and zero-division behaviour are stated
84
+ and recorded in every result, never silently assumed.
85
+ - **Validated.** Each metric is tested against scikit-learn where definitions coincide, plus property-based
86
+ tests and edge cases.
87
+ - **Documented.** Every metric carries its definition, formula, range, input requirements and references,
88
+ available programmatically through `metric_info()`.
89
+ - **Efficient.** `evaluate()` validates inputs once and computes the confusion matrix once for all metrics.
90
+ - **Lightweight.** Requires only NumPy, SciPy and pandas.
91
+
92
+ ## Quick start
93
+
94
+ ```python
95
+ import evalsuite as es
96
+
97
+ y_true = [0, 1, 1, 0, 1, 0]
98
+ y_pred = [0, 1, 0, 0, 1, 1]
99
+ y_prob = [0.1, 0.9, 0.4, 0.2, 0.8, 0.6]
100
+
101
+ result = es.evaluate(y_true, y_pred, y_prob=y_prob)
102
+ print(result.summary())
103
+
104
+ result["f1"] # MetricResult(f1=0.666667)
105
+ f"{result['mcc']:.3f}" # '0.333'
106
+ result.to_latex(caption="Test-set performance")
107
+ result.to_dataframe()
108
+
109
+ es.f1(y_true, y_pred) # individual metrics
110
+ es.roc_auc(y_true, y_prob)
111
+ es.metric_info("classification.mcc").formula # documentation
112
+ es.list_metrics("regression")
113
+ ```
114
+
115
+ ## Comparing models
116
+
117
+ ```python
118
+ result = es.compare(
119
+ y_true,
120
+ {"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
121
+ probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
122
+ random_state=0,
123
+ )
124
+ print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
125
+ result.to_latex(label="tab:models")
126
+
127
+ es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
128
+ es.accuracy_ci(y_true, y_pred) # Wilson interval
129
+ es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
130
+ es.mcnemar_test(y_true, pred_a, pred_b)
131
+ ```
132
+
133
+ Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
134
+ with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
135
+ p-values are adjusted for multiple comparisons (Holm by default).
136
+
137
+ ## Classification report
138
+
139
+ ```python
140
+ report = es.classification_report(y_true, y_pred)
141
+ print(report) # per-class precision, recall, F1, specificity, support + averages
142
+ report.save("report.html") # also .csv .md .tex .json .txt
143
+ ```
144
+
145
+ ## Plots
146
+
147
+ ```bash
148
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
149
+ ```
150
+
151
+ ```python
152
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
153
+ es.plot.pr(y_true, prob) # AP and the prevalence line
154
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
155
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
156
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
157
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
158
+ ```
159
+
160
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
161
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
162
+ *and* line styles, so figures stay readable in greyscale print.
163
+
164
+ ## Exports
165
+
166
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
167
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
168
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
169
+
170
+ ## Command line
171
+
172
+ ```bash
173
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
174
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
175
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
176
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
177
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
178
+ evalsuite metrics --category classification
179
+ evalsuite info classification.mcc
180
+ evalsuite benchmark --quick
181
+ ```
182
+
183
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
184
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
185
+
186
+ ## Performance
187
+
188
+ Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
189
+ scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
190
+ (largest difference 1.1e-16).
191
+
192
+ | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
193
+ | --- | ---: | ---: | ---: | ---: | ---: |
194
+ | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
195
+ | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
196
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
197
+ | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
198
+ | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
199
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
200
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
201
+
202
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
203
+ speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
204
+ infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
205
+ table and notes in
206
+ [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
207
+
208
+ ## Metrics in this release
209
+
210
+ **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
211
+ sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
212
+ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
213
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
214
+ calibration curve and expected calibration error.
215
+
216
+ **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
217
+ MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
218
+ loss, Huber loss, relative absolute error, relative squared error.
219
+
220
+ ## Conventions
221
+
222
+ - `average="auto"` resolves to `"binary"` for binary targets and `"macro"` otherwise; the resolved value
223
+ is stored in `result.params["average"]`.
224
+ - Labels are sorted unless you pass `labels=[...]`; that order defines per-class outputs and the columns
225
+ of 2-D `y_prob`.
226
+ - Undefined ratios (zero denominators) return 0 **with an `UndefinedMetricWarning`**; pass
227
+ `zero_division=np.nan` to propagate NaN, or `0`/`1` to choose silently.
228
+ - Domain violations raise clear errors instead of being patched over (for example MAPE with zero targets).
229
+
230
+ ## Development
231
+
232
+ ```bash
233
+ python -m venv .venv && source .venv/bin/activate
234
+ pip install -e ".[dev]"
235
+ pytest --cov=evalsuite
236
+ ruff check . && ruff format --check . && mypy
237
+ ```
238
+
239
+ ## Links
240
+
241
+ - PyPI: https://pypi.org/project/evalsuite-python/
242
+ - Website and documentation: https://evalsuite-nine.vercel.app
243
+ - Website source: https://github.com/mkcs28/evalsuite
244
+
245
+ ## License
246
+
247
+ MIT. See [LICENSE](LICENSE).
@@ -0,0 +1,195 @@
1
+ # EvalSuite
2
+
3
+ [![CI](https://github.com/mkcs28/evalsuite-python/actions/workflows/ci.yml/badge.svg)](https://github.com/mkcs28/evalsuite-python/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/evalsuite-python)](https://pypi.org/project/evalsuite-python/)
5
+ [![Python](https://img.shields.io/pypi/pyversions/evalsuite-python)](https://pypi.org/project/evalsuite-python/)
6
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
7
+
8
+ **Unified, reproducible evaluation for machine learning and research.**
9
+
10
+ EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
11
+ object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
+
13
+ > **Status: stable (0.1.0).** Every item on the 0.1.0 roadmap is implemented and verified.
14
+
15
+ ## Installation
16
+
17
+ ```bash
18
+ pip install evalsuite-python
19
+ ```
20
+
21
+ The package is installed as `evalsuite-python` and imported as `evalsuite`:
22
+
23
+ ```python
24
+ import evalsuite as es
25
+ ```
26
+
27
+ ## Why EvalSuite
28
+
29
+ - **One consistent API.** Every metric returns a result object that behaves like a number and exports to
30
+ JSON, pandas, Markdown and LaTeX.
31
+ - **Explicit conventions.** Averaging, label order, the positive class and zero-division behaviour are stated
32
+ and recorded in every result, never silently assumed.
33
+ - **Validated.** Each metric is tested against scikit-learn where definitions coincide, plus property-based
34
+ tests and edge cases.
35
+ - **Documented.** Every metric carries its definition, formula, range, input requirements and references,
36
+ available programmatically through `metric_info()`.
37
+ - **Efficient.** `evaluate()` validates inputs once and computes the confusion matrix once for all metrics.
38
+ - **Lightweight.** Requires only NumPy, SciPy and pandas.
39
+
40
+ ## Quick start
41
+
42
+ ```python
43
+ import evalsuite as es
44
+
45
+ y_true = [0, 1, 1, 0, 1, 0]
46
+ y_pred = [0, 1, 0, 0, 1, 1]
47
+ y_prob = [0.1, 0.9, 0.4, 0.2, 0.8, 0.6]
48
+
49
+ result = es.evaluate(y_true, y_pred, y_prob=y_prob)
50
+ print(result.summary())
51
+
52
+ result["f1"] # MetricResult(f1=0.666667)
53
+ f"{result['mcc']:.3f}" # '0.333'
54
+ result.to_latex(caption="Test-set performance")
55
+ result.to_dataframe()
56
+
57
+ es.f1(y_true, y_pred) # individual metrics
58
+ es.roc_auc(y_true, y_prob)
59
+ es.metric_info("classification.mcc").formula # documentation
60
+ es.list_metrics("regression")
61
+ ```
62
+
63
+ ## Comparing models
64
+
65
+ ```python
66
+ result = es.compare(
67
+ y_true,
68
+ {"logistic": pred_lr, "forest": pred_rf, "boosting": pred_gb},
69
+ probabilities={"logistic": prob_lr, "forest": prob_rf, "boosting": prob_gb},
70
+ random_state=0,
71
+ )
72
+ print(result.summary()) # estimates with 95% CIs, paired tests, Holm-adjusted p-values
73
+ result.to_latex(label="tab:models")
74
+
75
+ es.bootstrap_ci("f1", y_true, y_pred, average="macro", random_state=0) # BCa interval for any metric
76
+ es.accuracy_ci(y_true, y_pred) # Wilson interval
77
+ es.delong_test(y_true, prob_a, prob_b) # two correlated AUCs
78
+ es.mcnemar_test(y_true, pred_a, pred_b)
79
+ ```
80
+
81
+ Every model is evaluated on the same bootstrap resamples, so differences are paired. Accuracy is compared
82
+ with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
83
+ p-values are adjusted for multiple comparisons (Holm by default).
84
+
85
+ ## Classification report
86
+
87
+ ```python
88
+ report = es.classification_report(y_true, y_pred)
89
+ print(report) # per-class precision, recall, F1, specificity, support + averages
90
+ report.save("report.html") # also .csv .md .tex .json .txt
91
+ ```
92
+
93
+ ## Plots
94
+
95
+ ```bash
96
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
97
+ ```
98
+
99
+ ```python
100
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
101
+ es.plot.pr(y_true, prob) # AP and the prevalence line
102
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
103
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
104
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
105
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
106
+ ```
107
+
108
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
109
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
110
+ *and* line styles, so figures stay readable in greyscale print.
111
+
112
+ ## Exports
113
+
114
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
115
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
116
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
117
+
118
+ ## Command line
119
+
120
+ ```bash
121
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
122
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
123
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
124
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
125
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
126
+ evalsuite metrics --category classification
127
+ evalsuite info classification.mcc
128
+ evalsuite benchmark --quick
129
+ ```
130
+
131
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
132
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
133
+
134
+ ## Performance
135
+
136
+ Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
137
+ scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
138
+ (largest difference 1.1e-16).
139
+
140
+ | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
141
+ | --- | ---: | ---: | ---: | ---: | ---: |
142
+ | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
143
+ | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
144
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
145
+ | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
146
+ | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
147
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
148
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
149
+
150
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
151
+ speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
152
+ infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
153
+ table and notes in
154
+ [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
155
+
156
+ ## Metrics in this release
157
+
158
+ **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
159
+ sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
160
+ MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
161
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
162
+ calibration curve and expected calibration error.
163
+
164
+ **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
165
+ MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
166
+ loss, Huber loss, relative absolute error, relative squared error.
167
+
168
+ ## Conventions
169
+
170
+ - `average="auto"` resolves to `"binary"` for binary targets and `"macro"` otherwise; the resolved value
171
+ is stored in `result.params["average"]`.
172
+ - Labels are sorted unless you pass `labels=[...]`; that order defines per-class outputs and the columns
173
+ of 2-D `y_prob`.
174
+ - Undefined ratios (zero denominators) return 0 **with an `UndefinedMetricWarning`**; pass
175
+ `zero_division=np.nan` to propagate NaN, or `0`/`1` to choose silently.
176
+ - Domain violations raise clear errors instead of being patched over (for example MAPE with zero targets).
177
+
178
+ ## Development
179
+
180
+ ```bash
181
+ python -m venv .venv && source .venv/bin/activate
182
+ pip install -e ".[dev]"
183
+ pytest --cov=evalsuite
184
+ ruff check . && ruff format --check . && mypy
185
+ ```
186
+
187
+ ## Links
188
+
189
+ - PyPI: https://pypi.org/project/evalsuite-python/
190
+ - Website and documentation: https://evalsuite-nine.vercel.app
191
+ - Website source: https://github.com/mkcs28/evalsuite
192
+
193
+ ## License
194
+
195
+ MIT. See [LICENSE](LICENSE).