evalsuite-python 0.1.0b2__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/CHANGELOG.md +20 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/LICENSE +1 -1
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/PKG-INFO +28 -7
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/README.md +24 -3
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/pyproject.toml +3 -3
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/version.py +1 -1
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/.gitignore +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/CONTRIBUTING.md +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/__main__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/api.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/benchmarks.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/classification/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/classification/_common.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/classification/metrics.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/cli/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/cli/main.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/context.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/exceptions.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/export.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/registry.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/result.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/types.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/core/validation.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/plot.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/py.typed +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/regression/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/regression/metrics.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/reporting.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/_resolve.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/compare.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/effect.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/intervals.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/paired.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/stats/results.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/classification/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/classification/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/conftest.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/integration/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/output/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/output/test_benchmarks.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/output/test_cli.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/output/test_plot.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/output/test_reporting.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/regression/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/regression/test_against_sklearn.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/stats/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/stats/test_branches.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/stats/test_compare.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/stats/test_reference.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/unit/__init__.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/unit/test_core.py +0 -0
- {evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/unit/test_edges.py +0 -0
|
@@ -6,6 +6,26 @@ All notable changes to this project are documented here. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.2] - 2026-10-08
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
- LICENSE: copyright held by Manoj Kumar C S and Nikhil D Bharadwaj.
|
|
13
|
+
|
|
14
|
+
## [0.1.1] - 2026-10-08
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
- Credits: Manoj Kumar C S and Nikhil D Bharadwaj listed as authors and maintainers (README and package metadata).
|
|
18
|
+
|
|
19
|
+
## [0.1.0] - 2026-10-08
|
|
20
|
+
|
|
21
|
+
First stable release. Everything from the 0.1.0 roadmap: classification and regression metrics, result
|
|
22
|
+
system, input validation, metric registry, model comparison with confidence intervals and paired tests,
|
|
23
|
+
plots, HTML/CSV/LaTeX/Markdown reports, classification report, command-line tool and benchmarks.
|
|
24
|
+
|
|
25
|
+
### Changed
|
|
26
|
+
- Development status: Production/Stable.
|
|
27
|
+
- README (PyPI description) now includes the benchmark table against scikit-learn.
|
|
28
|
+
|
|
9
29
|
## [0.1.0b2]
|
|
10
30
|
|
|
11
31
|
### Changed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
MIT License
|
|
2
2
|
|
|
3
|
-
Copyright (c) 2026 Manoj Kumar C S
|
|
3
|
+
Copyright (c) 2026 Manoj Kumar C S and Nikhil D Bharadwaj
|
|
4
4
|
|
|
5
5
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
6
|
of this software and associated documentation files (the "Software"), to deal
|
|
@@ -1,18 +1,18 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: evalsuite-python
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
|
|
5
5
|
Project-URL: Homepage, https://evalsuite-nine.vercel.app
|
|
6
6
|
Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
|
|
7
7
|
Project-URL: Source, https://github.com/mkcs28/evalsuite-python
|
|
8
8
|
Project-URL: Issues, https://github.com/mkcs28/evalsuite-python/issues
|
|
9
9
|
Project-URL: Changelog, https://github.com/mkcs28/evalsuite-python/blob/main/CHANGELOG.md
|
|
10
|
-
Author: Manoj Kumar C S
|
|
11
|
-
Maintainer: Manoj Kumar C S
|
|
10
|
+
Author: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
11
|
+
Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
|
|
12
12
|
License-Expression: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
14
|
Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
|
|
15
|
-
Classifier: Development Status ::
|
|
15
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: Intended Audience :: Science/Research
|
|
18
18
|
Classifier: Operating System :: OS Independent
|
|
@@ -62,7 +62,7 @@ Description-Content-Type: text/markdown
|
|
|
62
62
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
63
63
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
64
64
|
|
|
65
|
-
> **Status:
|
|
65
|
+
> **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
|
|
66
66
|
|
|
67
67
|
## Installation
|
|
68
68
|
|
|
@@ -185,8 +185,25 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
185
185
|
|
|
186
186
|
## Performance
|
|
187
187
|
|
|
188
|
-
|
|
189
|
-
|
|
188
|
+
Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
189
|
+
scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
|
|
190
|
+
(largest difference 1.1e-16).
|
|
191
|
+
|
|
192
|
+
| Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
|
|
193
|
+
| --- | ---: | ---: | ---: | ---: | ---: |
|
|
194
|
+
| 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
|
|
195
|
+
| 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
|
|
196
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
|
|
197
|
+
| macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
|
|
198
|
+
| ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
|
|
199
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
|
|
200
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
|
|
201
|
+
|
|
202
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
|
|
203
|
+
speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
|
|
204
|
+
infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
|
|
205
|
+
table and notes in
|
|
206
|
+
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
190
207
|
|
|
191
208
|
## Metrics in this release
|
|
192
209
|
|
|
@@ -225,6 +242,10 @@ ruff check . && ruff format --check . && mypy
|
|
|
225
242
|
- Website and documentation: https://evalsuite-nine.vercel.app
|
|
226
243
|
- Website source: https://github.com/mkcs28/evalsuite
|
|
227
244
|
|
|
245
|
+
## Credits
|
|
246
|
+
|
|
247
|
+
Authors and maintainers: **Manoj Kumar C S** and **Nikhil D Bharadwaj**.
|
|
248
|
+
|
|
228
249
|
## License
|
|
229
250
|
|
|
230
251
|
MIT. See [LICENSE](LICENSE).
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
|
|
11
11
|
object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
|
|
12
12
|
|
|
13
|
-
> **Status:
|
|
13
|
+
> **Status: stable (0.1.2).** Every item on the 0.1.0 roadmap is implemented and verified.
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
@@ -133,8 +133,25 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
|
|
|
133
133
|
|
|
134
134
|
## Performance
|
|
135
135
|
|
|
136
|
-
|
|
137
|
-
|
|
136
|
+
Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
|
|
137
|
+
scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
|
|
138
|
+
(largest difference 1.1e-16).
|
|
139
|
+
|
|
140
|
+
| Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
|
|
141
|
+
| --- | ---: | ---: | ---: | ---: | ---: |
|
|
142
|
+
| 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
|
|
143
|
+
| 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
|
|
144
|
+
| 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
|
|
145
|
+
| macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
|
|
146
|
+
| ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
|
|
147
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
|
|
148
|
+
| MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
|
|
149
|
+
|
|
150
|
+
`evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
|
|
151
|
+
speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
|
|
152
|
+
infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
|
|
153
|
+
table and notes in
|
|
154
|
+
[BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
|
|
138
155
|
|
|
139
156
|
## Metrics in this release
|
|
140
157
|
|
|
@@ -173,6 +190,10 @@ ruff check . && ruff format --check . && mypy
|
|
|
173
190
|
- Website and documentation: https://evalsuite-nine.vercel.app
|
|
174
191
|
- Website source: https://github.com/mkcs28/evalsuite
|
|
175
192
|
|
|
193
|
+
## Credits
|
|
194
|
+
|
|
195
|
+
Authors and maintainers: **Manoj Kumar C S** and **Nikhil D Bharadwaj**.
|
|
196
|
+
|
|
176
197
|
## License
|
|
177
198
|
|
|
178
199
|
MIT. See [LICENSE](LICENSE).
|
|
@@ -10,11 +10,11 @@ readme = "README.md"
|
|
|
10
10
|
license = "MIT"
|
|
11
11
|
license-files = ["LICENSE"]
|
|
12
12
|
requires-python = ">=3.9"
|
|
13
|
-
authors = [{ name = "Manoj Kumar C S" }]
|
|
14
|
-
maintainers = [{ name = "Manoj Kumar C S" }]
|
|
13
|
+
authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
14
|
+
maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
|
|
15
15
|
keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
|
|
16
16
|
classifiers = [
|
|
17
|
-
"Development Status ::
|
|
17
|
+
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Science/Research",
|
|
19
19
|
"Intended Audience :: Developers",
|
|
20
20
|
"Operating System :: OS Independent",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/src/evalsuite/classification/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/classification/test_against_sklearn.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{evalsuite_python-0.1.0b2 → evalsuite_python-0.1.2}/tests/regression/test_against_sklearn.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|