evalsuite-python 0.1.0b1__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/CHANGELOG.md +21 -0
  2. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/PKG-INFO +28 -7
  3. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/README.md +24 -3
  4. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/pyproject.toml +3 -3
  5. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/plot.py +9 -2
  6. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/version.py +1 -1
  7. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/output/test_plot.py +10 -0
  8. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/.gitignore +0 -0
  9. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/CONTRIBUTING.md +0 -0
  10. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/LICENSE +0 -0
  11. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/__init__.py +0 -0
  12. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/__main__.py +0 -0
  13. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/api.py +0 -0
  14. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/benchmarks.py +0 -0
  15. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/classification/__init__.py +0 -0
  16. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/classification/_common.py +0 -0
  17. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/classification/metrics.py +0 -0
  18. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/cli/__init__.py +0 -0
  19. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/cli/main.py +0 -0
  20. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/__init__.py +0 -0
  21. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/context.py +0 -0
  22. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/exceptions.py +0 -0
  23. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/export.py +0 -0
  24. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/registry.py +0 -0
  25. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/result.py +0 -0
  26. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/types.py +0 -0
  27. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/core/validation.py +0 -0
  28. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/py.typed +0 -0
  29. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/regression/__init__.py +0 -0
  30. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/regression/metrics.py +0 -0
  31. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/reporting.py +0 -0
  32. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/__init__.py +0 -0
  33. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/_resolve.py +0 -0
  34. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/compare.py +0 -0
  35. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/effect.py +0 -0
  36. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/intervals.py +0 -0
  37. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/paired.py +0 -0
  38. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/src/evalsuite/stats/results.py +0 -0
  39. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/__init__.py +0 -0
  40. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/classification/__init__.py +0 -0
  41. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/classification/test_against_sklearn.py +0 -0
  42. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/conftest.py +0 -0
  43. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/integration/__init__.py +0 -0
  44. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/output/__init__.py +0 -0
  45. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/output/test_benchmarks.py +0 -0
  46. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/output/test_cli.py +0 -0
  47. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/output/test_reporting.py +0 -0
  48. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/regression/__init__.py +0 -0
  49. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/regression/test_against_sklearn.py +0 -0
  50. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/stats/__init__.py +0 -0
  51. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/stats/test_branches.py +0 -0
  52. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/stats/test_compare.py +0 -0
  53. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/stats/test_reference.py +0 -0
  54. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/unit/__init__.py +0 -0
  55. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/unit/test_core.py +0 -0
  56. {evalsuite_python-0.1.0b1 → evalsuite_python-0.1.1}/tests/unit/test_edges.py +0 -0
@@ -6,6 +6,27 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.1] - 2026-10-08
10
+
11
+ ### Changed
12
+ - Credits: Manoj Kumar C S and Nikhil D Bharadwaj listed as authors and maintainers (README and package metadata).
13
+
14
+ ## [0.1.0] - 2026-10-08
15
+
16
+ First stable release. Everything from the 0.1.0 roadmap: classification and regression metrics, result
17
+ system, input validation, metric registry, model comparison with confidence intervals and paired tests,
18
+ plots, HTML/CSV/LaTeX/Markdown reports, classification report, command-line tool and benchmarks.
19
+
20
+ ### Changed
21
+ - Development status: Production/Stable.
22
+ - README (PyPI description) now includes the benchmark table against scikit-learn.
23
+
24
+ ## [0.1.0b2]
25
+
26
+ ### Changed
27
+ - `es.plot.calibration`: the legend now sits below the axes by default so it no longer covers the
28
+ curves; new `legend_loc` argument (`"below"` or any matplotlib location).
29
+
9
30
  ## [0.1.0b1]
10
31
 
11
32
  Feature-complete for 0.1.0.
@@ -1,18 +1,18 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.1.0b1
3
+ Version: 0.1.1
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
7
7
  Project-URL: Source, https://github.com/mkcs28/evalsuite-python
8
8
  Project-URL: Issues, https://github.com/mkcs28/evalsuite-python/issues
9
9
  Project-URL: Changelog, https://github.com/mkcs28/evalsuite-python/blob/main/CHANGELOG.md
10
- Author: Manoj Kumar C S
11
- Maintainer: Manoj Kumar C S
10
+ Author: Manoj Kumar C S, Nikhil D Bharadwaj
11
+ Maintainer: Manoj Kumar C S, Nikhil D Bharadwaj
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
14
  Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
15
- Classifier: Development Status :: 4 - Beta
15
+ Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: Intended Audience :: Science/Research
18
18
  Classifier: Operating System :: OS Independent
@@ -62,7 +62,7 @@ Description-Content-Type: text/markdown
62
62
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
63
63
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
64
64
 
65
- > **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
65
+ > **Status: stable (0.1.1).** Every item on the 0.1.0 roadmap is implemented and verified.
66
66
 
67
67
  ## Installation
68
68
 
@@ -185,8 +185,25 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
185
185
 
186
186
  ## Performance
187
187
 
188
- `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
189
- than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
188
+ Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
189
+ scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
190
+ (largest difference 1.1e-16).
191
+
192
+ | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
193
+ | --- | ---: | ---: | ---: | ---: | ---: |
194
+ | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
195
+ | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
196
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
197
+ | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
198
+ | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
199
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
200
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
201
+
202
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
203
+ speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
204
+ infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
205
+ table and notes in
206
+ [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
190
207
 
191
208
  ## Metrics in this release
192
209
 
@@ -225,6 +242,10 @@ ruff check . && ruff format --check . && mypy
225
242
  - Website and documentation: https://evalsuite-nine.vercel.app
226
243
  - Website source: https://github.com/mkcs28/evalsuite
227
244
 
245
+ ## Credits
246
+
247
+ Authors and maintainers: **Manoj Kumar C S** and **Nikhil D Bharadwaj**.
248
+
228
249
  ## License
229
250
 
230
251
  MIT. See [LICENSE](LICENSE).
@@ -10,7 +10,7 @@
10
10
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
11
11
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
12
 
13
- > **Status: beta (0.1.0b1).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
13
+ > **Status: stable (0.1.1).** Every item on the 0.1.0 roadmap is implemented and verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -133,8 +133,25 @@ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` o
133
133
 
134
134
  ## Performance
135
135
 
136
- `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
137
- than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
136
+ Benchmarked against scikit-learn on the same data (fastest of 5 runs; Python 3.12, NumPy 2.5,
137
+ scikit-learn 1.9, Linux x86_64). Every result agrees with scikit-learn to floating-point rounding
138
+ (largest difference 1.1e-16).
139
+
140
+ | Case | n | EvalSuite (ms) | scikit-learn (ms) | Speed-up | Peak memory EvalSuite / sklearn (MiB) |
141
+ | --- | ---: | ---: | ---: | ---: | ---: |
142
+ | 8 binary label metrics via `evaluate()` | 1,000 | 0.38 | 11.70 | **31.2×** | 0.04 / 0.05 |
143
+ | 8 binary label metrics via `evaluate()` | 100,000 | 10.9 | 116.9 | **10.7×** | 3.2 / 3.1 |
144
+ | 8 binary label metrics via `evaluate()` | 1,000,000 | 108.8 | 1043.6 | **9.6×** | 31.5 / 30.5 |
145
+ | macro F1, 10 classes | 1,000,000 | 88.4 | 139.3 | **1.58×** | 30.5 / 21.8 |
146
+ | ROC AUC, binary | 1,000,000 | 247.4 | 352.5 | **1.42×** | 91.6 / 76.3 |
147
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000 | 0.12 | 0.90 | **7.2×** | 0.03 / 0.02 |
148
+ | MAE, MSE, RMSE, R² via `evaluate()` | 1,000,000 | 29.3 | 20.1 | 0.69× | 22.9 / 15.3 |
149
+
150
+ `evaluate()` validates inputs once and builds the confusion matrix once for all metrics, which is where the
151
+ speed-up comes from. Large regression arrays are slower because EvalSuite checks every value for NaN,
152
+ infinity, shape and dtype before computing. Reproduce on your machine with `evalsuite benchmark`; full
153
+ table and notes in
154
+ [BENCHMARKS.md](https://github.com/mkcs28/evalsuite-python/blob/main/BENCHMARKS.md).
138
155
 
139
156
  ## Metrics in this release
140
157
 
@@ -173,6 +190,10 @@ ruff check . && ruff format --check . && mypy
173
190
  - Website and documentation: https://evalsuite-nine.vercel.app
174
191
  - Website source: https://github.com/mkcs28/evalsuite
175
192
 
193
+ ## Credits
194
+
195
+ Authors and maintainers: **Manoj Kumar C S** and **Nikhil D Bharadwaj**.
196
+
176
197
  ## License
177
198
 
178
199
  MIT. See [LICENSE](LICENSE).
@@ -10,11 +10,11 @@ readme = "README.md"
10
10
  license = "MIT"
11
11
  license-files = ["LICENSE"]
12
12
  requires-python = ">=3.9"
13
- authors = [{ name = "Manoj Kumar C S" }]
14
- maintainers = [{ name = "Manoj Kumar C S" }]
13
+ authors = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
14
+ maintainers = [{ name = "Manoj Kumar C S" }, { name = "Nikhil D Bharadwaj" }]
15
15
  keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
- "Development Status :: 4 - Beta",
17
+ "Development Status :: 5 - Production/Stable",
18
18
  "Intended Audience :: Science/Research",
19
19
  "Intended Audience :: Developers",
20
20
  "Operating System :: OS Independent",
@@ -190,9 +190,13 @@ def calibration(
190
190
  strategy: str = "uniform",
191
191
  pos_label: Any = None,
192
192
  sample_weight: Optional[ArrayLike] = None,
193
+ legend_loc: str = "below",
193
194
  ) -> Axes:
194
195
  """Reliability diagram: observed frequency against mean predicted probability per bin, with ECE and Brier
195
- score in the legend; the diagonal is perfect calibration."""
196
+ score in the legend; the diagonal is perfect calibration.
197
+
198
+ ``legend_loc="below"`` (default) puts the legend under the axes so it never covers the curves; any
199
+ matplotlib location (e.g. ``"upper left"``) places it inside instead."""
196
200
  from .classification.metrics import brier_score, calibration_curve, expected_calibration_error
197
201
 
198
202
  ax = _axes(ax)
@@ -218,7 +222,10 @@ def calibration(
218
222
  ylim=(-0.01, 1.01),
219
223
  )
220
224
  ax.set_aspect("equal")
221
- ax.legend(loc="upper left", frameon=False)
225
+ if legend_loc == "below":
226
+ ax.legend(loc="upper center", bbox_to_anchor=(0.5, -0.14), frameon=False, fontsize="small")
227
+ else:
228
+ ax.legend(loc=legend_loc, frameon=False) # type: ignore[call-overload]
222
229
  return ax
223
230
 
224
231
 
@@ -1,3 +1,3 @@
1
1
  """Package version (single source of truth, read by the build backend)."""
2
2
 
3
- __version__ = "0.1.0b1"
3
+ __version__ = "0.1.1"
@@ -121,3 +121,13 @@ def test_matplotlib_is_optional(monkeypatch) -> None:
121
121
  monkeypatch.setattr(builtins, "__import__", fake)
122
122
  with pytest.raises(es.OptionalDependencyError, match=r"evalsuite-python\[plot\]"):
123
123
  es.plot.roc([0, 1], [0.2, 0.8])
124
+
125
+
126
+ def test_calibration_legend_below_by_default_and_configurable():
127
+ y = np.array([0, 1, 0, 1, 1, 0, 1, 0])
128
+ p = np.array([0.1, 0.8, 0.3, 0.7, 0.9, 0.2, 0.6, 0.4])
129
+ ax = es.plot.calibration(y, p)
130
+ anchor = ax.get_legend().get_bbox_to_anchor().transformed(ax.transAxes.inverted())
131
+ assert anchor.y0 < 0 # outside, under the axes
132
+ ax2 = es.plot.calibration(y, p, legend_loc="upper left")
133
+ assert ax2.get_legend()._loc == 2