evalsuite-python 0.1.0a2__tar.gz → 0.1.0b2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/CHANGELOG.md +29 -0
  2. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/PKG-INFO +60 -4
  3. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/README.md +57 -2
  4. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/pyproject.toml +4 -2
  5. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/__init__.py +9 -1
  6. evalsuite_python-0.1.0b2/src/evalsuite/__main__.py +5 -0
  7. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/api.py +22 -7
  8. evalsuite_python-0.1.0b2/src/evalsuite/benchmarks.py +272 -0
  9. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/classification/__init__.py +4 -0
  10. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/classification/metrics.py +71 -0
  11. evalsuite_python-0.1.0b2/src/evalsuite/cli/__init__.py +5 -0
  12. evalsuite_python-0.1.0b2/src/evalsuite/cli/main.py +389 -0
  13. evalsuite_python-0.1.0b2/src/evalsuite/core/export.py +100 -0
  14. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/result.py +102 -0
  15. evalsuite_python-0.1.0b2/src/evalsuite/plot.py +296 -0
  16. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/regression/metrics.py +36 -19
  17. evalsuite_python-0.1.0b2/src/evalsuite/reporting.py +241 -0
  18. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/compare.py +91 -0
  19. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/version.py +1 -1
  20. evalsuite_python-0.1.0b2/tests/output/test_benchmarks.py +32 -0
  21. evalsuite_python-0.1.0b2/tests/output/test_cli.py +195 -0
  22. evalsuite_python-0.1.0b2/tests/output/test_plot.py +133 -0
  23. evalsuite_python-0.1.0b2/tests/output/test_reporting.py +143 -0
  24. evalsuite_python-0.1.0b2/tests/unit/__init__.py +0 -0
  25. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/unit/test_core.py +21 -0
  26. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/.gitignore +0 -0
  27. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/CONTRIBUTING.md +0 -0
  28. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/LICENSE +0 -0
  29. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/classification/_common.py +0 -0
  30. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/__init__.py +0 -0
  31. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/context.py +0 -0
  32. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/exceptions.py +0 -0
  33. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/registry.py +0 -0
  34. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/types.py +0 -0
  35. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/core/validation.py +0 -0
  36. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/py.typed +0 -0
  37. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/regression/__init__.py +0 -0
  38. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/__init__.py +0 -0
  39. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/_resolve.py +0 -0
  40. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/effect.py +0 -0
  41. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/intervals.py +0 -0
  42. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/paired.py +0 -0
  43. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/src/evalsuite/stats/results.py +0 -0
  44. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/__init__.py +0 -0
  45. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/classification/__init__.py +0 -0
  46. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/classification/test_against_sklearn.py +0 -0
  47. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/conftest.py +0 -0
  48. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/integration/__init__.py +0 -0
  49. {evalsuite_python-0.1.0a2/tests/regression → evalsuite_python-0.1.0b2/tests/output}/__init__.py +0 -0
  50. {evalsuite_python-0.1.0a2/tests/stats → evalsuite_python-0.1.0b2/tests/regression}/__init__.py +0 -0
  51. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/regression/test_against_sklearn.py +0 -0
  52. {evalsuite_python-0.1.0a2/tests/unit → evalsuite_python-0.1.0b2/tests/stats}/__init__.py +0 -0
  53. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/stats/test_branches.py +0 -0
  54. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/stats/test_compare.py +0 -0
  55. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/stats/test_reference.py +0 -0
  56. {evalsuite_python-0.1.0a2 → evalsuite_python-0.1.0b2}/tests/unit/test_edges.py +0 -0
@@ -6,6 +6,35 @@ All notable changes to this project are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.0b2]
10
+
11
+ ### Changed
12
+ - `es.plot.calibration`: the legend now sits below the axes by default so it no longer covers the
13
+ curves; new `legend_loc` argument (`"below"` or any matplotlib location).
14
+
15
+ ## [0.1.0b1]
16
+
17
+ Feature-complete for 0.1.0.
18
+
19
+ ### Added
20
+
21
+ - Plots (`es.plot`, optional `[plot]` extra): ROC, precision-recall, calibration, confusion matrix, residuals /
22
+ predicted-vs-true, and model-comparison forest plots. Values come from EvalSuite's metrics; several models
23
+ get distinct colours and line styles. Importing `evalsuite` never imports matplotlib.
24
+ - Reporting: `classification_report` (per-class precision, recall, F1, specificity, support; accuracy, micro,
25
+ macro and weighted averages; matches scikit-learn); `to_html()` and `to_csv()` on every result; `save(path)`
26
+ choosing the format from the extension (.json .csv .md .tex .html .txt).
27
+ - `calibration_curve` and `expected_calibration_error` (matches scikit-learn's calibration curve).
28
+ - Command line: `evalsuite evaluate | report | compare | plot | metrics | info | benchmark`, and
29
+ `python -m evalsuite`. Reads CSV, TSV, Parquet and JSON; clear one-line errors with exit code 2.
30
+ - Benchmarks (`evalsuite.benchmarks.run_benchmarks`, `evalsuite benchmark`): time and peak memory against
31
+ scikit-learn, with a check that both libraries return the same numbers. Results in `BENCHMARKS.md`.
32
+
33
+ ### Changed
34
+
35
+ - Regression `evaluate()` validates inputs once for all metrics and uses a faster unweighted mean (2× faster
36
+ on large arrays).
37
+
9
38
  ## [0.1.0a2]
10
39
 
11
40
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: evalsuite-python
3
- Version: 0.1.0a2
3
+ Version: 0.1.0b2
4
4
  Summary: A unified Python framework for machine-learning, clinical, statistical, segmentation, object-detection, uncertainty, and model evaluation.
5
5
  Project-URL: Homepage, https://evalsuite-nine.vercel.app
6
6
  Project-URL: Documentation, https://evalsuite-nine.vercel.app/docs
@@ -12,7 +12,7 @@ Maintainer: Manoj Kumar C S
12
12
  License-Expression: MIT
13
13
  License-File: LICENSE
14
14
  Keywords: classification,evaluation,machine learning,metrics,regression,reproducibility,statistics
15
- Classifier: Development Status :: 3 - Alpha
15
+ Classifier: Development Status :: 4 - Beta
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: Intended Audience :: Science/Research
18
18
  Classifier: Operating System :: OS Independent
@@ -36,6 +36,7 @@ Requires-Dist: matplotlib>=3.5; extra == 'all'
36
36
  Provides-Extra: dev
37
37
  Requires-Dist: build; extra == 'dev'
38
38
  Requires-Dist: hypothesis>=6.80; extra == 'dev'
39
+ Requires-Dist: matplotlib>=3.5; extra == 'dev'
39
40
  Requires-Dist: mypy>=1.10; extra == 'dev'
40
41
  Requires-Dist: pandas-stubs; extra == 'dev'
41
42
  Requires-Dist: pip-audit; extra == 'dev'
@@ -61,7 +62,7 @@ Description-Content-Type: text/markdown
61
62
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
62
63
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
63
64
 
64
- > **Status: in development (v0.1.0 in progress).** The API may change before 0.1.0.
65
+ > **Status: beta (0.1.0b2).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
65
66
 
66
67
  ## Installation
67
68
 
@@ -133,12 +134,67 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
133
134
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
134
135
  p-values are adjusted for multiple comparisons (Holm by default).
135
136
 
137
+ ## Classification report
138
+
139
+ ```python
140
+ report = es.classification_report(y_true, y_pred)
141
+ print(report) # per-class precision, recall, F1, specificity, support + averages
142
+ report.save("report.html") # also .csv .md .tex .json .txt
143
+ ```
144
+
145
+ ## Plots
146
+
147
+ ```bash
148
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
149
+ ```
150
+
151
+ ```python
152
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
153
+ es.plot.pr(y_true, prob) # AP and the prevalence line
154
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
155
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
156
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
157
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
158
+ ```
159
+
160
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
161
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
162
+ *and* line styles, so figures stay readable in greyscale print.
163
+
164
+ ## Exports
165
+
166
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
167
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
168
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
169
+
170
+ ## Command line
171
+
172
+ ```bash
173
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
174
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
175
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
176
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
177
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
178
+ evalsuite metrics --category classification
179
+ evalsuite info classification.mcc
180
+ evalsuite benchmark --quick
181
+ ```
182
+
183
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
184
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
185
+
186
+ ## Performance
187
+
188
+ `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
189
+ than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
190
+
136
191
  ## Metrics in this release
137
192
 
138
193
  **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
139
194
  sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
140
195
  MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
141
- one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy.
196
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
197
+ calibration curve and expected calibration error.
142
198
 
143
199
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
144
200
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
@@ -10,7 +10,7 @@
10
10
  EvalSuite brings classification and regression metrics (with clinical, statistical, segmentation and
11
11
  object-detection evaluation on the roadmap) into one consistent, validated, documented framework.
12
12
 
13
- > **Status: in development (v0.1.0 in progress).** The API may change before 0.1.0.
13
+ > **Status: beta (0.1.0b2).** Feature-complete for 0.1.0; the stable release follows once this beta is verified.
14
14
 
15
15
  ## Installation
16
16
 
@@ -82,12 +82,67 @@ Every model is evaluated on the same bootstrap resamples, so differences are pai
82
82
  with McNemar's test, binary ROC AUC with DeLong's test and other metrics with a paired bootstrap test;
83
83
  p-values are adjusted for multiple comparisons (Holm by default).
84
84
 
85
+ ## Classification report
86
+
87
+ ```python
88
+ report = es.classification_report(y_true, y_pred)
89
+ print(report) # per-class precision, recall, F1, specificity, support + averages
90
+ report.save("report.html") # also .csv .md .tex .json .txt
91
+ ```
92
+
93
+ ## Plots
94
+
95
+ ```bash
96
+ pip install "evalsuite-python[plot]" # adds matplotlib; importing evalsuite never loads it
97
+ ```
98
+
99
+ ```python
100
+ es.plot.roc(y_true, {"logistic": prob_lr, "forest": prob_rf}) # AUC in the legend
101
+ es.plot.pr(y_true, prob) # AP and the prevalence line
102
+ es.plot.calibration(y_true, prob) # reliability diagram, ECE, Brier
103
+ es.plot.confusion_matrix(y_true, y_pred, normalize="true")
104
+ es.plot.residuals(y_reg, pred_reg) # or kind="predicted"
105
+ es.plot.comparison(es.compare(...)) # forest plot with CIs
106
+ ```
107
+
108
+ Each function returns a matplotlib `Axes` (pass `ax=` to draw into your own figure). The numbers shown are
109
+ computed with EvalSuite's metrics, so plots and tables always agree. Several models get distinct colours
110
+ *and* line styles, so figures stay readable in greyscale print.
111
+
112
+ ## Exports
113
+
114
+ Every result (`evaluate`, `classification_report`, `compare`, single metrics) exports to `summary()`,
115
+ `to_json()`, `to_csv()`, `to_dataframe()`, `to_markdown()`, `to_latex()` and `to_html()`, and `save(path)` picks
116
+ the format from the extension. HTML pages are standalone (inline CSS, no scripts) and escape all text.
117
+
118
+ ## Command line
119
+
120
+ ```bash
121
+ evalsuite evaluate predictions.csv --y-true label --y-pred pred --y-prob prob
122
+ evalsuite report predictions.csv --y-true label --y-pred pred -o report.html
123
+ evalsuite compare predictions.csv --y-true label --pred lr=pred_lr --pred rf=pred_rf \
124
+ --prob lr=p_lr --prob rf=p_rf --plot comparison.png
125
+ evalsuite plot roc predictions.csv --y-true label --y-prob prob -o roc.png
126
+ evalsuite metrics --category classification
127
+ evalsuite info classification.mcc
128
+ evalsuite benchmark --quick
129
+ ```
130
+
131
+ Input files can be CSV, TSV, Parquet or JSON. Output format follows `--format` or the `-o` extension
132
+ (text, json, csv, markdown, latex, html). Errors are reported in one line with exit code 2.
133
+
134
+ ## Performance
135
+
136
+ `evaluate()` validates inputs once and computes the confusion matrix once for all metrics: about 10× faster
137
+ than the equivalent separate scikit-learn calls, with identical results. See [BENCHMARKS.md](BENCHMARKS.md).
138
+
85
139
  ## Metrics in this release
86
140
 
87
141
  **Classification** (binary, multiclass, multilabel; micro/macro/weighted/samples/per-class averaging;
88
142
  sample weights): accuracy, balanced accuracy, precision, recall, specificity, NPV, F1, F-beta, Jaccard,
89
143
  MCC, Cohen's kappa (unweighted, linear, quadratic), Hamming loss, confusion matrix, ROC AUC (binary,
90
- one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy.
144
+ one-vs-rest, one-vs-one), average precision, ROC and PR curves, log loss, Brier score, top-k accuracy,
145
+ calibration curve and expected calibration error.
91
146
 
92
147
  **Regression** (single and multi-output; sample weights): MAE, MSE, RMSE, R², adjusted R², MAPE, sMAPE,
93
148
  MSLE, RMSLE, median absolute error, explained variance, max error, mean bias error, quantile (pinball)
@@ -14,7 +14,7 @@ authors = [{ name = "Manoj Kumar C S" }]
14
14
  maintainers = [{ name = "Manoj Kumar C S" }]
15
15
  keywords = ["evaluation", "metrics", "machine learning", "statistics", "classification", "regression", "reproducibility"]
16
16
  classifiers = [
17
- "Development Status :: 3 - Alpha",
17
+ "Development Status :: 4 - Beta",
18
18
  "Intended Audience :: Science/Research",
19
19
  "Intended Audience :: Developers",
20
20
  "Operating System :: OS Independent",
@@ -41,6 +41,7 @@ dev = [
41
41
  "hypothesis>=6.80",
42
42
  "scikit-learn>=1.2",
43
43
  "statsmodels>=0.13",
44
+ "matplotlib>=3.5",
44
45
  "ruff>=0.6",
45
46
  "mypy>=1.10",
46
47
  "pandas-stubs",
@@ -56,7 +57,8 @@ Source = "https://github.com/mkcs28/evalsuite-python"
56
57
  Issues = "https://github.com/mkcs28/evalsuite-python/issues"
57
58
  Changelog = "https://github.com/mkcs28/evalsuite-python/blob/main/CHANGELOG.md"
58
59
 
59
- # [project.scripts] evalsuite = "evalsuite.cli.main:main" is added together with the CLI itself.
60
+ [project.scripts]
61
+ evalsuite = "evalsuite.cli.main:main"
60
62
 
61
63
  [tool.hatch.version]
62
64
  path = "src/evalsuite/version.py"
@@ -5,15 +5,17 @@
5
5
  >>> print(result.summary()) # doctest: +SKIP
6
6
  """
7
7
 
8
- from . import classification, regression, stats
8
+ from . import classification, plot, regression, stats
9
9
  from .api import evaluate
10
10
  from .classification import (
11
11
  accuracy,
12
12
  average_precision,
13
13
  balanced_accuracy,
14
14
  brier_score,
15
+ calibration_curve,
15
16
  cohen_kappa,
16
17
  confusion_matrix,
18
+ expected_calibration_error,
17
19
  f1,
18
20
  fbeta,
19
21
  hamming_loss,
@@ -59,6 +61,7 @@ from .regression import (
59
61
  rse,
60
62
  smape,
61
63
  )
64
+ from .reporting import ClassificationReport, classification_report
62
65
  from .stats import (
63
66
  ComparisonResult,
64
67
  ConfidenceInterval,
@@ -79,6 +82,11 @@ from .stats import (
79
82
  from .version import __version__
80
83
 
81
84
  __all__ = [
85
+ "plot",
86
+ "expected_calibration_error",
87
+ "calibration_curve",
88
+ "classification_report",
89
+ "ClassificationReport",
82
90
  "stats",
83
91
  "roc_auc_ci",
84
92
  "proportion_ci",
@@ -0,0 +1,5 @@
1
+ """``python -m evalsuite`` runs the command-line interface."""
2
+
3
+ from .cli.main import main
4
+
5
+ raise SystemExit(main())
@@ -232,6 +232,28 @@ def _evaluate_regression(
232
232
  f"Unknown regression metric(s): {', '.join(unknown)}. Available: {', '.join(_REGRESSION)}."
233
233
  )
234
234
  out: dict[str, MetricResult] = {}
235
+ shared = reg._Inputs(y_true, y_pred, sample_weight) # validate once for every metric
236
+ token = reg._SHARED.set((id(y_true), id(y_pred), id(sample_weight), shared))
237
+ try:
238
+ _regression_loop(names, out, y_true, y_pred, sample_weight)
239
+ finally:
240
+ reg._SHARED.reset(token)
241
+ return EvaluationResult(
242
+ task="regression",
243
+ metrics=out,
244
+ n_samples=yt.shape[0],
245
+ target_type="continuous",
246
+ metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
247
+ )
248
+
249
+
250
+ def _regression_loop(
251
+ names: list[str],
252
+ out: dict[str, MetricResult],
253
+ y_true: ArrayLike,
254
+ y_pred: ArrayLike,
255
+ sample_weight: Optional[ArrayLike],
256
+ ) -> None:
235
257
  for name in names:
236
258
  fn = _REGRESSION[name]
237
259
  if name in _NO_WEIGHTS:
@@ -240,10 +262,3 @@ def _evaluate_regression(
240
262
  out[name] = fn(y_true, y_pred)
241
263
  else:
242
264
  out[name] = fn(y_true, y_pred, sample_weight=sample_weight)
243
- return EvaluationResult(
244
- task="regression",
245
- metrics=out,
246
- n_samples=yt.shape[0],
247
- target_type="continuous",
248
- metadata=_metadata(weighted=sample_weight is not None, outputs=yt.shape[1] if multi else 1),
249
- )
@@ -0,0 +1,272 @@
1
+ """Speed and memory benchmarks, against scikit-learn when it is installed.
2
+
3
+ Each case times the fastest of ``repeat`` runs (after one warm-up) and measures peak traced memory with
4
+ ``tracemalloc`` (NumPy reports its allocations to it). Both libraries compute the same metrics on the same
5
+ data, and the largest absolute difference between their results is reported, so speed is never shown for
6
+ numbers that disagree.
7
+
8
+ >>> from evalsuite.benchmarks import run_benchmarks
9
+ >>> print(run_benchmarks(sizes=(10_000,), repeat=3)) # doctest: +SKIP
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import platform
16
+ import time
17
+ import tracemalloc
18
+ import warnings
19
+ from collections.abc import Callable, Sequence
20
+ from dataclasses import dataclass, field
21
+ from types import MappingProxyType
22
+ from typing import TYPE_CHECKING, Any, Optional, cast
23
+
24
+ import numpy as np
25
+
26
+ from .core.result import _json_safe, _latex_escape, _latex_table
27
+
28
+ if TYPE_CHECKING:
29
+ import pandas as pd
30
+
31
+ __all__ = ["BenchmarkResult", "run_benchmarks"]
32
+
33
+ _HEADER = (
34
+ "case",
35
+ "n",
36
+ "evalsuite_ms",
37
+ "sklearn_ms",
38
+ "speedup",
39
+ "evalsuite_peak_mb",
40
+ "sklearn_peak_mb",
41
+ "max_abs_diff",
42
+ )
43
+
44
+
45
+ def _measure(fn: Callable[[], Any], repeat: int) -> tuple[float, float, Any]:
46
+ """(fastest seconds, peak MiB, result)."""
47
+ with warnings.catch_warnings():
48
+ warnings.simplefilter("ignore")
49
+ result = fn() # warm-up
50
+ best = float("inf")
51
+ for _ in range(repeat):
52
+ t0 = time.perf_counter()
53
+ fn()
54
+ best = min(best, time.perf_counter() - t0)
55
+ tracemalloc.start()
56
+ try:
57
+ fn()
58
+ _, peak = tracemalloc.get_traced_memory()
59
+ finally:
60
+ tracemalloc.stop()
61
+ return best, peak / 2**20, result
62
+
63
+
64
+ def _cases(n: int, rng: np.random.Generator) -> list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]]:
65
+ import evalsuite as es
66
+
67
+ y = rng.integers(0, 2, n)
68
+ p = np.where(rng.random(n) < 0.8, y, 1 - y)
69
+ prob = 1 / (1 + np.exp(-(2.0 * (y - 0.5) + rng.normal(0, 1, n))))
70
+ yk = rng.integers(0, 10, n)
71
+ pk = np.where(rng.random(n) < 0.7, yk, rng.integers(0, 10, n))
72
+ yr = rng.normal(10, 3, n)
73
+ pr = yr + rng.normal(0, 1, n)
74
+ names = ["accuracy", "balanced_accuracy", "precision", "recall", "f1", "specificity", "mcc", "cohen_kappa"]
75
+
76
+ def es_binary() -> list[float]:
77
+ r = es.evaluate(y, p, metrics=names)
78
+ return [float(r[m]) for m in names]
79
+
80
+ def es_reg() -> list[float]:
81
+ r = es.evaluate(yr, pr, metrics=["mae", "mse", "rmse", "r2"])
82
+ return [float(r[m]) for m in ("mae", "mse", "rmse", "r2")]
83
+
84
+ cases: list[tuple[str, Callable[[], Any], Optional[Callable[[], Any]]]] = [
85
+ ("binary: 8 label metrics via evaluate()", es_binary, None),
86
+ ("10 classes: macro F1", lambda: [float(es.f1(yk, pk, average="macro"))], None),
87
+ ("binary: ROC AUC", lambda: [float(es.roc_auc(y, prob))], None),
88
+ ("regression: MAE, MSE, RMSE, R² via evaluate()", es_reg, None),
89
+ ]
90
+ try:
91
+ import sklearn.metrics as skm # type: ignore[import-untyped]
92
+ except ImportError:
93
+ return cases
94
+
95
+ def sk_binary() -> list[float]:
96
+ return [
97
+ skm.accuracy_score(y, p),
98
+ skm.balanced_accuracy_score(y, p),
99
+ skm.precision_score(y, p),
100
+ skm.recall_score(y, p),
101
+ skm.f1_score(y, p),
102
+ skm.recall_score(y, p, pos_label=0),
103
+ skm.matthews_corrcoef(y, p),
104
+ skm.cohen_kappa_score(y, p),
105
+ ]
106
+
107
+ def sk_reg() -> list[float]:
108
+ mse = skm.mean_squared_error(yr, pr)
109
+ return [skm.mean_absolute_error(yr, pr), mse, float(np.sqrt(mse)), skm.r2_score(yr, pr)]
110
+
111
+ sk = [sk_binary, lambda: [skm.f1_score(yk, pk, average="macro")], lambda: [skm.roc_auc_score(y, prob)], sk_reg]
112
+ return [(name, es_fn, sk_fn) for (name, es_fn, _), sk_fn in zip(cases, sk)]
113
+
114
+
115
+ @dataclass(frozen=True, eq=False)
116
+ class BenchmarkResult:
117
+ """Rows of timings (milliseconds), peak memory (MiB) and agreement, plus the environment they ran in."""
118
+
119
+ rows: tuple[Any, ...]
120
+ environment: Any = field(default_factory=dict)
121
+
122
+ def __post_init__(self) -> None:
123
+ object.__setattr__(self, "rows", tuple(MappingProxyType(dict(r)) for r in self.rows))
124
+ object.__setattr__(self, "environment", MappingProxyType(dict(self.environment)))
125
+
126
+ def _cells(self, digits: int) -> list[list[str]]:
127
+ def f(v: Any, d: int = digits) -> str:
128
+ return "–" if v is None else f"{v:.{d}f}"
129
+
130
+ return [
131
+ [
132
+ r["case"],
133
+ f"{r['n']:,}",
134
+ f(r["evalsuite_ms"]),
135
+ f(r["sklearn_ms"]),
136
+ "–" if r["speedup"] is None else f"{r['speedup']:.2f}×",
137
+ f(r["evalsuite_peak_mb"], 2),
138
+ f(r["sklearn_peak_mb"], 2),
139
+ "–" if r["max_abs_diff"] is None else f"{r['max_abs_diff']:.1e}",
140
+ ]
141
+ for r in self.rows
142
+ ]
143
+
144
+ _TITLES = (
145
+ "Case",
146
+ "n",
147
+ "EvalSuite (ms)",
148
+ "scikit-learn (ms)",
149
+ "Speed-up",
150
+ "EvalSuite peak (MiB)",
151
+ "scikit-learn peak (MiB)",
152
+ "Max |difference|",
153
+ )
154
+
155
+ def summary(self, *, digits: int = 3) -> str:
156
+ cells = self._cells(digits)
157
+ widths = [max(len(t), *(len(c[i]) for c in cells)) for i, t in enumerate(self._TITLES)]
158
+
159
+ def line(c: Sequence[str]) -> str:
160
+ return " ".join(x.ljust(widths[i]) if i == 0 else x.rjust(widths[i]) for i, x in enumerate(c))
161
+
162
+ env = self.environment
163
+ head = (
164
+ f"EvalSuite {env['evalsuite']} benchmarks | Python {env['python']} | NumPy {env['numpy']}"
165
+ + (f" | scikit-learn {env['sklearn']}" if env.get("sklearn") else "")
166
+ + f" | {env['machine']} | fastest of {env['repeat']} runs"
167
+ )
168
+ note = "Speed-up > 1 means EvalSuite is faster. Max |difference| compares the two libraries' results."
169
+ return "\n".join([head, "", line(self._TITLES), *(line(c) for c in cells), "", note])
170
+
171
+ def __repr__(self) -> str:
172
+ return self.summary()
173
+
174
+ def to_dict(self) -> dict[str, Any]:
175
+ return cast(
176
+ "dict[str, Any]",
177
+ _json_safe({"environment": dict(self.environment), "rows": [dict(r) for r in self.rows]}),
178
+ )
179
+
180
+ def to_json(self, *, indent: Optional[int] = 2) -> str:
181
+ return json.dumps(self.to_dict(), indent=indent, allow_nan=False)
182
+
183
+ def to_dataframe(self) -> pd.DataFrame:
184
+ import pandas as pd
185
+
186
+ frame: pd.DataFrame = pd.DataFrame([dict(r) for r in self.rows])
187
+ return frame
188
+
189
+ def to_csv(self, path: Optional[str] = None) -> str:
190
+ from .core.export import csv_text
191
+
192
+ rows = [[r[h] if r[h] is not None else float("nan") for h in _HEADER] for r in self.rows]
193
+ text = csv_text(list(_HEADER), rows)
194
+ if path is not None:
195
+ with open(path, "w", encoding="utf-8", newline="") as fh:
196
+ fh.write(text)
197
+ return text
198
+
199
+ def to_markdown(self, *, digits: int = 3) -> str:
200
+ lines = ["| " + " | ".join(self._TITLES) + " |", "| --- |" + " ---: |" * (len(self._TITLES) - 1)]
201
+ return "\n".join(lines + ["| " + " | ".join(c) + " |" for c in self._cells(digits)])
202
+
203
+ def to_latex(self, *, digits: int = 3, caption: Optional[str] = None, label: Optional[str] = None) -> str:
204
+ return _latex_table(
205
+ [_latex_escape(t) for t in self._TITLES],
206
+ [[_latex_escape(x) for x in c] for c in self._cells(digits)],
207
+ caption=caption or "EvalSuite benchmarks.",
208
+ label=label,
209
+ )
210
+
211
+ def to_html(self, *, digits: int = 3, full: bool = False) -> str:
212
+ from .core.export import html_document, html_table
213
+
214
+ table = html_table(list(self._TITLES), self._cells(digits), caption="Benchmarks")
215
+ env = self.environment
216
+ meta = f"Python {env['python']}, NumPy {env['numpy']}, {env['machine']}"
217
+ return html_document("EvalSuite benchmarks", table, meta) if full else table
218
+
219
+
220
+ def run_benchmarks(
221
+ sizes: Sequence[int] = (1_000, 100_000, 1_000_000),
222
+ *,
223
+ repeat: int = 5,
224
+ compare_sklearn: bool = True,
225
+ random_state: Optional[int] = 0,
226
+ ) -> BenchmarkResult:
227
+ """Time and memory for common evaluation workloads at each size, against scikit-learn if installed."""
228
+ import evalsuite as es
229
+
230
+ if repeat < 1:
231
+ raise ValueError("repeat must be at least 1.")
232
+ rng = np.random.default_rng(random_state)
233
+ rows: list[dict[str, Any]] = []
234
+ sk_version: Optional[str] = None
235
+ if compare_sklearn:
236
+ try:
237
+ import sklearn
238
+
239
+ sk_version = sklearn.__version__
240
+ except ImportError:
241
+ compare_sklearn = False
242
+ for n in sizes:
243
+ for name, es_fn, sk_fn in _cases(int(n), rng):
244
+ es_t, es_mem, es_val = _measure(es_fn, repeat)
245
+ row: dict[str, Any] = {
246
+ "case": name,
247
+ "n": int(n),
248
+ "evalsuite_ms": es_t * 1000,
249
+ "evalsuite_peak_mb": es_mem,
250
+ "sklearn_ms": None,
251
+ "sklearn_peak_mb": None,
252
+ "speedup": None,
253
+ "max_abs_diff": None,
254
+ }
255
+ if compare_sklearn and sk_fn is not None:
256
+ sk_t, sk_mem, sk_val = _measure(sk_fn, repeat)
257
+ row.update(
258
+ sklearn_ms=sk_t * 1000,
259
+ sklearn_peak_mb=sk_mem,
260
+ speedup=sk_t / es_t if es_t else None,
261
+ max_abs_diff=float(np.max(np.abs(np.asarray(es_val, float) - np.asarray(sk_val, float)))),
262
+ )
263
+ rows.append(row)
264
+ env = {
265
+ "evalsuite": es.__version__,
266
+ "python": platform.python_version(),
267
+ "numpy": np.__version__,
268
+ "sklearn": sk_version,
269
+ "machine": f"{platform.system()} {platform.machine()}",
270
+ "repeat": repeat,
271
+ }
272
+ return BenchmarkResult(tuple(rows), env)
@@ -5,8 +5,10 @@ from .metrics import (
5
5
  average_precision,
6
6
  balanced_accuracy,
7
7
  brier_score,
8
+ calibration_curve,
8
9
  cohen_kappa,
9
10
  confusion_matrix,
11
+ expected_calibration_error,
10
12
  f1,
11
13
  fbeta,
12
14
  hamming_loss,
@@ -28,8 +30,10 @@ __all__ = [
28
30
  "average_precision",
29
31
  "balanced_accuracy",
30
32
  "brier_score",
33
+ "calibration_curve",
31
34
  "cohen_kappa",
32
35
  "confusion_matrix",
36
+ "expected_calibration_error",
33
37
  "f1",
34
38
  "fbeta",
35
39
  "hamming_loss",