saberlib 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. saberlib-0.1.0/.gitignore +26 -0
  2. saberlib-0.1.0/LICENSE +21 -0
  3. saberlib-0.1.0/PKG-INFO +209 -0
  4. saberlib-0.1.0/README.md +161 -0
  5. saberlib-0.1.0/pyproject.toml +174 -0
  6. saberlib-0.1.0/saber/__init__.py +71 -0
  7. saberlib-0.1.0/saber/__main__.py +5 -0
  8. saberlib-0.1.0/saber/_api/__init__.py +7 -0
  9. saberlib-0.1.0/saber/_api/_common.py +71 -0
  10. saberlib-0.1.0/saber/_api/evaluate.py +43 -0
  11. saberlib-0.1.0/saber/_api/predict.py +50 -0
  12. saberlib-0.1.0/saber/_api/train.py +68 -0
  13. saberlib-0.1.0/saber/_version.py +3 -0
  14. saberlib-0.1.0/saber/benchmark/__init__.py +13 -0
  15. saberlib-0.1.0/saber/benchmark/config.py +47 -0
  16. saberlib-0.1.0/saber/benchmark/engine.py +509 -0
  17. saberlib-0.1.0/saber/benchmark/results.py +199 -0
  18. saberlib-0.1.0/saber/classification/__init__.py +1 -0
  19. saberlib-0.1.0/saber/classification/lightgbm.py +25 -0
  20. saberlib-0.1.0/saber/classification/search_spaces.py +262 -0
  21. saberlib-0.1.0/saber/classification/sklearn.py +223 -0
  22. saberlib-0.1.0/saber/classification/xgboost.py +38 -0
  23. saberlib-0.1.0/saber/cli/__init__.py +5 -0
  24. saberlib-0.1.0/saber/cli/main.py +201 -0
  25. saberlib-0.1.0/saber/cli/render.py +306 -0
  26. saberlib-0.1.0/saber/config/__init__.py +14 -0
  27. saberlib-0.1.0/saber/config/builders.py +171 -0
  28. saberlib-0.1.0/saber/config/io.py +58 -0
  29. saberlib-0.1.0/saber/config/runner.py +314 -0
  30. saberlib-0.1.0/saber/config/schema.py +284 -0
  31. saberlib-0.1.0/saber/core/__init__.py +22 -0
  32. saberlib-0.1.0/saber/core/capabilities.py +76 -0
  33. saberlib-0.1.0/saber/core/metrics.py +234 -0
  34. saberlib-0.1.0/saber/core/prediction.py +254 -0
  35. saberlib-0.1.0/saber/core/registry.py +61 -0
  36. saberlib-0.1.0/saber/core/results.py +35 -0
  37. saberlib-0.1.0/saber/core/search_space.py +237 -0
  38. saberlib-0.1.0/saber/core/specs.py +138 -0
  39. saberlib-0.1.0/saber/core/task.py +7 -0
  40. saberlib-0.1.0/saber/datasets/__init__.py +16 -0
  41. saberlib-0.1.0/saber/datasets/_fingerprint.py +186 -0
  42. saberlib-0.1.0/saber/datasets/biosieve.py +343 -0
  43. saberlib-0.1.0/saber/datasets/folds.py +342 -0
  44. saberlib-0.1.0/saber/datasets/loaders.py +188 -0
  45. saberlib-0.1.0/saber/datasets/schemas.py +279 -0
  46. saberlib-0.1.0/saber/datasets/validation.py +206 -0
  47. saberlib-0.1.0/saber/evaluation/__init__.py +6 -0
  48. saberlib-0.1.0/saber/evaluation/classification.py +390 -0
  49. saberlib-0.1.0/saber/evaluation/evaluator.py +61 -0
  50. saberlib-0.1.0/saber/evaluation/regression.py +47 -0
  51. saberlib-0.1.0/saber/evaluation/results.py +30 -0
  52. saberlib-0.1.0/saber/exceptions.py +170 -0
  53. saberlib-0.1.0/saber/persistence/__init__.py +18 -0
  54. saberlib-0.1.0/saber/persistence/artifacts.py +117 -0
  55. saberlib-0.1.0/saber/persistence/checksums.py +75 -0
  56. saberlib-0.1.0/saber/persistence/environment.py +97 -0
  57. saberlib-0.1.0/saber/persistence/load.py +140 -0
  58. saberlib-0.1.0/saber/persistence/metadata.py +60 -0
  59. saberlib-0.1.0/saber/persistence/save.py +276 -0
  60. saberlib-0.1.0/saber/preprocessing/__init__.py +5 -0
  61. saberlib-0.1.0/saber/preprocessing/imputation.py +30 -0
  62. saberlib-0.1.0/saber/preprocessing/pipeline.py +130 -0
  63. saberlib-0.1.0/saber/preprocessing/scaling.py +57 -0
  64. saberlib-0.1.0/saber/preprocessing/validation.py +65 -0
  65. saberlib-0.1.0/saber/py.typed +0 -0
  66. saberlib-0.1.0/saber/regression/__init__.py +1 -0
  67. saberlib-0.1.0/saber/regression/lightgbm.py +25 -0
  68. saberlib-0.1.0/saber/regression/search_spaces.py +252 -0
  69. saberlib-0.1.0/saber/regression/sklearn.py +257 -0
  70. saberlib-0.1.0/saber/regression/xgboost.py +38 -0
  71. saberlib-0.1.0/saber/tuning/__init__.py +10 -0
  72. saberlib-0.1.0/saber/tuning/engine.py +601 -0
  73. saberlib-0.1.0/saber/tuning/results.py +93 -0
  74. saberlib-0.1.0/saber/utils/__init__.py +1 -0
  75. saberlib-0.1.0/saber/utils/serialization.py +65 -0
  76. saberlib-0.1.0/saber/utils/tabular.py +133 -0
  77. saberlib-0.1.0/saber/validation/__init__.py +10 -0
  78. saberlib-0.1.0/saber/validation/cross_validation.py +243 -0
  79. saberlib-0.1.0/saber/validation/partitioning.py +119 -0
  80. saberlib-0.1.0/saber/validation/results.py +134 -0
@@ -0,0 +1,26 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+
5
+ # Build
6
+ build/
7
+ dist/
8
+ *.egg-info/
9
+
10
+ # Environments
11
+ .venv/
12
+ .env
13
+
14
+ # Tooling caches and reports
15
+ .pytest_cache/
16
+ .ruff_cache/
17
+ .coverage
18
+ .coverage.*
19
+ htmlcov/
20
+
21
+ # Marimo
22
+ __marimo__/
23
+
24
+ # Editors
25
+ .idea/
26
+ .vscode/
saberlib-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Kren-AI Lab
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,209 @@
1
+ Metadata-Version: 2.5
2
+ Name: saberlib
3
+ Version: 0.1.0
4
+ Summary: Domain-agnostic classical supervised machine learning for classification, regression, tuning, and benchmarking.
5
+ Project-URL: Homepage, https://github.com/kren-ai-lab/saber
6
+ Project-URL: Repository, https://github.com/kren-ai-lab/saber
7
+ Project-URL: Issues, https://github.com/kren-ai-lab/saber/issues
8
+ Project-URL: Documentation, https://github.com/kren-ai-lab/saber#readme
9
+ Author: Diego Alvarez-Saravia, David Medina-Ortiz
10
+ Maintainer: Diego Alvarez-Saravia, David Medina-Ortiz
11
+ License-Expression: MIT
12
+ License-File: LICENSE
13
+ Keywords: benchmarking,classification,hyperparameter-optimization,machine-learning,regression,reproducibility,scientific-software,supervised-learning,tabular-data
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: Science/Research
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Programming Language :: Python :: 3.14
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Typing :: Typed
25
+ Requires-Python: <3.15,>=3.11
26
+ Requires-Dist: joblib<2,>=1.3
27
+ Requires-Dist: numpy<3,>=2.0
28
+ Requires-Dist: polars<2,>=1.44
29
+ Requires-Dist: pyyaml<7,>=6.0
30
+ Requires-Dist: rich<16,>=13.0
31
+ Requires-Dist: scikit-learn<2,>=1.8
32
+ Requires-Dist: scipy<2,>=1.16
33
+ Requires-Dist: typer<1,>=0.27
34
+ Provides-Extra: all
35
+ Requires-Dist: biosieve<0.2,>=0.1.2; extra == 'all'
36
+ Requires-Dist: lightgbm<5,>=4.6; extra == 'all'
37
+ Requires-Dist: optuna<5,>=4.6; extra == 'all'
38
+ Requires-Dist: xgboost<4,>=2.0; extra == 'all'
39
+ Provides-Extra: biosieve
40
+ Requires-Dist: biosieve<0.2,>=0.1.2; extra == 'biosieve'
41
+ Provides-Extra: lightgbm
42
+ Requires-Dist: lightgbm<5,>=4.6; extra == 'lightgbm'
43
+ Provides-Extra: optuna
44
+ Requires-Dist: optuna<5,>=4.6; extra == 'optuna'
45
+ Provides-Extra: xgboost
46
+ Requires-Dist: xgboost<4,>=2.0; extra == 'xgboost'
47
+ Description-Content-Type: text/markdown
48
+
49
+ # Saber
50
+
51
+ [![PyPI](https://img.shields.io/pypi/v/saberlib?style=flat-square)](https://pypi.org/project/saberlib/)
52
+ [![PyVersions](https://img.shields.io/pypi/pyversions/saberlib?style=flat-square)](https://github.com/kren-ai-lab/saber)
53
+ [![Tests](https://img.shields.io/github/actions/workflow/status/kren-ai-lab/saber/tests.yml?style=flat-square)](https://github.com/kren-ai-lab/saber/actions/workflows/tests.yml)
54
+ ![License](https://img.shields.io/github/license/kren-ai-lab/saber?style=flat-square)
55
+
56
+ Saber is a Python library for classical supervised machine learning
57
+ (classification and regression) on numerical tabular features. It trains,
58
+ validates, tunes, benchmarks and persists scikit-learn, XGBoost and LightGBM
59
+ models, with preprocessing fitted inside each fold and every result tied back
60
+ to its samples, partition and parameters.
61
+
62
+ Saber doesn't compute features. Bring a descriptor table, an embedding or any
63
+ numeric matrix, and Saber models it. Deep learning, AutoML, multilabel and
64
+ multi-output problems are out of scope.
65
+
66
+ ## Installation
67
+
68
+ Saber supports Python 3.11 to 3.14.
69
+
70
+ ```bash
71
+ python -m pip install saberlib
72
+ ```
73
+
74
+ Optional extras:
75
+
76
+ ```bash
77
+ python -m pip install "saberlib[biosieve]" # partition generation
78
+ python -m pip install "saberlib[optuna]" # Optuna tuning
79
+ python -m pip install "saberlib[xgboost]" # XGBoost models
80
+ python -m pip install "saberlib[lightgbm]" # LightGBM models
81
+ python -m pip install "saberlib[all]"
82
+ ```
83
+
84
+ For development setup with
85
+ `uv`, see [DEVELOPMENT.md](DEVELOPMENT.md).
86
+
87
+ ## Validate a model
88
+
89
+ ```python
90
+ import numpy as np
91
+ from sklearn.datasets import make_classification
92
+
93
+ import saber
94
+ from saber import DatasetBundle, PartitionPlan
95
+
96
+ X, y = make_classification(n_samples=120, n_features=12, random_state=42)
97
+ dataset = DatasetBundle(X=X, y=y, sample_ids=[f"s{i}" for i in range(len(y))])
98
+
99
+ plan = PartitionPlan.from_predefined_folds(
100
+ sample_ids=dataset.sample_ids,
101
+ fold_assignments=np.arange(len(y)) % 4,
102
+ dataset=dataset,
103
+ )
104
+
105
+ result = saber.validate(
106
+ dataset=dataset,
107
+ algorithm="logistic_regression",
108
+ partition_plan=plan,
109
+ metrics=("mcc", "balanced_accuracy", "roc_auc"),
110
+ random_state=42,
111
+ )
112
+ print(result.aggregate_metrics)
113
+ ```
114
+
115
+ `result.oof_prediction` holds the out-of-fold predictions, aligned by sample
116
+ ID. `X` can be a NumPy array, a Polars DataFrame or a pandas DataFrame. Partitions
117
+ refer to samples by ID, never by row position. If your data isn't split yet,
118
+ let BioSieve generate the partitions:
119
+
120
+ ```python
121
+ from saber import BioSievePartitionConfig
122
+
123
+ result = saber.validate(
124
+ dataset=dataset,
125
+ algorithm="random_forest_classifier",
126
+ partition_plan=BioSievePartitionConfig(
127
+ strategy="stratified_kfold",
128
+ params={"n_splits": 5, "seed": 42},
129
+ ),
130
+ metrics=("mcc", "roc_auc"),
131
+ )
132
+ ```
133
+
134
+ The [data and partitions](docs/data_and_partitions.md) guide covers holdouts,
135
+ external partition files and BioSieve.
136
+
137
+ ## Tune, benchmark and save
138
+
139
+ ```python
140
+ from saber import Categorical, LogFloat, SearchSpace, TuningConfig
141
+
142
+ tuned = saber.tune(
143
+ dataset=dataset,
144
+ algorithm="svc",
145
+ partition_plan=plan,
146
+ search_space=SearchSpace(
147
+ parameters={"C": LogFloat(1e-3, 1e2), "kernel": Categorical(["linear", "rbf"])},
148
+ ),
149
+ config=TuningConfig(optimizer="random", n_trials=20),
150
+ metrics=("mcc",),
151
+ random_state=42,
152
+ )
153
+ print(tuned.best_params)
154
+ ```
155
+
156
+ `saber.benchmark(...)` crosses representations, partitions, algorithms, seeds
157
+ and tuned/untuned modes, and returns long-form Polars tables of metrics and
158
+ per-sample predictions. `saber.train(...)` fits a final model, and
159
+ `saber.save_model(path, result, dataset=dataset)` writes a `train` or `tune` result as a directory with a feature schema,
160
+ provenance and checksums that `saber.load_model(...)` verifies before loading.
161
+ See the [tuning](docs/tuning.md), [benchmarking](docs/benchmarking.md) and
162
+ [persistence](docs/persistence.md) guides.
163
+
164
+ ## Command line
165
+
166
+ ```bash
167
+ saber models list --task classification
168
+ saber run experiment.yaml --dry-run
169
+ saber run study.yaml --json
170
+ saber artifact verify artifacts/model
171
+ ```
172
+
173
+ Every workflow can be written as a YAML or JSON file and run from the CLI or
174
+ with `saber.run_config("experiment.yaml")`. See the [configuration](docs/configuration.md)
175
+ and [CLI](docs/cli.md) references.
176
+
177
+ ## Design principles
178
+
179
+ - Imputation and scaling are fitted inside each training fold, never on the
180
+ whole dataset.
181
+ - Existing partitions are used exactly as given. New ones come from BioSieve;
182
+ Saber has no splitters of its own.
183
+ - Tuned results are reported on a protected test set, not on the folds used to
184
+ pick the hyperparameters.
185
+ - For binary problems the positive class is explicit, never inferred from a
186
+ probability column's position.
187
+ - Saved models carry dataset and partition fingerprints and are checksum
188
+ verified before loading. They use joblib, so load them only from trusted
189
+ sources.
190
+
191
+ Saber keeps the workflow honest, but it can't tell whether your partitions or
192
+ metrics answer your scientific question, or catch leakage that happened
193
+ upstream.
194
+
195
+ ## Documentation and examples
196
+
197
+ The [documentation index](docs/README.md) links all the guides. There are also
198
+ [13 example notebooks](examples/README.md) written in marimo. To open one:
199
+
200
+ ```bash
201
+ uv sync --all-extras --group examples
202
+ uv run marimo edit examples/01_binary_classification.py
203
+ ```
204
+
205
+ ## Citing and license
206
+
207
+ If you use Saber in published work, please cite it using
208
+ [`CITATION.cff`](https://github.com/kren-ai-lab/saber/blob/main/CITATION.cff), or the "Cite this repository" button on
209
+ GitHub. Saber is released under the [MIT license](https://github.com/kren-ai-lab/saber/blob/main/LICENSE).
@@ -0,0 +1,161 @@
1
+ # Saber
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/saberlib?style=flat-square)](https://pypi.org/project/saberlib/)
4
+ [![PyVersions](https://img.shields.io/pypi/pyversions/saberlib?style=flat-square)](https://github.com/kren-ai-lab/saber)
5
+ [![Tests](https://img.shields.io/github/actions/workflow/status/kren-ai-lab/saber/tests.yml?style=flat-square)](https://github.com/kren-ai-lab/saber/actions/workflows/tests.yml)
6
+ ![License](https://img.shields.io/github/license/kren-ai-lab/saber?style=flat-square)
7
+
8
+ Saber is a Python library for classical supervised machine learning
9
+ (classification and regression) on numerical tabular features. It trains,
10
+ validates, tunes, benchmarks and persists scikit-learn, XGBoost and LightGBM
11
+ models, with preprocessing fitted inside each fold and every result tied back
12
+ to its samples, partition and parameters.
13
+
14
+ Saber doesn't compute features. Bring a descriptor table, an embedding or any
15
+ numeric matrix, and Saber models it. Deep learning, AutoML, multilabel and
16
+ multi-output problems are out of scope.
17
+
18
+ ## Installation
19
+
20
+ Saber supports Python 3.11 to 3.14.
21
+
22
+ ```bash
23
+ python -m pip install saberlib
24
+ ```
25
+
26
+ Optional extras:
27
+
28
+ ```bash
29
+ python -m pip install "saberlib[biosieve]" # partition generation
30
+ python -m pip install "saberlib[optuna]" # Optuna tuning
31
+ python -m pip install "saberlib[xgboost]" # XGBoost models
32
+ python -m pip install "saberlib[lightgbm]" # LightGBM models
33
+ python -m pip install "saberlib[all]"
34
+ ```
35
+
36
+ For development setup with
37
+ `uv`, see [DEVELOPMENT.md](DEVELOPMENT.md).
38
+
39
+ ## Validate a model
40
+
41
+ ```python
42
+ import numpy as np
43
+ from sklearn.datasets import make_classification
44
+
45
+ import saber
46
+ from saber import DatasetBundle, PartitionPlan
47
+
48
+ X, y = make_classification(n_samples=120, n_features=12, random_state=42)
49
+ dataset = DatasetBundle(X=X, y=y, sample_ids=[f"s{i}" for i in range(len(y))])
50
+
51
+ plan = PartitionPlan.from_predefined_folds(
52
+ sample_ids=dataset.sample_ids,
53
+ fold_assignments=np.arange(len(y)) % 4,
54
+ dataset=dataset,
55
+ )
56
+
57
+ result = saber.validate(
58
+ dataset=dataset,
59
+ algorithm="logistic_regression",
60
+ partition_plan=plan,
61
+ metrics=("mcc", "balanced_accuracy", "roc_auc"),
62
+ random_state=42,
63
+ )
64
+ print(result.aggregate_metrics)
65
+ ```
66
+
67
+ `result.oof_prediction` holds the out-of-fold predictions, aligned by sample
68
+ ID. `X` can be a NumPy array, a Polars DataFrame or a pandas DataFrame. Partitions
69
+ refer to samples by ID, never by row position. If your data isn't split yet,
70
+ let BioSieve generate the partitions:
71
+
72
+ ```python
73
+ from saber import BioSievePartitionConfig
74
+
75
+ result = saber.validate(
76
+ dataset=dataset,
77
+ algorithm="random_forest_classifier",
78
+ partition_plan=BioSievePartitionConfig(
79
+ strategy="stratified_kfold",
80
+ params={"n_splits": 5, "seed": 42},
81
+ ),
82
+ metrics=("mcc", "roc_auc"),
83
+ )
84
+ ```
85
+
86
+ The [data and partitions](docs/data_and_partitions.md) guide covers holdouts,
87
+ external partition files and BioSieve.
88
+
89
+ ## Tune, benchmark and save
90
+
91
+ ```python
92
+ from saber import Categorical, LogFloat, SearchSpace, TuningConfig
93
+
94
+ tuned = saber.tune(
95
+ dataset=dataset,
96
+ algorithm="svc",
97
+ partition_plan=plan,
98
+ search_space=SearchSpace(
99
+ parameters={"C": LogFloat(1e-3, 1e2), "kernel": Categorical(["linear", "rbf"])},
100
+ ),
101
+ config=TuningConfig(optimizer="random", n_trials=20),
102
+ metrics=("mcc",),
103
+ random_state=42,
104
+ )
105
+ print(tuned.best_params)
106
+ ```
107
+
108
+ `saber.benchmark(...)` crosses representations, partitions, algorithms, seeds
109
+ and tuned/untuned modes, and returns long-form Polars tables of metrics and
110
+ per-sample predictions. `saber.train(...)` fits a final model, and
111
+ `saber.save_model(path, result, dataset=dataset)` writes a `train` or `tune` result as a directory with a feature schema,
112
+ provenance and checksums that `saber.load_model(...)` verifies before loading.
113
+ See the [tuning](docs/tuning.md), [benchmarking](docs/benchmarking.md) and
114
+ [persistence](docs/persistence.md) guides.
115
+
116
+ ## Command line
117
+
118
+ ```bash
119
+ saber models list --task classification
120
+ saber run experiment.yaml --dry-run
121
+ saber run study.yaml --json
122
+ saber artifact verify artifacts/model
123
+ ```
124
+
125
+ Every workflow can be written as a YAML or JSON file and run from the CLI or
126
+ with `saber.run_config("experiment.yaml")`. See the [configuration](docs/configuration.md)
127
+ and [CLI](docs/cli.md) references.
128
+
129
+ ## Design principles
130
+
131
+ - Imputation and scaling are fitted inside each training fold, never on the
132
+ whole dataset.
133
+ - Existing partitions are used exactly as given. New ones come from BioSieve;
134
+ Saber has no splitters of its own.
135
+ - Tuned results are reported on a protected test set, not on the folds used to
136
+ pick the hyperparameters.
137
+ - For binary problems the positive class is explicit, never inferred from a
138
+ probability column's position.
139
+ - Saved models carry dataset and partition fingerprints and are checksum
140
+ verified before loading. They use joblib, so load them only from trusted
141
+ sources.
142
+
143
+ Saber keeps the workflow honest, but it can't tell whether your partitions or
144
+ metrics answer your scientific question, or catch leakage that happened
145
+ upstream.
146
+
147
+ ## Documentation and examples
148
+
149
+ The [documentation index](docs/README.md) links all the guides. There are also
150
+ [13 example notebooks](examples/README.md) written in marimo. To open one:
151
+
152
+ ```bash
153
+ uv sync --all-extras --group examples
154
+ uv run marimo edit examples/01_binary_classification.py
155
+ ```
156
+
157
+ ## Citing and license
158
+
159
+ If you use Saber in published work, please cite it using
160
+ [`CITATION.cff`](https://github.com/kren-ai-lab/saber/blob/main/CITATION.cff), or the "Cite this repository" button on
161
+ GitHub. Saber is released under the [MIT license](https://github.com/kren-ai-lab/saber/blob/main/LICENSE).
@@ -0,0 +1,174 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "saberlib"
7
+ dynamic = ["version"]
8
+ description = "Domain-agnostic classical supervised machine learning for classification, regression, tuning, and benchmarking."
9
+ readme = { file = "README.md", content-type = "text/markdown" }
10
+ requires-python = ">=3.11,<3.15"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [
14
+ { name = "Diego Alvarez-Saravia" },
15
+ { name = "David Medina-Ortiz" },
16
+ ]
17
+ maintainers = [
18
+ { name = "Diego Alvarez-Saravia" },
19
+ { name = "David Medina-Ortiz" },
20
+ ]
21
+ keywords = [
22
+ "machine-learning",
23
+ "supervised-learning",
24
+ "tabular-data",
25
+ "classification",
26
+ "regression",
27
+ "benchmarking",
28
+ "hyperparameter-optimization",
29
+ "reproducibility",
30
+ "scientific-software",
31
+ ]
32
+ classifiers = [
33
+ "Development Status :: 4 - Beta",
34
+ "Intended Audience :: Science/Research",
35
+ "Intended Audience :: Developers",
36
+ "Operating System :: OS Independent",
37
+ "Programming Language :: Python :: 3",
38
+ "Programming Language :: Python :: 3.11",
39
+ "Programming Language :: Python :: 3.12",
40
+ "Programming Language :: Python :: 3.13",
41
+ "Programming Language :: Python :: 3.14",
42
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
43
+ "Typing :: Typed",
44
+ ]
45
+
46
+ dependencies = [
47
+ "numpy>=2.0,<3",
48
+ "scikit-learn>=1.8,<2",
49
+ "scipy>=1.16,<2",
50
+ "joblib>=1.3,<2",
51
+ "rich>=13.0,<16",
52
+ "pyyaml>=6.0,<7",
53
+ "typer>=0.27,<1",
54
+ "polars>=1.44,<2",
55
+ ]
56
+
57
+ [project.scripts]
58
+ saber = "saber.cli.main:main"
59
+
60
+ [project.optional-dependencies]
61
+ biosieve = ["biosieve>=0.1.2,<0.2"]
62
+ xgboost = ["xgboost>=2.0,<4"]
63
+ lightgbm = ["lightgbm>=4.6,<5"]
64
+ optuna = ["optuna>=4.6,<5"]
65
+ all = [
66
+ "biosieve>=0.1.2,<0.2",
67
+ "xgboost>=2.0,<4",
68
+ "lightgbm>=4.6,<5",
69
+ "optuna>=4.6,<5",
70
+ ]
71
+
72
+ [project.urls]
73
+ Homepage = "https://github.com/kren-ai-lab/saber"
74
+ Repository = "https://github.com/kren-ai-lab/saber"
75
+ Issues = "https://github.com/kren-ai-lab/saber/issues"
76
+ Documentation = "https://github.com/kren-ai-lab/saber#readme"
77
+
78
+ [dependency-groups]
79
+ dev = [
80
+ "pytest>=9.0",
81
+ "pytest-cov>=7.1",
82
+ "ruff>=0.16",
83
+ "pyrefly>1.0",
84
+ "taskipy>=1.14",
85
+ "prek",
86
+ "pandas>=2.2.2",
87
+ "pyarrow>=25.0.1",
88
+ ]
89
+ examples = [
90
+ "marimo>=0.24",
91
+ "matplotlib>=3.8",
92
+ ]
93
+
94
+ [tool.hatch.version]
95
+ path = "saber/_version.py"
96
+ pattern = '__version__ = "(?P<version>[^"]+)"'
97
+
98
+ [tool.hatch.build.targets.sdist]
99
+ include = ["/saber"]
100
+
101
+ [tool.hatch.build.targets.wheel]
102
+ packages = ["saber"]
103
+
104
+ [tool.pytest.ini_options]
105
+ testpaths = ["tests"]
106
+ addopts = "-ra"
107
+
108
+ [tool.ruff]
109
+ line-length = 110
110
+ target-version = "py311"
111
+
112
+ [tool.ruff.lint]
113
+ select = ["ALL"]
114
+ ignore = [
115
+ "COM812",
116
+ "CPY001",
117
+ "D203",
118
+ "D213",
119
+ "N803",
120
+ "N806",
121
+ "PLR0913",
122
+ # Deliberate exceptions, not pending work.
123
+ "TRY003", # a raise names the offending data, not a class per failure mode
124
+ "EM101",
125
+ "EM102",
126
+ "ANN401", # coercion helpers take anything; `object` only moves it to pyrefly
127
+ "PLR2004", # `if n < 3` reads better than a named constant
128
+ "C901", # workflows branch on contracts; splitting them hides the execution path
129
+ "PLR0911",
130
+ "PLR0912",
131
+ "PLR0915",
132
+ "PLR0917",
133
+ ]
134
+
135
+ [tool.ruff.lint.per-file-ignores]
136
+ "saber/cli/*" = [
137
+ "FBT001", # bool flags are standard in CLI functions
138
+ "FBT002",
139
+ "FBT003",
140
+ "TC001", # typer resolves annotations at runtime, so they stay imported
141
+ "TC002",
142
+ "TC003",
143
+ ]
144
+ "tests/*" = [
145
+ "ANN", # test helpers are not part of the typed public surface
146
+ "D", # docstrings are not required in tests
147
+ "INP001", # test packages are intentionally implicit namespaces
148
+ "PLC0415",
149
+ "PT011",
150
+ "PT017",
151
+ "PT018",
152
+ "S101", # assert is how a test asserts
153
+ "TC003", # pytest resolves fixture annotations at runtime
154
+ ]
155
+ "examples/*" = [
156
+ "INP001", # example scripts are intentionally standalone modules
157
+ "T201", # examples print result snippets by design
158
+ ]
159
+
160
+ [tool.pyrefly]
161
+ # Without a config section Pyrefly falls back to the `basic` preset, which
162
+ # silences most of its checks; `default` is the real baseline.
163
+ preset = "default"
164
+ project-includes = ["saber", "tests"]
165
+
166
+ [tool.taskipy.tasks]
167
+ sort-imports = "ruff check --fix --select I,F401 saber/ tests/"
168
+ format = "task sort-imports && ruff format saber/ tests/"
169
+ lint = "ruff check saber/ tests/"
170
+ lint-fix = "ruff check --fix saber/ tests/"
171
+ test = "pytest -q"
172
+ test-v = "pytest -v"
173
+ test-cov = "pytest --cov=saber --cov-report=html"
174
+ pyrefly = "pyrefly check saber/ tests/"
@@ -0,0 +1,71 @@
1
+ """saber — classical supervised machine learning infrastructure."""
2
+
3
+ from saber._api import evaluate, predict, train
4
+ from saber._version import __version__
5
+ from saber.benchmark import BenchmarkConfig, BenchmarkResult, benchmark
6
+ from saber.config import run_config
7
+ from saber.core import (
8
+ ALGORITHMS,
9
+ Categorical,
10
+ Float,
11
+ Integer,
12
+ LogFloat,
13
+ PredictionResult,
14
+ SearchSpace,
15
+ TrainResult,
16
+ )
17
+ from saber.datasets import BioSievePartitionConfig, DatasetBundle, PartitionPlan
18
+ from saber.evaluation import EvaluationResult
19
+ from saber.exceptions import SaberError
20
+ from saber.persistence import (
21
+ LoadedModelArtifact,
22
+ inspect_artifact,
23
+ load_benchmark,
24
+ load_model,
25
+ save_benchmark,
26
+ save_model,
27
+ )
28
+ from saber.preprocessing import PreprocessingConfig
29
+ from saber.tuning import OptimizationResult, TuningConfig, tune
30
+ from saber.validation import ValidationResult, validate
31
+
32
+ __all__ = [ # noqa: RUF022 # grouped by purpose, not alphabetical
33
+ # Workflows
34
+ "train",
35
+ "validate",
36
+ "tune",
37
+ "benchmark",
38
+ "evaluate",
39
+ "predict",
40
+ "run_config",
41
+ # Persistence
42
+ "save_model",
43
+ "load_model",
44
+ "save_benchmark",
45
+ "load_benchmark",
46
+ "inspect_artifact",
47
+ # Inputs
48
+ "DatasetBundle",
49
+ "PartitionPlan",
50
+ "BioSievePartitionConfig",
51
+ "PreprocessingConfig",
52
+ "TuningConfig",
53
+ "BenchmarkConfig",
54
+ "SearchSpace",
55
+ "Categorical",
56
+ "Integer",
57
+ "Float",
58
+ "LogFloat",
59
+ # Results
60
+ "TrainResult",
61
+ "PredictionResult",
62
+ "EvaluationResult",
63
+ "ValidationResult",
64
+ "OptimizationResult",
65
+ "BenchmarkResult",
66
+ "LoadedModelArtifact",
67
+ # Catalog / misc
68
+ "ALGORITHMS",
69
+ "SaberError",
70
+ "__version__",
71
+ ]
@@ -0,0 +1,5 @@
1
+ """Entry point for running saber's CLI as a module."""
2
+
3
+ from saber.cli.main import main
4
+
5
+ raise SystemExit(main())
@@ -0,0 +1,7 @@
1
+ """Private implementation of the train/predict/evaluate workflows."""
2
+
3
+ from saber._api.evaluate import evaluate
4
+ from saber._api.predict import predict
5
+ from saber._api.train import train
6
+
7
+ __all__ = ["evaluate", "predict", "train"]