saberlib 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- saberlib-0.1.0/.gitignore +26 -0
- saberlib-0.1.0/LICENSE +21 -0
- saberlib-0.1.0/PKG-INFO +209 -0
- saberlib-0.1.0/README.md +161 -0
- saberlib-0.1.0/pyproject.toml +174 -0
- saberlib-0.1.0/saber/__init__.py +71 -0
- saberlib-0.1.0/saber/__main__.py +5 -0
- saberlib-0.1.0/saber/_api/__init__.py +7 -0
- saberlib-0.1.0/saber/_api/_common.py +71 -0
- saberlib-0.1.0/saber/_api/evaluate.py +43 -0
- saberlib-0.1.0/saber/_api/predict.py +50 -0
- saberlib-0.1.0/saber/_api/train.py +68 -0
- saberlib-0.1.0/saber/_version.py +3 -0
- saberlib-0.1.0/saber/benchmark/__init__.py +13 -0
- saberlib-0.1.0/saber/benchmark/config.py +47 -0
- saberlib-0.1.0/saber/benchmark/engine.py +509 -0
- saberlib-0.1.0/saber/benchmark/results.py +199 -0
- saberlib-0.1.0/saber/classification/__init__.py +1 -0
- saberlib-0.1.0/saber/classification/lightgbm.py +25 -0
- saberlib-0.1.0/saber/classification/search_spaces.py +262 -0
- saberlib-0.1.0/saber/classification/sklearn.py +223 -0
- saberlib-0.1.0/saber/classification/xgboost.py +38 -0
- saberlib-0.1.0/saber/cli/__init__.py +5 -0
- saberlib-0.1.0/saber/cli/main.py +201 -0
- saberlib-0.1.0/saber/cli/render.py +306 -0
- saberlib-0.1.0/saber/config/__init__.py +14 -0
- saberlib-0.1.0/saber/config/builders.py +171 -0
- saberlib-0.1.0/saber/config/io.py +58 -0
- saberlib-0.1.0/saber/config/runner.py +314 -0
- saberlib-0.1.0/saber/config/schema.py +284 -0
- saberlib-0.1.0/saber/core/__init__.py +22 -0
- saberlib-0.1.0/saber/core/capabilities.py +76 -0
- saberlib-0.1.0/saber/core/metrics.py +234 -0
- saberlib-0.1.0/saber/core/prediction.py +254 -0
- saberlib-0.1.0/saber/core/registry.py +61 -0
- saberlib-0.1.0/saber/core/results.py +35 -0
- saberlib-0.1.0/saber/core/search_space.py +237 -0
- saberlib-0.1.0/saber/core/specs.py +138 -0
- saberlib-0.1.0/saber/core/task.py +7 -0
- saberlib-0.1.0/saber/datasets/__init__.py +16 -0
- saberlib-0.1.0/saber/datasets/_fingerprint.py +186 -0
- saberlib-0.1.0/saber/datasets/biosieve.py +343 -0
- saberlib-0.1.0/saber/datasets/folds.py +342 -0
- saberlib-0.1.0/saber/datasets/loaders.py +188 -0
- saberlib-0.1.0/saber/datasets/schemas.py +279 -0
- saberlib-0.1.0/saber/datasets/validation.py +206 -0
- saberlib-0.1.0/saber/evaluation/__init__.py +6 -0
- saberlib-0.1.0/saber/evaluation/classification.py +390 -0
- saberlib-0.1.0/saber/evaluation/evaluator.py +61 -0
- saberlib-0.1.0/saber/evaluation/regression.py +47 -0
- saberlib-0.1.0/saber/evaluation/results.py +30 -0
- saberlib-0.1.0/saber/exceptions.py +170 -0
- saberlib-0.1.0/saber/persistence/__init__.py +18 -0
- saberlib-0.1.0/saber/persistence/artifacts.py +117 -0
- saberlib-0.1.0/saber/persistence/checksums.py +75 -0
- saberlib-0.1.0/saber/persistence/environment.py +97 -0
- saberlib-0.1.0/saber/persistence/load.py +140 -0
- saberlib-0.1.0/saber/persistence/metadata.py +60 -0
- saberlib-0.1.0/saber/persistence/save.py +276 -0
- saberlib-0.1.0/saber/preprocessing/__init__.py +5 -0
- saberlib-0.1.0/saber/preprocessing/imputation.py +30 -0
- saberlib-0.1.0/saber/preprocessing/pipeline.py +130 -0
- saberlib-0.1.0/saber/preprocessing/scaling.py +57 -0
- saberlib-0.1.0/saber/preprocessing/validation.py +65 -0
- saberlib-0.1.0/saber/py.typed +0 -0
- saberlib-0.1.0/saber/regression/__init__.py +1 -0
- saberlib-0.1.0/saber/regression/lightgbm.py +25 -0
- saberlib-0.1.0/saber/regression/search_spaces.py +252 -0
- saberlib-0.1.0/saber/regression/sklearn.py +257 -0
- saberlib-0.1.0/saber/regression/xgboost.py +38 -0
- saberlib-0.1.0/saber/tuning/__init__.py +10 -0
- saberlib-0.1.0/saber/tuning/engine.py +601 -0
- saberlib-0.1.0/saber/tuning/results.py +93 -0
- saberlib-0.1.0/saber/utils/__init__.py +1 -0
- saberlib-0.1.0/saber/utils/serialization.py +65 -0
- saberlib-0.1.0/saber/utils/tabular.py +133 -0
- saberlib-0.1.0/saber/validation/__init__.py +10 -0
- saberlib-0.1.0/saber/validation/cross_validation.py +243 -0
- saberlib-0.1.0/saber/validation/partitioning.py +119 -0
- saberlib-0.1.0/saber/validation/results.py +134 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
|
|
5
|
+
# Build
|
|
6
|
+
build/
|
|
7
|
+
dist/
|
|
8
|
+
*.egg-info/
|
|
9
|
+
|
|
10
|
+
# Environments
|
|
11
|
+
.venv/
|
|
12
|
+
.env
|
|
13
|
+
|
|
14
|
+
# Tooling caches and reports
|
|
15
|
+
.pytest_cache/
|
|
16
|
+
.ruff_cache/
|
|
17
|
+
.coverage
|
|
18
|
+
.coverage.*
|
|
19
|
+
htmlcov/
|
|
20
|
+
|
|
21
|
+
# Marimo
|
|
22
|
+
__marimo__/
|
|
23
|
+
|
|
24
|
+
# Editors
|
|
25
|
+
.idea/
|
|
26
|
+
.vscode/
|
saberlib-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kren-AI Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
saberlib-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: saberlib
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Domain-agnostic classical supervised machine learning for classification, regression, tuning, and benchmarking.
|
|
5
|
+
Project-URL: Homepage, https://github.com/kren-ai-lab/saber
|
|
6
|
+
Project-URL: Repository, https://github.com/kren-ai-lab/saber
|
|
7
|
+
Project-URL: Issues, https://github.com/kren-ai-lab/saber/issues
|
|
8
|
+
Project-URL: Documentation, https://github.com/kren-ai-lab/saber#readme
|
|
9
|
+
Author: Diego Alvarez-Saravia, David Medina-Ortiz
|
|
10
|
+
Maintainer: Diego Alvarez-Saravia, David Medina-Ortiz
|
|
11
|
+
License-Expression: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: benchmarking,classification,hyperparameter-optimization,machine-learning,regression,reproducibility,scientific-software,supervised-learning,tabular-data
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: <3.15,>=3.11
|
|
26
|
+
Requires-Dist: joblib<2,>=1.3
|
|
27
|
+
Requires-Dist: numpy<3,>=2.0
|
|
28
|
+
Requires-Dist: polars<2,>=1.44
|
|
29
|
+
Requires-Dist: pyyaml<7,>=6.0
|
|
30
|
+
Requires-Dist: rich<16,>=13.0
|
|
31
|
+
Requires-Dist: scikit-learn<2,>=1.8
|
|
32
|
+
Requires-Dist: scipy<2,>=1.16
|
|
33
|
+
Requires-Dist: typer<1,>=0.27
|
|
34
|
+
Provides-Extra: all
|
|
35
|
+
Requires-Dist: biosieve<0.2,>=0.1.2; extra == 'all'
|
|
36
|
+
Requires-Dist: lightgbm<5,>=4.6; extra == 'all'
|
|
37
|
+
Requires-Dist: optuna<5,>=4.6; extra == 'all'
|
|
38
|
+
Requires-Dist: xgboost<4,>=2.0; extra == 'all'
|
|
39
|
+
Provides-Extra: biosieve
|
|
40
|
+
Requires-Dist: biosieve<0.2,>=0.1.2; extra == 'biosieve'
|
|
41
|
+
Provides-Extra: lightgbm
|
|
42
|
+
Requires-Dist: lightgbm<5,>=4.6; extra == 'lightgbm'
|
|
43
|
+
Provides-Extra: optuna
|
|
44
|
+
Requires-Dist: optuna<5,>=4.6; extra == 'optuna'
|
|
45
|
+
Provides-Extra: xgboost
|
|
46
|
+
Requires-Dist: xgboost<4,>=2.0; extra == 'xgboost'
|
|
47
|
+
Description-Content-Type: text/markdown
|
|
48
|
+
|
|
49
|
+
# Saber
|
|
50
|
+
|
|
51
|
+
[](https://pypi.org/project/saberlib/)
|
|
52
|
+
[](https://github.com/kren-ai-lab/saber)
|
|
53
|
+
[](https://github.com/kren-ai-lab/saber/actions/workflows/tests.yml)
|
|
54
|
+

|
|
55
|
+
|
|
56
|
+
Saber is a Python library for classical supervised machine learning
|
|
57
|
+
(classification and regression) on numerical tabular features. It trains,
|
|
58
|
+
validates, tunes, benchmarks and persists scikit-learn, XGBoost and LightGBM
|
|
59
|
+
models, with preprocessing fitted inside each fold and every result tied back
|
|
60
|
+
to its samples, partition and parameters.
|
|
61
|
+
|
|
62
|
+
Saber doesn't compute features. Bring a descriptor table, an embedding or any
|
|
63
|
+
numeric matrix, and Saber models it. Deep learning, AutoML, multilabel and
|
|
64
|
+
multi-output problems are out of scope.
|
|
65
|
+
|
|
66
|
+
## Installation
|
|
67
|
+
|
|
68
|
+
Saber supports Python 3.11 to 3.14.
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
python -m pip install saberlib
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Optional extras:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
python -m pip install "saberlib[biosieve]" # partition generation
|
|
78
|
+
python -m pip install "saberlib[optuna]" # Optuna tuning
|
|
79
|
+
python -m pip install "saberlib[xgboost]" # XGBoost models
|
|
80
|
+
python -m pip install "saberlib[lightgbm]" # LightGBM models
|
|
81
|
+
python -m pip install "saberlib[all]"
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
For development setup with
|
|
85
|
+
`uv`, see [DEVELOPMENT.md](DEVELOPMENT.md).
|
|
86
|
+
|
|
87
|
+
## Validate a model
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
import numpy as np
|
|
91
|
+
from sklearn.datasets import make_classification
|
|
92
|
+
|
|
93
|
+
import saber
|
|
94
|
+
from saber import DatasetBundle, PartitionPlan
|
|
95
|
+
|
|
96
|
+
X, y = make_classification(n_samples=120, n_features=12, random_state=42)
|
|
97
|
+
dataset = DatasetBundle(X=X, y=y, sample_ids=[f"s{i}" for i in range(len(y))])
|
|
98
|
+
|
|
99
|
+
plan = PartitionPlan.from_predefined_folds(
|
|
100
|
+
sample_ids=dataset.sample_ids,
|
|
101
|
+
fold_assignments=np.arange(len(y)) % 4,
|
|
102
|
+
dataset=dataset,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
result = saber.validate(
|
|
106
|
+
dataset=dataset,
|
|
107
|
+
algorithm="logistic_regression",
|
|
108
|
+
partition_plan=plan,
|
|
109
|
+
metrics=("mcc", "balanced_accuracy", "roc_auc"),
|
|
110
|
+
random_state=42,
|
|
111
|
+
)
|
|
112
|
+
print(result.aggregate_metrics)
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
`result.oof_prediction` holds the out-of-fold predictions, aligned by sample
|
|
116
|
+
ID. `X` can be a NumPy array, a Polars DataFrame or a pandas DataFrame. Partitions
|
|
117
|
+
refer to samples by ID, never by row position. If your data isn't split yet,
|
|
118
|
+
let BioSieve generate the partitions:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
from saber import BioSievePartitionConfig
|
|
122
|
+
|
|
123
|
+
result = saber.validate(
|
|
124
|
+
dataset=dataset,
|
|
125
|
+
algorithm="random_forest_classifier",
|
|
126
|
+
partition_plan=BioSievePartitionConfig(
|
|
127
|
+
strategy="stratified_kfold",
|
|
128
|
+
params={"n_splits": 5, "seed": 42},
|
|
129
|
+
),
|
|
130
|
+
metrics=("mcc", "roc_auc"),
|
|
131
|
+
)
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
The [data and partitions](docs/data_and_partitions.md) guide covers holdouts,
|
|
135
|
+
external partition files and BioSieve.
|
|
136
|
+
|
|
137
|
+
## Tune, benchmark and save
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
from saber import Categorical, LogFloat, SearchSpace, TuningConfig
|
|
141
|
+
|
|
142
|
+
tuned = saber.tune(
|
|
143
|
+
dataset=dataset,
|
|
144
|
+
algorithm="svc",
|
|
145
|
+
partition_plan=plan,
|
|
146
|
+
search_space=SearchSpace(
|
|
147
|
+
parameters={"C": LogFloat(1e-3, 1e2), "kernel": Categorical(["linear", "rbf"])},
|
|
148
|
+
),
|
|
149
|
+
config=TuningConfig(optimizer="random", n_trials=20),
|
|
150
|
+
metrics=("mcc",),
|
|
151
|
+
random_state=42,
|
|
152
|
+
)
|
|
153
|
+
print(tuned.best_params)
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`saber.benchmark(...)` crosses representations, partitions, algorithms, seeds
|
|
157
|
+
and tuned/untuned modes, and returns long-form Polars tables of metrics and
|
|
158
|
+
per-sample predictions. `saber.train(...)` fits a final model, and
|
|
159
|
+
`saber.save_model(path, result, dataset=dataset)` writes a `train` or `tune` result as a directory with a feature schema,
|
|
160
|
+
provenance and checksums that `saber.load_model(...)` verifies before loading.
|
|
161
|
+
See the [tuning](docs/tuning.md), [benchmarking](docs/benchmarking.md) and
|
|
162
|
+
[persistence](docs/persistence.md) guides.
|
|
163
|
+
|
|
164
|
+
## Command line
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
saber models list --task classification
|
|
168
|
+
saber run experiment.yaml --dry-run
|
|
169
|
+
saber run study.yaml --json
|
|
170
|
+
saber artifact verify artifacts/model
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
Every workflow can be written as a YAML or JSON file and run from the CLI or
|
|
174
|
+
with `saber.run_config("experiment.yaml")`. See the [configuration](docs/configuration.md)
|
|
175
|
+
and [CLI](docs/cli.md) references.
|
|
176
|
+
|
|
177
|
+
## Design principles
|
|
178
|
+
|
|
179
|
+
- Imputation and scaling are fitted inside each training fold, never on the
|
|
180
|
+
whole dataset.
|
|
181
|
+
- Existing partitions are used exactly as given. New ones come from BioSieve;
|
|
182
|
+
Saber has no splitters of its own.
|
|
183
|
+
- Tuned results are reported on a protected test set, not on the folds used to
|
|
184
|
+
pick the hyperparameters.
|
|
185
|
+
- For binary problems the positive class is explicit, never inferred from a
|
|
186
|
+
probability column's position.
|
|
187
|
+
- Saved models carry dataset and partition fingerprints and are checksum
|
|
188
|
+
verified before loading. They use joblib, so load them only from trusted
|
|
189
|
+
sources.
|
|
190
|
+
|
|
191
|
+
Saber keeps the workflow honest, but it can't tell whether your partitions or
|
|
192
|
+
metrics answer your scientific question, or catch leakage that happened
|
|
193
|
+
upstream.
|
|
194
|
+
|
|
195
|
+
## Documentation and examples
|
|
196
|
+
|
|
197
|
+
The [documentation index](docs/README.md) links all the guides. There are also
|
|
198
|
+
[13 example notebooks](examples/README.md) written in marimo. To open one:
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
uv sync --all-extras --group examples
|
|
202
|
+
uv run marimo edit examples/01_binary_classification.py
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
## Citing and license
|
|
206
|
+
|
|
207
|
+
If you use Saber in published work, please cite it using
|
|
208
|
+
[`CITATION.cff`](https://github.com/kren-ai-lab/saber/blob/main/CITATION.cff), or the "Cite this repository" button on
|
|
209
|
+
GitHub. Saber is released under the [MIT license](https://github.com/kren-ai-lab/saber/blob/main/LICENSE).
|
saberlib-0.1.0/README.md
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Saber
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/saberlib/)
|
|
4
|
+
[](https://github.com/kren-ai-lab/saber)
|
|
5
|
+
[](https://github.com/kren-ai-lab/saber/actions/workflows/tests.yml)
|
|
6
|
+

|
|
7
|
+
|
|
8
|
+
Saber is a Python library for classical supervised machine learning
|
|
9
|
+
(classification and regression) on numerical tabular features. It trains,
|
|
10
|
+
validates, tunes, benchmarks and persists scikit-learn, XGBoost and LightGBM
|
|
11
|
+
models, with preprocessing fitted inside each fold and every result tied back
|
|
12
|
+
to its samples, partition and parameters.
|
|
13
|
+
|
|
14
|
+
Saber doesn't compute features. Bring a descriptor table, an embedding or any
|
|
15
|
+
numeric matrix, and Saber models it. Deep learning, AutoML, multilabel and
|
|
16
|
+
multi-output problems are out of scope.
|
|
17
|
+
|
|
18
|
+
## Installation
|
|
19
|
+
|
|
20
|
+
Saber supports Python 3.11 to 3.14.
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
python -m pip install saberlib
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Optional extras:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
python -m pip install "saberlib[biosieve]" # partition generation
|
|
30
|
+
python -m pip install "saberlib[optuna]" # Optuna tuning
|
|
31
|
+
python -m pip install "saberlib[xgboost]" # XGBoost models
|
|
32
|
+
python -m pip install "saberlib[lightgbm]" # LightGBM models
|
|
33
|
+
python -m pip install "saberlib[all]"
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
For development setup with
|
|
37
|
+
`uv`, see [DEVELOPMENT.md](DEVELOPMENT.md).
|
|
38
|
+
|
|
39
|
+
## Validate a model
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
import numpy as np
|
|
43
|
+
from sklearn.datasets import make_classification
|
|
44
|
+
|
|
45
|
+
import saber
|
|
46
|
+
from saber import DatasetBundle, PartitionPlan
|
|
47
|
+
|
|
48
|
+
X, y = make_classification(n_samples=120, n_features=12, random_state=42)
|
|
49
|
+
dataset = DatasetBundle(X=X, y=y, sample_ids=[f"s{i}" for i in range(len(y))])
|
|
50
|
+
|
|
51
|
+
plan = PartitionPlan.from_predefined_folds(
|
|
52
|
+
sample_ids=dataset.sample_ids,
|
|
53
|
+
fold_assignments=np.arange(len(y)) % 4,
|
|
54
|
+
dataset=dataset,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
result = saber.validate(
|
|
58
|
+
dataset=dataset,
|
|
59
|
+
algorithm="logistic_regression",
|
|
60
|
+
partition_plan=plan,
|
|
61
|
+
metrics=("mcc", "balanced_accuracy", "roc_auc"),
|
|
62
|
+
random_state=42,
|
|
63
|
+
)
|
|
64
|
+
print(result.aggregate_metrics)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`result.oof_prediction` holds the out-of-fold predictions, aligned by sample
|
|
68
|
+
ID. `X` can be a NumPy array, a Polars DataFrame or a pandas DataFrame. Partitions
|
|
69
|
+
refer to samples by ID, never by row position. If your data isn't split yet,
|
|
70
|
+
let BioSieve generate the partitions:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from saber import BioSievePartitionConfig
|
|
74
|
+
|
|
75
|
+
result = saber.validate(
|
|
76
|
+
dataset=dataset,
|
|
77
|
+
algorithm="random_forest_classifier",
|
|
78
|
+
partition_plan=BioSievePartitionConfig(
|
|
79
|
+
strategy="stratified_kfold",
|
|
80
|
+
params={"n_splits": 5, "seed": 42},
|
|
81
|
+
),
|
|
82
|
+
metrics=("mcc", "roc_auc"),
|
|
83
|
+
)
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The [data and partitions](docs/data_and_partitions.md) guide covers holdouts,
|
|
87
|
+
external partition files and BioSieve.
|
|
88
|
+
|
|
89
|
+
## Tune, benchmark and save
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from saber import Categorical, LogFloat, SearchSpace, TuningConfig
|
|
93
|
+
|
|
94
|
+
tuned = saber.tune(
|
|
95
|
+
dataset=dataset,
|
|
96
|
+
algorithm="svc",
|
|
97
|
+
partition_plan=plan,
|
|
98
|
+
search_space=SearchSpace(
|
|
99
|
+
parameters={"C": LogFloat(1e-3, 1e2), "kernel": Categorical(["linear", "rbf"])},
|
|
100
|
+
),
|
|
101
|
+
config=TuningConfig(optimizer="random", n_trials=20),
|
|
102
|
+
metrics=("mcc",),
|
|
103
|
+
random_state=42,
|
|
104
|
+
)
|
|
105
|
+
print(tuned.best_params)
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
`saber.benchmark(...)` crosses representations, partitions, algorithms, seeds
|
|
109
|
+
and tuned/untuned modes, and returns long-form Polars tables of metrics and
|
|
110
|
+
per-sample predictions. `saber.train(...)` fits a final model, and
|
|
111
|
+
`saber.save_model(path, result, dataset=dataset)` writes a `train` or `tune` result as a directory with a feature schema,
|
|
112
|
+
provenance and checksums that `saber.load_model(...)` verifies before loading.
|
|
113
|
+
See the [tuning](docs/tuning.md), [benchmarking](docs/benchmarking.md) and
|
|
114
|
+
[persistence](docs/persistence.md) guides.
|
|
115
|
+
|
|
116
|
+
## Command line
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
saber models list --task classification
|
|
120
|
+
saber run experiment.yaml --dry-run
|
|
121
|
+
saber run study.yaml --json
|
|
122
|
+
saber artifact verify artifacts/model
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Every workflow can be written as a YAML or JSON file and run from the CLI or
|
|
126
|
+
with `saber.run_config("experiment.yaml")`. See the [configuration](docs/configuration.md)
|
|
127
|
+
and [CLI](docs/cli.md) references.
|
|
128
|
+
|
|
129
|
+
## Design principles
|
|
130
|
+
|
|
131
|
+
- Imputation and scaling are fitted inside each training fold, never on the
|
|
132
|
+
whole dataset.
|
|
133
|
+
- Existing partitions are used exactly as given. New ones come from BioSieve;
|
|
134
|
+
Saber has no splitters of its own.
|
|
135
|
+
- Tuned results are reported on a protected test set, not on the folds used to
|
|
136
|
+
pick the hyperparameters.
|
|
137
|
+
- For binary problems the positive class is explicit, never inferred from a
|
|
138
|
+
probability column's position.
|
|
139
|
+
- Saved models carry dataset and partition fingerprints and are checksum
|
|
140
|
+
verified before loading. They use joblib, so load them only from trusted
|
|
141
|
+
sources.
|
|
142
|
+
|
|
143
|
+
Saber keeps the workflow honest, but it can't tell whether your partitions or
|
|
144
|
+
metrics answer your scientific question, or catch leakage that happened
|
|
145
|
+
upstream.
|
|
146
|
+
|
|
147
|
+
## Documentation and examples
|
|
148
|
+
|
|
149
|
+
The [documentation index](docs/README.md) links all the guides. There are also
|
|
150
|
+
[13 example notebooks](examples/README.md) written in marimo. To open one:
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
uv sync --all-extras --group examples
|
|
154
|
+
uv run marimo edit examples/01_binary_classification.py
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
## Citing and license
|
|
158
|
+
|
|
159
|
+
If you use Saber in published work, please cite it using
|
|
160
|
+
[`CITATION.cff`](https://github.com/kren-ai-lab/saber/blob/main/CITATION.cff), or the "Cite this repository" button on
|
|
161
|
+
GitHub. Saber is released under the [MIT license](https://github.com/kren-ai-lab/saber/blob/main/LICENSE).
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "saberlib"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Domain-agnostic classical supervised machine learning for classification, regression, tuning, and benchmarking."
|
|
9
|
+
readme = { file = "README.md", content-type = "text/markdown" }
|
|
10
|
+
requires-python = ">=3.11,<3.15"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Diego Alvarez-Saravia" },
|
|
15
|
+
{ name = "David Medina-Ortiz" },
|
|
16
|
+
]
|
|
17
|
+
maintainers = [
|
|
18
|
+
{ name = "Diego Alvarez-Saravia" },
|
|
19
|
+
{ name = "David Medina-Ortiz" },
|
|
20
|
+
]
|
|
21
|
+
keywords = [
|
|
22
|
+
"machine-learning",
|
|
23
|
+
"supervised-learning",
|
|
24
|
+
"tabular-data",
|
|
25
|
+
"classification",
|
|
26
|
+
"regression",
|
|
27
|
+
"benchmarking",
|
|
28
|
+
"hyperparameter-optimization",
|
|
29
|
+
"reproducibility",
|
|
30
|
+
"scientific-software",
|
|
31
|
+
]
|
|
32
|
+
classifiers = [
|
|
33
|
+
"Development Status :: 4 - Beta",
|
|
34
|
+
"Intended Audience :: Science/Research",
|
|
35
|
+
"Intended Audience :: Developers",
|
|
36
|
+
"Operating System :: OS Independent",
|
|
37
|
+
"Programming Language :: Python :: 3",
|
|
38
|
+
"Programming Language :: Python :: 3.11",
|
|
39
|
+
"Programming Language :: Python :: 3.12",
|
|
40
|
+
"Programming Language :: Python :: 3.13",
|
|
41
|
+
"Programming Language :: Python :: 3.14",
|
|
42
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
43
|
+
"Typing :: Typed",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
dependencies = [
|
|
47
|
+
"numpy>=2.0,<3",
|
|
48
|
+
"scikit-learn>=1.8,<2",
|
|
49
|
+
"scipy>=1.16,<2",
|
|
50
|
+
"joblib>=1.3,<2",
|
|
51
|
+
"rich>=13.0,<16",
|
|
52
|
+
"pyyaml>=6.0,<7",
|
|
53
|
+
"typer>=0.27,<1",
|
|
54
|
+
"polars>=1.44,<2",
|
|
55
|
+
]
|
|
56
|
+
|
|
57
|
+
[project.scripts]
|
|
58
|
+
saber = "saber.cli.main:main"
|
|
59
|
+
|
|
60
|
+
[project.optional-dependencies]
|
|
61
|
+
biosieve = ["biosieve>=0.1.2,<0.2"]
|
|
62
|
+
xgboost = ["xgboost>=2.0,<4"]
|
|
63
|
+
lightgbm = ["lightgbm>=4.6,<5"]
|
|
64
|
+
optuna = ["optuna>=4.6,<5"]
|
|
65
|
+
all = [
|
|
66
|
+
"biosieve>=0.1.2,<0.2",
|
|
67
|
+
"xgboost>=2.0,<4",
|
|
68
|
+
"lightgbm>=4.6,<5",
|
|
69
|
+
"optuna>=4.6,<5",
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
[project.urls]
|
|
73
|
+
Homepage = "https://github.com/kren-ai-lab/saber"
|
|
74
|
+
Repository = "https://github.com/kren-ai-lab/saber"
|
|
75
|
+
Issues = "https://github.com/kren-ai-lab/saber/issues"
|
|
76
|
+
Documentation = "https://github.com/kren-ai-lab/saber#readme"
|
|
77
|
+
|
|
78
|
+
[dependency-groups]
|
|
79
|
+
dev = [
|
|
80
|
+
"pytest>=9.0",
|
|
81
|
+
"pytest-cov>=7.1",
|
|
82
|
+
"ruff>=0.16",
|
|
83
|
+
"pyrefly>1.0",
|
|
84
|
+
"taskipy>=1.14",
|
|
85
|
+
"prek",
|
|
86
|
+
"pandas>=2.2.2",
|
|
87
|
+
"pyarrow>=25.0.1",
|
|
88
|
+
]
|
|
89
|
+
examples = [
|
|
90
|
+
"marimo>=0.24",
|
|
91
|
+
"matplotlib>=3.8",
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
[tool.hatch.version]
|
|
95
|
+
path = "saber/_version.py"
|
|
96
|
+
pattern = '__version__ = "(?P<version>[^"]+)"'
|
|
97
|
+
|
|
98
|
+
[tool.hatch.build.targets.sdist]
|
|
99
|
+
include = ["/saber"]
|
|
100
|
+
|
|
101
|
+
[tool.hatch.build.targets.wheel]
|
|
102
|
+
packages = ["saber"]
|
|
103
|
+
|
|
104
|
+
[tool.pytest.ini_options]
|
|
105
|
+
testpaths = ["tests"]
|
|
106
|
+
addopts = "-ra"
|
|
107
|
+
|
|
108
|
+
[tool.ruff]
|
|
109
|
+
line-length = 110
|
|
110
|
+
target-version = "py311"
|
|
111
|
+
|
|
112
|
+
[tool.ruff.lint]
|
|
113
|
+
select = ["ALL"]
|
|
114
|
+
ignore = [
|
|
115
|
+
"COM812",
|
|
116
|
+
"CPY001",
|
|
117
|
+
"D203",
|
|
118
|
+
"D213",
|
|
119
|
+
"N803",
|
|
120
|
+
"N806",
|
|
121
|
+
"PLR0913",
|
|
122
|
+
# Deliberate exceptions, not pending work.
|
|
123
|
+
"TRY003", # a raise names the offending data, not a class per failure mode
|
|
124
|
+
"EM101",
|
|
125
|
+
"EM102",
|
|
126
|
+
"ANN401", # coercion helpers take anything; `object` only moves it to pyrefly
|
|
127
|
+
"PLR2004", # `if n < 3` reads better than a named constant
|
|
128
|
+
"C901", # workflows branch on contracts; splitting them hides the execution path
|
|
129
|
+
"PLR0911",
|
|
130
|
+
"PLR0912",
|
|
131
|
+
"PLR0915",
|
|
132
|
+
"PLR0917",
|
|
133
|
+
]
|
|
134
|
+
|
|
135
|
+
[tool.ruff.lint.per-file-ignores]
|
|
136
|
+
"saber/cli/*" = [
|
|
137
|
+
"FBT001", # bool flags are standard in CLI functions
|
|
138
|
+
"FBT002",
|
|
139
|
+
"FBT003",
|
|
140
|
+
"TC001", # typer resolves annotations at runtime, so they stay imported
|
|
141
|
+
"TC002",
|
|
142
|
+
"TC003",
|
|
143
|
+
]
|
|
144
|
+
"tests/*" = [
|
|
145
|
+
"ANN", # test helpers are not part of the typed public surface
|
|
146
|
+
"D", # docstrings are not required in tests
|
|
147
|
+
"INP001", # test packages are intentionally implicit namespaces
|
|
148
|
+
"PLC0415",
|
|
149
|
+
"PT011",
|
|
150
|
+
"PT017",
|
|
151
|
+
"PT018",
|
|
152
|
+
"S101", # assert is how a test asserts
|
|
153
|
+
"TC003", # pytest resolves fixture annotations at runtime
|
|
154
|
+
]
|
|
155
|
+
"examples/*" = [
|
|
156
|
+
"INP001", # example scripts are intentionally standalone modules
|
|
157
|
+
"T201", # examples print result snippets by design
|
|
158
|
+
]
|
|
159
|
+
|
|
160
|
+
[tool.pyrefly]
|
|
161
|
+
# Without a config section Pyrefly falls back to the `basic` preset, which
|
|
162
|
+
# silences most of its checks; `default` is the real baseline.
|
|
163
|
+
preset = "default"
|
|
164
|
+
project-includes = ["saber", "tests"]
|
|
165
|
+
|
|
166
|
+
[tool.taskipy.tasks]
|
|
167
|
+
sort-imports = "ruff check --fix --select I,F401 saber/ tests/"
|
|
168
|
+
format = "task sort-imports && ruff format saber/ tests/"
|
|
169
|
+
lint = "ruff check saber/ tests/"
|
|
170
|
+
lint-fix = "ruff check --fix saber/ tests/"
|
|
171
|
+
test = "pytest -q"
|
|
172
|
+
test-v = "pytest -v"
|
|
173
|
+
test-cov = "pytest --cov=saber --cov-report=html"
|
|
174
|
+
pyrefly = "pyrefly check saber/ tests/"
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""saber — classical supervised machine learning infrastructure."""
|
|
2
|
+
|
|
3
|
+
from saber._api import evaluate, predict, train
|
|
4
|
+
from saber._version import __version__
|
|
5
|
+
from saber.benchmark import BenchmarkConfig, BenchmarkResult, benchmark
|
|
6
|
+
from saber.config import run_config
|
|
7
|
+
from saber.core import (
|
|
8
|
+
ALGORITHMS,
|
|
9
|
+
Categorical,
|
|
10
|
+
Float,
|
|
11
|
+
Integer,
|
|
12
|
+
LogFloat,
|
|
13
|
+
PredictionResult,
|
|
14
|
+
SearchSpace,
|
|
15
|
+
TrainResult,
|
|
16
|
+
)
|
|
17
|
+
from saber.datasets import BioSievePartitionConfig, DatasetBundle, PartitionPlan
|
|
18
|
+
from saber.evaluation import EvaluationResult
|
|
19
|
+
from saber.exceptions import SaberError
|
|
20
|
+
from saber.persistence import (
|
|
21
|
+
LoadedModelArtifact,
|
|
22
|
+
inspect_artifact,
|
|
23
|
+
load_benchmark,
|
|
24
|
+
load_model,
|
|
25
|
+
save_benchmark,
|
|
26
|
+
save_model,
|
|
27
|
+
)
|
|
28
|
+
from saber.preprocessing import PreprocessingConfig
|
|
29
|
+
from saber.tuning import OptimizationResult, TuningConfig, tune
|
|
30
|
+
from saber.validation import ValidationResult, validate
|
|
31
|
+
|
|
32
|
+
__all__ = [ # noqa: RUF022 # grouped by purpose, not alphabetical
|
|
33
|
+
# Workflows
|
|
34
|
+
"train",
|
|
35
|
+
"validate",
|
|
36
|
+
"tune",
|
|
37
|
+
"benchmark",
|
|
38
|
+
"evaluate",
|
|
39
|
+
"predict",
|
|
40
|
+
"run_config",
|
|
41
|
+
# Persistence
|
|
42
|
+
"save_model",
|
|
43
|
+
"load_model",
|
|
44
|
+
"save_benchmark",
|
|
45
|
+
"load_benchmark",
|
|
46
|
+
"inspect_artifact",
|
|
47
|
+
# Inputs
|
|
48
|
+
"DatasetBundle",
|
|
49
|
+
"PartitionPlan",
|
|
50
|
+
"BioSievePartitionConfig",
|
|
51
|
+
"PreprocessingConfig",
|
|
52
|
+
"TuningConfig",
|
|
53
|
+
"BenchmarkConfig",
|
|
54
|
+
"SearchSpace",
|
|
55
|
+
"Categorical",
|
|
56
|
+
"Integer",
|
|
57
|
+
"Float",
|
|
58
|
+
"LogFloat",
|
|
59
|
+
# Results
|
|
60
|
+
"TrainResult",
|
|
61
|
+
"PredictionResult",
|
|
62
|
+
"EvaluationResult",
|
|
63
|
+
"ValidationResult",
|
|
64
|
+
"OptimizationResult",
|
|
65
|
+
"BenchmarkResult",
|
|
66
|
+
"LoadedModelArtifact",
|
|
67
|
+
# Catalog / misc
|
|
68
|
+
"ALGORITHMS",
|
|
69
|
+
"SaberError",
|
|
70
|
+
"__version__",
|
|
71
|
+
]
|