saberlib 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- saber/__init__.py +71 -0
- saber/__main__.py +5 -0
- saber/_api/__init__.py +7 -0
- saber/_api/_common.py +71 -0
- saber/_api/evaluate.py +43 -0
- saber/_api/predict.py +50 -0
- saber/_api/train.py +68 -0
- saber/_version.py +3 -0
- saber/benchmark/__init__.py +13 -0
- saber/benchmark/config.py +47 -0
- saber/benchmark/engine.py +509 -0
- saber/benchmark/results.py +199 -0
- saber/classification/__init__.py +1 -0
- saber/classification/lightgbm.py +25 -0
- saber/classification/search_spaces.py +262 -0
- saber/classification/sklearn.py +223 -0
- saber/classification/xgboost.py +38 -0
- saber/cli/__init__.py +5 -0
- saber/cli/main.py +201 -0
- saber/cli/render.py +306 -0
- saber/config/__init__.py +14 -0
- saber/config/builders.py +171 -0
- saber/config/io.py +58 -0
- saber/config/runner.py +314 -0
- saber/config/schema.py +284 -0
- saber/core/__init__.py +22 -0
- saber/core/capabilities.py +76 -0
- saber/core/metrics.py +234 -0
- saber/core/prediction.py +254 -0
- saber/core/registry.py +61 -0
- saber/core/results.py +35 -0
- saber/core/search_space.py +237 -0
- saber/core/specs.py +138 -0
- saber/core/task.py +7 -0
- saber/datasets/__init__.py +16 -0
- saber/datasets/_fingerprint.py +186 -0
- saber/datasets/biosieve.py +343 -0
- saber/datasets/folds.py +342 -0
- saber/datasets/loaders.py +188 -0
- saber/datasets/schemas.py +279 -0
- saber/datasets/validation.py +206 -0
- saber/evaluation/__init__.py +6 -0
- saber/evaluation/classification.py +390 -0
- saber/evaluation/evaluator.py +61 -0
- saber/evaluation/regression.py +47 -0
- saber/evaluation/results.py +30 -0
- saber/exceptions.py +170 -0
- saber/persistence/__init__.py +18 -0
- saber/persistence/artifacts.py +117 -0
- saber/persistence/checksums.py +75 -0
- saber/persistence/environment.py +97 -0
- saber/persistence/load.py +140 -0
- saber/persistence/metadata.py +60 -0
- saber/persistence/save.py +276 -0
- saber/preprocessing/__init__.py +5 -0
- saber/preprocessing/imputation.py +30 -0
- saber/preprocessing/pipeline.py +130 -0
- saber/preprocessing/scaling.py +57 -0
- saber/preprocessing/validation.py +65 -0
- saber/py.typed +0 -0
- saber/regression/__init__.py +1 -0
- saber/regression/lightgbm.py +25 -0
- saber/regression/search_spaces.py +252 -0
- saber/regression/sklearn.py +257 -0
- saber/regression/xgboost.py +38 -0
- saber/tuning/__init__.py +10 -0
- saber/tuning/engine.py +601 -0
- saber/tuning/results.py +93 -0
- saber/utils/__init__.py +1 -0
- saber/utils/serialization.py +65 -0
- saber/utils/tabular.py +133 -0
- saber/validation/__init__.py +10 -0
- saber/validation/cross_validation.py +243 -0
- saber/validation/partitioning.py +119 -0
- saber/validation/results.py +134 -0
- saberlib-0.1.0.dist-info/METADATA +209 -0
- saberlib-0.1.0.dist-info/RECORD +80 -0
- saberlib-0.1.0.dist-info/WHEEL +4 -0
- saberlib-0.1.0.dist-info/entry_points.txt +2 -0
- saberlib-0.1.0.dist-info/licenses/LICENSE +21 -0
saber/__init__.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""saber — classical supervised machine learning infrastructure."""
|
|
2
|
+
|
|
3
|
+
from saber._api import evaluate, predict, train
|
|
4
|
+
from saber._version import __version__
|
|
5
|
+
from saber.benchmark import BenchmarkConfig, BenchmarkResult, benchmark
|
|
6
|
+
from saber.config import run_config
|
|
7
|
+
from saber.core import (
|
|
8
|
+
ALGORITHMS,
|
|
9
|
+
Categorical,
|
|
10
|
+
Float,
|
|
11
|
+
Integer,
|
|
12
|
+
LogFloat,
|
|
13
|
+
PredictionResult,
|
|
14
|
+
SearchSpace,
|
|
15
|
+
TrainResult,
|
|
16
|
+
)
|
|
17
|
+
from saber.datasets import BioSievePartitionConfig, DatasetBundle, PartitionPlan
|
|
18
|
+
from saber.evaluation import EvaluationResult
|
|
19
|
+
from saber.exceptions import SaberError
|
|
20
|
+
from saber.persistence import (
|
|
21
|
+
LoadedModelArtifact,
|
|
22
|
+
inspect_artifact,
|
|
23
|
+
load_benchmark,
|
|
24
|
+
load_model,
|
|
25
|
+
save_benchmark,
|
|
26
|
+
save_model,
|
|
27
|
+
)
|
|
28
|
+
from saber.preprocessing import PreprocessingConfig
|
|
29
|
+
from saber.tuning import OptimizationResult, TuningConfig, tune
|
|
30
|
+
from saber.validation import ValidationResult, validate
|
|
31
|
+
|
|
32
|
+
__all__ = [ # noqa: RUF022 # grouped by purpose, not alphabetical
|
|
33
|
+
# Workflows
|
|
34
|
+
"train",
|
|
35
|
+
"validate",
|
|
36
|
+
"tune",
|
|
37
|
+
"benchmark",
|
|
38
|
+
"evaluate",
|
|
39
|
+
"predict",
|
|
40
|
+
"run_config",
|
|
41
|
+
# Persistence
|
|
42
|
+
"save_model",
|
|
43
|
+
"load_model",
|
|
44
|
+
"save_benchmark",
|
|
45
|
+
"load_benchmark",
|
|
46
|
+
"inspect_artifact",
|
|
47
|
+
# Inputs
|
|
48
|
+
"DatasetBundle",
|
|
49
|
+
"PartitionPlan",
|
|
50
|
+
"BioSievePartitionConfig",
|
|
51
|
+
"PreprocessingConfig",
|
|
52
|
+
"TuningConfig",
|
|
53
|
+
"BenchmarkConfig",
|
|
54
|
+
"SearchSpace",
|
|
55
|
+
"Categorical",
|
|
56
|
+
"Integer",
|
|
57
|
+
"Float",
|
|
58
|
+
"LogFloat",
|
|
59
|
+
# Results
|
|
60
|
+
"TrainResult",
|
|
61
|
+
"PredictionResult",
|
|
62
|
+
"EvaluationResult",
|
|
63
|
+
"ValidationResult",
|
|
64
|
+
"OptimizationResult",
|
|
65
|
+
"BenchmarkResult",
|
|
66
|
+
"LoadedModelArtifact",
|
|
67
|
+
# Catalog / misc
|
|
68
|
+
"ALGORITHMS",
|
|
69
|
+
"SaberError",
|
|
70
|
+
"__version__",
|
|
71
|
+
]
|
saber/__main__.py
ADDED
saber/_api/__init__.py
ADDED
saber/_api/_common.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Shared helpers for the public high-level API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
from saber.core.prediction import PredictionResult, collect_model_outputs
|
|
11
|
+
from saber.core.results import TrainResult
|
|
12
|
+
from saber.exceptions import ValidationContractError
|
|
13
|
+
from saber.persistence import LoadedModelArtifact, load_model
|
|
14
|
+
from saber.preprocessing.pipeline import pipeline_input
|
|
15
|
+
|
|
16
|
+
if TYPE_CHECKING:
|
|
17
|
+
from collections.abc import Sequence
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load_active_model(model: Any) -> TrainResult | LoadedModelArtifact:
|
|
21
|
+
"""Resolve an artifact path, TrainResult, or LoadedModelArtifact to a fitted model."""
|
|
22
|
+
if isinstance(model, (str, Path)):
|
|
23
|
+
return load_model(model)
|
|
24
|
+
if isinstance(model, (TrainResult, LoadedModelArtifact)):
|
|
25
|
+
return model
|
|
26
|
+
raise ValidationContractError("model must be an artifact path, LoadedModelArtifact, or TrainResult.")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def predict_model(
|
|
30
|
+
model: TrainResult | LoadedModelArtifact,
|
|
31
|
+
X: Any,
|
|
32
|
+
*,
|
|
33
|
+
feature_names: Sequence[str] | None = None,
|
|
34
|
+
sample_ids: Sequence[Any] | None = None,
|
|
35
|
+
positive_class: Any | None = None,
|
|
36
|
+
) -> PredictionResult:
|
|
37
|
+
"""Generate a structured prediction from a resolved fitted model."""
|
|
38
|
+
if X is None:
|
|
39
|
+
raise ValidationContractError("Prediction requires a feature matrix or DatasetBundle.")
|
|
40
|
+
if isinstance(model, LoadedModelArtifact):
|
|
41
|
+
return model._predict_result( # noqa: SLF001 # the artifact's own inference path
|
|
42
|
+
X,
|
|
43
|
+
feature_names=feature_names,
|
|
44
|
+
sample_ids=sample_ids,
|
|
45
|
+
positive_class=positive_class,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
fitted = model.model
|
|
49
|
+
if fitted is None:
|
|
50
|
+
raise ValidationContractError("TrainResult does not contain a fitted model.")
|
|
51
|
+
|
|
52
|
+
if model.feature_schema is not None:
|
|
53
|
+
# Validate against the training schema before the NumPy conversion.
|
|
54
|
+
model.feature_schema.validate_compatible(X, feature_names=feature_names)
|
|
55
|
+
X = pipeline_input(fitted, X)
|
|
56
|
+
capabilities = model.spec.resolved_capabilities
|
|
57
|
+
outputs = collect_model_outputs(
|
|
58
|
+
fitted,
|
|
59
|
+
X,
|
|
60
|
+
task=model.spec.task,
|
|
61
|
+
use_proba=capabilities.predict_proba,
|
|
62
|
+
use_decision=capabilities.decision_function,
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
return PredictionResult(
|
|
66
|
+
task=model.spec.task,
|
|
67
|
+
**outputs,
|
|
68
|
+
positive_class=model.positive_class if positive_class is None else positive_class,
|
|
69
|
+
sample_ids=None if sample_ids is None else np.asarray(sample_ids),
|
|
70
|
+
metadata={"algorithm": model.spec.name},
|
|
71
|
+
)
|
saber/_api/evaluate.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""High-level direct evaluation API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
from saber._api._common import load_active_model, predict_model
|
|
8
|
+
from saber.core.results import TrainResult
|
|
9
|
+
from saber.evaluation import EvaluationResult, evaluate_prediction
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from collections.abc import Sequence
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from saber.datasets import DatasetBundle
|
|
16
|
+
from saber.persistence import LoadedModelArtifact
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def evaluate(
|
|
20
|
+
model: str | Path | TrainResult | LoadedModelArtifact,
|
|
21
|
+
dataset: DatasetBundle,
|
|
22
|
+
*,
|
|
23
|
+
metrics: Sequence[str] | None = None,
|
|
24
|
+
positive_class: Any | None = None,
|
|
25
|
+
) -> EvaluationResult:
|
|
26
|
+
"""Evaluate one already-fitted model on a labeled prepared dataset."""
|
|
27
|
+
active = load_active_model(model)
|
|
28
|
+
dataset.validate(task=active.spec.task if isinstance(active, TrainResult) else active.task)
|
|
29
|
+
prediction = predict_model(
|
|
30
|
+
active,
|
|
31
|
+
dataset.X,
|
|
32
|
+
feature_names=dataset.feature_names,
|
|
33
|
+
sample_ids=dataset.sample_ids,
|
|
34
|
+
positive_class=positive_class,
|
|
35
|
+
)
|
|
36
|
+
result = evaluate_prediction(dataset.y, prediction, metrics=metrics)
|
|
37
|
+
result.metadata.update(
|
|
38
|
+
algorithm=prediction.metadata.get("algorithm"), dataset_fingerprint=dataset.fingerprint
|
|
39
|
+
)
|
|
40
|
+
return result
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
__all__ = ["evaluate"]
|
saber/_api/predict.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""High-level structured prediction API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
from saber._api._common import load_active_model, predict_model
|
|
8
|
+
from saber.datasets import DatasetBundle
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from collections.abc import Sequence
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from saber.core.prediction import PredictionResult
|
|
15
|
+
from saber.core.results import TrainResult
|
|
16
|
+
from saber.persistence import LoadedModelArtifact
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def predict(
|
|
20
|
+
model: str | Path | TrainResult | LoadedModelArtifact,
|
|
21
|
+
X: Any,
|
|
22
|
+
*,
|
|
23
|
+
sample_ids: Sequence[Any] | None = None,
|
|
24
|
+
positive_class: Any | None = None,
|
|
25
|
+
) -> PredictionResult:
|
|
26
|
+
"""Generate a structured PredictionResult from a fitted or persisted model.
|
|
27
|
+
|
|
28
|
+
``X`` is a :class:`DatasetBundle`, a NumPy array, or a Polars/pandas
|
|
29
|
+
DataFrame. DataFrames and datasets are checked against the training
|
|
30
|
+
feature names/order; a NumPy array only against the feature count. A path
|
|
31
|
+
is loaded with :func:`saber.load_model` defaults; call ``load_model``
|
|
32
|
+
yourself for ``strict_environment`` or ``verify`` control.
|
|
33
|
+
"""
|
|
34
|
+
active = load_active_model(model)
|
|
35
|
+
feature_names = None
|
|
36
|
+
if isinstance(X, DatasetBundle):
|
|
37
|
+
feature_names = X.feature_names
|
|
38
|
+
if sample_ids is None:
|
|
39
|
+
sample_ids = X.sample_ids
|
|
40
|
+
X = X.X
|
|
41
|
+
return predict_model(
|
|
42
|
+
active,
|
|
43
|
+
X,
|
|
44
|
+
feature_names=feature_names,
|
|
45
|
+
sample_ids=sample_ids,
|
|
46
|
+
positive_class=positive_class,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
__all__ = ["predict"]
|
saber/_api/train.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""High-level final-model training API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
from saber.core.registry import get_algorithm
|
|
8
|
+
from saber.core.results import TrainResult
|
|
9
|
+
from saber.preprocessing.pipeline import build_model_pipeline, pipeline_input, preprocessing_summary
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from collections.abc import Mapping
|
|
13
|
+
|
|
14
|
+
from saber.datasets import DatasetBundle
|
|
15
|
+
from saber.preprocessing import PreprocessingConfig
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def train(
|
|
19
|
+
*,
|
|
20
|
+
dataset: DatasetBundle,
|
|
21
|
+
algorithm: str,
|
|
22
|
+
model_params: Mapping[str, Any] | None = None,
|
|
23
|
+
preprocessing: PreprocessingConfig | None = None,
|
|
24
|
+
random_state: int | None = None,
|
|
25
|
+
positive_class: Any | None = None,
|
|
26
|
+
) -> TrainResult:
|
|
27
|
+
"""Fit a final supervised model on all supplied samples.
|
|
28
|
+
|
|
29
|
+
This is a final-fit operation. Model assessment belongs to ``validate`` or
|
|
30
|
+
``benchmark``; hyperparameter selection belongs to ``tune``. Persist the
|
|
31
|
+
result with :func:`saber.save_model`.
|
|
32
|
+
"""
|
|
33
|
+
spec = get_algorithm(algorithm)
|
|
34
|
+
dataset.validate(task=spec.task)
|
|
35
|
+
|
|
36
|
+
estimator = spec.build_estimator(random_state=random_state, **(model_params or {}))
|
|
37
|
+
pipeline = build_model_pipeline(
|
|
38
|
+
spec=spec,
|
|
39
|
+
estimator=estimator,
|
|
40
|
+
training_data=dataset,
|
|
41
|
+
preprocessing=preprocessing,
|
|
42
|
+
)
|
|
43
|
+
pipeline.fit(
|
|
44
|
+
pipeline_input(pipeline, dataset.X),
|
|
45
|
+
dataset.y,
|
|
46
|
+
**spec.sample_weight_fit_params(dataset.sample_weight),
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
return TrainResult(
|
|
50
|
+
model=pipeline,
|
|
51
|
+
spec=spec,
|
|
52
|
+
parameters=dict(pipeline.named_steps["estimator"].get_params(deep=False)),
|
|
53
|
+
feature_schema=dataset.feature_schema,
|
|
54
|
+
positive_class=positive_class,
|
|
55
|
+
metadata={
|
|
56
|
+
"algorithm": spec.name,
|
|
57
|
+
"provider": spec.provider,
|
|
58
|
+
"task": spec.task,
|
|
59
|
+
"dataset_fingerprint": dataset.fingerprint,
|
|
60
|
+
"n_samples": dataset.n_samples,
|
|
61
|
+
"n_features": dataset.n_features,
|
|
62
|
+
"random_state": random_state,
|
|
63
|
+
"preprocessing": preprocessing_summary(preprocessing),
|
|
64
|
+
},
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
__all__ = ["train"]
|
saber/_version.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Systematic supervised model benchmarking."""
|
|
2
|
+
|
|
3
|
+
from saber.benchmark.config import BenchmarkConfig, BenchmarkMode
|
|
4
|
+
from saber.benchmark.engine import benchmark
|
|
5
|
+
from saber.benchmark.results import BenchmarkResult, BenchmarkRun
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"BenchmarkConfig",
|
|
9
|
+
"BenchmarkMode",
|
|
10
|
+
"BenchmarkResult",
|
|
11
|
+
"BenchmarkRun",
|
|
12
|
+
"benchmark",
|
|
13
|
+
]
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Benchmark execution configuration."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from typing import TYPE_CHECKING, Literal, get_args
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from saber.tuning import TuningConfig
|
|
10
|
+
|
|
11
|
+
BenchmarkMode = Literal["untuned", "tuned"]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True, slots=True)
|
|
15
|
+
class BenchmarkConfig:
|
|
16
|
+
"""Benchmark-specific controls of the algorithm x representation x partition matrix.
|
|
17
|
+
|
|
18
|
+
Workflow knobs shared with ``validate``/``tune`` (``metrics``,
|
|
19
|
+
``evaluation_role``, ...) are keyword arguments of :func:`benchmark`.
|
|
20
|
+
Each seed in ``seeds`` is the ``random_state`` of one set of runs.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
seeds: tuple[int | None, ...] = (42,)
|
|
24
|
+
modes: tuple[BenchmarkMode, ...] = ("untuned",)
|
|
25
|
+
include_baselines: bool = True
|
|
26
|
+
fail_fast: bool = False
|
|
27
|
+
tuning: TuningConfig | None = None
|
|
28
|
+
metadata: dict[str, object] = field(default_factory=dict, compare=False)
|
|
29
|
+
|
|
30
|
+
def __post_init__(self) -> None:
|
|
31
|
+
"""Deduplicate seeds/modes and validate the tuned-mode contract."""
|
|
32
|
+
seeds = tuple(dict.fromkeys(self.seeds))
|
|
33
|
+
modes = tuple(dict.fromkeys(self.modes))
|
|
34
|
+
|
|
35
|
+
if not seeds:
|
|
36
|
+
raise ValueError("BenchmarkConfig.seeds must contain at least one seed.")
|
|
37
|
+
if not modes:
|
|
38
|
+
raise ValueError("BenchmarkConfig.modes must contain at least one mode.")
|
|
39
|
+
invalid = set(modes) - set(get_args(BenchmarkMode))
|
|
40
|
+
if invalid:
|
|
41
|
+
raise ValueError(f"Unsupported benchmark modes: {sorted(invalid)!r}.")
|
|
42
|
+
if "tuned" in modes and self.tuning is None:
|
|
43
|
+
raise ValueError("Tuned benchmarks require an explicit TuningConfig.")
|
|
44
|
+
|
|
45
|
+
object.__setattr__(self, "seeds", seeds)
|
|
46
|
+
object.__setattr__(self, "modes", modes)
|
|
47
|
+
object.__setattr__(self, "metadata", dict(self.metadata))
|