saberlib 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. saber/__init__.py +71 -0
  2. saber/__main__.py +5 -0
  3. saber/_api/__init__.py +7 -0
  4. saber/_api/_common.py +71 -0
  5. saber/_api/evaluate.py +43 -0
  6. saber/_api/predict.py +50 -0
  7. saber/_api/train.py +68 -0
  8. saber/_version.py +3 -0
  9. saber/benchmark/__init__.py +13 -0
  10. saber/benchmark/config.py +47 -0
  11. saber/benchmark/engine.py +509 -0
  12. saber/benchmark/results.py +199 -0
  13. saber/classification/__init__.py +1 -0
  14. saber/classification/lightgbm.py +25 -0
  15. saber/classification/search_spaces.py +262 -0
  16. saber/classification/sklearn.py +223 -0
  17. saber/classification/xgboost.py +38 -0
  18. saber/cli/__init__.py +5 -0
  19. saber/cli/main.py +201 -0
  20. saber/cli/render.py +306 -0
  21. saber/config/__init__.py +14 -0
  22. saber/config/builders.py +171 -0
  23. saber/config/io.py +58 -0
  24. saber/config/runner.py +314 -0
  25. saber/config/schema.py +284 -0
  26. saber/core/__init__.py +22 -0
  27. saber/core/capabilities.py +76 -0
  28. saber/core/metrics.py +234 -0
  29. saber/core/prediction.py +254 -0
  30. saber/core/registry.py +61 -0
  31. saber/core/results.py +35 -0
  32. saber/core/search_space.py +237 -0
  33. saber/core/specs.py +138 -0
  34. saber/core/task.py +7 -0
  35. saber/datasets/__init__.py +16 -0
  36. saber/datasets/_fingerprint.py +186 -0
  37. saber/datasets/biosieve.py +343 -0
  38. saber/datasets/folds.py +342 -0
  39. saber/datasets/loaders.py +188 -0
  40. saber/datasets/schemas.py +279 -0
  41. saber/datasets/validation.py +206 -0
  42. saber/evaluation/__init__.py +6 -0
  43. saber/evaluation/classification.py +390 -0
  44. saber/evaluation/evaluator.py +61 -0
  45. saber/evaluation/regression.py +47 -0
  46. saber/evaluation/results.py +30 -0
  47. saber/exceptions.py +170 -0
  48. saber/persistence/__init__.py +18 -0
  49. saber/persistence/artifacts.py +117 -0
  50. saber/persistence/checksums.py +75 -0
  51. saber/persistence/environment.py +97 -0
  52. saber/persistence/load.py +140 -0
  53. saber/persistence/metadata.py +60 -0
  54. saber/persistence/save.py +276 -0
  55. saber/preprocessing/__init__.py +5 -0
  56. saber/preprocessing/imputation.py +30 -0
  57. saber/preprocessing/pipeline.py +130 -0
  58. saber/preprocessing/scaling.py +57 -0
  59. saber/preprocessing/validation.py +65 -0
  60. saber/py.typed +0 -0
  61. saber/regression/__init__.py +1 -0
  62. saber/regression/lightgbm.py +25 -0
  63. saber/regression/search_spaces.py +252 -0
  64. saber/regression/sklearn.py +257 -0
  65. saber/regression/xgboost.py +38 -0
  66. saber/tuning/__init__.py +10 -0
  67. saber/tuning/engine.py +601 -0
  68. saber/tuning/results.py +93 -0
  69. saber/utils/__init__.py +1 -0
  70. saber/utils/serialization.py +65 -0
  71. saber/utils/tabular.py +133 -0
  72. saber/validation/__init__.py +10 -0
  73. saber/validation/cross_validation.py +243 -0
  74. saber/validation/partitioning.py +119 -0
  75. saber/validation/results.py +134 -0
  76. saberlib-0.1.0.dist-info/METADATA +209 -0
  77. saberlib-0.1.0.dist-info/RECORD +80 -0
  78. saberlib-0.1.0.dist-info/WHEEL +4 -0
  79. saberlib-0.1.0.dist-info/entry_points.txt +2 -0
  80. saberlib-0.1.0.dist-info/licenses/LICENSE +21 -0
saber/__init__.py ADDED
@@ -0,0 +1,71 @@
1
+ """saber — classical supervised machine learning infrastructure."""
2
+
3
+ from saber._api import evaluate, predict, train
4
+ from saber._version import __version__
5
+ from saber.benchmark import BenchmarkConfig, BenchmarkResult, benchmark
6
+ from saber.config import run_config
7
+ from saber.core import (
8
+ ALGORITHMS,
9
+ Categorical,
10
+ Float,
11
+ Integer,
12
+ LogFloat,
13
+ PredictionResult,
14
+ SearchSpace,
15
+ TrainResult,
16
+ )
17
+ from saber.datasets import BioSievePartitionConfig, DatasetBundle, PartitionPlan
18
+ from saber.evaluation import EvaluationResult
19
+ from saber.exceptions import SaberError
20
+ from saber.persistence import (
21
+ LoadedModelArtifact,
22
+ inspect_artifact,
23
+ load_benchmark,
24
+ load_model,
25
+ save_benchmark,
26
+ save_model,
27
+ )
28
+ from saber.preprocessing import PreprocessingConfig
29
+ from saber.tuning import OptimizationResult, TuningConfig, tune
30
+ from saber.validation import ValidationResult, validate
31
+
32
+ __all__ = [ # noqa: RUF022 # grouped by purpose, not alphabetical
33
+ # Workflows
34
+ "train",
35
+ "validate",
36
+ "tune",
37
+ "benchmark",
38
+ "evaluate",
39
+ "predict",
40
+ "run_config",
41
+ # Persistence
42
+ "save_model",
43
+ "load_model",
44
+ "save_benchmark",
45
+ "load_benchmark",
46
+ "inspect_artifact",
47
+ # Inputs
48
+ "DatasetBundle",
49
+ "PartitionPlan",
50
+ "BioSievePartitionConfig",
51
+ "PreprocessingConfig",
52
+ "TuningConfig",
53
+ "BenchmarkConfig",
54
+ "SearchSpace",
55
+ "Categorical",
56
+ "Integer",
57
+ "Float",
58
+ "LogFloat",
59
+ # Results
60
+ "TrainResult",
61
+ "PredictionResult",
62
+ "EvaluationResult",
63
+ "ValidationResult",
64
+ "OptimizationResult",
65
+ "BenchmarkResult",
66
+ "LoadedModelArtifact",
67
+ # Catalog / misc
68
+ "ALGORITHMS",
69
+ "SaberError",
70
+ "__version__",
71
+ ]
saber/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Entry point for running saber's CLI as a module."""
2
+
3
+ from saber.cli.main import main
4
+
5
+ raise SystemExit(main())
saber/_api/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Private implementation of the train/predict/evaluate workflows."""
2
+
3
+ from saber._api.evaluate import evaluate
4
+ from saber._api.predict import predict
5
+ from saber._api.train import train
6
+
7
+ __all__ = ["evaluate", "predict", "train"]
saber/_api/_common.py ADDED
@@ -0,0 +1,71 @@
1
+ """Shared helpers for the public high-level API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import TYPE_CHECKING, Any
7
+
8
+ import numpy as np
9
+
10
+ from saber.core.prediction import PredictionResult, collect_model_outputs
11
+ from saber.core.results import TrainResult
12
+ from saber.exceptions import ValidationContractError
13
+ from saber.persistence import LoadedModelArtifact, load_model
14
+ from saber.preprocessing.pipeline import pipeline_input
15
+
16
+ if TYPE_CHECKING:
17
+ from collections.abc import Sequence
18
+
19
+
20
+ def load_active_model(model: Any) -> TrainResult | LoadedModelArtifact:
21
+ """Resolve an artifact path, TrainResult, or LoadedModelArtifact to a fitted model."""
22
+ if isinstance(model, (str, Path)):
23
+ return load_model(model)
24
+ if isinstance(model, (TrainResult, LoadedModelArtifact)):
25
+ return model
26
+ raise ValidationContractError("model must be an artifact path, LoadedModelArtifact, or TrainResult.")
27
+
28
+
29
+ def predict_model(
30
+ model: TrainResult | LoadedModelArtifact,
31
+ X: Any,
32
+ *,
33
+ feature_names: Sequence[str] | None = None,
34
+ sample_ids: Sequence[Any] | None = None,
35
+ positive_class: Any | None = None,
36
+ ) -> PredictionResult:
37
+ """Generate a structured prediction from a resolved fitted model."""
38
+ if X is None:
39
+ raise ValidationContractError("Prediction requires a feature matrix or DatasetBundle.")
40
+ if isinstance(model, LoadedModelArtifact):
41
+ return model._predict_result( # noqa: SLF001 # the artifact's own inference path
42
+ X,
43
+ feature_names=feature_names,
44
+ sample_ids=sample_ids,
45
+ positive_class=positive_class,
46
+ )
47
+
48
+ fitted = model.model
49
+ if fitted is None:
50
+ raise ValidationContractError("TrainResult does not contain a fitted model.")
51
+
52
+ if model.feature_schema is not None:
53
+ # Validate against the training schema before the NumPy conversion.
54
+ model.feature_schema.validate_compatible(X, feature_names=feature_names)
55
+ X = pipeline_input(fitted, X)
56
+ capabilities = model.spec.resolved_capabilities
57
+ outputs = collect_model_outputs(
58
+ fitted,
59
+ X,
60
+ task=model.spec.task,
61
+ use_proba=capabilities.predict_proba,
62
+ use_decision=capabilities.decision_function,
63
+ )
64
+
65
+ return PredictionResult(
66
+ task=model.spec.task,
67
+ **outputs,
68
+ positive_class=model.positive_class if positive_class is None else positive_class,
69
+ sample_ids=None if sample_ids is None else np.asarray(sample_ids),
70
+ metadata={"algorithm": model.spec.name},
71
+ )
saber/_api/evaluate.py ADDED
@@ -0,0 +1,43 @@
1
+ """High-level direct evaluation API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from saber._api._common import load_active_model, predict_model
8
+ from saber.core.results import TrainResult
9
+ from saber.evaluation import EvaluationResult, evaluate_prediction
10
+
11
+ if TYPE_CHECKING:
12
+ from collections.abc import Sequence
13
+ from pathlib import Path
14
+
15
+ from saber.datasets import DatasetBundle
16
+ from saber.persistence import LoadedModelArtifact
17
+
18
+
19
+ def evaluate(
20
+ model: str | Path | TrainResult | LoadedModelArtifact,
21
+ dataset: DatasetBundle,
22
+ *,
23
+ metrics: Sequence[str] | None = None,
24
+ positive_class: Any | None = None,
25
+ ) -> EvaluationResult:
26
+ """Evaluate one already-fitted model on a labeled prepared dataset."""
27
+ active = load_active_model(model)
28
+ dataset.validate(task=active.spec.task if isinstance(active, TrainResult) else active.task)
29
+ prediction = predict_model(
30
+ active,
31
+ dataset.X,
32
+ feature_names=dataset.feature_names,
33
+ sample_ids=dataset.sample_ids,
34
+ positive_class=positive_class,
35
+ )
36
+ result = evaluate_prediction(dataset.y, prediction, metrics=metrics)
37
+ result.metadata.update(
38
+ algorithm=prediction.metadata.get("algorithm"), dataset_fingerprint=dataset.fingerprint
39
+ )
40
+ return result
41
+
42
+
43
+ __all__ = ["evaluate"]
saber/_api/predict.py ADDED
@@ -0,0 +1,50 @@
1
+ """High-level structured prediction API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from saber._api._common import load_active_model, predict_model
8
+ from saber.datasets import DatasetBundle
9
+
10
+ if TYPE_CHECKING:
11
+ from collections.abc import Sequence
12
+ from pathlib import Path
13
+
14
+ from saber.core.prediction import PredictionResult
15
+ from saber.core.results import TrainResult
16
+ from saber.persistence import LoadedModelArtifact
17
+
18
+
19
+ def predict(
20
+ model: str | Path | TrainResult | LoadedModelArtifact,
21
+ X: Any,
22
+ *,
23
+ sample_ids: Sequence[Any] | None = None,
24
+ positive_class: Any | None = None,
25
+ ) -> PredictionResult:
26
+ """Generate a structured PredictionResult from a fitted or persisted model.
27
+
28
+ ``X`` is a :class:`DatasetBundle`, a NumPy array, or a Polars/pandas
29
+ DataFrame. DataFrames and datasets are checked against the training
30
+ feature names/order; a NumPy array only against the feature count. A path
31
+ is loaded with :func:`saber.load_model` defaults; call ``load_model``
32
+ yourself for ``strict_environment`` or ``verify`` control.
33
+ """
34
+ active = load_active_model(model)
35
+ feature_names = None
36
+ if isinstance(X, DatasetBundle):
37
+ feature_names = X.feature_names
38
+ if sample_ids is None:
39
+ sample_ids = X.sample_ids
40
+ X = X.X
41
+ return predict_model(
42
+ active,
43
+ X,
44
+ feature_names=feature_names,
45
+ sample_ids=sample_ids,
46
+ positive_class=positive_class,
47
+ )
48
+
49
+
50
+ __all__ = ["predict"]
saber/_api/train.py ADDED
@@ -0,0 +1,68 @@
1
+ """High-level final-model training API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from saber.core.registry import get_algorithm
8
+ from saber.core.results import TrainResult
9
+ from saber.preprocessing.pipeline import build_model_pipeline, pipeline_input, preprocessing_summary
10
+
11
+ if TYPE_CHECKING:
12
+ from collections.abc import Mapping
13
+
14
+ from saber.datasets import DatasetBundle
15
+ from saber.preprocessing import PreprocessingConfig
16
+
17
+
18
+ def train(
19
+ *,
20
+ dataset: DatasetBundle,
21
+ algorithm: str,
22
+ model_params: Mapping[str, Any] | None = None,
23
+ preprocessing: PreprocessingConfig | None = None,
24
+ random_state: int | None = None,
25
+ positive_class: Any | None = None,
26
+ ) -> TrainResult:
27
+ """Fit a final supervised model on all supplied samples.
28
+
29
+ This is a final-fit operation. Model assessment belongs to ``validate`` or
30
+ ``benchmark``; hyperparameter selection belongs to ``tune``. Persist the
31
+ result with :func:`saber.save_model`.
32
+ """
33
+ spec = get_algorithm(algorithm)
34
+ dataset.validate(task=spec.task)
35
+
36
+ estimator = spec.build_estimator(random_state=random_state, **(model_params or {}))
37
+ pipeline = build_model_pipeline(
38
+ spec=spec,
39
+ estimator=estimator,
40
+ training_data=dataset,
41
+ preprocessing=preprocessing,
42
+ )
43
+ pipeline.fit(
44
+ pipeline_input(pipeline, dataset.X),
45
+ dataset.y,
46
+ **spec.sample_weight_fit_params(dataset.sample_weight),
47
+ )
48
+
49
+ return TrainResult(
50
+ model=pipeline,
51
+ spec=spec,
52
+ parameters=dict(pipeline.named_steps["estimator"].get_params(deep=False)),
53
+ feature_schema=dataset.feature_schema,
54
+ positive_class=positive_class,
55
+ metadata={
56
+ "algorithm": spec.name,
57
+ "provider": spec.provider,
58
+ "task": spec.task,
59
+ "dataset_fingerprint": dataset.fingerprint,
60
+ "n_samples": dataset.n_samples,
61
+ "n_features": dataset.n_features,
62
+ "random_state": random_state,
63
+ "preprocessing": preprocessing_summary(preprocessing),
64
+ },
65
+ )
66
+
67
+
68
+ __all__ = ["train"]
saber/_version.py ADDED
@@ -0,0 +1,3 @@
1
+ """Package version."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,13 @@
1
+ """Systematic supervised model benchmarking."""
2
+
3
+ from saber.benchmark.config import BenchmarkConfig, BenchmarkMode
4
+ from saber.benchmark.engine import benchmark
5
+ from saber.benchmark.results import BenchmarkResult, BenchmarkRun
6
+
7
+ __all__ = [
8
+ "BenchmarkConfig",
9
+ "BenchmarkMode",
10
+ "BenchmarkResult",
11
+ "BenchmarkRun",
12
+ "benchmark",
13
+ ]
@@ -0,0 +1,47 @@
1
+ """Benchmark execution configuration."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from typing import TYPE_CHECKING, Literal, get_args
7
+
8
+ if TYPE_CHECKING:
9
+ from saber.tuning import TuningConfig
10
+
11
+ BenchmarkMode = Literal["untuned", "tuned"]
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class BenchmarkConfig:
16
+ """Benchmark-specific controls of the algorithm x representation x partition matrix.
17
+
18
+ Workflow knobs shared with ``validate``/``tune`` (``metrics``,
19
+ ``evaluation_role``, ...) are keyword arguments of :func:`benchmark`.
20
+ Each seed in ``seeds`` is the ``random_state`` of one set of runs.
21
+ """
22
+
23
+ seeds: tuple[int | None, ...] = (42,)
24
+ modes: tuple[BenchmarkMode, ...] = ("untuned",)
25
+ include_baselines: bool = True
26
+ fail_fast: bool = False
27
+ tuning: TuningConfig | None = None
28
+ metadata: dict[str, object] = field(default_factory=dict, compare=False)
29
+
30
+ def __post_init__(self) -> None:
31
+ """Deduplicate seeds/modes and validate the tuned-mode contract."""
32
+ seeds = tuple(dict.fromkeys(self.seeds))
33
+ modes = tuple(dict.fromkeys(self.modes))
34
+
35
+ if not seeds:
36
+ raise ValueError("BenchmarkConfig.seeds must contain at least one seed.")
37
+ if not modes:
38
+ raise ValueError("BenchmarkConfig.modes must contain at least one mode.")
39
+ invalid = set(modes) - set(get_args(BenchmarkMode))
40
+ if invalid:
41
+ raise ValueError(f"Unsupported benchmark modes: {sorted(invalid)!r}.")
42
+ if "tuned" in modes and self.tuning is None:
43
+ raise ValueError("Tuned benchmarks require an explicit TuningConfig.")
44
+
45
+ object.__setattr__(self, "seeds", seeds)
46
+ object.__setattr__(self, "modes", modes)
47
+ object.__setattr__(self, "metadata", dict(self.metadata))