leanhebo 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. leanhebo/__init__.py +27 -0
  2. leanhebo/acquisition/__init__.py +8 -0
  3. leanhebo/acquisition/mace.py +91 -0
  4. leanhebo/acquisition/posterior.py +119 -0
  5. leanhebo/checkpoint.py +73 -0
  6. leanhebo/config.py +263 -0
  7. leanhebo/data/__init__.py +29 -0
  8. leanhebo/data/adapters/__init__.py +11 -0
  9. leanhebo/data/adapters/numpy.py +30 -0
  10. leanhebo/data/adapters/pandas.py +15 -0
  11. leanhebo/data/adapters/polars.py +15 -0
  12. leanhebo/data/adapters/registry.py +125 -0
  13. leanhebo/data/batch.py +255 -0
  14. leanhebo/data/store.py +355 -0
  15. leanhebo/diagnostics.py +134 -0
  16. leanhebo/errors.py +23 -0
  17. leanhebo/gp/__init__.py +8 -0
  18. leanhebo/gp/exact.py +742 -0
  19. leanhebo/gp/kernel.py +130 -0
  20. leanhebo/gp/optimizer.py +127 -0
  21. leanhebo/gp/reports.py +7 -0
  22. leanhebo/optimizer.py +628 -0
  23. leanhebo/py.typed +1 -0
  24. leanhebo/runtime/__init__.py +8 -0
  25. leanhebo/runtime/process.py +44 -0
  26. leanhebo/runtime/rng.py +82 -0
  27. leanhebo/search/__init__.py +82 -0
  28. leanhebo/search/duplicates.py +228 -0
  29. leanhebo/search/nsga2.py +599 -0
  30. leanhebo/search/operators.py +451 -0
  31. leanhebo/search/repair.py +307 -0
  32. leanhebo/search/sorting.py +187 -0
  33. leanhebo/search/survival.py +127 -0
  34. leanhebo/space/__init__.py +22 -0
  35. leanhebo/space/compiled.py +732 -0
  36. leanhebo/space/keys.py +92 -0
  37. leanhebo/space/parameters.py +488 -0
  38. leanhebo/space/space.py +101 -0
  39. leanhebo/transforms/__init__.py +53 -0
  40. leanhebo/transforms/power.py +957 -0
  41. leanhebo/transforms/scalers.py +362 -0
  42. leanhebo-0.1.0.dist-info/METADATA +192 -0
  43. leanhebo-0.1.0.dist-info/RECORD +46 -0
  44. leanhebo-0.1.0.dist-info/WHEEL +4 -0
  45. leanhebo-0.1.0.dist-info/licenses/LICENSE +22 -0
  46. leanhebo-0.1.0.dist-info/licenses/NOTICE.md +14 -0
leanhebo/__init__.py ADDED
@@ -0,0 +1,27 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Portions derived from Huawei HEBO; see NOTICE.md.
3
+
4
+ """LeanHEBO public package."""
5
+
6
+ from leanhebo.config import (
7
+ AcquisitionConfig,
8
+ GPConfig,
9
+ LeanHEBOConfig,
10
+ RuntimeConfig,
11
+ SearchConfig,
12
+ WarpConfig,
13
+ )
14
+ from leanhebo.optimizer import LeanHEBO
15
+
16
+ __version__ = "0.1.0"
17
+
18
+ __all__ = [
19
+ "AcquisitionConfig",
20
+ "GPConfig",
21
+ "LeanHEBO",
22
+ "LeanHEBOConfig",
23
+ "RuntimeConfig",
24
+ "SearchConfig",
25
+ "WarpConfig",
26
+ "__version__",
27
+ ]
@@ -0,0 +1,8 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """Posterior and MACE evaluation."""
4
+
5
+ from leanhebo.acquisition.mace import MACEEvaluator
6
+ from leanhebo.acquisition.posterior import PosteriorEvaluator, PosteriorStats
7
+
8
+ __all__ = ["MACEEvaluator", "PosteriorEvaluator", "PosteriorStats"]
@@ -0,0 +1,91 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Portions derived from Huawei HEBO; see NOTICE.md.
3
+
4
+ """HEBO's three-objective MACE acquisition from shared posterior statistics."""
5
+
6
+ from __future__ import annotations
7
+
8
+ import math
9
+
10
+ import torch
11
+
12
+ from leanhebo.acquisition.posterior import PosteriorEvaluator, PosteriorStats
13
+ from leanhebo.errors import NumericalError
14
+
15
+
16
+ class MACEEvaluator:
17
+ """Compute stochastic LCB, negative log-EI, and negative log-PI."""
18
+
19
+ num_objectives = 3
20
+
21
+ def __init__(
22
+ self,
23
+ posterior: PosteriorEvaluator,
24
+ *,
25
+ best_y: torch.Tensor | float,
26
+ kappa: float,
27
+ epsilon: float = 1e-4,
28
+ stochastic: bool = True,
29
+ generator: torch.Generator | None = None,
30
+ ) -> None:
31
+ if kappa < 0:
32
+ raise ValueError("kappa cannot be negative")
33
+ if epsilon < 0:
34
+ raise ValueError("epsilon cannot be negative")
35
+ self.posterior = posterior
36
+ self.best_y = best_y
37
+ self.kappa = kappa
38
+ self.epsilon = epsilon
39
+ self.stochastic = stochastic
40
+ self.generator = generator
41
+
42
+ def evaluate(self, continuous: torch.Tensor, categorical: torch.Tensor) -> torch.Tensor:
43
+ return self.from_stats(self.posterior.evaluate(continuous, categorical))
44
+
45
+ def from_stats(self, stats: PosteriorStats) -> torch.Tensor:
46
+ mean = stats.mean
47
+ stddev = stats.stddev.clamp_min(torch.finfo(stats.stddev.dtype).eps)
48
+ tau = torch.as_tensor(self.best_y, device=mean.device, dtype=mean.dtype)
49
+ if self.stochastic:
50
+ noise_stddev = (2.0 * stats.noise_variance).sqrt()
51
+ lcb_noise = torch.randn(
52
+ mean.shape,
53
+ device=mean.device,
54
+ dtype=mean.dtype,
55
+ generator=self.generator,
56
+ )
57
+ improvement_noise = torch.randn(
58
+ mean.shape,
59
+ device=mean.device,
60
+ dtype=mean.dtype,
61
+ generator=self.generator,
62
+ )
63
+ noisy_lcb_mean = mean + noise_stddev * lcb_noise
64
+ improvement_mean = mean + noise_stddev * improvement_noise
65
+ else:
66
+ noisy_lcb_mean = mean
67
+ improvement_mean = mean
68
+ lcb = noisy_lcb_mean - self.kappa * stddev
69
+ normalized = (tau - self.epsilon - improvement_mean) / stddev
70
+
71
+ log_phi = -0.5 * normalized.square() - 0.5 * math.log(2.0 * math.pi)
72
+ probability = torch.special.ndtr(normalized)
73
+ expected_improvement = stddev * (probability * normalized + torch.exp(log_phi))
74
+ log_ei = torch.log(expected_improvement)
75
+ log_pi = torch.log(probability)
76
+ log_ei_approx = (
77
+ torch.log(stddev) - 0.5 * normalized.square() - torch.log(normalized.square() - 1.0)
78
+ )
79
+ log_pi_approx = (
80
+ -0.5 * normalized.square() - torch.log(-normalized) - 0.5 * math.log(2.0 * math.pi)
81
+ )
82
+ direct = (normalized > -6.0) & torch.isfinite(log_ei) & torch.isfinite(log_pi)
83
+ negative_log_ei = -torch.where(direct, log_ei, log_ei_approx)
84
+ negative_log_pi = -torch.where(direct, log_pi, log_pi_approx)
85
+ objectives = torch.stack((lcb, negative_log_ei, negative_log_pi), dim=-1)
86
+ if not torch.isfinite(objectives).all():
87
+ bad = int((~torch.isfinite(objectives)).sum().item())
88
+ raise NumericalError(f"MACE produced {bad} non-finite objective values")
89
+ return objectives
90
+
91
+ __call__ = evaluate
@@ -0,0 +1,119 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """One-call-per-chunk exact-GP posterior evaluation."""
4
+
5
+ from __future__ import annotations
6
+
7
+ from dataclasses import dataclass
8
+ from typing import Protocol
9
+
10
+ import torch
11
+
12
+
13
+ class PosteriorProvider(Protocol):
14
+ posterior_cache_version: int
15
+
16
+ def predict(
17
+ self, continuous: torch.Tensor, categorical: torch.Tensor
18
+ ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]: ...
19
+
20
+
21
+ @dataclass(frozen=True, slots=True)
22
+ class PosteriorStats:
23
+ mean: torch.Tensor
24
+ variance: torch.Tensor
25
+ stddev: torch.Tensor
26
+ noise_variance: torch.Tensor
27
+
28
+ def index_select(self, indices: torch.Tensor) -> PosteriorStats:
29
+ return PosteriorStats(
30
+ mean=self.mean.index_select(0, indices),
31
+ variance=self.variance.index_select(0, indices),
32
+ stddev=self.stddev.index_select(0, indices),
33
+ noise_variance=self.noise_variance,
34
+ )
35
+
36
+
37
+ class PosteriorEvaluator:
38
+ """Evaluate and optionally cache shared posterior statistics."""
39
+
40
+ def __init__(
41
+ self,
42
+ provider: PosteriorProvider,
43
+ *,
44
+ batch_size: int | None = 4096,
45
+ cache: bool = True,
46
+ ) -> None:
47
+ if batch_size is not None and batch_size < 1:
48
+ raise ValueError("batch_size must be positive or None")
49
+ self.provider = provider
50
+ self.batch_size = batch_size
51
+ self.cache = cache
52
+ self._cache_key: tuple[object, ...] | None = None
53
+ self._cache_value: PosteriorStats | None = None
54
+ self._cache_inputs: tuple[torch.Tensor, torch.Tensor] | None = None
55
+
56
+ def invalidate(self) -> None:
57
+ self._cache_key = None
58
+ self._cache_value = None
59
+ self._cache_inputs = None
60
+
61
+ def evaluate(self, continuous: torch.Tensor, categorical: torch.Tensor) -> PosteriorStats:
62
+ if continuous.shape[0] != categorical.shape[0]:
63
+ raise ValueError("continuous and categorical batch lengths differ")
64
+ key = self._key(continuous, categorical)
65
+ if self.cache and key == self._cache_key and self._cache_value is not None:
66
+ return self._cache_value
67
+ count = continuous.shape[0]
68
+ if count == 0:
69
+ empty = continuous.new_empty((0,))
70
+ result = PosteriorStats(empty, empty, empty, continuous.new_zeros(()))
71
+ if self.cache:
72
+ self._cache_key = key
73
+ self._cache_value = result
74
+ self._cache_inputs = (continuous, categorical)
75
+ return result
76
+ chunk_size = count if self.batch_size is None else self.batch_size
77
+ means: list[torch.Tensor] = []
78
+ variances: list[torch.Tensor] = []
79
+ noise: torch.Tensor | None = None
80
+ for start in range(0, count, chunk_size):
81
+ end = min(start + chunk_size, count)
82
+ mean, variance, chunk_noise = self.provider.predict(
83
+ continuous[start:end], categorical[start:end]
84
+ )
85
+ means.append(mean)
86
+ variances.append(variance)
87
+ noise = chunk_noise if noise is None else noise
88
+ mean = torch.cat(means)
89
+ variance = torch.cat(variances).clamp_min(torch.finfo(continuous.dtype).eps)
90
+ assert noise is not None
91
+ result = PosteriorStats(mean, variance, variance.sqrt(), noise)
92
+ if self.cache:
93
+ self._cache_key = key
94
+ self._cache_value = result
95
+ # Retaining the input objects prevents allocator pointer reuse from
96
+ # making unrelated tensors look like the cached evaluation.
97
+ self._cache_inputs = (continuous, categorical)
98
+ return result
99
+
100
+ def _key(self, continuous: torch.Tensor, categorical: torch.Tensor) -> tuple[object, ...]:
101
+ return (
102
+ self.provider.posterior_cache_version,
103
+ id(continuous),
104
+ continuous.data_ptr(),
105
+ tuple(continuous.shape),
106
+ tuple(continuous.stride()),
107
+ continuous.storage_offset(),
108
+ continuous.dtype,
109
+ continuous.device,
110
+ continuous._version,
111
+ id(categorical),
112
+ categorical.data_ptr(),
113
+ tuple(categorical.shape),
114
+ tuple(categorical.stride()),
115
+ categorical.storage_offset(),
116
+ categorical.dtype,
117
+ categorical.device,
118
+ categorical._version,
119
+ )
leanhebo/checkpoint.py ADDED
@@ -0,0 +1,73 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """Versioned, tensor-and-primitive-only LeanHEBO checkpoints."""
4
+
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ import pickle
9
+ import struct
10
+ import tempfile
11
+ from collections.abc import Mapping
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ import torch
16
+
17
+ from leanhebo.errors import CheckpointError
18
+
19
+ CHECKPOINT_KIND = "leanhebo.optimizer"
20
+ CHECKPOINT_SCHEMA_VERSION = 1
21
+
22
+
23
+ def make_checkpoint(payload: Mapping[str, Any]) -> dict[str, Any]:
24
+ return {
25
+ "kind": CHECKPOINT_KIND,
26
+ "schema_version": CHECKPOINT_SCHEMA_VERSION,
27
+ "payload": dict(payload),
28
+ }
29
+
30
+
31
+ def save_checkpoint(path: str | os.PathLike[str], payload: Mapping[str, Any]) -> None:
32
+ """Atomically save a versioned checkpoint using Torch's tensor-aware format."""
33
+
34
+ destination = Path(path).expanduser().resolve()
35
+ destination.parent.mkdir(parents=True, exist_ok=True)
36
+ handle, temporary_name = tempfile.mkstemp(
37
+ prefix=f".{destination.name}.", suffix=".tmp", dir=destination.parent
38
+ )
39
+ os.close(handle)
40
+ temporary = Path(temporary_name)
41
+ try:
42
+ torch.save(make_checkpoint(payload), temporary)
43
+ os.replace(temporary, destination)
44
+ finally:
45
+ temporary.unlink(missing_ok=True)
46
+
47
+
48
+ def load_checkpoint(
49
+ path: str | os.PathLike[str],
50
+ *,
51
+ map_location: str | torch.device | None = None,
52
+ ) -> dict[str, Any]:
53
+ """Load and validate a checkpoint without permitting arbitrary pickled objects."""
54
+
55
+ try:
56
+ state = torch.load(path, map_location=map_location, weights_only=True)
57
+ except (
58
+ EOFError,
59
+ OSError,
60
+ RuntimeError,
61
+ ValueError,
62
+ pickle.UnpicklingError,
63
+ struct.error,
64
+ ) as exc:
65
+ raise CheckpointError(f"failed to load LeanHEBO checkpoint: {exc}") from exc
66
+ if not isinstance(state, Mapping) or state.get("kind") != CHECKPOINT_KIND:
67
+ raise CheckpointError("file is not a LeanHEBO optimizer checkpoint")
68
+ if state.get("schema_version") != CHECKPOINT_SCHEMA_VERSION:
69
+ raise CheckpointError(f"unsupported checkpoint schema: {state.get('schema_version')!r}")
70
+ payload = state.get("payload")
71
+ if not isinstance(payload, Mapping):
72
+ raise CheckpointError("checkpoint payload is missing or malformed")
73
+ return dict(payload)
leanhebo/config.py ADDED
@@ -0,0 +1,263 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Portions derived from Huawei HEBO; see NOTICE.md.
3
+
4
+ """Explicit, serializable configuration for LeanHEBO."""
5
+
6
+ from __future__ import annotations
7
+
8
+ import math
9
+ from collections.abc import Mapping
10
+ from dataclasses import asdict, dataclass, field
11
+ from typing import Any, Literal, TypeVar
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class RuntimeConfig:
16
+ """Device, precision, randomness, and evaluation controls."""
17
+
18
+ device: str = "cpu"
19
+ dtype: Literal["float32", "float64"] = "float32"
20
+ seed: int | None = None
21
+ deterministic: bool = False
22
+ acquisition_batch_size: int | None = 4096
23
+ synchronize_device_for_timing: bool = True
24
+ enable_torch_compile: bool = False
25
+
26
+ def __post_init__(self) -> None:
27
+ if not self.device:
28
+ raise ValueError("device must be a non-empty Torch device string")
29
+ if self.dtype not in ("float32", "float64"):
30
+ raise ValueError("dtype must be 'float32' or 'float64'")
31
+ if self.acquisition_batch_size is not None and self.acquisition_batch_size < 1:
32
+ raise ValueError("acquisition_batch_size must be positive or None")
33
+
34
+
35
+ @dataclass(frozen=True, slots=True)
36
+ class GPConfig:
37
+ """Exact-GP fitting and lifecycle controls."""
38
+
39
+ learning_rate: float = 1e-2
40
+ optimizer: Literal["psgld", "adam", "lbfgs"] = "psgld"
41
+ initial_steps: int = 100
42
+ update_steps: int = 10
43
+ full_refit_interval: int | None = 25
44
+ full_refit_growth_factor: float | None = 1.5
45
+ reuse_parameters: bool = True
46
+ reuse_optimizer_state: bool = True
47
+ use_set_train_data: bool = True
48
+ use_fantasy_updates: bool = False
49
+ noise_lower_bound: float = 8e-4
50
+ noise_initial: float = 1e-2
51
+ predict_observation_noise: bool = False
52
+ ard: bool = True
53
+ early_stopping: bool = False
54
+ patience: int = 10
55
+ relative_tolerance: float = 1e-4
56
+ max_cholesky_size: int | None = None
57
+ max_preconditioner_size: int | None = None
58
+ cg_tolerance: float | None = None
59
+ eval_cg_tolerance: float | None = None
60
+ fast_pred_var: bool = True
61
+ kernel_initialization_samples: int = 1000
62
+ lengthscale_lower_bound: float = 0.02
63
+ jitter_initial: float = 1e-8
64
+ jitter_multiplier: float = 10.0
65
+ jitter_max: float = 1.0
66
+ max_jitter_retries: int = 9
67
+ lbfgs_max_iter: int = 5
68
+
69
+ def __post_init__(self) -> None:
70
+ if not math.isfinite(self.learning_rate) or self.learning_rate <= 0:
71
+ raise ValueError("learning_rate must be positive and finite")
72
+ if self.initial_steps < 0 or self.update_steps < 0:
73
+ raise ValueError("GP step counts cannot be negative")
74
+ if self.full_refit_interval is not None and self.full_refit_interval < 1:
75
+ raise ValueError("full_refit_interval must be positive or None")
76
+ if self.full_refit_growth_factor is not None and (
77
+ not math.isfinite(self.full_refit_growth_factor) or self.full_refit_growth_factor <= 1
78
+ ):
79
+ raise ValueError("full_refit_growth_factor must exceed 1 or be None")
80
+ if (
81
+ not math.isfinite(self.noise_lower_bound)
82
+ or not math.isfinite(self.noise_initial)
83
+ or self.noise_lower_bound <= 0
84
+ or self.noise_initial <= 0
85
+ ):
86
+ raise ValueError("noise bounds and initial value must be positive and finite")
87
+ if self.noise_initial <= self.noise_lower_bound:
88
+ raise ValueError("noise_initial must be strictly greater than noise_lower_bound")
89
+ if self.use_fantasy_updates and self.update_steps != 0:
90
+ raise ValueError("fantasy updates require update_steps=0 so their cache remains valid")
91
+ if self.use_fantasy_updates and not self.reuse_parameters:
92
+ raise ValueError("fantasy updates require reuse_parameters=True")
93
+ if self.patience < 1:
94
+ raise ValueError("patience must be positive")
95
+ if not math.isfinite(self.relative_tolerance) or self.relative_tolerance < 0:
96
+ raise ValueError("relative_tolerance must be finite and non-negative")
97
+ if self.kernel_initialization_samples < 2:
98
+ raise ValueError("kernel_initialization_samples must be at least 2")
99
+ if not math.isfinite(self.lengthscale_lower_bound) or self.lengthscale_lower_bound <= 0:
100
+ raise ValueError("lengthscale_lower_bound must be positive and finite")
101
+ if (
102
+ not math.isfinite(self.jitter_initial)
103
+ or not math.isfinite(self.jitter_multiplier)
104
+ or self.jitter_initial <= 0
105
+ or self.jitter_multiplier <= 1
106
+ ):
107
+ raise ValueError("invalid jitter schedule")
108
+ if (
109
+ not math.isfinite(self.jitter_max)
110
+ or self.jitter_max < self.jitter_initial
111
+ or self.max_jitter_retries < 0
112
+ ):
113
+ raise ValueError("invalid maximum jitter settings")
114
+ if self.lbfgs_max_iter < 1:
115
+ raise ValueError("lbfgs_max_iter must be positive")
116
+ if self.max_cholesky_size is not None and self.max_cholesky_size < 0:
117
+ raise ValueError("max_cholesky_size cannot be negative")
118
+ if self.max_preconditioner_size is not None and self.max_preconditioner_size < 0:
119
+ raise ValueError("max_preconditioner_size cannot be negative")
120
+ for name, value in (
121
+ ("cg_tolerance", self.cg_tolerance),
122
+ ("eval_cg_tolerance", self.eval_cg_tolerance),
123
+ ):
124
+ if value is not None and (not math.isfinite(value) or value <= 0):
125
+ raise ValueError(f"{name} must be positive and finite or None")
126
+
127
+
128
+ @dataclass(frozen=True, slots=True)
129
+ class WarpConfig:
130
+ """Output standardization and power-transform controls."""
131
+
132
+ method: Literal["auto", "none", "box-cox", "yeo-johnson"] = "auto"
133
+ standardize_before_warp: bool = True
134
+ refit_interval: int | None = 1
135
+ minimum_points: int = 3
136
+ minimum_transformed_std: float = 0.5
137
+ lambda_lower_bound: float = -5.0
138
+ lambda_upper_bound: float = 5.0
139
+ lambda_tolerance: float = 1e-5
140
+
141
+ def __post_init__(self) -> None:
142
+ if self.refit_interval is not None and self.refit_interval < 1:
143
+ raise ValueError("refit_interval must be positive or None")
144
+ if self.minimum_points < 1:
145
+ raise ValueError("minimum_points must be positive")
146
+ if not math.isfinite(self.minimum_transformed_std) or self.minimum_transformed_std < 0:
147
+ raise ValueError("minimum_transformed_std must be finite and non-negative")
148
+ if (
149
+ not math.isfinite(self.lambda_lower_bound)
150
+ or not math.isfinite(self.lambda_upper_bound)
151
+ or self.lambda_lower_bound >= self.lambda_upper_bound
152
+ ):
153
+ raise ValueError("lambda bounds must be strictly increasing")
154
+ if not math.isfinite(self.lambda_tolerance) or self.lambda_tolerance <= 0:
155
+ raise ValueError("lambda_tolerance must be positive and finite")
156
+
157
+
158
+ @dataclass(frozen=True, slots=True)
159
+ class AcquisitionConfig:
160
+ """MACE policy controls."""
161
+
162
+ epsilon: float = 1e-4
163
+ upsi: float = 0.5
164
+ delta: float = 0.01
165
+ kappa: float | None = None
166
+ stochastic: bool = True
167
+ posterior_cache: bool = True
168
+
169
+ def __post_init__(self) -> None:
170
+ if not math.isfinite(self.epsilon) or self.epsilon < 0:
171
+ raise ValueError("epsilon must be finite and non-negative")
172
+ if not math.isfinite(self.upsi) or self.upsi <= 0:
173
+ raise ValueError("upsi must be positive and finite")
174
+ if not math.isfinite(self.delta) or not 0 < self.delta < 1:
175
+ raise ValueError("delta must lie strictly between zero and one")
176
+ if self.kappa is not None and (not math.isfinite(self.kappa) or self.kappa < 0):
177
+ raise ValueError("kappa must be finite and non-negative or None")
178
+
179
+
180
+ @dataclass(frozen=True, slots=True)
181
+ class SearchConfig:
182
+ """Tensor-native NSGA-II controls."""
183
+
184
+ population_size: int = 100
185
+ generations: int = 100
186
+ crossover_probability: float = 0.9
187
+ crossover_eta: float = 15.0
188
+ mutation_probability: float | None = None
189
+ mutation_eta: float = 20.0
190
+ tournament_size: int = 2
191
+ eliminate_duplicates: bool = True
192
+ reuse_previous_population: bool = False
193
+ keep_history: bool = False
194
+ seed: int | None = None
195
+
196
+ def __post_init__(self) -> None:
197
+ if self.population_size < 2:
198
+ raise ValueError("population_size must be at least 2")
199
+ if self.generations < 0:
200
+ raise ValueError("generations cannot be negative")
201
+ if not 0 <= self.crossover_probability <= 1:
202
+ raise ValueError("crossover_probability must be between zero and one")
203
+ if (
204
+ not math.isfinite(self.crossover_eta)
205
+ or not math.isfinite(self.mutation_eta)
206
+ or self.crossover_eta <= 0
207
+ or self.mutation_eta <= 0
208
+ ):
209
+ raise ValueError("crossover_eta and mutation_eta must be positive and finite")
210
+ if self.mutation_probability is not None and not 0 <= self.mutation_probability <= 1:
211
+ raise ValueError("mutation_probability must be between zero and one")
212
+ if self.tournament_size < 2:
213
+ raise ValueError("tournament_size must be at least 2")
214
+
215
+
216
+ @dataclass(frozen=True, slots=True)
217
+ class LeanHEBOConfig:
218
+ """Complete LeanHEBO configuration; there are deliberately no presets."""
219
+
220
+ random_samples: int | None = None
221
+ nonfinite_policy: Literal["drop", "raise"] = "drop"
222
+ runtime: RuntimeConfig = field(default_factory=RuntimeConfig)
223
+ gp: GPConfig = field(default_factory=GPConfig)
224
+ warp: WarpConfig = field(default_factory=WarpConfig)
225
+ acquisition: AcquisitionConfig = field(default_factory=AcquisitionConfig)
226
+ search: SearchConfig = field(default_factory=SearchConfig)
227
+
228
+ def __post_init__(self) -> None:
229
+ if self.random_samples is not None and self.random_samples < 2:
230
+ raise ValueError("random_samples must be at least 2 or None")
231
+ if self.nonfinite_policy not in ("drop", "raise"):
232
+ raise ValueError("nonfinite_policy must be 'drop' or 'raise'")
233
+
234
+ def to_dict(self) -> dict[str, Any]:
235
+ """Return a checkpoint- and JSON-friendly representation."""
236
+
237
+ return asdict(self)
238
+
239
+ @classmethod
240
+ def from_dict(cls, value: Mapping[str, Any]) -> LeanHEBOConfig:
241
+ """Reconstruct a configuration from :meth:`to_dict` output."""
242
+
243
+ root = dict(value)
244
+ return cls(
245
+ random_samples=root.get("random_samples"),
246
+ nonfinite_policy=root.get("nonfinite_policy", "drop"),
247
+ runtime=_construct(RuntimeConfig, root.get("runtime", {})),
248
+ gp=_construct(GPConfig, root.get("gp", {})),
249
+ warp=_construct(WarpConfig, root.get("warp", {})),
250
+ acquisition=_construct(AcquisitionConfig, root.get("acquisition", {})),
251
+ search=_construct(SearchConfig, root.get("search", {})),
252
+ )
253
+
254
+
255
+ ConfigT = TypeVar("ConfigT", RuntimeConfig, GPConfig, WarpConfig, AcquisitionConfig, SearchConfig)
256
+
257
+
258
+ def _construct(config_type: type[ConfigT], value: object) -> ConfigT:
259
+ if isinstance(value, config_type):
260
+ return value
261
+ if not isinstance(value, Mapping):
262
+ raise TypeError(f"expected a mapping for {config_type.__name__}")
263
+ return config_type(**dict(value))
@@ -0,0 +1,29 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """Tensor-native batch and observation data APIs."""
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from leanhebo.data.batch import CandidateBatch, EncodedBatch
8
+
9
+ if TYPE_CHECKING:
10
+ from leanhebo.data.store import NonFinitePolicy, ObservationBatch, ObservationStore
11
+
12
+ __all__ = [
13
+ "CandidateBatch",
14
+ "EncodedBatch",
15
+ "NonFinitePolicy",
16
+ "ObservationBatch",
17
+ "ObservationStore",
18
+ ]
19
+
20
+
21
+ def __getattr__(name: str) -> Any:
22
+ # CompiledSpace imports the adapter package. Keeping store exports lazy
23
+ # avoids a package-initialization cycle while preserving the concise public
24
+ # ``from leanhebo.data import ObservationStore`` spelling.
25
+ if name in {"NonFinitePolicy", "ObservationBatch", "ObservationStore"}:
26
+ from leanhebo.data import store
27
+
28
+ return getattr(store, name)
29
+ raise AttributeError(name)
@@ -0,0 +1,11 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """Public adapter registry exports."""
4
+
5
+ from leanhebo.data.adapters.registry import (
6
+ DEFAULT_ADAPTERS,
7
+ InputAdapterRegistry,
8
+ columns_from_input,
9
+ )
10
+
11
+ __all__ = ["DEFAULT_ADAPTERS", "InputAdapterRegistry", "columns_from_input"]
@@ -0,0 +1,30 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """NumPy input adapter."""
4
+
5
+ from __future__ import annotations
6
+
7
+ from collections.abc import Sequence
8
+
9
+ import numpy as np
10
+
11
+
12
+ def numpy_to_columns(value: np.ndarray, names: Sequence[str]) -> dict[str, list[object]]:
13
+ """Convert a regular or structured ndarray to named Python columns."""
14
+
15
+ if value.dtype.names is not None:
16
+ missing = [name for name in names if name not in value.dtype.names]
17
+ if missing:
18
+ raise ValueError(f"NumPy structured array is missing columns: {missing}")
19
+ return {name: value[name].reshape(-1).tolist() for name in names}
20
+ if value.ndim == 1:
21
+ if value.size != len(names):
22
+ raise ValueError(
23
+ f"one-dimensional NumPy input must contain {len(names)} values, got {value.size}"
24
+ )
25
+ value = value.reshape(1, -1)
26
+ if value.ndim != 2:
27
+ raise ValueError("NumPy input must have one or two dimensions")
28
+ if value.shape[1] != len(names):
29
+ raise ValueError(f"NumPy input must have {len(names)} columns, got {value.shape[1]}")
30
+ return {name: value[:, index].tolist() for index, name in enumerate(names)}
@@ -0,0 +1,15 @@
1
+ # SPDX-License-Identifier: MIT
2
+
3
+ """Pandas input adapter with no import-time Pandas dependency."""
4
+
5
+ from __future__ import annotations
6
+
7
+ from collections.abc import Sequence
8
+ from typing import Any
9
+
10
+
11
+ def pandas_to_columns(value: Any, names: Sequence[str]) -> dict[str, list[object]]:
12
+ missing = [name for name in names if name not in value.columns]
13
+ if missing:
14
+ raise ValueError(f"Pandas DataFrame is missing columns: {missing}")
15
+ return {name: value[name].tolist() for name in names}