mlsampler 0.3.3__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mlsampler-0.3.3/src/mlsampler.egg-info → mlsampler-0.3.4}/PKG-INFO +6 -2
- {mlsampler-0.3.3 → mlsampler-0.3.4}/pyproject.toml +6 -3
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/constraints.py +17 -5
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/hypergrid.py +28 -27
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/validate.py +3 -1
- {mlsampler-0.3.3 → mlsampler-0.3.4/src/mlsampler.egg-info}/PKG-INFO +6 -2
- mlsampler-0.3.4/src/mlsampler.egg-info/requires.txt +7 -0
- mlsampler-0.3.3/src/mlsampler.egg-info/requires.txt +0 -2
- {mlsampler-0.3.3 → mlsampler-0.3.4}/LICENSE +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/README.md +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/setup.cfg +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/__init__.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/base.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/__init__.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/base.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/__init__.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/random.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/errors.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/terminals.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/types.py +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/SOURCES.txt +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/dependency_links.txt +0 -0
- {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mlsampler
|
|
3
|
-
Version: 0.3.
|
|
4
|
-
Summary: Flexible constrained random sampler for reverse analysis in machine learning
|
|
3
|
+
Version: 0.3.4
|
|
4
|
+
Summary: Flexible constrained random sampler for doe sampling or reverse analysis in machine learning
|
|
5
5
|
Author: Jugai O
|
|
6
6
|
License: LICENSE
|
|
7
7
|
Requires-Python: >=3.11
|
|
@@ -9,6 +9,10 @@ Description-Content-Type: text/markdown
|
|
|
9
9
|
License-File: LICENSE
|
|
10
10
|
Requires-Dist: numpy
|
|
11
11
|
Requires-Dist: joblib
|
|
12
|
+
Provides-Extra: docs
|
|
13
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
14
|
+
Requires-Dist: sphinx-rtd-theme; extra == "docs"
|
|
15
|
+
Requires-Dist: scipy; extra == "docs"
|
|
12
16
|
Dynamic: license-file
|
|
13
17
|
|
|
14
18
|
# RandomSampler
|
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "mlsampler"
|
|
7
|
-
version = "0.3.
|
|
8
|
-
description = "Flexible constrained random sampler for reverse analysis in machine learning"
|
|
7
|
+
version = "0.3.4"
|
|
8
|
+
description = "Flexible constrained random sampler for doe sampling or reverse analysis in machine learning"
|
|
9
9
|
authors = [
|
|
10
10
|
{ name="Jugai O" }
|
|
11
11
|
]
|
|
@@ -19,4 +19,7 @@ dependencies = [
|
|
|
19
19
|
license = { text= "LICENSE" }
|
|
20
20
|
|
|
21
21
|
[tool.setuptools.packages.find]
|
|
22
|
-
where = ["src"]
|
|
22
|
+
where = ["src"]
|
|
23
|
+
|
|
24
|
+
[project.optional-dependencies]
|
|
25
|
+
docs = ["sphinx", "sphinx-rtd-theme", "scipy"]
|
|
@@ -165,6 +165,7 @@ class RangeConstraint(Constraints):
|
|
|
165
165
|
super().__init__(cols, **kwargs)
|
|
166
166
|
self.low = low
|
|
167
167
|
self.high = high
|
|
168
|
+
v.validate_range(low, high)
|
|
168
169
|
|
|
169
170
|
def _constrain(self, row: np.ndarray, rng: Optional[np.random.Generator] = None) -> np.ndarray:
|
|
170
171
|
# No need to reset cols here
|
|
@@ -175,13 +176,20 @@ class RangeConstraint(Constraints):
|
|
|
175
176
|
class StepConstraint(Constraints):
|
|
176
177
|
def __init__(self, col:int, step: float, low: float, high: float, **kwargs):
|
|
177
178
|
super().__init__(cols=[col], **kwargs)
|
|
178
|
-
self.values = np.arange(low, high + step, step)
|
|
179
179
|
self.col = col
|
|
180
|
+
|
|
181
|
+
n_steps = int(np.floor((high - low) / step))
|
|
182
|
+
self.values = low + np.arange(n_steps + 1) * step
|
|
183
|
+
|
|
184
|
+
self.low = low
|
|
185
|
+
self.high = high
|
|
180
186
|
self.step = step
|
|
181
187
|
|
|
188
|
+
v.validate_range(low, high, step)
|
|
189
|
+
|
|
182
190
|
def _constrain(
|
|
183
|
-
self,
|
|
184
|
-
row: np.ndarray,
|
|
191
|
+
self,
|
|
192
|
+
row: np.ndarray,
|
|
185
193
|
rng: Optional[np.random.Generator] = None
|
|
186
194
|
) -> np.ndarray:
|
|
187
195
|
rng = self._rng(rng)
|
|
@@ -246,7 +254,9 @@ class SumStepConstraint(StepConstraint):
|
|
|
246
254
|
]
|
|
247
255
|
|
|
248
256
|
if not eligible_indices:
|
|
249
|
-
raise ConstraintViolationError(
|
|
257
|
+
raise ConstraintViolationError(
|
|
258
|
+
"Target sum_value is unreachable within defined highs."
|
|
259
|
+
)
|
|
250
260
|
|
|
251
261
|
target_idx = rng.choice(eligible_indices)
|
|
252
262
|
current_values[target_idx] += self.step
|
|
@@ -274,4 +284,6 @@ class FunctionConstraint(Constraints):
|
|
|
274
284
|
row[self.cols] = result
|
|
275
285
|
return row
|
|
276
286
|
else:
|
|
277
|
-
raise ConstraintTypeError(
|
|
287
|
+
raise ConstraintTypeError(
|
|
288
|
+
"Constraint function must return either a boolean or a numpy array"
|
|
289
|
+
)
|
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
import numpy as np
|
|
2
2
|
from typing import Optional
|
|
3
|
-
from scipy.stats import qmc
|
|
4
3
|
from ..base import BaseSampler, DtypeMeta as dm
|
|
5
4
|
from ..errors import ConstraintViolationError
|
|
6
5
|
from ..terminals import spinning
|
|
6
|
+
try:
|
|
7
|
+
from scipy.stats import qmc
|
|
8
|
+
except:
|
|
9
|
+
qmc = None
|
|
7
10
|
|
|
8
11
|
class HyperGridSampler(BaseSampler):
|
|
9
12
|
"""
|
|
@@ -47,10 +50,14 @@ class HyperGridSampler(BaseSampler):
|
|
|
47
50
|
|
|
48
51
|
def _lhs(self, n_samples: int, float_features: list):
|
|
49
52
|
"""LHS sampling for float columns"""
|
|
53
|
+
if qmc is None:
|
|
54
|
+
raise ImportError("scipy is required for HyperGridSampler")
|
|
55
|
+
|
|
50
56
|
if not float_features:
|
|
51
57
|
return np.empty((n_samples, 0))
|
|
52
58
|
|
|
53
59
|
dim = len(float_features)
|
|
60
|
+
|
|
54
61
|
sampler = qmc.LatinHypercube(d=dim, seed=self.config.random_state)
|
|
55
62
|
sample = sampler.random(n=n_samples)
|
|
56
63
|
|
|
@@ -82,32 +89,6 @@ class HyperGridSampler(BaseSampler):
|
|
|
82
89
|
return np.column_stack(cols) if cols else np.empty((n_samples, 0))
|
|
83
90
|
|
|
84
91
|
def _sample(self, n_samples: int) -> np.ndarray:
|
|
85
|
-
"""
|
|
86
|
-
Generate samples based on feature data types.
|
|
87
|
-
|
|
88
|
-
Continuous features are sampled using Latin Hypercube Sampling (LHS),
|
|
89
|
-
while discrete features (e.g., "int", "binary", "categorical", "constant") are
|
|
90
|
-
sampled using uniform random selection over their respective value spaces.
|
|
91
|
-
|
|
92
|
-
Parameters
|
|
93
|
-
----------
|
|
94
|
-
n_samples : int
|
|
95
|
-
Total number of samples to generate.
|
|
96
|
-
|
|
97
|
-
Returns
|
|
98
|
-
-------
|
|
99
|
-
np.ndarray
|
|
100
|
-
Array of shape (n_samples, n_features) containing the generated samples.
|
|
101
|
-
The column order matches the input feature configuration.
|
|
102
|
-
|
|
103
|
-
Notes
|
|
104
|
-
-----
|
|
105
|
-
- Float features are scaled to their respective [low, high] ranges.
|
|
106
|
-
- Integer features are sampled from the inclusive range [low, high].
|
|
107
|
-
- Categorical features are sampled from the provided category list.
|
|
108
|
-
- Constant features return the same value for all samples.
|
|
109
|
-
"""
|
|
110
|
-
|
|
111
92
|
float_feats = [f for f in self.config.features if f.dtype == dm.float]
|
|
112
93
|
discrete_feats = [f for f in self.config.features if f.dtype != dm.float]
|
|
113
94
|
|
|
@@ -130,6 +111,26 @@ class HyperGridSampler(BaseSampler):
|
|
|
130
111
|
return result
|
|
131
112
|
|
|
132
113
|
def sample(self, n_samples: int) -> np.ndarray:
|
|
114
|
+
"""
|
|
115
|
+
Parameters
|
|
116
|
+
----------
|
|
117
|
+
n_samples : int
|
|
118
|
+
Total number of samples to generate.
|
|
119
|
+
|
|
120
|
+
Returns
|
|
121
|
+
-------
|
|
122
|
+
np.ndarray
|
|
123
|
+
Array of shape (n_samples, n_features) containing the generated samples.
|
|
124
|
+
The column order matches the input feature configuration.
|
|
125
|
+
|
|
126
|
+
Notes
|
|
127
|
+
-----
|
|
128
|
+
- Float features are scaled to their respective [low, high] ranges.
|
|
129
|
+
- Integer features are sampled from the inclusive range [low, high].
|
|
130
|
+
- Categorical features are sampled from the provided category list.
|
|
131
|
+
- Constant features return the same value for all samples.
|
|
132
|
+
"""
|
|
133
|
+
|
|
133
134
|
with spinning():
|
|
134
135
|
samples = self._sample(n_samples)
|
|
135
136
|
return samples
|
|
@@ -35,7 +35,7 @@ def validate_values(value: Numeric):
|
|
|
35
35
|
if value < 0:
|
|
36
36
|
warnings.warn("Value should be non-negative", ConstraintWarning)
|
|
37
37
|
|
|
38
|
-
def validate_range(low: Numeric, high: Numeric):
|
|
38
|
+
def validate_range(low: Numeric, high: Numeric, step: Numeric = 0):
|
|
39
39
|
if not isinstance(low, Numeric):
|
|
40
40
|
raise ConstraintTypeError("low must be a numeric")
|
|
41
41
|
|
|
@@ -45,4 +45,6 @@ def validate_range(low: Numeric, high: Numeric):
|
|
|
45
45
|
if low > high:
|
|
46
46
|
raise ConstraintValidationError("low must be <= high")
|
|
47
47
|
|
|
48
|
+
if step < 0:
|
|
49
|
+
raise ConstraintValidationError("step must be non-negative")
|
|
48
50
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mlsampler
|
|
3
|
-
Version: 0.3.
|
|
4
|
-
Summary: Flexible constrained random sampler for reverse analysis in machine learning
|
|
3
|
+
Version: 0.3.4
|
|
4
|
+
Summary: Flexible constrained random sampler for doe sampling or reverse analysis in machine learning
|
|
5
5
|
Author: Jugai O
|
|
6
6
|
License: LICENSE
|
|
7
7
|
Requires-Python: >=3.11
|
|
@@ -9,6 +9,10 @@ Description-Content-Type: text/markdown
|
|
|
9
9
|
License-File: LICENSE
|
|
10
10
|
Requires-Dist: numpy
|
|
11
11
|
Requires-Dist: joblib
|
|
12
|
+
Provides-Extra: docs
|
|
13
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
14
|
+
Requires-Dist: sphinx-rtd-theme; extra == "docs"
|
|
15
|
+
Requires-Dist: scipy; extra == "docs"
|
|
12
16
|
Dynamic: license-file
|
|
13
17
|
|
|
14
18
|
# RandomSampler
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|