mlsampler 0.3.3__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {mlsampler-0.3.3/src/mlsampler.egg-info → mlsampler-0.3.4}/PKG-INFO +6 -2
  2. {mlsampler-0.3.3 → mlsampler-0.3.4}/pyproject.toml +6 -3
  3. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/constraints.py +17 -5
  4. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/hypergrid.py +28 -27
  5. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/validate.py +3 -1
  6. {mlsampler-0.3.3 → mlsampler-0.3.4/src/mlsampler.egg-info}/PKG-INFO +6 -2
  7. mlsampler-0.3.4/src/mlsampler.egg-info/requires.txt +7 -0
  8. mlsampler-0.3.3/src/mlsampler.egg-info/requires.txt +0 -2
  9. {mlsampler-0.3.3 → mlsampler-0.3.4}/LICENSE +0 -0
  10. {mlsampler-0.3.3 → mlsampler-0.3.4}/README.md +0 -0
  11. {mlsampler-0.3.3 → mlsampler-0.3.4}/setup.cfg +0 -0
  12. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/__init__.py +0 -0
  13. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/base.py +0 -0
  14. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/__init__.py +0 -0
  15. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/constraints/base.py +0 -0
  16. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/__init__.py +0 -0
  17. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/engine/random.py +0 -0
  18. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/errors.py +0 -0
  19. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/terminals.py +0 -0
  20. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler/types.py +0 -0
  21. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/SOURCES.txt +0 -0
  22. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/dependency_links.txt +0 -0
  23. {mlsampler-0.3.3 → mlsampler-0.3.4}/src/mlsampler.egg-info/top_level.txt +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mlsampler
3
- Version: 0.3.3
4
- Summary: Flexible constrained random sampler for reverse analysis in machine learning
3
+ Version: 0.3.4
4
+ Summary: Flexible constrained random sampler for doe sampling or reverse analysis in machine learning
5
5
  Author: Jugai O
6
6
  License: LICENSE
7
7
  Requires-Python: >=3.11
@@ -9,6 +9,10 @@ Description-Content-Type: text/markdown
9
9
  License-File: LICENSE
10
10
  Requires-Dist: numpy
11
11
  Requires-Dist: joblib
12
+ Provides-Extra: docs
13
+ Requires-Dist: sphinx; extra == "docs"
14
+ Requires-Dist: sphinx-rtd-theme; extra == "docs"
15
+ Requires-Dist: scipy; extra == "docs"
12
16
  Dynamic: license-file
13
17
 
14
18
  # RandomSampler
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "mlsampler"
7
- version = "0.3.3"
8
- description = "Flexible constrained random sampler for reverse analysis in machine learning"
7
+ version = "0.3.4"
8
+ description = "Flexible constrained random sampler for doe sampling or reverse analysis in machine learning"
9
9
  authors = [
10
10
  { name="Jugai O" }
11
11
  ]
@@ -19,4 +19,7 @@ dependencies = [
19
19
  license = { text= "LICENSE" }
20
20
 
21
21
  [tool.setuptools.packages.find]
22
- where = ["src"]
22
+ where = ["src"]
23
+
24
+ [project.optional-dependencies]
25
+ docs = ["sphinx", "sphinx-rtd-theme", "scipy"]
@@ -165,6 +165,7 @@ class RangeConstraint(Constraints):
165
165
  super().__init__(cols, **kwargs)
166
166
  self.low = low
167
167
  self.high = high
168
+ v.validate_range(low, high)
168
169
 
169
170
  def _constrain(self, row: np.ndarray, rng: Optional[np.random.Generator] = None) -> np.ndarray:
170
171
  # No need to reset cols here
@@ -175,13 +176,20 @@ class RangeConstraint(Constraints):
175
176
  class StepConstraint(Constraints):
176
177
  def __init__(self, col:int, step: float, low: float, high: float, **kwargs):
177
178
  super().__init__(cols=[col], **kwargs)
178
- self.values = np.arange(low, high + step, step)
179
179
  self.col = col
180
+
181
+ n_steps = int(np.floor((high - low) / step))
182
+ self.values = low + np.arange(n_steps + 1) * step
183
+
184
+ self.low = low
185
+ self.high = high
180
186
  self.step = step
181
187
 
188
+ v.validate_range(low, high, step)
189
+
182
190
  def _constrain(
183
- self,
184
- row: np.ndarray,
191
+ self,
192
+ row: np.ndarray,
185
193
  rng: Optional[np.random.Generator] = None
186
194
  ) -> np.ndarray:
187
195
  rng = self._rng(rng)
@@ -246,7 +254,9 @@ class SumStepConstraint(StepConstraint):
246
254
  ]
247
255
 
248
256
  if not eligible_indices:
249
- raise ConstraintViolationError("Target sum_value is unreachable within defined highs.")
257
+ raise ConstraintViolationError(
258
+ "Target sum_value is unreachable within defined highs."
259
+ )
250
260
 
251
261
  target_idx = rng.choice(eligible_indices)
252
262
  current_values[target_idx] += self.step
@@ -274,4 +284,6 @@ class FunctionConstraint(Constraints):
274
284
  row[self.cols] = result
275
285
  return row
276
286
  else:
277
- raise ConstraintTypeError("Constraint function must return either a boolean or a numpy array")
287
+ raise ConstraintTypeError(
288
+ "Constraint function must return either a boolean or a numpy array"
289
+ )
@@ -1,9 +1,12 @@
1
1
  import numpy as np
2
2
  from typing import Optional
3
- from scipy.stats import qmc
4
3
  from ..base import BaseSampler, DtypeMeta as dm
5
4
  from ..errors import ConstraintViolationError
6
5
  from ..terminals import spinning
6
+ try:
7
+ from scipy.stats import qmc
8
+ except:
9
+ qmc = None
7
10
 
8
11
  class HyperGridSampler(BaseSampler):
9
12
  """
@@ -47,10 +50,14 @@ class HyperGridSampler(BaseSampler):
47
50
 
48
51
  def _lhs(self, n_samples: int, float_features: list):
49
52
  """LHS sampling for float columns"""
53
+ if qmc is None:
54
+ raise ImportError("scipy is required for HyperGridSampler")
55
+
50
56
  if not float_features:
51
57
  return np.empty((n_samples, 0))
52
58
 
53
59
  dim = len(float_features)
60
+
54
61
  sampler = qmc.LatinHypercube(d=dim, seed=self.config.random_state)
55
62
  sample = sampler.random(n=n_samples)
56
63
 
@@ -82,32 +89,6 @@ class HyperGridSampler(BaseSampler):
82
89
  return np.column_stack(cols) if cols else np.empty((n_samples, 0))
83
90
 
84
91
  def _sample(self, n_samples: int) -> np.ndarray:
85
- """
86
- Generate samples based on feature data types.
87
-
88
- Continuous features are sampled using Latin Hypercube Sampling (LHS),
89
- while discrete features (e.g., "int", "binary", "categorical", "constant") are
90
- sampled using uniform random selection over their respective value spaces.
91
-
92
- Parameters
93
- ----------
94
- n_samples : int
95
- Total number of samples to generate.
96
-
97
- Returns
98
- -------
99
- np.ndarray
100
- Array of shape (n_samples, n_features) containing the generated samples.
101
- The column order matches the input feature configuration.
102
-
103
- Notes
104
- -----
105
- - Float features are scaled to their respective [low, high] ranges.
106
- - Integer features are sampled from the inclusive range [low, high].
107
- - Categorical features are sampled from the provided category list.
108
- - Constant features return the same value for all samples.
109
- """
110
-
111
92
  float_feats = [f for f in self.config.features if f.dtype == dm.float]
112
93
  discrete_feats = [f for f in self.config.features if f.dtype != dm.float]
113
94
 
@@ -130,6 +111,26 @@ class HyperGridSampler(BaseSampler):
130
111
  return result
131
112
 
132
113
  def sample(self, n_samples: int) -> np.ndarray:
114
+ """
115
+ Parameters
116
+ ----------
117
+ n_samples : int
118
+ Total number of samples to generate.
119
+
120
+ Returns
121
+ -------
122
+ np.ndarray
123
+ Array of shape (n_samples, n_features) containing the generated samples.
124
+ The column order matches the input feature configuration.
125
+
126
+ Notes
127
+ -----
128
+ - Float features are scaled to their respective [low, high] ranges.
129
+ - Integer features are sampled from the inclusive range [low, high].
130
+ - Categorical features are sampled from the provided category list.
131
+ - Constant features return the same value for all samples.
132
+ """
133
+
133
134
  with spinning():
134
135
  samples = self._sample(n_samples)
135
136
  return samples
@@ -35,7 +35,7 @@ def validate_values(value: Numeric):
35
35
  if value < 0:
36
36
  warnings.warn("Value should be non-negative", ConstraintWarning)
37
37
 
38
- def validate_range(low: Numeric, high: Numeric):
38
+ def validate_range(low: Numeric, high: Numeric, step: Numeric = 0):
39
39
  if not isinstance(low, Numeric):
40
40
  raise ConstraintTypeError("low must be a numeric")
41
41
 
@@ -45,4 +45,6 @@ def validate_range(low: Numeric, high: Numeric):
45
45
  if low > high:
46
46
  raise ConstraintValidationError("low must be <= high")
47
47
 
48
+ if step < 0:
49
+ raise ConstraintValidationError("step must be non-negative")
48
50
 
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mlsampler
3
- Version: 0.3.3
4
- Summary: Flexible constrained random sampler for reverse analysis in machine learning
3
+ Version: 0.3.4
4
+ Summary: Flexible constrained random sampler for doe sampling or reverse analysis in machine learning
5
5
  Author: Jugai O
6
6
  License: LICENSE
7
7
  Requires-Python: >=3.11
@@ -9,6 +9,10 @@ Description-Content-Type: text/markdown
9
9
  License-File: LICENSE
10
10
  Requires-Dist: numpy
11
11
  Requires-Dist: joblib
12
+ Provides-Extra: docs
13
+ Requires-Dist: sphinx; extra == "docs"
14
+ Requires-Dist: sphinx-rtd-theme; extra == "docs"
15
+ Requires-Dist: scipy; extra == "docs"
12
16
  Dynamic: license-file
13
17
 
14
18
  # RandomSampler
@@ -0,0 +1,7 @@
1
+ numpy
2
+ joblib
3
+
4
+ [docs]
5
+ sphinx
6
+ sphinx-rtd-theme
7
+ scipy
@@ -1,2 +0,0 @@
1
- numpy
2
- joblib
File without changes
File without changes
File without changes