fakerforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fakerforge/__init__.py +7 -0
- fakerforge/constraints/__init__.py +5 -0
- fakerforge/constraints/base.py +35 -0
- fakerforge/distributions/__init__.py +6 -0
- fakerforge/distributions/base.py +26 -0
- fakerforge/distributions/engine.py +335 -0
- fakerforge/faker.py +200 -0
- fakerforge/generator/__init__.py +5 -0
- fakerforge/generator/generator.py +68 -0
- fakerforge/providers/__init__.py +6 -0
- fakerforge/providers/base.py +16 -0
- fakerforge/providers/finance.py +112 -0
- fakerforge/py.typed +0 -0
- fakerforge/schema/__init__.py +25 -0
- fakerforge/schema/dataset.py +813 -0
- fakerforge/schema/generate.py +506 -0
- fakerforge/schema/result.py +137 -0
- fakerforge/schema/schema.py +59 -0
- fakerforge/validation/__init__.py +14 -0
- fakerforge/validation/dataset.py +438 -0
- fakerforge/validation/report.py +65 -0
- fakerforge/validation/validator.py +82 -0
- fakerforge-0.1.0.dist-info/METADATA +246 -0
- fakerforge-0.1.0.dist-info/RECORD +27 -0
- fakerforge-0.1.0.dist-info/WHEEL +5 -0
- fakerforge-0.1.0.dist-info/licenses/LICENSE +21 -0
- fakerforge-0.1.0.dist-info/top_level.txt +1 -0
fakerforge/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Base class for value constraints."""
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Constraint(ABC):
|
|
8
|
+
"""A predicate that accepts or rejects a single generated value.
|
|
9
|
+
|
|
10
|
+
Subclasses implement :meth:`check`. Built-in constraint types are not
|
|
11
|
+
part of this release.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
@abstractmethod
|
|
15
|
+
def check(self, value: Any) -> bool:
|
|
16
|
+
"""Return whether ``value`` satisfies this constraint.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
value: Generated field value.
|
|
20
|
+
|
|
21
|
+
Returns:
|
|
22
|
+
``True`` when the value is accepted.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def message(self, field: str, value: Any) -> str:
|
|
26
|
+
"""Describe a failed check.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
field: Record field name.
|
|
30
|
+
value: Value that failed :meth:`check`.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
A short failure message.
|
|
34
|
+
"""
|
|
35
|
+
return f"{field!r} failed {type(self).__name__}: {value!r}"
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Base class for sampling distributions."""
|
|
2
|
+
|
|
3
|
+
import random
|
|
4
|
+
from abc import ABC, abstractmethod
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Distribution(ABC):
|
|
9
|
+
"""A reproducible sampling strategy.
|
|
10
|
+
|
|
11
|
+
Implementations must draw randomness only from the :class:`random.Random`
|
|
12
|
+
instance passed to :meth:`sample`. Built-in uniform, normal, log-normal,
|
|
13
|
+
and categorical sampling lives on
|
|
14
|
+
:class:`fakerforge.distributions.DistributionEngine`.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
def sample(self, rng: random.Random) -> Any:
|
|
19
|
+
"""Draw one value.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
rng: Random instance owned by the caller.
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
The sampled value.
|
|
26
|
+
"""
|
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
"""Seeded sampling for numeric and categorical values.
|
|
2
|
+
|
|
3
|
+
``DistributionEngine`` is the API the schema generator should call. It keeps
|
|
4
|
+
its own random stream, seeded from the same value as ``FakerForge``, so
|
|
5
|
+
distribution draws stay reproducible and do not consume Faker provider draws.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import hashlib
|
|
9
|
+
import math
|
|
10
|
+
import random
|
|
11
|
+
from collections.abc import Sequence
|
|
12
|
+
from importlib import import_module
|
|
13
|
+
from types import ModuleType
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
# Matches ``fakerforge.faker.Seed`` without importing that module.
|
|
17
|
+
Seed = int | float | str | bytes | bytearray | None
|
|
18
|
+
|
|
19
|
+
MAX_REJECTION_ATTEMPTS = 1000
|
|
20
|
+
|
|
21
|
+
_NUMERIC = ("uniform", "normal", "lognormal")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class DistributionError(ValueError):
|
|
25
|
+
"""Raised when sampling cannot satisfy a hard bound.
|
|
26
|
+
|
|
27
|
+
Normal and log-normal bounds are inclusive. Draws outside the bounds are
|
|
28
|
+
discarded and sampled again. They are never clipped to ``min`` or ``max``.
|
|
29
|
+
This error means every attempt fell outside the bounds.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class DistributionEngine:
|
|
34
|
+
"""Sample uniform, normal, log-normal, and categorical values.
|
|
35
|
+
|
|
36
|
+
NumPy is used when it is installed. Without NumPy, the engine samples
|
|
37
|
+
with :mod:`random`. The two backends are each deterministic for a seed,
|
|
38
|
+
and they do not produce the same series.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
seed: Seed for this engine's random stream. ``None`` draws entropy
|
|
42
|
+
from the operating system.
|
|
43
|
+
use_numpy: ``True`` requires NumPy, ``False`` uses the standard
|
|
44
|
+
library, and ``None`` uses NumPy when it imports.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(
|
|
48
|
+
self, seed: Seed | None = None, *, use_numpy: bool | None = None
|
|
49
|
+
) -> None:
|
|
50
|
+
numpy = _import_numpy()
|
|
51
|
+
if use_numpy is True and numpy is None:
|
|
52
|
+
raise ImportError(
|
|
53
|
+
"NumPy is required for this distribution engine. "
|
|
54
|
+
"Install it with: pip install 'fakerforge[numpy]'"
|
|
55
|
+
)
|
|
56
|
+
self._numpy = numpy if use_numpy is not False else None
|
|
57
|
+
self._py_rng = random.Random(seed)
|
|
58
|
+
self._np_rng: Any = _numpy_generator(self._numpy, seed) if self._numpy else None
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def uses_numpy(self) -> bool:
|
|
62
|
+
"""Return whether sampling is going through NumPy."""
|
|
63
|
+
return self._numpy is not None
|
|
64
|
+
|
|
65
|
+
def reseed(self, seed: Seed | None = None) -> None:
|
|
66
|
+
"""Replace this engine's random stream.
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
seed: New seed. ``None`` draws entropy from the operating system.
|
|
70
|
+
"""
|
|
71
|
+
self._py_rng.seed(seed)
|
|
72
|
+
if self._numpy is not None:
|
|
73
|
+
self._np_rng = _numpy_generator(self._numpy, seed)
|
|
74
|
+
|
|
75
|
+
def number(
|
|
76
|
+
self,
|
|
77
|
+
*,
|
|
78
|
+
distribution: str,
|
|
79
|
+
mean: float | None = None,
|
|
80
|
+
std: float | None = None,
|
|
81
|
+
min: float | None = None,
|
|
82
|
+
max: float | None = None,
|
|
83
|
+
) -> float:
|
|
84
|
+
"""Draw one float from a numeric distribution.
|
|
85
|
+
|
|
86
|
+
``min`` and ``max`` are inclusive hard bounds for normal and
|
|
87
|
+
log-normal sampling. A draw outside those bounds is thrown away and
|
|
88
|
+
another draw is taken, up to ``MAX_REJECTION_ATTEMPTS``. The engine
|
|
89
|
+
does not clip a value to the nearest bound. If every attempt misses,
|
|
90
|
+
this method raises :class:`DistributionError`.
|
|
91
|
+
|
|
92
|
+
Uniform samples lie in the half-open interval ``[min, max)``. When
|
|
93
|
+
``min == max``, uniform returns that value.
|
|
94
|
+
|
|
95
|
+
Log-normal ``mean`` and ``std`` are the mean and standard deviation
|
|
96
|
+
of the underlying normal distribution, matching NumPy's
|
|
97
|
+
``Generator.lognormal``. Every untruncated log-normal sample is
|
|
98
|
+
greater than zero. ``min`` must be ``>= 0`` and ``max`` must be
|
|
99
|
+
``> 0``.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
distribution: ``"uniform"``, ``"normal"``, ``"lognormal"``, or
|
|
103
|
+
``"log-normal"``.
|
|
104
|
+
mean: Required for normal and log-normal. Not accepted for uniform.
|
|
105
|
+
std: Required for normal and log-normal, and must be ``> 0``.
|
|
106
|
+
Not accepted for uniform.
|
|
107
|
+
min: Lower inclusive bound. Required for uniform.
|
|
108
|
+
max: Upper inclusive bound for normal and log-normal. Required
|
|
109
|
+
for uniform, where the upper end is exclusive unless it
|
|
110
|
+
equals ``min``.
|
|
111
|
+
|
|
112
|
+
Returns:
|
|
113
|
+
A Python ``float``.
|
|
114
|
+
|
|
115
|
+
Raises:
|
|
116
|
+
ValueError: The distribution name or parameters are invalid.
|
|
117
|
+
DistributionError: Normal or log-normal rejection sampling could
|
|
118
|
+
not find a value inside the bounds.
|
|
119
|
+
"""
|
|
120
|
+
kind = _distribution_name(distribution)
|
|
121
|
+
lower = _optional_float(min, "min")
|
|
122
|
+
upper = _optional_float(max, "max")
|
|
123
|
+
_require_ordered(lower, upper)
|
|
124
|
+
if kind == "uniform":
|
|
125
|
+
_reject_unused(mean, "mean", kind)
|
|
126
|
+
_reject_unused(std, "std", kind)
|
|
127
|
+
if lower is None or upper is None:
|
|
128
|
+
raise ValueError("uniform requires min and max")
|
|
129
|
+
if lower == upper:
|
|
130
|
+
return lower
|
|
131
|
+
return self._uniform(lower, upper)
|
|
132
|
+
center = _required_float(mean, "mean")
|
|
133
|
+
scale = _required_float(std, "std")
|
|
134
|
+
if scale <= 0:
|
|
135
|
+
raise ValueError("std must be greater than 0")
|
|
136
|
+
if lower is not None and upper is not None and lower == upper:
|
|
137
|
+
raise ValueError(
|
|
138
|
+
f"{kind} bounds must span a range; values are not clipped to a point"
|
|
139
|
+
)
|
|
140
|
+
if kind == "lognormal":
|
|
141
|
+
_validate_lognormal_bounds(lower, upper)
|
|
142
|
+
draw = (
|
|
143
|
+
self._normal_draw(center, scale)
|
|
144
|
+
if kind == "normal"
|
|
145
|
+
else self._lognormal_draw(center, scale)
|
|
146
|
+
)
|
|
147
|
+
return _draw_until_in_bounds(draw, lower, upper, kind)
|
|
148
|
+
|
|
149
|
+
def categorical(
|
|
150
|
+
self,
|
|
151
|
+
values: Sequence[Any],
|
|
152
|
+
weights: Sequence[float] | None = None,
|
|
153
|
+
) -> Any:
|
|
154
|
+
"""Draw one of ``values``.
|
|
155
|
+
|
|
156
|
+
Omit ``weights`` for a uniform categorical distribution. Weights do
|
|
157
|
+
not need to sum to 1; they are normalized. Selection follows the
|
|
158
|
+
position in ``values``, so duplicate labels stay distinct.
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
values: Non-empty sequence of choices. Strings are not treated
|
|
162
|
+
as sequences of characters.
|
|
163
|
+
weights: Non-negative weights, one per value. At least one
|
|
164
|
+
weight must be positive.
|
|
165
|
+
|
|
166
|
+
Returns:
|
|
167
|
+
The chosen element of ``values``, unchanged.
|
|
168
|
+
|
|
169
|
+
Raises:
|
|
170
|
+
ValueError: ``values`` or ``weights`` are empty, misaligned, or
|
|
171
|
+
not a usable weight vector.
|
|
172
|
+
TypeError: A weight is not numeric.
|
|
173
|
+
"""
|
|
174
|
+
choices = _choices(values)
|
|
175
|
+
if weights is None:
|
|
176
|
+
index = self._weighted_index(len(choices), None)
|
|
177
|
+
else:
|
|
178
|
+
index = self._weighted_index(len(choices), _weights(weights, len(choices)))
|
|
179
|
+
return choices[index]
|
|
180
|
+
|
|
181
|
+
def _uniform(self, low: float, high: float) -> float:
|
|
182
|
+
if self._np_rng is not None:
|
|
183
|
+
return float(self._np_rng.uniform(low, high))
|
|
184
|
+
return low + (high - low) * self._py_rng.random()
|
|
185
|
+
|
|
186
|
+
def _normal_draw(self, mean: float, std: float) -> Any:
|
|
187
|
+
if self._np_rng is not None:
|
|
188
|
+
rng = self._np_rng
|
|
189
|
+
return lambda: float(rng.normal(mean, std))
|
|
190
|
+
rng_py = self._py_rng
|
|
191
|
+
return lambda: rng_py.normalvariate(mean, std)
|
|
192
|
+
|
|
193
|
+
def _lognormal_draw(self, mean: float, std: float) -> Any:
|
|
194
|
+
if self._np_rng is not None:
|
|
195
|
+
rng = self._np_rng
|
|
196
|
+
return lambda: float(rng.lognormal(mean, std))
|
|
197
|
+
rng_py = self._py_rng
|
|
198
|
+
return lambda: rng_py.lognormvariate(mean, std)
|
|
199
|
+
|
|
200
|
+
def _weighted_index(self, count: int, weights: list[float] | None) -> int:
|
|
201
|
+
numpy = self._numpy
|
|
202
|
+
if self._np_rng is not None and numpy is not None:
|
|
203
|
+
probabilities = (
|
|
204
|
+
None if weights is None else _numpy_probabilities(numpy, weights)
|
|
205
|
+
)
|
|
206
|
+
return int(self._np_rng.choice(count, p=probabilities))
|
|
207
|
+
if weights is None:
|
|
208
|
+
return self._py_rng.randrange(count)
|
|
209
|
+
return self._py_rng.choices(range(count), weights=weights, k=1)[0]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _draw_until_in_bounds(
|
|
213
|
+
draw: Any,
|
|
214
|
+
lower: float | None,
|
|
215
|
+
upper: float | None,
|
|
216
|
+
distribution: str,
|
|
217
|
+
) -> float:
|
|
218
|
+
for _ in range(MAX_REJECTION_ATTEMPTS):
|
|
219
|
+
value = float(draw())
|
|
220
|
+
if _inside(value, lower, upper):
|
|
221
|
+
return value
|
|
222
|
+
raise DistributionError(
|
|
223
|
+
f"{distribution} sample fell outside {_bound_label(lower, upper)} on "
|
|
224
|
+
f"{MAX_REJECTION_ATTEMPTS} consecutive draws. Bounds are inclusive "
|
|
225
|
+
"and out-of-range draws are discarded, not clipped."
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _inside(value: float, lower: float | None, upper: float | None) -> bool:
|
|
230
|
+
if lower is not None and value < lower:
|
|
231
|
+
return False
|
|
232
|
+
if upper is not None and value > upper:
|
|
233
|
+
return False
|
|
234
|
+
return True
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _bound_label(lower: float | None, upper: float | None) -> str:
|
|
238
|
+
low = "-inf" if lower is None else repr(lower)
|
|
239
|
+
high = "inf" if upper is None else repr(upper)
|
|
240
|
+
return f"[{low}, {high}]"
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _distribution_name(distribution: str) -> str:
|
|
244
|
+
if not isinstance(distribution, str):
|
|
245
|
+
raise TypeError("distribution must be a string")
|
|
246
|
+
name = distribution.strip().lower().replace("-", "")
|
|
247
|
+
if name not in _NUMERIC:
|
|
248
|
+
allowed = "uniform, normal, lognormal"
|
|
249
|
+
raise ValueError(
|
|
250
|
+
f"unknown distribution {distribution!r}; expected one of {allowed}"
|
|
251
|
+
)
|
|
252
|
+
return name
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _optional_float(value: object, label: str) -> float | None:
|
|
256
|
+
if value is None:
|
|
257
|
+
return None
|
|
258
|
+
return _finite_float(value, label)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _required_float(value: object, label: str) -> float:
|
|
262
|
+
if value is None:
|
|
263
|
+
raise ValueError(f"{label} is required")
|
|
264
|
+
return _finite_float(value, label)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _finite_float(value: object, label: str) -> float:
|
|
268
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
269
|
+
raise TypeError(f"{label} must be an int or float")
|
|
270
|
+
number = float(value)
|
|
271
|
+
if not math.isfinite(number):
|
|
272
|
+
raise ValueError(f"{label} must be finite")
|
|
273
|
+
return number
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _reject_unused(value: object, label: str, distribution: str) -> None:
|
|
277
|
+
if value is not None:
|
|
278
|
+
raise ValueError(f"{distribution} does not accept {label}")
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _require_ordered(lower: float | None, upper: float | None) -> None:
|
|
282
|
+
if lower is not None and upper is not None and lower > upper:
|
|
283
|
+
raise ValueError("min must be less than or equal to max")
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _validate_lognormal_bounds(lower: float | None, upper: float | None) -> None:
|
|
287
|
+
if lower is not None and lower < 0:
|
|
288
|
+
raise ValueError("lognormal min must be greater than or equal to 0")
|
|
289
|
+
if upper is not None and upper <= 0:
|
|
290
|
+
raise ValueError("lognormal max must be greater than 0")
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _choices(values: Sequence[Any]) -> Sequence[Any]:
|
|
294
|
+
if isinstance(values, (str, bytes)) or not isinstance(values, Sequence):
|
|
295
|
+
raise TypeError("values must be a sequence of choices")
|
|
296
|
+
if len(values) == 0:
|
|
297
|
+
raise ValueError("values must not be empty")
|
|
298
|
+
return values
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _weights(weights: Sequence[float], count: int) -> list[float]:
|
|
302
|
+
if isinstance(weights, (str, bytes)) or not isinstance(weights, Sequence):
|
|
303
|
+
raise TypeError("weights must be a sequence of numbers")
|
|
304
|
+
if len(weights) != count:
|
|
305
|
+
raise ValueError("weights must have the same length as values")
|
|
306
|
+
parsed = [_finite_float(weight, "weight") for weight in weights]
|
|
307
|
+
if any(weight < 0 for weight in parsed):
|
|
308
|
+
raise ValueError("weights must be greater than or equal to 0")
|
|
309
|
+
if sum(parsed) <= 0:
|
|
310
|
+
raise ValueError("at least one weight must be greater than 0")
|
|
311
|
+
return parsed
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _numpy_probabilities(numpy: ModuleType, weights: list[float]) -> Any:
|
|
315
|
+
probabilities = numpy.asarray(weights, dtype=float)
|
|
316
|
+
total = float(probabilities.sum())
|
|
317
|
+
return probabilities / total
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _import_numpy() -> ModuleType | None:
|
|
321
|
+
try:
|
|
322
|
+
return import_module("numpy")
|
|
323
|
+
except ImportError:
|
|
324
|
+
return None
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _numpy_generator(numpy: ModuleType, seed: Seed | None) -> Any:
|
|
328
|
+
return numpy.random.default_rng(_numpy_seed(seed))
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _numpy_seed(seed: Seed | None) -> int | None:
|
|
332
|
+
if seed is None:
|
|
333
|
+
return None
|
|
334
|
+
digest = hashlib.sha256(repr(seed).encode("utf-8")).digest()
|
|
335
|
+
return int.from_bytes(digest[:8], "big")
|
fakerforge/faker.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""Faker-compatible entry point."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping, Sequence
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from faker import Faker
|
|
7
|
+
from faker.providers import BaseProvider
|
|
8
|
+
|
|
9
|
+
from fakerforge.distributions.engine import DistributionEngine
|
|
10
|
+
from fakerforge.providers.finance import FinanceProvider
|
|
11
|
+
from fakerforge.schema.dataset import DatabaseSchema, DatasetSchema
|
|
12
|
+
from fakerforge.schema.generate import generate_dataset
|
|
13
|
+
from fakerforge.schema.result import DatabaseResult, DatasetResult
|
|
14
|
+
|
|
15
|
+
# Matches ``faker.typing.SeedType`` without depending on a private module.
|
|
16
|
+
Seed = int | float | str | bytes | bytearray | None
|
|
17
|
+
|
|
18
|
+
# Registered on every instance. Later ``add_provider`` calls take precedence.
|
|
19
|
+
_BUILTIN_PROVIDERS: tuple[type[BaseProvider], ...] = (FinanceProvider,)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class FakerForge:
|
|
23
|
+
"""Wrapper around a :class:`faker.Faker` instance.
|
|
24
|
+
|
|
25
|
+
Attributes and methods that FakerForge does not define are delegated to
|
|
26
|
+
the underlying Faker instance. Built-in domain providers, including
|
|
27
|
+
finance, are registered with ``add_provider`` and use that instance's
|
|
28
|
+
random generator. Numeric and categorical distributions use a separate
|
|
29
|
+
:class:`~fakerforge.distributions.DistributionEngine` seeded with the
|
|
30
|
+
same value. Seeding uses ``seed_instance``, so it affects only this
|
|
31
|
+
object and leaves Faker's shared random state alone.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
locale: Locale or sequence of locales forwarded to Faker. ``None``
|
|
35
|
+
selects Faker's default locale.
|
|
36
|
+
seed: Seed forwarded to ``seed_instance``. ``None`` leaves this
|
|
37
|
+
instance unseeded.
|
|
38
|
+
providers: Extra providers registered after the built-in providers.
|
|
39
|
+
Each item may be a provider class or an already constructed
|
|
40
|
+
provider. A provider added here overrides a built-in method of
|
|
41
|
+
the same name.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
def __init__(
|
|
45
|
+
self,
|
|
46
|
+
locale: str | Sequence[str] | None = None,
|
|
47
|
+
*,
|
|
48
|
+
seed: Seed | None = None,
|
|
49
|
+
providers: Sequence[type[BaseProvider] | BaseProvider] | None = None,
|
|
50
|
+
) -> None:
|
|
51
|
+
if locale is None:
|
|
52
|
+
self._faker = Faker()
|
|
53
|
+
else:
|
|
54
|
+
self._faker = Faker(locale)
|
|
55
|
+
self._seed = seed
|
|
56
|
+
self._distributions = DistributionEngine(seed)
|
|
57
|
+
if seed is not None:
|
|
58
|
+
self._faker.seed_instance(seed)
|
|
59
|
+
registered: tuple[type[BaseProvider] | BaseProvider, ...] = _BUILTIN_PROVIDERS
|
|
60
|
+
if providers:
|
|
61
|
+
registered = (*registered, *providers)
|
|
62
|
+
for provider in registered:
|
|
63
|
+
self.add_provider(provider)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def faker(self) -> Faker:
|
|
67
|
+
"""Return the wrapped Faker instance."""
|
|
68
|
+
return self._faker
|
|
69
|
+
|
|
70
|
+
def seed_instance(self, seed: Seed | None = None) -> None:
|
|
71
|
+
"""Reseed this instance only.
|
|
72
|
+
|
|
73
|
+
This reseeds both the wrapped Faker generator and the distribution
|
|
74
|
+
engine.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
seed: Seed forwarded to ``Faker.seed_instance`` and the
|
|
78
|
+
distribution engine.
|
|
79
|
+
"""
|
|
80
|
+
self._seed = seed
|
|
81
|
+
self._faker.seed_instance(seed)
|
|
82
|
+
self._distributions.reseed(seed)
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def distributions(self) -> DistributionEngine:
|
|
86
|
+
"""Return the seeded distribution engine.
|
|
87
|
+
|
|
88
|
+
Schema generation should sample through this object so numeric draws
|
|
89
|
+
stay on the engine's random stream.
|
|
90
|
+
"""
|
|
91
|
+
return self._distributions
|
|
92
|
+
|
|
93
|
+
def number(
|
|
94
|
+
self,
|
|
95
|
+
*,
|
|
96
|
+
distribution: str,
|
|
97
|
+
mean: float | None = None,
|
|
98
|
+
std: float | None = None,
|
|
99
|
+
min: float | None = None,
|
|
100
|
+
max: float | None = None,
|
|
101
|
+
) -> float:
|
|
102
|
+
"""Draw one float from ``distribution``.
|
|
103
|
+
|
|
104
|
+
See :meth:`DistributionEngine.number` for parameters, inclusive
|
|
105
|
+
normal and log-normal bounds, and the rejection rule. Out-of-range
|
|
106
|
+
normal and log-normal draws are discarded, not clipped.
|
|
107
|
+
|
|
108
|
+
Args:
|
|
109
|
+
distribution: ``"uniform"``, ``"normal"``, ``"lognormal"``, or
|
|
110
|
+
``"log-normal"``.
|
|
111
|
+
mean: Required for normal and log-normal.
|
|
112
|
+
std: Required for normal and log-normal, and must be ``> 0``.
|
|
113
|
+
min: Lower bound. Required for uniform.
|
|
114
|
+
max: Upper bound. Required for uniform.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
A Python ``float``.
|
|
118
|
+
"""
|
|
119
|
+
return self._distributions.number(
|
|
120
|
+
distribution=distribution,
|
|
121
|
+
mean=mean,
|
|
122
|
+
std=std,
|
|
123
|
+
min=min,
|
|
124
|
+
max=max,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
def categorical(
|
|
128
|
+
self,
|
|
129
|
+
values: Sequence[Any],
|
|
130
|
+
weights: Sequence[float] | None = None,
|
|
131
|
+
) -> Any:
|
|
132
|
+
"""Draw one of ``values``, optionally with weights.
|
|
133
|
+
|
|
134
|
+
Args:
|
|
135
|
+
values: Non-empty sequence of choices.
|
|
136
|
+
weights: Non-negative weights, one per value. Omit them for a
|
|
137
|
+
uniform categorical draw.
|
|
138
|
+
|
|
139
|
+
Returns:
|
|
140
|
+
The chosen element of ``values``.
|
|
141
|
+
"""
|
|
142
|
+
return self._distributions.categorical(values, weights)
|
|
143
|
+
|
|
144
|
+
def generate(
|
|
145
|
+
self,
|
|
146
|
+
schema: Mapping[str, Any] | DatasetSchema | DatabaseSchema,
|
|
147
|
+
*,
|
|
148
|
+
rows: int | Mapping[str, int] | None = None,
|
|
149
|
+
) -> DatasetResult | DatabaseResult:
|
|
150
|
+
"""Generate a dataset from a declarative schema.
|
|
151
|
+
|
|
152
|
+
The schema is validated before any row is produced. Integer ``min``
|
|
153
|
+
and ``max`` are inclusive. Float fields use the half-open uniform
|
|
154
|
+
interval ``[min, max)``. ``nullable`` fields are ``None`` with
|
|
155
|
+
probability ``null_probability`` (default ``0.1``). Unique fields
|
|
156
|
+
never repeat a non-null value; nulls may repeat. Derived fields are
|
|
157
|
+
computed from an earlier field. A date that depends on a date is
|
|
158
|
+
drawn on or after that date. Foreign keys are copied from parent
|
|
159
|
+
rows that were already generated.
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
schema: A field mapping, a parsed dataset, or a relational
|
|
163
|
+
mapping of tables.
|
|
164
|
+
rows: Row count for a single table, or a mapping of table name
|
|
165
|
+
to row count. Relational tables may instead set ``rows``
|
|
166
|
+
on each table.
|
|
167
|
+
|
|
168
|
+
Returns:
|
|
169
|
+
A ``DatasetResult`` for one table, or a ``DatabaseResult`` whose
|
|
170
|
+
keys are table names. Both expose ``validate()``. ``rows`` holds
|
|
171
|
+
the records, and ``frame`` is a DataFrame when pandas is installed.
|
|
172
|
+
|
|
173
|
+
Raises:
|
|
174
|
+
ValueError: ``rows`` is missing or not a non-negative integer.
|
|
175
|
+
SchemaError: The schema is invalid for this instance.
|
|
176
|
+
GenerationError: A unique column or foreign key cannot be filled
|
|
177
|
+
without inventing a value.
|
|
178
|
+
"""
|
|
179
|
+
return generate_dataset(self, schema, rows=rows)
|
|
180
|
+
|
|
181
|
+
def add_provider(self, provider: type[BaseProvider] | BaseProvider) -> None:
|
|
182
|
+
"""Register a Faker provider on the underlying instance.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
provider: Provider class or instance accepted by
|
|
186
|
+
``Faker.add_provider``.
|
|
187
|
+
"""
|
|
188
|
+
self._faker.add_provider(provider)
|
|
189
|
+
|
|
190
|
+
def __getattr__(self, name: str) -> Any:
|
|
191
|
+
"""Delegate unknown attribute lookup to the wrapped Faker instance."""
|
|
192
|
+
return getattr(self._faker, name)
|
|
193
|
+
|
|
194
|
+
def __dir__(self) -> list[str]:
|
|
195
|
+
"""Include delegated Faker attributes in interactive completion."""
|
|
196
|
+
return sorted(set(super().__dir__()) | set(dir(self._faker)))
|
|
197
|
+
|
|
198
|
+
def __repr__(self) -> str:
|
|
199
|
+
locales = getattr(self._faker, "locales", None)
|
|
200
|
+
return f"FakerForge(locales={locales!r}, seed={self._seed!r})"
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Generate records by calling Faker provider methods."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from fakerforge.faker import FakerForge
|
|
6
|
+
from fakerforge.schema.schema import Schema
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Generator:
|
|
10
|
+
"""Build records from a :class:`~fakerforge.schema.Schema`.
|
|
11
|
+
|
|
12
|
+
Each field calls the named provider on the supplied
|
|
13
|
+
:class:`~fakerforge.FakerForge`. The generator does not reseed; pass a
|
|
14
|
+
seeded forge when results must be reproducible.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
forge: FakerForge instance used to resolve provider methods.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(self, forge: FakerForge) -> None:
|
|
21
|
+
if not isinstance(forge, FakerForge):
|
|
22
|
+
raise TypeError("forge must be a FakerForge instance")
|
|
23
|
+
self._forge = forge
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def forge(self) -> FakerForge:
|
|
27
|
+
"""Return the forge used to populate records."""
|
|
28
|
+
return self._forge
|
|
29
|
+
|
|
30
|
+
def generate_one(self, schema: Schema) -> dict[str, Any]:
|
|
31
|
+
"""Generate a single record.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
schema: Fields to populate.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
A dictionary keyed by field name.
|
|
38
|
+
|
|
39
|
+
Raises:
|
|
40
|
+
TypeError: ``schema`` is not a :class:`Schema`, or a provider
|
|
41
|
+
attribute is not callable.
|
|
42
|
+
"""
|
|
43
|
+
if not isinstance(schema, Schema):
|
|
44
|
+
raise TypeError("schema must be a Schema instance")
|
|
45
|
+
record: dict[str, Any] = {}
|
|
46
|
+
for column in schema.fields:
|
|
47
|
+
method = getattr(self._forge, column.provider)
|
|
48
|
+
if not callable(method):
|
|
49
|
+
raise TypeError(f"provider {column.provider!r} is not callable")
|
|
50
|
+
record[column.name] = method(**dict(column.kwargs))
|
|
51
|
+
return record
|
|
52
|
+
|
|
53
|
+
def generate(self, schema: Schema, *, count: int) -> list[dict[str, Any]]:
|
|
54
|
+
"""Generate ``count`` records.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
schema: Fields to populate.
|
|
58
|
+
count: Number of records. Zero yields an empty list.
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
Records in generation order.
|
|
62
|
+
|
|
63
|
+
Raises:
|
|
64
|
+
ValueError: ``count`` is not a non-negative integer.
|
|
65
|
+
"""
|
|
66
|
+
if isinstance(count, bool) or not isinstance(count, int) or count < 0:
|
|
67
|
+
raise ValueError("count must be a non-negative integer")
|
|
68
|
+
return [self.generate_one(schema) for _ in range(count)]
|