fakerforge 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fakerforge-0.1.0/CHANGELOG.md +16 -0
- fakerforge-0.1.0/LICENSE +21 -0
- fakerforge-0.1.0/MANIFEST.in +4 -0
- fakerforge-0.1.0/PKG-INFO +246 -0
- fakerforge-0.1.0/README.md +213 -0
- fakerforge-0.1.0/docs/architecture.md +31 -0
- fakerforge-0.1.0/examples/quickstart.py +42 -0
- fakerforge-0.1.0/pyproject.toml +76 -0
- fakerforge-0.1.0/setup.cfg +4 -0
- fakerforge-0.1.0/src/fakerforge/__init__.py +7 -0
- fakerforge-0.1.0/src/fakerforge/constraints/__init__.py +5 -0
- fakerforge-0.1.0/src/fakerforge/constraints/base.py +35 -0
- fakerforge-0.1.0/src/fakerforge/distributions/__init__.py +6 -0
- fakerforge-0.1.0/src/fakerforge/distributions/base.py +26 -0
- fakerforge-0.1.0/src/fakerforge/distributions/engine.py +335 -0
- fakerforge-0.1.0/src/fakerforge/faker.py +200 -0
- fakerforge-0.1.0/src/fakerforge/generator/__init__.py +5 -0
- fakerforge-0.1.0/src/fakerforge/generator/generator.py +68 -0
- fakerforge-0.1.0/src/fakerforge/providers/__init__.py +6 -0
- fakerforge-0.1.0/src/fakerforge/providers/base.py +16 -0
- fakerforge-0.1.0/src/fakerforge/providers/finance.py +112 -0
- fakerforge-0.1.0/src/fakerforge/py.typed +0 -0
- fakerforge-0.1.0/src/fakerforge/schema/__init__.py +25 -0
- fakerforge-0.1.0/src/fakerforge/schema/dataset.py +813 -0
- fakerforge-0.1.0/src/fakerforge/schema/generate.py +506 -0
- fakerforge-0.1.0/src/fakerforge/schema/result.py +137 -0
- fakerforge-0.1.0/src/fakerforge/schema/schema.py +59 -0
- fakerforge-0.1.0/src/fakerforge/validation/__init__.py +14 -0
- fakerforge-0.1.0/src/fakerforge/validation/dataset.py +438 -0
- fakerforge-0.1.0/src/fakerforge/validation/report.py +65 -0
- fakerforge-0.1.0/src/fakerforge/validation/validator.py +82 -0
- fakerforge-0.1.0/src/fakerforge.egg-info/SOURCES.txt +42 -0
- fakerforge-0.1.0/tests/test_constraints.py +26 -0
- fakerforge-0.1.0/tests/test_dataset_schema.py +224 -0
- fakerforge-0.1.0/tests/test_dataset_validation.py +480 -0
- fakerforge-0.1.0/tests/test_distribution_engine.py +215 -0
- fakerforge-0.1.0/tests/test_distributions.py +29 -0
- fakerforge-0.1.0/tests/test_faker.py +74 -0
- fakerforge-0.1.0/tests/test_finance.py +127 -0
- fakerforge-0.1.0/tests/test_generator.py +57 -0
- fakerforge-0.1.0/tests/test_packaging.py +9 -0
- fakerforge-0.1.0/tests/test_providers.py +30 -0
- fakerforge-0.1.0/tests/test_relationships.py +273 -0
- fakerforge-0.1.0/tests/test_schema.py +44 -0
- fakerforge-0.1.0/tests/test_validator.py +46 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 - 2026-09-29
|
|
4
|
+
|
|
5
|
+
First public release. Requires Python 3.10 or newer.
|
|
6
|
+
|
|
7
|
+
- Add `FakerForge`, a seeded wrapper that delegates unknown attributes to `faker.Faker`.
|
|
8
|
+
- Add `ForgeProvider` as a subclass of Faker's `BaseProvider`.
|
|
9
|
+
- Add abstract `Constraint` and `Distribution` bases.
|
|
10
|
+
- Add `Field`, `Schema`, and a schema `Generator` that calls Faker provider methods.
|
|
11
|
+
- Add `Validator` for per-field constraint checks.
|
|
12
|
+
- Register `FinanceProvider` on every `FakerForge` instance through Faker's `BaseProvider` mechanism.
|
|
13
|
+
- Add a seeded `DistributionEngine` with uniform, normal, log-normal, and weighted categorical sampling.
|
|
14
|
+
- Add declarative schema generation with primitive types, Faker provider references, bounds, nulls, uniqueness, and optional pandas output.
|
|
15
|
+
- Add derived fields, dependent fields, and foreign keys that preserve referential integrity.
|
|
16
|
+
- Add dataset validation for types, bounds, nulls, uniqueness, provider output, foreign keys, and row counts.
|
fakerforge-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Gore Shardul
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fakerforge
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Synthetic data generation built on the Faker library.
|
|
5
|
+
Author: Shardul
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: faker,synthetic-data,testing,fixtures
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Software Development :: Testing
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: Faker>=24.0.0
|
|
22
|
+
Provides-Extra: numpy
|
|
23
|
+
Requires-Dist: numpy>=1.24; extra == "numpy"
|
|
24
|
+
Provides-Extra: pandas
|
|
25
|
+
Requires-Dist: pandas>=2; extra == "pandas"
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
28
|
+
Requires-Dist: ruff>=0.8.0; extra == "dev"
|
|
29
|
+
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
30
|
+
Requires-Dist: numpy>=1.24; extra == "dev"
|
|
31
|
+
Requires-Dist: pandas>=2; extra == "dev"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# FakerForge
|
|
35
|
+
|
|
36
|
+
FakerForge is a Python framework for synthetic data, built on top of [Faker](https://faker.readthedocs.io/). It does not fork Faker. Custom providers are normal Faker providers, and every other Faker method stays available.
|
|
37
|
+
|
|
38
|
+
Requires Python 3.10 or newer.
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install fakerforge
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
NumPy and pandas are optional:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install "fakerforge[numpy]"
|
|
50
|
+
pip install "fakerforge[pandas]"
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Quick start
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from fakerforge import FakerForge
|
|
57
|
+
|
|
58
|
+
fake = FakerForge(seed=42)
|
|
59
|
+
|
|
60
|
+
fake.name()
|
|
61
|
+
fake.email()
|
|
62
|
+
fake.address()
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
`FakerForge` forwards unknown attributes to its `faker.Faker` instance. `seed` calls `seed_instance`, so the seed applies only to that object. Built-in providers are registered with Faker's `add_provider` and read the same random generator.
|
|
66
|
+
|
|
67
|
+
## Finance
|
|
68
|
+
|
|
69
|
+
`FinanceProvider` is registered on every `FakerForge` instance.
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from fakerforge import FakerForge
|
|
73
|
+
|
|
74
|
+
fake = FakerForge(seed=42)
|
|
75
|
+
|
|
76
|
+
fake.credit_score() # int, 300 through 850
|
|
77
|
+
fake.transaction_amount() # Decimal quantized to cents, 1.00 through 10000.00
|
|
78
|
+
fake.account_number() # 12-digit string
|
|
79
|
+
|
|
80
|
+
fake.credit_score(min_score=700, max_score=750)
|
|
81
|
+
fake.transaction_amount(min_amount=10, max_amount=25)
|
|
82
|
+
fake.account_number(length=8)
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
The same seed reproduces the same sequence. `min_score` and `max_score` are inclusive. Amount bounds are inclusive after rounding to cents.
|
|
86
|
+
|
|
87
|
+
## Distributions
|
|
88
|
+
|
|
89
|
+
`number` and `categorical` sample from a `DistributionEngine` created with the same seed as the `FakerForge` instance. That stream is separate from Faker provider calls, so a later schema generator can replay distribution draws without depending on how many names or emails were produced. `seed_instance` reseeds both streams.
|
|
90
|
+
|
|
91
|
+
NumPy is used when it is installed (`pip install fakerforge[numpy]`). Without NumPy, sampling uses the standard library. Each backend is deterministic for a given seed. They do not emit the same series.
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
from fakerforge import FakerForge
|
|
95
|
+
|
|
96
|
+
fake = FakerForge(seed=42)
|
|
97
|
+
|
|
98
|
+
fake.number(distribution="uniform", min=0, max=1)
|
|
99
|
+
fake.number(distribution="normal", mean=50, std=10, min=18, max=80)
|
|
100
|
+
fake.number(distribution="lognormal", mean=0, std=0.25, min=0.5, max=3)
|
|
101
|
+
fake.categorical(values=["A", "B", "C"])
|
|
102
|
+
fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
`min` and `max` on normal and log-normal draws are inclusive. A sample outside that interval is discarded and another sample is drawn. Values are not clipped to the boundary. If 1000 draws in a row miss the interval, `number` raises `DistributionError`.
|
|
106
|
+
|
|
107
|
+
Log-normal `mean` and `std` describe the underlying normal distribution, matching NumPy's `Generator.lognormal`. Samples are greater than zero. Uniform draws use the half-open interval `[min, max)`, except when `min == max`, which returns that value. Weights do not need to sum to 1.
|
|
108
|
+
|
|
109
|
+
## Schemas
|
|
110
|
+
|
|
111
|
+
`generate` validates a declarative schema, then builds rows. The return value is a dataset result. `rows` is the list of records. `frame` is a pandas DataFrame when pandas is installed (`pip install fakerforge[pandas]`). `validate()` checks the rows against the schema.
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from fakerforge import FakerForge
|
|
115
|
+
|
|
116
|
+
fake = FakerForge(seed=42)
|
|
117
|
+
|
|
118
|
+
schema = {
|
|
119
|
+
"customer_id": {"type": "uuid", "unique": True},
|
|
120
|
+
"name": {"provider": "person.name"},
|
|
121
|
+
"age": {"type": "integer", "min": 18, "max": 80},
|
|
122
|
+
"email": {"provider": "internet.email", "unique": True},
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
result = fake.generate(schema=schema, rows=1000)
|
|
126
|
+
report = result.validate()
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Primitive types are `string`, `integer`, `float`, `boolean`, `uuid`, and `date`. Provider references use Faker's `provider.method` form, such as `person.name`. Integer bounds are inclusive. Float bounds use the half-open interval `[min, max)`. Omitted numeric bounds default to `0..100` for integers and `0.0..1.0` for floats.
|
|
130
|
+
|
|
131
|
+
`nullable` fields are `None` with probability `null_probability`, which defaults to `0.1`. Unique columns do not repeat non-null values. Nulls may repeat. If a unique value cannot be found, generation raises `GenerationError` instead of inserting a duplicate. Dates are drawn from 1990-01-01 through 2030-12-31 so a seed stays stable.
|
|
132
|
+
|
|
133
|
+
Invalid schemas raise `SchemaError` before any row is produced. `errors` lists every problem found in that pass.
|
|
134
|
+
|
|
135
|
+
### Derived and dependent fields
|
|
136
|
+
|
|
137
|
+
`derived_from` computes a field from an earlier one. A date source produces completed years as of 2030-12-31, so the result does not change with the calendar day. `depends_on` generates the named fields first. A date that depends on a date is drawn on or after that date. Cycles are rejected.
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
schema = {
|
|
141
|
+
"date_of_birth": {"provider": "date.date_of_birth"},
|
|
142
|
+
"age": {"derived_from": "date_of_birth"},
|
|
143
|
+
"start_date": {"type": "date"},
|
|
144
|
+
"end_date": {"type": "date", "depends_on": "start_date"},
|
|
145
|
+
}
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
### Foreign keys
|
|
149
|
+
|
|
150
|
+
A relational schema maps each table to `fields` and `rows`. `references` copies a value from a unique parent column. Child rows are generated after the parent, so a foreign key is never invented. A unique foreign key uses each parent value at most once. Many-to-many relationships are not supported.
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
schema = {
|
|
154
|
+
"customers": {
|
|
155
|
+
"rows": 10,
|
|
156
|
+
"fields": {
|
|
157
|
+
"customer_id": {"type": "uuid", "unique": True},
|
|
158
|
+
"name": {"provider": "person.name"},
|
|
159
|
+
},
|
|
160
|
+
},
|
|
161
|
+
"orders": {
|
|
162
|
+
"rows": 40,
|
|
163
|
+
"fields": {
|
|
164
|
+
"order_id": {"type": "uuid", "unique": True},
|
|
165
|
+
"customer_id": {"references": "customers.customer_id"},
|
|
166
|
+
},
|
|
167
|
+
},
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
tables = fake.generate(schema=schema)
|
|
171
|
+
report = tables.validate()
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
### Validation
|
|
175
|
+
|
|
176
|
+
`validate()` checks the generated rows against the schema that produced them. The report has a `passed` or `failed` status, the row count, the violation count, and violations grouped by field. Each violation has a human-readable message. A failed report is returned; it is not raised.
|
|
177
|
+
|
|
178
|
+
Built-in checks cover type, min/max, nullable, uniqueness, known provider output, foreign keys, and row count. Subclass `DatasetCheck` and pass instances as `extra` to add further data-quality checks:
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
from fakerforge.validation import DatasetCheck, ValidationContext, Violation
|
|
182
|
+
|
|
183
|
+
class RejectPlaceholder(DatasetCheck):
|
|
184
|
+
name = "placeholder"
|
|
185
|
+
|
|
186
|
+
def run(self, context: ValidationContext) -> list[Violation]:
|
|
187
|
+
return []
|
|
188
|
+
|
|
189
|
+
report = result.validate(extra=(RejectPlaceholder(),))
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## Custom providers
|
|
193
|
+
|
|
194
|
+
Subclass `faker.providers.BaseProvider` and pass the class to `FakerForge` or `add_provider`. `ForgeProvider` is a thin subclass of that base, re-exported for convenience. Provider methods use Faker helpers such as `random_int` and `numerify`, which read the instance generator.
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
from faker.providers import BaseProvider
|
|
198
|
+
|
|
199
|
+
from fakerforge import FakerForge
|
|
200
|
+
from fakerforge.generator import Generator
|
|
201
|
+
from fakerforge.schema import Field, Schema
|
|
202
|
+
|
|
203
|
+
class StatusProvider(BaseProvider):
|
|
204
|
+
def status_code(self) -> str:
|
|
205
|
+
return self.random_element(("new", "open", "closed"))
|
|
206
|
+
|
|
207
|
+
fake = FakerForge(seed=42, providers=[StatusProvider])
|
|
208
|
+
fake.status_code()
|
|
209
|
+
|
|
210
|
+
schema = Schema(
|
|
211
|
+
fields=(
|
|
212
|
+
Field("name", "name"),
|
|
213
|
+
Field("email", "email"),
|
|
214
|
+
Field("status", "status_code"),
|
|
215
|
+
)
|
|
216
|
+
)
|
|
217
|
+
rows = Generator(fake).generate(schema, count=3)
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
|
|
221
|
+
|
|
222
|
+
## This release
|
|
223
|
+
|
|
224
|
+
| Piece | What 0.1 provides |
|
|
225
|
+
| --- | --- |
|
|
226
|
+
| Providers | Built-in `FinanceProvider`, plus `add_provider` for custom `BaseProvider` subclasses |
|
|
227
|
+
| Constraints | `Constraint` base and `Validator` |
|
|
228
|
+
| Distributions | Uniform, normal, log-normal, and weighted categorical sampling |
|
|
229
|
+
| Schema | Declarative datasets via `FakerForge.generate` |
|
|
230
|
+
| Generation | Seeded rows, with a DataFrame when pandas is installed |
|
|
231
|
+
| Validation | Dataset checks on generated rows, plus per-field `Validator` |
|
|
232
|
+
| Reproducibility | Instance seeding |
|
|
233
|
+
|
|
234
|
+
Later releases can add more domain providers and many-to-many relationships.
|
|
235
|
+
|
|
236
|
+
## Development
|
|
237
|
+
|
|
238
|
+
```bash
|
|
239
|
+
python -m venv .venv
|
|
240
|
+
source .venv/bin/activate
|
|
241
|
+
pip install -e ".[dev]"
|
|
242
|
+
pytest
|
|
243
|
+
ruff check src tests examples
|
|
244
|
+
ruff format --check src tests examples
|
|
245
|
+
mypy
|
|
246
|
+
```
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
# FakerForge
|
|
2
|
+
|
|
3
|
+
FakerForge is a Python framework for synthetic data, built on top of [Faker](https://faker.readthedocs.io/). It does not fork Faker. Custom providers are normal Faker providers, and every other Faker method stays available.
|
|
4
|
+
|
|
5
|
+
Requires Python 3.10 or newer.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install fakerforge
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
NumPy and pandas are optional:
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install "fakerforge[numpy]"
|
|
17
|
+
pip install "fakerforge[pandas]"
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Quick start
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
from fakerforge import FakerForge
|
|
24
|
+
|
|
25
|
+
fake = FakerForge(seed=42)
|
|
26
|
+
|
|
27
|
+
fake.name()
|
|
28
|
+
fake.email()
|
|
29
|
+
fake.address()
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
`FakerForge` forwards unknown attributes to its `faker.Faker` instance. `seed` calls `seed_instance`, so the seed applies only to that object. Built-in providers are registered with Faker's `add_provider` and read the same random generator.
|
|
33
|
+
|
|
34
|
+
## Finance
|
|
35
|
+
|
|
36
|
+
`FinanceProvider` is registered on every `FakerForge` instance.
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from fakerforge import FakerForge
|
|
40
|
+
|
|
41
|
+
fake = FakerForge(seed=42)
|
|
42
|
+
|
|
43
|
+
fake.credit_score() # int, 300 through 850
|
|
44
|
+
fake.transaction_amount() # Decimal quantized to cents, 1.00 through 10000.00
|
|
45
|
+
fake.account_number() # 12-digit string
|
|
46
|
+
|
|
47
|
+
fake.credit_score(min_score=700, max_score=750)
|
|
48
|
+
fake.transaction_amount(min_amount=10, max_amount=25)
|
|
49
|
+
fake.account_number(length=8)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
The same seed reproduces the same sequence. `min_score` and `max_score` are inclusive. Amount bounds are inclusive after rounding to cents.
|
|
53
|
+
|
|
54
|
+
## Distributions
|
|
55
|
+
|
|
56
|
+
`number` and `categorical` sample from a `DistributionEngine` created with the same seed as the `FakerForge` instance. That stream is separate from Faker provider calls, so a later schema generator can replay distribution draws without depending on how many names or emails were produced. `seed_instance` reseeds both streams.
|
|
57
|
+
|
|
58
|
+
NumPy is used when it is installed (`pip install fakerforge[numpy]`). Without NumPy, sampling uses the standard library. Each backend is deterministic for a given seed. They do not emit the same series.
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from fakerforge import FakerForge
|
|
62
|
+
|
|
63
|
+
fake = FakerForge(seed=42)
|
|
64
|
+
|
|
65
|
+
fake.number(distribution="uniform", min=0, max=1)
|
|
66
|
+
fake.number(distribution="normal", mean=50, std=10, min=18, max=80)
|
|
67
|
+
fake.number(distribution="lognormal", mean=0, std=0.25, min=0.5, max=3)
|
|
68
|
+
fake.categorical(values=["A", "B", "C"])
|
|
69
|
+
fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1])
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`min` and `max` on normal and log-normal draws are inclusive. A sample outside that interval is discarded and another sample is drawn. Values are not clipped to the boundary. If 1000 draws in a row miss the interval, `number` raises `DistributionError`.
|
|
73
|
+
|
|
74
|
+
Log-normal `mean` and `std` describe the underlying normal distribution, matching NumPy's `Generator.lognormal`. Samples are greater than zero. Uniform draws use the half-open interval `[min, max)`, except when `min == max`, which returns that value. Weights do not need to sum to 1.
|
|
75
|
+
|
|
76
|
+
## Schemas
|
|
77
|
+
|
|
78
|
+
`generate` validates a declarative schema, then builds rows. The return value is a dataset result. `rows` is the list of records. `frame` is a pandas DataFrame when pandas is installed (`pip install fakerforge[pandas]`). `validate()` checks the rows against the schema.
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from fakerforge import FakerForge
|
|
82
|
+
|
|
83
|
+
fake = FakerForge(seed=42)
|
|
84
|
+
|
|
85
|
+
schema = {
|
|
86
|
+
"customer_id": {"type": "uuid", "unique": True},
|
|
87
|
+
"name": {"provider": "person.name"},
|
|
88
|
+
"age": {"type": "integer", "min": 18, "max": 80},
|
|
89
|
+
"email": {"provider": "internet.email", "unique": True},
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
result = fake.generate(schema=schema, rows=1000)
|
|
93
|
+
report = result.validate()
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Primitive types are `string`, `integer`, `float`, `boolean`, `uuid`, and `date`. Provider references use Faker's `provider.method` form, such as `person.name`. Integer bounds are inclusive. Float bounds use the half-open interval `[min, max)`. Omitted numeric bounds default to `0..100` for integers and `0.0..1.0` for floats.
|
|
97
|
+
|
|
98
|
+
`nullable` fields are `None` with probability `null_probability`, which defaults to `0.1`. Unique columns do not repeat non-null values. Nulls may repeat. If a unique value cannot be found, generation raises `GenerationError` instead of inserting a duplicate. Dates are drawn from 1990-01-01 through 2030-12-31 so a seed stays stable.
|
|
99
|
+
|
|
100
|
+
Invalid schemas raise `SchemaError` before any row is produced. `errors` lists every problem found in that pass.
|
|
101
|
+
|
|
102
|
+
### Derived and dependent fields
|
|
103
|
+
|
|
104
|
+
`derived_from` computes a field from an earlier one. A date source produces completed years as of 2030-12-31, so the result does not change with the calendar day. `depends_on` generates the named fields first. A date that depends on a date is drawn on or after that date. Cycles are rejected.
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
schema = {
|
|
108
|
+
"date_of_birth": {"provider": "date.date_of_birth"},
|
|
109
|
+
"age": {"derived_from": "date_of_birth"},
|
|
110
|
+
"start_date": {"type": "date"},
|
|
111
|
+
"end_date": {"type": "date", "depends_on": "start_date"},
|
|
112
|
+
}
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
### Foreign keys
|
|
116
|
+
|
|
117
|
+
A relational schema maps each table to `fields` and `rows`. `references` copies a value from a unique parent column. Child rows are generated after the parent, so a foreign key is never invented. A unique foreign key uses each parent value at most once. Many-to-many relationships are not supported.
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
schema = {
|
|
121
|
+
"customers": {
|
|
122
|
+
"rows": 10,
|
|
123
|
+
"fields": {
|
|
124
|
+
"customer_id": {"type": "uuid", "unique": True},
|
|
125
|
+
"name": {"provider": "person.name"},
|
|
126
|
+
},
|
|
127
|
+
},
|
|
128
|
+
"orders": {
|
|
129
|
+
"rows": 40,
|
|
130
|
+
"fields": {
|
|
131
|
+
"order_id": {"type": "uuid", "unique": True},
|
|
132
|
+
"customer_id": {"references": "customers.customer_id"},
|
|
133
|
+
},
|
|
134
|
+
},
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
tables = fake.generate(schema=schema)
|
|
138
|
+
report = tables.validate()
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### Validation
|
|
142
|
+
|
|
143
|
+
`validate()` checks the generated rows against the schema that produced them. The report has a `passed` or `failed` status, the row count, the violation count, and violations grouped by field. Each violation has a human-readable message. A failed report is returned; it is not raised.
|
|
144
|
+
|
|
145
|
+
Built-in checks cover type, min/max, nullable, uniqueness, known provider output, foreign keys, and row count. Subclass `DatasetCheck` and pass instances as `extra` to add further data-quality checks:
|
|
146
|
+
|
|
147
|
+
```python
|
|
148
|
+
from fakerforge.validation import DatasetCheck, ValidationContext, Violation
|
|
149
|
+
|
|
150
|
+
class RejectPlaceholder(DatasetCheck):
|
|
151
|
+
name = "placeholder"
|
|
152
|
+
|
|
153
|
+
def run(self, context: ValidationContext) -> list[Violation]:
|
|
154
|
+
return []
|
|
155
|
+
|
|
156
|
+
report = result.validate(extra=(RejectPlaceholder(),))
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
## Custom providers
|
|
160
|
+
|
|
161
|
+
Subclass `faker.providers.BaseProvider` and pass the class to `FakerForge` or `add_provider`. `ForgeProvider` is a thin subclass of that base, re-exported for convenience. Provider methods use Faker helpers such as `random_int` and `numerify`, which read the instance generator.
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
from faker.providers import BaseProvider
|
|
165
|
+
|
|
166
|
+
from fakerforge import FakerForge
|
|
167
|
+
from fakerforge.generator import Generator
|
|
168
|
+
from fakerforge.schema import Field, Schema
|
|
169
|
+
|
|
170
|
+
class StatusProvider(BaseProvider):
|
|
171
|
+
def status_code(self) -> str:
|
|
172
|
+
return self.random_element(("new", "open", "closed"))
|
|
173
|
+
|
|
174
|
+
fake = FakerForge(seed=42, providers=[StatusProvider])
|
|
175
|
+
fake.status_code()
|
|
176
|
+
|
|
177
|
+
schema = Schema(
|
|
178
|
+
fields=(
|
|
179
|
+
Field("name", "name"),
|
|
180
|
+
Field("email", "email"),
|
|
181
|
+
Field("status", "status_code"),
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
rows = Generator(fake).generate(schema, count=3)
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
|
|
188
|
+
|
|
189
|
+
## This release
|
|
190
|
+
|
|
191
|
+
| Piece | What 0.1 provides |
|
|
192
|
+
| --- | --- |
|
|
193
|
+
| Providers | Built-in `FinanceProvider`, plus `add_provider` for custom `BaseProvider` subclasses |
|
|
194
|
+
| Constraints | `Constraint` base and `Validator` |
|
|
195
|
+
| Distributions | Uniform, normal, log-normal, and weighted categorical sampling |
|
|
196
|
+
| Schema | Declarative datasets via `FakerForge.generate` |
|
|
197
|
+
| Generation | Seeded rows, with a DataFrame when pandas is installed |
|
|
198
|
+
| Validation | Dataset checks on generated rows, plus per-field `Validator` |
|
|
199
|
+
| Reproducibility | Instance seeding |
|
|
200
|
+
|
|
201
|
+
Later releases can add more domain providers and many-to-many relationships.
|
|
202
|
+
|
|
203
|
+
## Development
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
python -m venv .venv
|
|
207
|
+
source .venv/bin/activate
|
|
208
|
+
pip install -e ".[dev]"
|
|
209
|
+
pytest
|
|
210
|
+
ruff check src tests examples
|
|
211
|
+
ruff format --check src tests examples
|
|
212
|
+
mypy
|
|
213
|
+
```
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Architecture
|
|
2
|
+
|
|
3
|
+
FakerForge 0.1 is a wrapper and a set of extension points. It imports Faker and does not vendor or patch it.
|
|
4
|
+
|
|
5
|
+
```text
|
|
6
|
+
FakerForge
|
|
7
|
+
└── faker.Faker attribute delegation, seed_instance, add_provider
|
|
8
|
+
├── FinanceProvider, a faker.providers.BaseProvider
|
|
9
|
+
└── caller providers, also BaseProvider subclasses
|
|
10
|
+
|
|
11
|
+
DistributionEngine seeded uniform, normal, log-normal, categorical
|
|
12
|
+
└── FakerForge.number / FakerForge.categorical
|
|
13
|
+
|
|
14
|
+
DatasetSchema / DatabaseSchema
|
|
15
|
+
└── FakerForge.generate validates order, then fills rows
|
|
16
|
+
├── derived fields and date dependencies
|
|
17
|
+
├── foreign keys sampled from parent rows
|
|
18
|
+
└── DatasetResult / DatabaseResult
|
|
19
|
+
├── frame is a pandas.DataFrame when pandas is installed
|
|
20
|
+
└── validate() runs DatasetCheck subclasses
|
|
21
|
+
|
|
22
|
+
Schema (Field...)
|
|
23
|
+
└── Generator calls provider methods on a FakerForge
|
|
24
|
+
|
|
25
|
+
Constraint subclasses
|
|
26
|
+
└── Validator checks a finished record against constraints
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
`DistributionEngine` keeps its own random stream so schema generation can sample numbers without consuming Faker provider draws. NumPy is optional.
|
|
30
|
+
|
|
31
|
+
Finance is the first built-in domain provider. Many-to-many relationships are still later work.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Generate a few seeded records with FakerForge."""
|
|
2
|
+
|
|
3
|
+
from fakerforge import FakerForge
|
|
4
|
+
from fakerforge.generator import Generator
|
|
5
|
+
from fakerforge.schema import Field, Schema
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def main() -> None:
|
|
9
|
+
fake = FakerForge(seed=42)
|
|
10
|
+
print(fake.name())
|
|
11
|
+
print(fake.email())
|
|
12
|
+
print(fake.address())
|
|
13
|
+
print(fake.credit_score())
|
|
14
|
+
print(fake.transaction_amount())
|
|
15
|
+
print(fake.account_number())
|
|
16
|
+
print(fake.number(distribution="normal", mean=50, std=10, min=18, max=80))
|
|
17
|
+
print(fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1]))
|
|
18
|
+
generated = fake.generate(
|
|
19
|
+
schema={
|
|
20
|
+
"customer_id": {"type": "uuid", "unique": True},
|
|
21
|
+
"name": {"provider": "person.name"},
|
|
22
|
+
"age": {"type": "integer", "min": 18, "max": 80},
|
|
23
|
+
"email": {"provider": "internet.email", "unique": True},
|
|
24
|
+
},
|
|
25
|
+
rows=2,
|
|
26
|
+
)
|
|
27
|
+
print(generated.validate())
|
|
28
|
+
|
|
29
|
+
schema = Schema(
|
|
30
|
+
fields=(
|
|
31
|
+
Field("name", "name"),
|
|
32
|
+
Field("email", "email"),
|
|
33
|
+
Field("address", "address"),
|
|
34
|
+
)
|
|
35
|
+
)
|
|
36
|
+
rows = Generator(FakerForge(seed=42)).generate(schema, count=2)
|
|
37
|
+
for row in rows:
|
|
38
|
+
print(row)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
if __name__ == "__main__":
|
|
42
|
+
main()
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=69", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "fakerforge"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Synthetic data generation built on the Faker library."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Shardul" }]
|
|
14
|
+
keywords = ["faker", "synthetic-data", "testing", "fixtures"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Topic :: Software Development :: Testing",
|
|
25
|
+
"Typing :: Typed",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"Faker>=24.0.0",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
numpy = ["numpy>=1.24"]
|
|
33
|
+
pandas = ["pandas>=2"]
|
|
34
|
+
dev = [
|
|
35
|
+
"pytest>=8.0",
|
|
36
|
+
"ruff>=0.8.0",
|
|
37
|
+
"mypy>=1.13",
|
|
38
|
+
"numpy>=1.24",
|
|
39
|
+
"pandas>=2",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
[tool.setuptools.packages.find]
|
|
43
|
+
where = ["src"]
|
|
44
|
+
include = ["fakerforge*"]
|
|
45
|
+
|
|
46
|
+
[tool.setuptools.package-data]
|
|
47
|
+
fakerforge = ["py.typed"]
|
|
48
|
+
|
|
49
|
+
[tool.mypy]
|
|
50
|
+
python_version = "3.10"
|
|
51
|
+
packages = ["fakerforge"]
|
|
52
|
+
mypy_path = "src"
|
|
53
|
+
warn_unused_ignores = true
|
|
54
|
+
warn_redundant_casts = true
|
|
55
|
+
check_untyped_defs = true
|
|
56
|
+
|
|
57
|
+
[tool.pytest.ini_options]
|
|
58
|
+
testpaths = ["tests"]
|
|
59
|
+
addopts = ["-q", "--strict-markers"]
|
|
60
|
+
|
|
61
|
+
[tool.ruff]
|
|
62
|
+
target-version = "py310"
|
|
63
|
+
line-length = 88
|
|
64
|
+
src = ["src", "tests", "examples"]
|
|
65
|
+
|
|
66
|
+
[tool.ruff.lint]
|
|
67
|
+
select = ["E", "F", "I", "UP", "B", "D"]
|
|
68
|
+
# Constructor parameters are documented on the class docstring.
|
|
69
|
+
ignore = ["D105", "D107"]
|
|
70
|
+
|
|
71
|
+
[tool.ruff.lint.pydocstyle]
|
|
72
|
+
convention = "google"
|
|
73
|
+
|
|
74
|
+
[tool.ruff.lint.per-file-ignores]
|
|
75
|
+
"tests/**" = ["D"]
|
|
76
|
+
"examples/**" = ["D"]
|