fakerforge 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. fakerforge-0.1.0/CHANGELOG.md +16 -0
  2. fakerforge-0.1.0/LICENSE +21 -0
  3. fakerforge-0.1.0/MANIFEST.in +4 -0
  4. fakerforge-0.1.0/PKG-INFO +246 -0
  5. fakerforge-0.1.0/README.md +213 -0
  6. fakerforge-0.1.0/docs/architecture.md +31 -0
  7. fakerforge-0.1.0/examples/quickstart.py +42 -0
  8. fakerforge-0.1.0/pyproject.toml +76 -0
  9. fakerforge-0.1.0/setup.cfg +4 -0
  10. fakerforge-0.1.0/src/fakerforge/__init__.py +7 -0
  11. fakerforge-0.1.0/src/fakerforge/constraints/__init__.py +5 -0
  12. fakerforge-0.1.0/src/fakerforge/constraints/base.py +35 -0
  13. fakerforge-0.1.0/src/fakerforge/distributions/__init__.py +6 -0
  14. fakerforge-0.1.0/src/fakerforge/distributions/base.py +26 -0
  15. fakerforge-0.1.0/src/fakerforge/distributions/engine.py +335 -0
  16. fakerforge-0.1.0/src/fakerforge/faker.py +200 -0
  17. fakerforge-0.1.0/src/fakerforge/generator/__init__.py +5 -0
  18. fakerforge-0.1.0/src/fakerforge/generator/generator.py +68 -0
  19. fakerforge-0.1.0/src/fakerforge/providers/__init__.py +6 -0
  20. fakerforge-0.1.0/src/fakerforge/providers/base.py +16 -0
  21. fakerforge-0.1.0/src/fakerforge/providers/finance.py +112 -0
  22. fakerforge-0.1.0/src/fakerforge/py.typed +0 -0
  23. fakerforge-0.1.0/src/fakerforge/schema/__init__.py +25 -0
  24. fakerforge-0.1.0/src/fakerforge/schema/dataset.py +813 -0
  25. fakerforge-0.1.0/src/fakerforge/schema/generate.py +506 -0
  26. fakerforge-0.1.0/src/fakerforge/schema/result.py +137 -0
  27. fakerforge-0.1.0/src/fakerforge/schema/schema.py +59 -0
  28. fakerforge-0.1.0/src/fakerforge/validation/__init__.py +14 -0
  29. fakerforge-0.1.0/src/fakerforge/validation/dataset.py +438 -0
  30. fakerforge-0.1.0/src/fakerforge/validation/report.py +65 -0
  31. fakerforge-0.1.0/src/fakerforge/validation/validator.py +82 -0
  32. fakerforge-0.1.0/src/fakerforge.egg-info/SOURCES.txt +42 -0
  33. fakerforge-0.1.0/tests/test_constraints.py +26 -0
  34. fakerforge-0.1.0/tests/test_dataset_schema.py +224 -0
  35. fakerforge-0.1.0/tests/test_dataset_validation.py +480 -0
  36. fakerforge-0.1.0/tests/test_distribution_engine.py +215 -0
  37. fakerforge-0.1.0/tests/test_distributions.py +29 -0
  38. fakerforge-0.1.0/tests/test_faker.py +74 -0
  39. fakerforge-0.1.0/tests/test_finance.py +127 -0
  40. fakerforge-0.1.0/tests/test_generator.py +57 -0
  41. fakerforge-0.1.0/tests/test_packaging.py +9 -0
  42. fakerforge-0.1.0/tests/test_providers.py +30 -0
  43. fakerforge-0.1.0/tests/test_relationships.py +273 -0
  44. fakerforge-0.1.0/tests/test_schema.py +44 -0
  45. fakerforge-0.1.0/tests/test_validator.py +46 -0
@@ -0,0 +1,16 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 - 2026-09-29
4
+
5
+ First public release. Requires Python 3.10 or newer.
6
+
7
+ - Add `FakerForge`, a seeded wrapper that delegates unknown attributes to `faker.Faker`.
8
+ - Add `ForgeProvider` as a subclass of Faker's `BaseProvider`.
9
+ - Add abstract `Constraint` and `Distribution` bases.
10
+ - Add `Field`, `Schema`, and a schema `Generator` that calls Faker provider methods.
11
+ - Add `Validator` for per-field constraint checks.
12
+ - Register `FinanceProvider` on every `FakerForge` instance through Faker's `BaseProvider` mechanism.
13
+ - Add a seeded `DistributionEngine` with uniform, normal, log-normal, and weighted categorical sampling.
14
+ - Add declarative schema generation with primitive types, Faker provider references, bounds, nulls, uniqueness, and optional pandas output.
15
+ - Add derived fields, dependent fields, and foreign keys that preserve referential integrity.
16
+ - Add dataset validation for types, bounds, nulls, uniqueness, provider output, foreign keys, and row counts.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Gore Shardul
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,4 @@
1
+ include CHANGELOG.md
2
+ recursive-include docs *.md
3
+ recursive-include examples *.py
4
+ prune src/fakerforge.egg-info
@@ -0,0 +1,246 @@
1
+ Metadata-Version: 2.4
2
+ Name: fakerforge
3
+ Version: 0.1.0
4
+ Summary: Synthetic data generation built on the Faker library.
5
+ Author: Shardul
6
+ License-Expression: MIT
7
+ Keywords: faker,synthetic-data,testing,fixtures
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Software Development :: Testing
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: Faker>=24.0.0
22
+ Provides-Extra: numpy
23
+ Requires-Dist: numpy>=1.24; extra == "numpy"
24
+ Provides-Extra: pandas
25
+ Requires-Dist: pandas>=2; extra == "pandas"
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest>=8.0; extra == "dev"
28
+ Requires-Dist: ruff>=0.8.0; extra == "dev"
29
+ Requires-Dist: mypy>=1.13; extra == "dev"
30
+ Requires-Dist: numpy>=1.24; extra == "dev"
31
+ Requires-Dist: pandas>=2; extra == "dev"
32
+ Dynamic: license-file
33
+
34
+ # FakerForge
35
+
36
+ FakerForge is a Python framework for synthetic data, built on top of [Faker](https://faker.readthedocs.io/). It does not fork Faker. Custom providers are normal Faker providers, and every other Faker method stays available.
37
+
38
+ Requires Python 3.10 or newer.
39
+
40
+ ## Install
41
+
42
+ ```bash
43
+ pip install fakerforge
44
+ ```
45
+
46
+ NumPy and pandas are optional:
47
+
48
+ ```bash
49
+ pip install "fakerforge[numpy]"
50
+ pip install "fakerforge[pandas]"
51
+ ```
52
+
53
+ ## Quick start
54
+
55
+ ```python
56
+ from fakerforge import FakerForge
57
+
58
+ fake = FakerForge(seed=42)
59
+
60
+ fake.name()
61
+ fake.email()
62
+ fake.address()
63
+ ```
64
+
65
+ `FakerForge` forwards unknown attributes to its `faker.Faker` instance. `seed` calls `seed_instance`, so the seed applies only to that object. Built-in providers are registered with Faker's `add_provider` and read the same random generator.
66
+
67
+ ## Finance
68
+
69
+ `FinanceProvider` is registered on every `FakerForge` instance.
70
+
71
+ ```python
72
+ from fakerforge import FakerForge
73
+
74
+ fake = FakerForge(seed=42)
75
+
76
+ fake.credit_score() # int, 300 through 850
77
+ fake.transaction_amount() # Decimal quantized to cents, 1.00 through 10000.00
78
+ fake.account_number() # 12-digit string
79
+
80
+ fake.credit_score(min_score=700, max_score=750)
81
+ fake.transaction_amount(min_amount=10, max_amount=25)
82
+ fake.account_number(length=8)
83
+ ```
84
+
85
+ The same seed reproduces the same sequence. `min_score` and `max_score` are inclusive. Amount bounds are inclusive after rounding to cents.
86
+
87
+ ## Distributions
88
+
89
+ `number` and `categorical` sample from a `DistributionEngine` created with the same seed as the `FakerForge` instance. That stream is separate from Faker provider calls, so a later schema generator can replay distribution draws without depending on how many names or emails were produced. `seed_instance` reseeds both streams.
90
+
91
+ NumPy is used when it is installed (`pip install fakerforge[numpy]`). Without NumPy, sampling uses the standard library. Each backend is deterministic for a given seed. They do not emit the same series.
92
+
93
+ ```python
94
+ from fakerforge import FakerForge
95
+
96
+ fake = FakerForge(seed=42)
97
+
98
+ fake.number(distribution="uniform", min=0, max=1)
99
+ fake.number(distribution="normal", mean=50, std=10, min=18, max=80)
100
+ fake.number(distribution="lognormal", mean=0, std=0.25, min=0.5, max=3)
101
+ fake.categorical(values=["A", "B", "C"])
102
+ fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1])
103
+ ```
104
+
105
+ `min` and `max` on normal and log-normal draws are inclusive. A sample outside that interval is discarded and another sample is drawn. Values are not clipped to the boundary. If 1000 draws in a row miss the interval, `number` raises `DistributionError`.
106
+
107
+ Log-normal `mean` and `std` describe the underlying normal distribution, matching NumPy's `Generator.lognormal`. Samples are greater than zero. Uniform draws use the half-open interval `[min, max)`, except when `min == max`, which returns that value. Weights do not need to sum to 1.
108
+
109
+ ## Schemas
110
+
111
+ `generate` validates a declarative schema, then builds rows. The return value is a dataset result. `rows` is the list of records. `frame` is a pandas DataFrame when pandas is installed (`pip install fakerforge[pandas]`). `validate()` checks the rows against the schema.
112
+
113
+ ```python
114
+ from fakerforge import FakerForge
115
+
116
+ fake = FakerForge(seed=42)
117
+
118
+ schema = {
119
+ "customer_id": {"type": "uuid", "unique": True},
120
+ "name": {"provider": "person.name"},
121
+ "age": {"type": "integer", "min": 18, "max": 80},
122
+ "email": {"provider": "internet.email", "unique": True},
123
+ }
124
+
125
+ result = fake.generate(schema=schema, rows=1000)
126
+ report = result.validate()
127
+ ```
128
+
129
+ Primitive types are `string`, `integer`, `float`, `boolean`, `uuid`, and `date`. Provider references use Faker's `provider.method` form, such as `person.name`. Integer bounds are inclusive. Float bounds use the half-open interval `[min, max)`. Omitted numeric bounds default to `0..100` for integers and `0.0..1.0` for floats.
130
+
131
+ `nullable` fields are `None` with probability `null_probability`, which defaults to `0.1`. Unique columns do not repeat non-null values. Nulls may repeat. If a unique value cannot be found, generation raises `GenerationError` instead of inserting a duplicate. Dates are drawn from 1990-01-01 through 2030-12-31 so a seed stays stable.
132
+
133
+ Invalid schemas raise `SchemaError` before any row is produced. `errors` lists every problem found in that pass.
134
+
135
+ ### Derived and dependent fields
136
+
137
+ `derived_from` computes a field from an earlier one. A date source produces completed years as of 2030-12-31, so the result does not change with the calendar day. `depends_on` generates the named fields first. A date that depends on a date is drawn on or after that date. Cycles are rejected.
138
+
139
+ ```python
140
+ schema = {
141
+ "date_of_birth": {"provider": "date.date_of_birth"},
142
+ "age": {"derived_from": "date_of_birth"},
143
+ "start_date": {"type": "date"},
144
+ "end_date": {"type": "date", "depends_on": "start_date"},
145
+ }
146
+ ```
147
+
148
+ ### Foreign keys
149
+
150
+ A relational schema maps each table to `fields` and `rows`. `references` copies a value from a unique parent column. Child rows are generated after the parent, so a foreign key is never invented. A unique foreign key uses each parent value at most once. Many-to-many relationships are not supported.
151
+
152
+ ```python
153
+ schema = {
154
+ "customers": {
155
+ "rows": 10,
156
+ "fields": {
157
+ "customer_id": {"type": "uuid", "unique": True},
158
+ "name": {"provider": "person.name"},
159
+ },
160
+ },
161
+ "orders": {
162
+ "rows": 40,
163
+ "fields": {
164
+ "order_id": {"type": "uuid", "unique": True},
165
+ "customer_id": {"references": "customers.customer_id"},
166
+ },
167
+ },
168
+ }
169
+
170
+ tables = fake.generate(schema=schema)
171
+ report = tables.validate()
172
+ ```
173
+
174
+ ### Validation
175
+
176
+ `validate()` checks the generated rows against the schema that produced them. The report has a `passed` or `failed` status, the row count, the violation count, and violations grouped by field. Each violation has a human-readable message. A failed report is returned; it is not raised.
177
+
178
+ Built-in checks cover type, min/max, nullable, uniqueness, known provider output, foreign keys, and row count. Subclass `DatasetCheck` and pass instances as `extra` to add further data-quality checks:
179
+
180
+ ```python
181
+ from fakerforge.validation import DatasetCheck, ValidationContext, Violation
182
+
183
+ class RejectPlaceholder(DatasetCheck):
184
+ name = "placeholder"
185
+
186
+ def run(self, context: ValidationContext) -> list[Violation]:
187
+ return []
188
+
189
+ report = result.validate(extra=(RejectPlaceholder(),))
190
+ ```
191
+
192
+ ## Custom providers
193
+
194
+ Subclass `faker.providers.BaseProvider` and pass the class to `FakerForge` or `add_provider`. `ForgeProvider` is a thin subclass of that base, re-exported for convenience. Provider methods use Faker helpers such as `random_int` and `numerify`, which read the instance generator.
195
+
196
+ ```python
197
+ from faker.providers import BaseProvider
198
+
199
+ from fakerforge import FakerForge
200
+ from fakerforge.generator import Generator
201
+ from fakerforge.schema import Field, Schema
202
+
203
+ class StatusProvider(BaseProvider):
204
+ def status_code(self) -> str:
205
+ return self.random_element(("new", "open", "closed"))
206
+
207
+ fake = FakerForge(seed=42, providers=[StatusProvider])
208
+ fake.status_code()
209
+
210
+ schema = Schema(
211
+ fields=(
212
+ Field("name", "name"),
213
+ Field("email", "email"),
214
+ Field("status", "status_code"),
215
+ )
216
+ )
217
+ rows = Generator(fake).generate(schema, count=3)
218
+ ```
219
+
220
+ The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
221
+
222
+ ## This release
223
+
224
+ | Piece | What 0.1 provides |
225
+ | --- | --- |
226
+ | Providers | Built-in `FinanceProvider`, plus `add_provider` for custom `BaseProvider` subclasses |
227
+ | Constraints | `Constraint` base and `Validator` |
228
+ | Distributions | Uniform, normal, log-normal, and weighted categorical sampling |
229
+ | Schema | Declarative datasets via `FakerForge.generate` |
230
+ | Generation | Seeded rows, with a DataFrame when pandas is installed |
231
+ | Validation | Dataset checks on generated rows, plus per-field `Validator` |
232
+ | Reproducibility | Instance seeding |
233
+
234
+ Later releases can add more domain providers and many-to-many relationships.
235
+
236
+ ## Development
237
+
238
+ ```bash
239
+ python -m venv .venv
240
+ source .venv/bin/activate
241
+ pip install -e ".[dev]"
242
+ pytest
243
+ ruff check src tests examples
244
+ ruff format --check src tests examples
245
+ mypy
246
+ ```
@@ -0,0 +1,213 @@
1
+ # FakerForge
2
+
3
+ FakerForge is a Python framework for synthetic data, built on top of [Faker](https://faker.readthedocs.io/). It does not fork Faker. Custom providers are normal Faker providers, and every other Faker method stays available.
4
+
5
+ Requires Python 3.10 or newer.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pip install fakerforge
11
+ ```
12
+
13
+ NumPy and pandas are optional:
14
+
15
+ ```bash
16
+ pip install "fakerforge[numpy]"
17
+ pip install "fakerforge[pandas]"
18
+ ```
19
+
20
+ ## Quick start
21
+
22
+ ```python
23
+ from fakerforge import FakerForge
24
+
25
+ fake = FakerForge(seed=42)
26
+
27
+ fake.name()
28
+ fake.email()
29
+ fake.address()
30
+ ```
31
+
32
+ `FakerForge` forwards unknown attributes to its `faker.Faker` instance. `seed` calls `seed_instance`, so the seed applies only to that object. Built-in providers are registered with Faker's `add_provider` and read the same random generator.
33
+
34
+ ## Finance
35
+
36
+ `FinanceProvider` is registered on every `FakerForge` instance.
37
+
38
+ ```python
39
+ from fakerforge import FakerForge
40
+
41
+ fake = FakerForge(seed=42)
42
+
43
+ fake.credit_score() # int, 300 through 850
44
+ fake.transaction_amount() # Decimal quantized to cents, 1.00 through 10000.00
45
+ fake.account_number() # 12-digit string
46
+
47
+ fake.credit_score(min_score=700, max_score=750)
48
+ fake.transaction_amount(min_amount=10, max_amount=25)
49
+ fake.account_number(length=8)
50
+ ```
51
+
52
+ The same seed reproduces the same sequence. `min_score` and `max_score` are inclusive. Amount bounds are inclusive after rounding to cents.
53
+
54
+ ## Distributions
55
+
56
+ `number` and `categorical` sample from a `DistributionEngine` created with the same seed as the `FakerForge` instance. That stream is separate from Faker provider calls, so a later schema generator can replay distribution draws without depending on how many names or emails were produced. `seed_instance` reseeds both streams.
57
+
58
+ NumPy is used when it is installed (`pip install fakerforge[numpy]`). Without NumPy, sampling uses the standard library. Each backend is deterministic for a given seed. They do not emit the same series.
59
+
60
+ ```python
61
+ from fakerforge import FakerForge
62
+
63
+ fake = FakerForge(seed=42)
64
+
65
+ fake.number(distribution="uniform", min=0, max=1)
66
+ fake.number(distribution="normal", mean=50, std=10, min=18, max=80)
67
+ fake.number(distribution="lognormal", mean=0, std=0.25, min=0.5, max=3)
68
+ fake.categorical(values=["A", "B", "C"])
69
+ fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1])
70
+ ```
71
+
72
+ `min` and `max` on normal and log-normal draws are inclusive. A sample outside that interval is discarded and another sample is drawn. Values are not clipped to the boundary. If 1000 draws in a row miss the interval, `number` raises `DistributionError`.
73
+
74
+ Log-normal `mean` and `std` describe the underlying normal distribution, matching NumPy's `Generator.lognormal`. Samples are greater than zero. Uniform draws use the half-open interval `[min, max)`, except when `min == max`, which returns that value. Weights do not need to sum to 1.
75
+
76
+ ## Schemas
77
+
78
+ `generate` validates a declarative schema, then builds rows. The return value is a dataset result. `rows` is the list of records. `frame` is a pandas DataFrame when pandas is installed (`pip install fakerforge[pandas]`). `validate()` checks the rows against the schema.
79
+
80
+ ```python
81
+ from fakerforge import FakerForge
82
+
83
+ fake = FakerForge(seed=42)
84
+
85
+ schema = {
86
+ "customer_id": {"type": "uuid", "unique": True},
87
+ "name": {"provider": "person.name"},
88
+ "age": {"type": "integer", "min": 18, "max": 80},
89
+ "email": {"provider": "internet.email", "unique": True},
90
+ }
91
+
92
+ result = fake.generate(schema=schema, rows=1000)
93
+ report = result.validate()
94
+ ```
95
+
96
+ Primitive types are `string`, `integer`, `float`, `boolean`, `uuid`, and `date`. Provider references use Faker's `provider.method` form, such as `person.name`. Integer bounds are inclusive. Float bounds use the half-open interval `[min, max)`. Omitted numeric bounds default to `0..100` for integers and `0.0..1.0` for floats.
97
+
98
+ `nullable` fields are `None` with probability `null_probability`, which defaults to `0.1`. Unique columns do not repeat non-null values. Nulls may repeat. If a unique value cannot be found, generation raises `GenerationError` instead of inserting a duplicate. Dates are drawn from 1990-01-01 through 2030-12-31 so a seed stays stable.
99
+
100
+ Invalid schemas raise `SchemaError` before any row is produced. `errors` lists every problem found in that pass.
101
+
102
+ ### Derived and dependent fields
103
+
104
+ `derived_from` computes a field from an earlier one. A date source produces completed years as of 2030-12-31, so the result does not change with the calendar day. `depends_on` generates the named fields first. A date that depends on a date is drawn on or after that date. Cycles are rejected.
105
+
106
+ ```python
107
+ schema = {
108
+ "date_of_birth": {"provider": "date.date_of_birth"},
109
+ "age": {"derived_from": "date_of_birth"},
110
+ "start_date": {"type": "date"},
111
+ "end_date": {"type": "date", "depends_on": "start_date"},
112
+ }
113
+ ```
114
+
115
+ ### Foreign keys
116
+
117
+ A relational schema maps each table to `fields` and `rows`. `references` copies a value from a unique parent column. Child rows are generated after the parent, so a foreign key is never invented. A unique foreign key uses each parent value at most once. Many-to-many relationships are not supported.
118
+
119
+ ```python
120
+ schema = {
121
+ "customers": {
122
+ "rows": 10,
123
+ "fields": {
124
+ "customer_id": {"type": "uuid", "unique": True},
125
+ "name": {"provider": "person.name"},
126
+ },
127
+ },
128
+ "orders": {
129
+ "rows": 40,
130
+ "fields": {
131
+ "order_id": {"type": "uuid", "unique": True},
132
+ "customer_id": {"references": "customers.customer_id"},
133
+ },
134
+ },
135
+ }
136
+
137
+ tables = fake.generate(schema=schema)
138
+ report = tables.validate()
139
+ ```
140
+
141
+ ### Validation
142
+
143
+ `validate()` checks the generated rows against the schema that produced them. The report has a `passed` or `failed` status, the row count, the violation count, and violations grouped by field. Each violation has a human-readable message. A failed report is returned; it is not raised.
144
+
145
+ Built-in checks cover type, min/max, nullable, uniqueness, known provider output, foreign keys, and row count. Subclass `DatasetCheck` and pass instances as `extra` to add further data-quality checks:
146
+
147
+ ```python
148
+ from fakerforge.validation import DatasetCheck, ValidationContext, Violation
149
+
150
+ class RejectPlaceholder(DatasetCheck):
151
+ name = "placeholder"
152
+
153
+ def run(self, context: ValidationContext) -> list[Violation]:
154
+ return []
155
+
156
+ report = result.validate(extra=(RejectPlaceholder(),))
157
+ ```
158
+
159
+ ## Custom providers
160
+
161
+ Subclass `faker.providers.BaseProvider` and pass the class to `FakerForge` or `add_provider`. `ForgeProvider` is a thin subclass of that base, re-exported for convenience. Provider methods use Faker helpers such as `random_int` and `numerify`, which read the instance generator.
162
+
163
+ ```python
164
+ from faker.providers import BaseProvider
165
+
166
+ from fakerforge import FakerForge
167
+ from fakerforge.generator import Generator
168
+ from fakerforge.schema import Field, Schema
169
+
170
+ class StatusProvider(BaseProvider):
171
+ def status_code(self) -> str:
172
+ return self.random_element(("new", "open", "closed"))
173
+
174
+ fake = FakerForge(seed=42, providers=[StatusProvider])
175
+ fake.status_code()
176
+
177
+ schema = Schema(
178
+ fields=(
179
+ Field("name", "name"),
180
+ Field("email", "email"),
181
+ Field("status", "status_code"),
182
+ )
183
+ )
184
+ rows = Generator(fake).generate(schema, count=3)
185
+ ```
186
+
187
+ The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
188
+
189
+ ## This release
190
+
191
+ | Piece | What 0.1 provides |
192
+ | --- | --- |
193
+ | Providers | Built-in `FinanceProvider`, plus `add_provider` for custom `BaseProvider` subclasses |
194
+ | Constraints | `Constraint` base and `Validator` |
195
+ | Distributions | Uniform, normal, log-normal, and weighted categorical sampling |
196
+ | Schema | Declarative datasets via `FakerForge.generate` |
197
+ | Generation | Seeded rows, with a DataFrame when pandas is installed |
198
+ | Validation | Dataset checks on generated rows, plus per-field `Validator` |
199
+ | Reproducibility | Instance seeding |
200
+
201
+ Later releases can add more domain providers and many-to-many relationships.
202
+
203
+ ## Development
204
+
205
+ ```bash
206
+ python -m venv .venv
207
+ source .venv/bin/activate
208
+ pip install -e ".[dev]"
209
+ pytest
210
+ ruff check src tests examples
211
+ ruff format --check src tests examples
212
+ mypy
213
+ ```
@@ -0,0 +1,31 @@
1
+ # Architecture
2
+
3
+ FakerForge 0.1 is a wrapper and a set of extension points. It imports Faker and does not vendor or patch it.
4
+
5
+ ```text
6
+ FakerForge
7
+ └── faker.Faker attribute delegation, seed_instance, add_provider
8
+ ├── FinanceProvider, a faker.providers.BaseProvider
9
+ └── caller providers, also BaseProvider subclasses
10
+
11
+ DistributionEngine seeded uniform, normal, log-normal, categorical
12
+ └── FakerForge.number / FakerForge.categorical
13
+
14
+ DatasetSchema / DatabaseSchema
15
+ └── FakerForge.generate validates order, then fills rows
16
+ ├── derived fields and date dependencies
17
+ ├── foreign keys sampled from parent rows
18
+ └── DatasetResult / DatabaseResult
19
+ ├── frame is a pandas.DataFrame when pandas is installed
20
+ └── validate() runs DatasetCheck subclasses
21
+
22
+ Schema (Field...)
23
+ └── Generator calls provider methods on a FakerForge
24
+
25
+ Constraint subclasses
26
+ └── Validator checks a finished record against constraints
27
+ ```
28
+
29
+ `DistributionEngine` keeps its own random stream so schema generation can sample numbers without consuming Faker provider draws. NumPy is optional.
30
+
31
+ Finance is the first built-in domain provider. Many-to-many relationships are still later work.
@@ -0,0 +1,42 @@
1
+ """Generate a few seeded records with FakerForge."""
2
+
3
+ from fakerforge import FakerForge
4
+ from fakerforge.generator import Generator
5
+ from fakerforge.schema import Field, Schema
6
+
7
+
8
+ def main() -> None:
9
+ fake = FakerForge(seed=42)
10
+ print(fake.name())
11
+ print(fake.email())
12
+ print(fake.address())
13
+ print(fake.credit_score())
14
+ print(fake.transaction_amount())
15
+ print(fake.account_number())
16
+ print(fake.number(distribution="normal", mean=50, std=10, min=18, max=80))
17
+ print(fake.categorical(values=["A", "B", "C"], weights=[0.6, 0.3, 0.1]))
18
+ generated = fake.generate(
19
+ schema={
20
+ "customer_id": {"type": "uuid", "unique": True},
21
+ "name": {"provider": "person.name"},
22
+ "age": {"type": "integer", "min": 18, "max": 80},
23
+ "email": {"provider": "internet.email", "unique": True},
24
+ },
25
+ rows=2,
26
+ )
27
+ print(generated.validate())
28
+
29
+ schema = Schema(
30
+ fields=(
31
+ Field("name", "name"),
32
+ Field("email", "email"),
33
+ Field("address", "address"),
34
+ )
35
+ )
36
+ rows = Generator(FakerForge(seed=42)).generate(schema, count=2)
37
+ for row in rows:
38
+ print(row)
39
+
40
+
41
+ if __name__ == "__main__":
42
+ main()
@@ -0,0 +1,76 @@
1
+ [build-system]
2
+ requires = ["setuptools>=69", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "fakerforge"
7
+ version = "0.1.0"
8
+ description = "Synthetic data generation built on the Faker library."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Shardul" }]
14
+ keywords = ["faker", "synthetic-data", "testing", "fixtures"]
15
+ classifiers = [
16
+ "Development Status :: 3 - Alpha",
17
+ "Intended Audience :: Developers",
18
+ "Operating System :: OS Independent",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3.10",
21
+ "Programming Language :: Python :: 3.11",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Topic :: Software Development :: Testing",
25
+ "Typing :: Typed",
26
+ ]
27
+ dependencies = [
28
+ "Faker>=24.0.0",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ numpy = ["numpy>=1.24"]
33
+ pandas = ["pandas>=2"]
34
+ dev = [
35
+ "pytest>=8.0",
36
+ "ruff>=0.8.0",
37
+ "mypy>=1.13",
38
+ "numpy>=1.24",
39
+ "pandas>=2",
40
+ ]
41
+
42
+ [tool.setuptools.packages.find]
43
+ where = ["src"]
44
+ include = ["fakerforge*"]
45
+
46
+ [tool.setuptools.package-data]
47
+ fakerforge = ["py.typed"]
48
+
49
+ [tool.mypy]
50
+ python_version = "3.10"
51
+ packages = ["fakerforge"]
52
+ mypy_path = "src"
53
+ warn_unused_ignores = true
54
+ warn_redundant_casts = true
55
+ check_untyped_defs = true
56
+
57
+ [tool.pytest.ini_options]
58
+ testpaths = ["tests"]
59
+ addopts = ["-q", "--strict-markers"]
60
+
61
+ [tool.ruff]
62
+ target-version = "py310"
63
+ line-length = 88
64
+ src = ["src", "tests", "examples"]
65
+
66
+ [tool.ruff.lint]
67
+ select = ["E", "F", "I", "UP", "B", "D"]
68
+ # Constructor parameters are documented on the class docstring.
69
+ ignore = ["D105", "D107"]
70
+
71
+ [tool.ruff.lint.pydocstyle]
72
+ convention = "google"
73
+
74
+ [tool.ruff.lint.per-file-ignores]
75
+ "tests/**" = ["D"]
76
+ "examples/**" = ["D"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,7 @@
1
+ """Synthetic data generation built on the Faker library."""
2
+
3
+ from fakerforge.faker import FakerForge
4
+
5
+ __version__ = "0.1.0"
6
+
7
+ __all__ = ["FakerForge", "__version__"]
@@ -0,0 +1,5 @@
1
+ """Constraint extension point."""
2
+
3
+ from fakerforge.constraints.base import Constraint
4
+
5
+ __all__ = ["Constraint"]