fakerforge 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fakerforge-0.1.0 → fakerforge-0.2.0}/CHANGELOG.md +6 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/PKG-INFO +24 -1
- {fakerforge-0.1.0 → fakerforge-0.2.0}/README.md +21 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/docs/architecture.md +10 -2
- {fakerforge-0.1.0 → fakerforge-0.2.0}/pyproject.toml +2 -1
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/__init__.py +1 -1
- fakerforge-0.2.0/src/fakerforge/core/__init__.py +14 -0
- fakerforge-0.2.0/src/fakerforge/core/errors.py +36 -0
- fakerforge-0.2.0/src/fakerforge/core/plan.py +39 -0
- fakerforge-0.2.0/src/fakerforge/core/project.py +247 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/faker.py +76 -1
- fakerforge-0.2.0/src/fakerforge/generation/__init__.py +6 -0
- fakerforge-0.2.0/src/fakerforge/generation/batch.py +167 -0
- fakerforge-0.2.0/src/fakerforge/generation/keys.py +78 -0
- fakerforge-0.2.0/src/fakerforge/graph/__init__.py +17 -0
- fakerforge-0.2.0/src/fakerforge/graph/dag.py +286 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/schema/dataset.py +3 -32
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge.egg-info/SOURCES.txt +12 -0
- fakerforge-0.2.0/tests/test_batch_generation.py +104 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_faker.py +1 -1
- fakerforge-0.2.0/tests/test_graph.py +76 -0
- fakerforge-0.2.0/tests/test_project_plan.py +144 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/LICENSE +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/MANIFEST.in +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/examples/quickstart.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/setup.cfg +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/constraints/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/constraints/base.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/distributions/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/distributions/base.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/distributions/engine.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/generator/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/generator/generator.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/providers/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/providers/base.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/providers/finance.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/py.typed +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/schema/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/schema/generate.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/schema/result.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/schema/schema.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/validation/__init__.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/validation/dataset.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/validation/report.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/src/fakerforge/validation/validator.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_constraints.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_dataset_schema.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_dataset_validation.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_distribution_engine.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_distributions.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_finance.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_generator.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_packaging.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_providers.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_relationships.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_schema.py +0 -0
- {fakerforge-0.1.0 → fakerforge-0.2.0}/tests/test_validator.py +0 -0
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.2.0 - 2026-10-05
|
|
4
|
+
|
|
5
|
+
- Add a dependency graph and use it for declarative schema order. Parallel groups are reported and execution stays single-threaded.
|
|
6
|
+
- Add `FakerForge.load` and `FakerForge.plan` for JSON projects, and for YAML when PyYAML is installed. Planning validates providers and dependencies and does not generate rows.
|
|
7
|
+
- Add `FakerForge.iter_batches`. Concatenating chunks matches `generate` for the same seed. `generate` still returns the full in-memory result.
|
|
8
|
+
|
|
3
9
|
## 0.1.0 - 2026-09-29
|
|
4
10
|
|
|
5
11
|
First public release. Requires Python 3.10 or newer.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fakerforge
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Synthetic data generation built on the Faker library.
|
|
5
5
|
Author: Shardul
|
|
6
6
|
License-Expression: MIT
|
|
@@ -23,6 +23,8 @@ Provides-Extra: numpy
|
|
|
23
23
|
Requires-Dist: numpy>=1.24; extra == "numpy"
|
|
24
24
|
Provides-Extra: pandas
|
|
25
25
|
Requires-Dist: pandas>=2; extra == "pandas"
|
|
26
|
+
Provides-Extra: yaml
|
|
27
|
+
Requires-Dist: PyYAML>=6; extra == "yaml"
|
|
26
28
|
Provides-Extra: dev
|
|
27
29
|
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
28
30
|
Requires-Dist: ruff>=0.8.0; extra == "dev"
|
|
@@ -219,6 +221,27 @@ rows = Generator(fake).generate(schema, count=3)
|
|
|
219
221
|
|
|
220
222
|
The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
|
|
221
223
|
|
|
224
|
+
## Project plan and batches
|
|
225
|
+
|
|
226
|
+
`load` reads a project document. `plan` checks providers and dependencies and reports generation order, row counts, and tables that can be generated together. It does not draw rows. `estimated_bytes_per_row` is a placeholder of 64 bytes per column, not a measured size.
|
|
227
|
+
|
|
228
|
+
`iter_batches` yields chunks. Concatenating them matches `generate` for the same seed and schema. `generate` still returns the full in-memory dataset. Execution stays single-threaded. The dependency graph names parallel groups for later releases.
|
|
229
|
+
|
|
230
|
+
JSON works with the base install. YAML needs `pip install "fakerforge[yaml]"`. Documents are parsed as data and are not executed. Distribution draws stay reproducible for a given backend. NumPy and the standard library do not emit the same series, so reproducibility includes whether NumPy is installed.
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
from fakerforge import FakerForge
|
|
234
|
+
|
|
235
|
+
forge = FakerForge(seed=42)
|
|
236
|
+
project = forge.load("examples/projects/minimal/project.json")
|
|
237
|
+
plan = forge.plan(project)
|
|
238
|
+
|
|
239
|
+
for batch in forge.iter_batches(project.tables, batch_size=2):
|
|
240
|
+
print(batch.table, len(batch))
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
`ARCHITECTURE_AUDIT.md` describes the current architecture and the roadmap.
|
|
244
|
+
|
|
222
245
|
## This release
|
|
223
246
|
|
|
224
247
|
| Piece | What 0.1 provides |
|
|
@@ -186,6 +186,27 @@ rows = Generator(fake).generate(schema, count=3)
|
|
|
186
186
|
|
|
187
187
|
The source distribution includes `examples/quickstart.py` and `docs/architecture.md`.
|
|
188
188
|
|
|
189
|
+
## Project plan and batches
|
|
190
|
+
|
|
191
|
+
`load` reads a project document. `plan` checks providers and dependencies and reports generation order, row counts, and tables that can be generated together. It does not draw rows. `estimated_bytes_per_row` is a placeholder of 64 bytes per column, not a measured size.
|
|
192
|
+
|
|
193
|
+
`iter_batches` yields chunks. Concatenating them matches `generate` for the same seed and schema. `generate` still returns the full in-memory dataset. Execution stays single-threaded. The dependency graph names parallel groups for later releases.
|
|
194
|
+
|
|
195
|
+
JSON works with the base install. YAML needs `pip install "fakerforge[yaml]"`. Documents are parsed as data and are not executed. Distribution draws stay reproducible for a given backend. NumPy and the standard library do not emit the same series, so reproducibility includes whether NumPy is installed.
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from fakerforge import FakerForge
|
|
199
|
+
|
|
200
|
+
forge = FakerForge(seed=42)
|
|
201
|
+
project = forge.load("examples/projects/minimal/project.json")
|
|
202
|
+
plan = forge.plan(project)
|
|
203
|
+
|
|
204
|
+
for batch in forge.iter_batches(project.tables, batch_size=2):
|
|
205
|
+
print(batch.table, len(batch))
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
`ARCHITECTURE_AUDIT.md` describes the current architecture and the roadmap.
|
|
209
|
+
|
|
189
210
|
## This release
|
|
190
211
|
|
|
191
212
|
| Piece | What 0.1 provides |
|
|
@@ -12,6 +12,7 @@ DistributionEngine seeded uniform, normal, log-normal, categorical
|
|
|
12
12
|
└── FakerForge.number / FakerForge.categorical
|
|
13
13
|
|
|
14
14
|
DatasetSchema / DatabaseSchema
|
|
15
|
+
└── DependencyGraph stable topological order and parallel groups
|
|
15
16
|
└── FakerForge.generate validates order, then fills rows
|
|
16
17
|
├── derived fields and date dependencies
|
|
17
18
|
├── foreign keys sampled from parent rows
|
|
@@ -19,6 +20,9 @@ DatasetSchema / DatabaseSchema
|
|
|
19
20
|
├── frame is a pandas.DataFrame when pandas is installed
|
|
20
21
|
└── validate() runs DatasetCheck subclasses
|
|
21
22
|
|
|
23
|
+
FakerForge.load / plan project document, no row draws
|
|
24
|
+
FakerForge.iter_batches same row function, fixed-size chunks
|
|
25
|
+
|
|
22
26
|
Schema (Field...)
|
|
23
27
|
└── Generator calls provider methods on a FakerForge
|
|
24
28
|
|
|
@@ -26,6 +30,10 @@ Constraint subclasses
|
|
|
26
30
|
└── Validator checks a finished record against constraints
|
|
27
31
|
```
|
|
28
32
|
|
|
29
|
-
`DistributionEngine` keeps its own random stream so schema generation can sample numbers without consuming Faker provider draws. NumPy is optional.
|
|
33
|
+
`DistributionEngine` keeps its own random stream so schema generation can sample numbers without consuming Faker provider draws. NumPy is optional. Each backend is deterministic, and the two backends do not emit the same series.
|
|
34
|
+
|
|
35
|
+
`FakerForge.load` reads a project document (`project.name`, optional `project.seed`, and `tables`). `FakerForge.plan` checks providers and dependencies and returns generation order, row counts, and parallel groups. It does not draw rows. `estimated_bytes_per_row` is a placeholder of 64 bytes per column.
|
|
36
|
+
|
|
37
|
+
`FakerForge.iter_batches` yields fixed-size chunks through the same row function as `generate`. `generate` still materializes every row. `DependencyGraph` is the topological sort used by declarative schemas.
|
|
30
38
|
|
|
31
|
-
Finance is the first built-in domain provider. Many-to-many relationships are still later work.
|
|
39
|
+
Finance is the first built-in domain provider. Many-to-many relationships are still later work. The roadmap is in `ARCHITECTURE_AUDIT.md`.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "fakerforge"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Synthetic data generation built on the Faker library."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -31,6 +31,7 @@ dependencies = [
|
|
|
31
31
|
[project.optional-dependencies]
|
|
32
32
|
numpy = ["numpy>=1.24"]
|
|
33
33
|
pandas = ["pandas>=2"]
|
|
34
|
+
yaml = ["PyYAML>=6"]
|
|
34
35
|
dev = [
|
|
35
36
|
"pytest>=8.0",
|
|
36
37
|
"ruff>=0.8.0",
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Project documents and generation plans."""
|
|
2
|
+
|
|
3
|
+
from fakerforge.core.errors import FakerForgeError, PlanError
|
|
4
|
+
from fakerforge.core.plan import ESTIMATED_BYTES_PER_COLUMN, GenerationPlan
|
|
5
|
+
from fakerforge.core.project import Project, load_project
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"ESTIMATED_BYTES_PER_COLUMN",
|
|
9
|
+
"FakerForgeError",
|
|
10
|
+
"GenerationPlan",
|
|
11
|
+
"PlanError",
|
|
12
|
+
"Project",
|
|
13
|
+
"load_project",
|
|
14
|
+
]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Errors for project documents and generation plans.
|
|
2
|
+
|
|
3
|
+
``SchemaError`` and ``GenerationError`` keep their original bases.
|
|
4
|
+
``isinstance`` checks against ``ValueError`` and ``RuntimeError`` stay valid.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class FakerForgeError(Exception):
|
|
9
|
+
"""Base for project loading and planning failures."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class PlanError(FakerForgeError):
|
|
13
|
+
"""Raised when a project document or dependency graph is invalid.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
message: Human-readable failure.
|
|
17
|
+
project: Project name, when it is known.
|
|
18
|
+
table: Table name, when the failure is in one table.
|
|
19
|
+
column: Column name, when the failure is in one column.
|
|
20
|
+
dependency: Related table, column, or cycle description.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
def __init__(
|
|
24
|
+
self,
|
|
25
|
+
message: str,
|
|
26
|
+
*,
|
|
27
|
+
project: str | None = None,
|
|
28
|
+
table: str | None = None,
|
|
29
|
+
column: str | None = None,
|
|
30
|
+
dependency: str | None = None,
|
|
31
|
+
) -> None:
|
|
32
|
+
self.project = project
|
|
33
|
+
self.table = table
|
|
34
|
+
self.column = column
|
|
35
|
+
self.dependency = dependency
|
|
36
|
+
super().__init__(message)
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Dry-run summary of a project."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
# Placeholder allowance used until a measured size exists. Not a benchmark.
|
|
8
|
+
ESTIMATED_BYTES_PER_COLUMN = 64
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True, slots=True)
|
|
12
|
+
class GenerationPlan:
|
|
13
|
+
"""Tables, row counts, and parallel groups for a project.
|
|
14
|
+
|
|
15
|
+
``estimated_bytes_per_row`` multiplies
|
|
16
|
+
:data:`ESTIMATED_BYTES_PER_COLUMN` by the column count. That figure is
|
|
17
|
+
a placeholder, not a measurement of generated data or Parquet size.
|
|
18
|
+
|
|
19
|
+
Building a plan does not draw provider values or distribution samples.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
project: Project name from the document.
|
|
23
|
+
seed: Seed recorded on the project. ``None`` when the document
|
|
24
|
+
omitted it.
|
|
25
|
+
order: Table names in generation order.
|
|
26
|
+
row_counts: Requested rows for each table.
|
|
27
|
+
parallel_groups: Tables that do not depend on each other. Names
|
|
28
|
+
inside a group stay in declaration order.
|
|
29
|
+
estimated_bytes_per_row: Placeholder bytes for one row of each table.
|
|
30
|
+
total_rows: Sum of the requested row counts.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
project: str
|
|
34
|
+
seed: Any
|
|
35
|
+
order: tuple[str, ...]
|
|
36
|
+
row_counts: Mapping[str, int]
|
|
37
|
+
parallel_groups: tuple[tuple[str, ...], ...]
|
|
38
|
+
estimated_bytes_per_row: Mapping[str, int]
|
|
39
|
+
total_rows: int
|
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""Project documents in memory, JSON, and YAML."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from importlib import import_module
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from types import MappingProxyType
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from fakerforge.core.errors import PlanError
|
|
11
|
+
from fakerforge.core.plan import ESTIMATED_BYTES_PER_COLUMN, GenerationPlan
|
|
12
|
+
from fakerforge.graph.dag import DependencyGraph
|
|
13
|
+
from fakerforge.schema.dataset import DatabaseSchema, SchemaError
|
|
14
|
+
from fakerforge.schema.generate import _resolve_providers
|
|
15
|
+
|
|
16
|
+
# Same values ``FakerForge`` accepts, without importing that module.
|
|
17
|
+
Seed = int | float | str | bytes | bytearray | None
|
|
18
|
+
|
|
19
|
+
_TOP_KEYS = frozenset({"project", "tables"})
|
|
20
|
+
_PROJECT_KEYS = frozenset({"name", "seed"})
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Project:
|
|
24
|
+
"""A named relational schema plus the seed recorded for it.
|
|
25
|
+
|
|
26
|
+
The ``tables`` mapping uses the same shape as ``FakerForge.generate``:
|
|
27
|
+
each table has ``rows`` and ``fields``. Loading a document does not
|
|
28
|
+
generate rows and does not reseed a forge.
|
|
29
|
+
|
|
30
|
+
Args:
|
|
31
|
+
name: Project name.
|
|
32
|
+
seed: Seed from the document. ``None`` when the document omitted it.
|
|
33
|
+
tables: Relational schema mapping.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
name: str,
|
|
39
|
+
seed: Seed,
|
|
40
|
+
tables: Mapping[str, Any],
|
|
41
|
+
) -> None:
|
|
42
|
+
if not isinstance(name, str) or name == "":
|
|
43
|
+
raise PlanError("project name must be a non-empty string")
|
|
44
|
+
if not isinstance(tables, Mapping):
|
|
45
|
+
raise PlanError("tables must be a mapping", project=name)
|
|
46
|
+
self.name = name
|
|
47
|
+
self.seed = seed
|
|
48
|
+
self.tables = dict(tables)
|
|
49
|
+
|
|
50
|
+
def schema(self) -> DatabaseSchema:
|
|
51
|
+
"""Parse ``tables`` into a relational schema.
|
|
52
|
+
|
|
53
|
+
Circular dependencies raise :class:`PlanError`. Other schema
|
|
54
|
+
problems raise :class:`~fakerforge.schema.SchemaError`.
|
|
55
|
+
|
|
56
|
+
Returns:
|
|
57
|
+
The parsed database schema.
|
|
58
|
+
"""
|
|
59
|
+
try:
|
|
60
|
+
return DatabaseSchema.from_dict(self.tables)
|
|
61
|
+
except SchemaError as exc:
|
|
62
|
+
circular = _circular_error(self.name, exc)
|
|
63
|
+
if circular is not None:
|
|
64
|
+
raise circular from exc
|
|
65
|
+
raise
|
|
66
|
+
|
|
67
|
+
def build_dependency_graph(self) -> DependencyGraph:
|
|
68
|
+
"""Return the table and column dependency graph.
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
A graph whose table order matches the schema generation order.
|
|
72
|
+
|
|
73
|
+
Raises:
|
|
74
|
+
PlanError: The tables contain a circular dependency.
|
|
75
|
+
SchemaError: The tables are otherwise invalid.
|
|
76
|
+
"""
|
|
77
|
+
return DependencyGraph.from_database(self.schema())
|
|
78
|
+
|
|
79
|
+
def validate_dependency_graph(self) -> DependencyGraph:
|
|
80
|
+
"""Return the graph when every table can be ordered.
|
|
81
|
+
|
|
82
|
+
Returns:
|
|
83
|
+
The dependency graph.
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
PlanError: A table or field cycle is present.
|
|
87
|
+
SchemaError: The tables are otherwise invalid.
|
|
88
|
+
"""
|
|
89
|
+
graph = self.build_dependency_graph()
|
|
90
|
+
if graph.cycle:
|
|
91
|
+
raise PlanError(
|
|
92
|
+
f"circular table dependency: {', '.join(graph.cycle)}",
|
|
93
|
+
project=self.name,
|
|
94
|
+
dependency=", ".join(graph.cycle),
|
|
95
|
+
)
|
|
96
|
+
return graph
|
|
97
|
+
|
|
98
|
+
def generation_plan(self, forge: Any) -> GenerationPlan:
|
|
99
|
+
"""Validate providers and dependencies without generating rows.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
forge: FakerForge instance whose registered providers are checked.
|
|
103
|
+
|
|
104
|
+
Returns:
|
|
105
|
+
Row counts, generation order, and parallel groups.
|
|
106
|
+
|
|
107
|
+
Raises:
|
|
108
|
+
PlanError: The document has a circular dependency.
|
|
109
|
+
SchemaError: A field, provider, or row count is invalid.
|
|
110
|
+
"""
|
|
111
|
+
graph = self.validate_dependency_graph()
|
|
112
|
+
database = self.schema()
|
|
113
|
+
for table in database.in_generation_order():
|
|
114
|
+
_resolve_providers(forge, table.dataset, table.rows)
|
|
115
|
+
counts = {table.name: table.rows for table in database.tables}
|
|
116
|
+
estimates = {
|
|
117
|
+
table.name: ESTIMATED_BYTES_PER_COLUMN * len(table.dataset.columns)
|
|
118
|
+
for table in database.tables
|
|
119
|
+
}
|
|
120
|
+
return GenerationPlan(
|
|
121
|
+
project=self.name,
|
|
122
|
+
seed=self.seed,
|
|
123
|
+
order=graph.order,
|
|
124
|
+
row_counts=MappingProxyType(counts),
|
|
125
|
+
parallel_groups=graph.levels,
|
|
126
|
+
estimated_bytes_per_row=MappingProxyType(estimates),
|
|
127
|
+
total_rows=sum(counts[name] for name in graph.order),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def load_project(source: str | Path | Mapping[str, Any]) -> Project:
|
|
132
|
+
"""Load a project from a mapping or a JSON or YAML file.
|
|
133
|
+
|
|
134
|
+
YAML requires PyYAML (``pip install 'fakerforge[yaml]'``). Files are
|
|
135
|
+
parsed as data. They are not executed.
|
|
136
|
+
|
|
137
|
+
Args:
|
|
138
|
+
source: A project mapping, or a path to ``.json``, ``.yaml``, or
|
|
139
|
+
``.yml``.
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
The parsed project. Tables are not validated until
|
|
143
|
+
:meth:`Project.schema` or :meth:`Project.generation_plan`.
|
|
144
|
+
|
|
145
|
+
Raises:
|
|
146
|
+
PlanError: The document, path, or YAML dependency is invalid.
|
|
147
|
+
"""
|
|
148
|
+
document = _read_document(source)
|
|
149
|
+
if not isinstance(document, Mapping):
|
|
150
|
+
raise PlanError("project document must be a mapping")
|
|
151
|
+
unknown = sorted(set(document) - _TOP_KEYS)
|
|
152
|
+
if unknown:
|
|
153
|
+
raise PlanError(f"unknown project keys {unknown}")
|
|
154
|
+
if "project" not in document or "tables" not in document:
|
|
155
|
+
raise PlanError("project document requires 'project' and 'tables'")
|
|
156
|
+
meta = document["project"]
|
|
157
|
+
if not isinstance(meta, Mapping):
|
|
158
|
+
raise PlanError("project must be a mapping")
|
|
159
|
+
unknown_meta = sorted(str(key) for key in set(meta) - _PROJECT_KEYS)
|
|
160
|
+
if unknown_meta:
|
|
161
|
+
raise PlanError(f"unknown keys in project: {unknown_meta}")
|
|
162
|
+
name = meta.get("name")
|
|
163
|
+
if not isinstance(name, str) or name == "":
|
|
164
|
+
raise PlanError("project name must be a non-empty string")
|
|
165
|
+
seed = _parse_seed(meta["seed"], name) if "seed" in meta else None
|
|
166
|
+
tables = document["tables"]
|
|
167
|
+
if not isinstance(tables, Mapping):
|
|
168
|
+
raise PlanError("tables must be a mapping", project=name)
|
|
169
|
+
return Project(name=name, seed=seed, tables=tables)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _read_document(source: str | Path | Mapping[str, Any]) -> Any:
|
|
173
|
+
if isinstance(source, Mapping):
|
|
174
|
+
return source
|
|
175
|
+
path = Path(source)
|
|
176
|
+
if not path.is_file():
|
|
177
|
+
raise PlanError(f"project file {path} does not exist")
|
|
178
|
+
text = path.read_text(encoding="utf-8")
|
|
179
|
+
suffix = path.suffix.lower()
|
|
180
|
+
if suffix == ".json":
|
|
181
|
+
try:
|
|
182
|
+
return json.loads(text)
|
|
183
|
+
except json.JSONDecodeError as exc:
|
|
184
|
+
raise PlanError(f"invalid JSON in {path}: {exc.msg}") from exc
|
|
185
|
+
if suffix in {".yaml", ".yml"}:
|
|
186
|
+
return _read_yaml(path, text)
|
|
187
|
+
raise PlanError(f"project file {path} must be JSON (.json) or YAML (.yaml, .yml)")
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _read_yaml(path: Path, text: str) -> Any:
|
|
191
|
+
yaml = _import_yaml()
|
|
192
|
+
if yaml is None:
|
|
193
|
+
raise PlanError(
|
|
194
|
+
"YAML projects require PyYAML. "
|
|
195
|
+
"Install it with: pip install 'fakerforge[yaml]'"
|
|
196
|
+
)
|
|
197
|
+
try:
|
|
198
|
+
loaded = yaml.safe_load(text)
|
|
199
|
+
except Exception as exc:
|
|
200
|
+
if type(exc).__module__.split(".")[0] != "yaml":
|
|
201
|
+
raise
|
|
202
|
+
raise PlanError(f"invalid YAML in {path}") from exc
|
|
203
|
+
return loaded
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _import_yaml() -> Any:
|
|
207
|
+
try:
|
|
208
|
+
return import_module("yaml")
|
|
209
|
+
except ImportError:
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _parse_seed(value: object, project: str) -> Seed:
|
|
214
|
+
if value is None or isinstance(value, (str, bytes, bytearray)):
|
|
215
|
+
return value
|
|
216
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
217
|
+
raise PlanError(
|
|
218
|
+
"project seed must be an int, float, str, or null",
|
|
219
|
+
project=project,
|
|
220
|
+
)
|
|
221
|
+
return value
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _circular_error(project: str, exc: SchemaError) -> PlanError | None:
|
|
225
|
+
circular = [item for item in exc.errors if "circular" in item]
|
|
226
|
+
if not circular:
|
|
227
|
+
return None
|
|
228
|
+
message = circular[0]
|
|
229
|
+
dependency = message.split(":", 1)[1].strip() if ":" in message else message
|
|
230
|
+
table = _table_name(message)
|
|
231
|
+
return PlanError(
|
|
232
|
+
message,
|
|
233
|
+
project=project,
|
|
234
|
+
table=table,
|
|
235
|
+
dependency=dependency,
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _table_name(message: str) -> str | None:
|
|
240
|
+
marker = "table '"
|
|
241
|
+
if marker not in message:
|
|
242
|
+
return None
|
|
243
|
+
start = message.index(marker) + len(marker)
|
|
244
|
+
end = message.find("'", start)
|
|
245
|
+
if end == -1:
|
|
246
|
+
return None
|
|
247
|
+
return message[start:end]
|
|
@@ -1,12 +1,16 @@
|
|
|
1
1
|
"""Faker-compatible entry point."""
|
|
2
2
|
|
|
3
|
-
from collections.abc import Mapping, Sequence
|
|
3
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
4
|
+
from pathlib import Path
|
|
4
5
|
from typing import Any
|
|
5
6
|
|
|
6
7
|
from faker import Faker
|
|
7
8
|
from faker.providers import BaseProvider
|
|
8
9
|
|
|
10
|
+
from fakerforge.core.plan import GenerationPlan
|
|
11
|
+
from fakerforge.core.project import Project, load_project
|
|
9
12
|
from fakerforge.distributions.engine import DistributionEngine
|
|
13
|
+
from fakerforge.generation.batch import Batch, iter_batches
|
|
10
14
|
from fakerforge.providers.finance import FinanceProvider
|
|
11
15
|
from fakerforge.schema.dataset import DatabaseSchema, DatasetSchema
|
|
12
16
|
from fakerforge.schema.generate import generate_dataset
|
|
@@ -178,6 +182,77 @@ class FakerForge:
|
|
|
178
182
|
"""
|
|
179
183
|
return generate_dataset(self, schema, rows=rows)
|
|
180
184
|
|
|
185
|
+
def load(self, source: str | Path | Mapping[str, Any]) -> Project:
|
|
186
|
+
"""Load a project document.
|
|
187
|
+
|
|
188
|
+
The document has ``project.name``, an optional ``project.seed``, and
|
|
189
|
+
``tables`` in the same shape ``generate`` accepts for a relational
|
|
190
|
+
schema. JSON and YAML files are parsed as data. This method does not
|
|
191
|
+
generate rows and does not reseed this instance.
|
|
192
|
+
|
|
193
|
+
Args:
|
|
194
|
+
source: A project mapping, or a path to ``.json``, ``.yaml``, or
|
|
195
|
+
``.yml``. YAML requires ``pip install 'fakerforge[yaml]'``.
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
The loaded project. Call :meth:`plan` to validate it.
|
|
199
|
+
|
|
200
|
+
Raises:
|
|
201
|
+
PlanError: The document or file is invalid.
|
|
202
|
+
"""
|
|
203
|
+
return load_project(source)
|
|
204
|
+
|
|
205
|
+
def plan(self, project: Project) -> GenerationPlan:
|
|
206
|
+
"""Validate a project and describe how it would be generated.
|
|
207
|
+
|
|
208
|
+
The plan reports row counts, generation order, and tables that do
|
|
209
|
+
not depend on each other. It checks that providers are registered.
|
|
210
|
+
It does not draw values. ``estimated_bytes_per_row`` is a placeholder
|
|
211
|
+
of 64 bytes per column, not a measured size.
|
|
212
|
+
|
|
213
|
+
Args:
|
|
214
|
+
project: Project returned by :meth:`load`.
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
The generation plan.
|
|
218
|
+
|
|
219
|
+
Raises:
|
|
220
|
+
PlanError: The project has a circular dependency.
|
|
221
|
+
SchemaError: A table, field, or provider is invalid.
|
|
222
|
+
"""
|
|
223
|
+
return project.generation_plan(self)
|
|
224
|
+
|
|
225
|
+
def iter_batches(
|
|
226
|
+
self,
|
|
227
|
+
schema: Mapping[str, Any] | DatasetSchema | DatabaseSchema,
|
|
228
|
+
*,
|
|
229
|
+
rows: int | Mapping[str, int] | None = None,
|
|
230
|
+
batch_size: int = 1000,
|
|
231
|
+
) -> Iterator[Batch]:
|
|
232
|
+
"""Yield generated rows in chunks of at most ``batch_size``.
|
|
233
|
+
|
|
234
|
+
Concatenating the chunks matches :meth:`generate` for the same
|
|
235
|
+
seed and schema. This method does not keep earlier chunks. Use
|
|
236
|
+
:meth:`generate` when the full in-memory result is required.
|
|
237
|
+
|
|
238
|
+
Args:
|
|
239
|
+
schema: A field mapping, dataset, table mapping, or database.
|
|
240
|
+
rows: Row count for one table, or per-table counts. Relational
|
|
241
|
+
tables can store ``rows`` on each table instead.
|
|
242
|
+
batch_size: Maximum rows in each chunk. The default is 1000.
|
|
243
|
+
|
|
244
|
+
Yields:
|
|
245
|
+
:class:`~fakerforge.generation.Batch` objects in generation
|
|
246
|
+
order.
|
|
247
|
+
|
|
248
|
+
Raises:
|
|
249
|
+
ValueError: ``rows`` or ``batch_size`` is invalid.
|
|
250
|
+
SchemaError: The schema or its providers are invalid.
|
|
251
|
+
GenerationError: A unique value or foreign key cannot be filled
|
|
252
|
+
without inventing a value.
|
|
253
|
+
"""
|
|
254
|
+
return iter_batches(self, schema, rows=rows, batch_size=batch_size)
|
|
255
|
+
|
|
181
256
|
def add_provider(self, provider: type[BaseProvider] | BaseProvider) -> None:
|
|
182
257
|
"""Register a Faker provider on the underlying instance.
|
|
183
258
|
|