model2data 1.3.1__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.3.1/model2data.egg-info → model2data-1.4.0}/PKG-INFO +17 -3
- {model2data-1.3.1 → model2data-1.4.0}/README.md +28 -2
- {model2data-1.3.1 → model2data-1.4.0}/README_PYPI.md +16 -2
- {model2data-1.3.1 → model2data-1.4.0}/model2data/cli.py +82 -2
- {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/core.py +102 -5
- {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/faker.py +67 -12
- {model2data-1.3.1 → model2data-1.4.0/model2data.egg-info}/PKG-INFO +17 -3
- {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/SOURCES.txt +3 -1
- {model2data-1.3.1 → model2data-1.4.0}/pyproject.toml +1 -1
- model2data-1.4.0/tests/test_as_of_anchor.py +139 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_cli.py +95 -0
- model2data-1.4.0/tests/test_table_seeds.py +225 -0
- {model2data-1.3.1 → model2data-1.4.0}/LICENSE +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/__init__.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/__init__.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/project.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/tests.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/__init__.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/relationships.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/parse/__init__.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/parse/dbml.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data/utils.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/setup.cfg +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbml_parser.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_integration.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_naming.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_project.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_tests.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_faker_name_inference.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_generation.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_release_stress.py +0 -0
- {model2data-1.3.1 → model2data-1.4.0}/tests/test_row_identity.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Author: JB Analytica
|
|
6
6
|
License-Expression: MIT
|
|
@@ -85,7 +85,10 @@ access required.
|
|
|
85
85
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
86
86
|
dependency order.
|
|
87
87
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
88
|
-
commit fixtures, safe to diff across CI runs.
|
|
88
|
+
commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
|
|
89
|
+
on, and the run reproduces on any later day rather than only on the day it first ran.
|
|
90
|
+
- **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
|
|
91
|
+
leaves every other table byte-identical, so you can keep the four tables that look right.
|
|
89
92
|
- **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
|
|
90
93
|
and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
|
|
91
94
|
A single `dbt build` loads, transforms, and tests the whole thing.
|
|
@@ -136,6 +139,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
136
139
|
|
|
137
140
|
This creates a `dbt_ecommerce/` folder with your data and dbt setup.
|
|
138
141
|
|
|
142
|
+
`--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
|
|
143
|
+
current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
|
|
144
|
+
anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
|
|
145
|
+
fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
|
|
146
|
+
re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
|
|
147
|
+
the country every generated person and address comes from:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
|
|
151
|
+
--as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
|
|
152
|
+
```
|
|
153
|
+
|
|
139
154
|
Run dbt to load, transform, and test the data:
|
|
140
155
|
|
|
141
156
|
```bash
|
|
@@ -274,7 +289,6 @@ wants to pick them up as a contribution:
|
|
|
274
289
|
parsed schema shape.
|
|
275
290
|
- Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
|
|
276
291
|
ready-to-uncomment `marts` schema/materialization config for this).
|
|
277
|
-
- Locale-aware generation (`--locale`) for non-English/US synthetic data.
|
|
278
292
|
|
|
279
293
|
See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
|
|
280
294
|
|
|
@@ -44,7 +44,10 @@ access required.
|
|
|
44
44
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
45
45
|
dependency order.
|
|
46
46
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
47
|
-
commit fixtures, safe to diff across CI runs.
|
|
47
|
+
commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
|
|
48
|
+
on, and the run reproduces on any later day rather than only on the day it first ran.
|
|
49
|
+
- **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
|
|
50
|
+
leaves every other table byte-identical, so you can keep the four tables that look right.
|
|
48
51
|
- **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
|
|
49
52
|
and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
|
|
50
53
|
A single `dbt build` loads, transforms, and tests the whole thing.
|
|
@@ -138,6 +141,30 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
|
|
|
138
141
|
--rows-for customers=50 --rows-for order_items=5000
|
|
139
142
|
```
|
|
140
143
|
|
|
144
|
+
`--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
|
|
145
|
+
current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
|
|
146
|
+
anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
|
|
147
|
+
fixture safe to commit:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --as-of 2026-01-31
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
If one table comes out wrong and the rest looks right, `--table-seed` re-rolls just that table.
|
|
154
|
+
Every other table's seed CSV stays byte-identical, and children of the re-rolled table still
|
|
155
|
+
reference rows that exist, so there's nothing to re-check but the table you asked to change:
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --table-seed orders=7
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
`--locale` picks the country every generated person and address comes from (`en_US` by default);
|
|
162
|
+
it's a per-run setting, so a table can't end up holding one Belgian and one American address:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
|
|
166
|
+
```
|
|
167
|
+
|
|
141
168
|
Run dbt to load, transform, and test the data:
|
|
142
169
|
|
|
143
170
|
```bash
|
|
@@ -277,7 +304,6 @@ wants to pick them up as a contribution:
|
|
|
277
304
|
parsed schema shape.
|
|
278
305
|
- Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
|
|
279
306
|
ready-to-uncomment `marts` schema/materialization config for this).
|
|
280
|
-
- Locale-aware generation (`--locale`) for non-English/US synthetic data.
|
|
281
307
|
|
|
282
308
|
See [CONTRIBUTING.md](CONTRIBUTING.md) if you'd like to work on any of these.
|
|
283
309
|
|
|
@@ -41,7 +41,10 @@ access required.
|
|
|
41
41
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
42
42
|
dependency order.
|
|
43
43
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
44
|
-
commit fixtures, safe to diff across CI runs.
|
|
44
|
+
commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
|
|
45
|
+
on, and the run reproduces on any later day rather than only on the day it first ran.
|
|
46
|
+
- **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
|
|
47
|
+
leaves every other table byte-identical, so you can keep the four tables that look right.
|
|
45
48
|
- **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
|
|
46
49
|
and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
|
|
47
50
|
A single `dbt build` loads, transforms, and tests the whole thing.
|
|
@@ -92,6 +95,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
92
95
|
|
|
93
96
|
This creates a `dbt_ecommerce/` folder with your data and dbt setup.
|
|
94
97
|
|
|
98
|
+
`--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
|
|
99
|
+
current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
|
|
100
|
+
anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
|
|
101
|
+
fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
|
|
102
|
+
re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
|
|
103
|
+
the country every generated person and address comes from:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
|
|
107
|
+
--as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
|
|
108
|
+
```
|
|
109
|
+
|
|
95
110
|
Run dbt to load, transform, and test the data:
|
|
96
111
|
|
|
97
112
|
```bash
|
|
@@ -230,7 +245,6 @@ wants to pick them up as a contribution:
|
|
|
230
245
|
parsed schema shape.
|
|
231
246
|
- Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
|
|
232
247
|
ready-to-uncomment `marts` schema/materialization config for this).
|
|
233
|
-
- Locale-aware generation (`--locale`) for non-English/US synthetic data.
|
|
234
248
|
|
|
235
249
|
See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
|
|
236
250
|
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import random
|
|
2
2
|
import shutil
|
|
3
|
+
from datetime import datetime
|
|
3
4
|
from pathlib import Path
|
|
4
5
|
from typing import Optional
|
|
5
6
|
|
|
@@ -75,6 +76,43 @@ def _parse_row_overrides(
|
|
|
75
76
|
return overrides
|
|
76
77
|
|
|
77
78
|
|
|
79
|
+
def _parse_table_seeds(
|
|
80
|
+
raw: Optional[list[str]],
|
|
81
|
+
tables: dict,
|
|
82
|
+
) -> dict[str, int]:
|
|
83
|
+
"""Turn repeated `--table-seed TABLE=N` values into a {table: seed} mapping.
|
|
84
|
+
|
|
85
|
+
Same shape and the same loud failure on an unknown name as `--rows-for`:
|
|
86
|
+
the whole point of naming a table here is to change that table, so a typo
|
|
87
|
+
would otherwise produce a run in which nothing moved and nothing was said.
|
|
88
|
+
"""
|
|
89
|
+
if not isinstance(raw, (list, tuple)):
|
|
90
|
+
return {}
|
|
91
|
+
|
|
92
|
+
table_seeds: dict[str, int] = {}
|
|
93
|
+
for item in raw:
|
|
94
|
+
table_name, separator, value = item.partition("=")
|
|
95
|
+
table_name = table_name.strip()
|
|
96
|
+
if not separator or not table_name:
|
|
97
|
+
raise typer.BadParameter(f"Expected TABLE=N, got {item!r}.", param_hint="--table-seed")
|
|
98
|
+
|
|
99
|
+
try:
|
|
100
|
+
table_seed = int(value)
|
|
101
|
+
except ValueError:
|
|
102
|
+
raise typer.BadParameter(
|
|
103
|
+
f"Seed for {table_name!r} must be a whole number, got {value!r}.",
|
|
104
|
+
param_hint="--table-seed",
|
|
105
|
+
) from None
|
|
106
|
+
if table_name not in tables:
|
|
107
|
+
known = ", ".join(sorted(tables)) or "none"
|
|
108
|
+
raise typer.BadParameter(
|
|
109
|
+
f"No table named {table_name!r} in this schema. Tables: {known}.",
|
|
110
|
+
param_hint="--table-seed",
|
|
111
|
+
)
|
|
112
|
+
table_seeds[table_name] = table_seed
|
|
113
|
+
return table_seeds
|
|
114
|
+
|
|
115
|
+
|
|
78
116
|
app = typer.Typer(
|
|
79
117
|
help=(
|
|
80
118
|
"model2data: Generate analytics-ready datasets from DBML models.\n\n"
|
|
@@ -107,8 +145,11 @@ def main(
|
|
|
107
145
|
min=10,
|
|
108
146
|
help="Number of rows to generate per table.",
|
|
109
147
|
),
|
|
110
|
-
# noqa: B008 is
|
|
111
|
-
#
|
|
148
|
+
# noqa: B008 is needed on some options and not others because ruff waves a
|
|
149
|
+
# call through in a default only when the annotation is one of the types it
|
|
150
|
+
# knows to be immutable. `str`, `int`, `bool` and `Path` are on that list;
|
|
151
|
+
# the `list` a repeatable option must be annotated with, and `datetime`,
|
|
152
|
+
# are not -- neither is actually mutated here.
|
|
112
153
|
rows_for: Optional[list[str]] = typer.Option( # noqa: B008
|
|
113
154
|
None,
|
|
114
155
|
"--rows-for",
|
|
@@ -126,6 +167,25 @@ def main(
|
|
|
126
167
|
"Using the same seed will always produce identical datasets."
|
|
127
168
|
),
|
|
128
169
|
),
|
|
170
|
+
table_seed: Optional[list[str]] = typer.Option( # noqa: B008
|
|
171
|
+
None,
|
|
172
|
+
"--table-seed",
|
|
173
|
+
metavar="TABLE=N",
|
|
174
|
+
help=(
|
|
175
|
+
"Re-roll one table without disturbing the others, keeping --seed for the rest.\n"
|
|
176
|
+
"Repeatable, e.g. --table-seed orders=7. Requires --seed."
|
|
177
|
+
),
|
|
178
|
+
),
|
|
179
|
+
as_of: Optional[datetime] = typer.Option( # noqa: B008
|
|
180
|
+
None,
|
|
181
|
+
"--as-of",
|
|
182
|
+
formats=["%Y-%m-%d"],
|
|
183
|
+
metavar="YYYY-MM-DD",
|
|
184
|
+
help=(
|
|
185
|
+
"Date to anchor generated dates and timestamps on (default: today).\n"
|
|
186
|
+
"Pin it and a --seed run reproduces on any later day, not just the day it first ran."
|
|
187
|
+
),
|
|
188
|
+
),
|
|
129
189
|
locale: Optional[str] = typer.Option(
|
|
130
190
|
None,
|
|
131
191
|
"--locale",
|
|
@@ -183,6 +243,14 @@ def main(
|
|
|
183
243
|
Faker.seed(seed)
|
|
184
244
|
typer.echo(f"🔁 Using deterministic seed: {seed}")
|
|
185
245
|
|
|
246
|
+
# Same reason as `_parse_row_overrides`'s isinstance guard: `main` is also
|
|
247
|
+
# called directly as a plain function, which leaves this holding its
|
|
248
|
+
# `OptionInfo` default rather than None. Anything that isn't a real
|
|
249
|
+
# datetime means "not supplied", i.e. anchor on today.
|
|
250
|
+
as_of = as_of if isinstance(as_of, datetime) else None
|
|
251
|
+
if as_of is not None:
|
|
252
|
+
typer.echo(f"📅 Anchoring generated dates on: {as_of.date()}")
|
|
253
|
+
|
|
186
254
|
# -------------------------
|
|
187
255
|
# Parse DBML (names untouched)
|
|
188
256
|
# -------------------------
|
|
@@ -196,6 +264,16 @@ def main(
|
|
|
196
264
|
# should not leave a half-scaffolded project behind for the next run to trip
|
|
197
265
|
# over with a confusing "destination already exists".
|
|
198
266
|
row_overrides = _parse_row_overrides(rows_for, tables)
|
|
267
|
+
table_seeds = _parse_table_seeds(table_seed, tables)
|
|
268
|
+
if table_seeds and seed is None:
|
|
269
|
+
raise typer.BadParameter(
|
|
270
|
+
"--table-seed re-rolls one table out of the run's seed, so there has to "
|
|
271
|
+
"be one. Add --seed.",
|
|
272
|
+
param_hint="--table-seed",
|
|
273
|
+
)
|
|
274
|
+
if table_seeds:
|
|
275
|
+
rolled = ", ".join(f"{name}={value}" for name, value in sorted(table_seeds.items()))
|
|
276
|
+
typer.echo(f"🎲 Re-rolling with a table seed of its own: {rolled}")
|
|
199
277
|
|
|
200
278
|
project_name = normalize_identifier(name or file.stem)
|
|
201
279
|
dest = Path.cwd() / f"dbt_{project_name}"
|
|
@@ -225,6 +303,8 @@ def main(
|
|
|
225
303
|
seed=seed,
|
|
226
304
|
row_overrides=row_overrides,
|
|
227
305
|
locale=locale,
|
|
306
|
+
as_of=as_of,
|
|
307
|
+
table_seeds=table_seeds,
|
|
228
308
|
)
|
|
229
309
|
|
|
230
310
|
# -------------------------
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import hashlib
|
|
3
4
|
import random
|
|
4
5
|
from collections import defaultdict, deque
|
|
5
6
|
from collections.abc import Mapping
|
|
@@ -9,6 +10,7 @@ import pandas as pd
|
|
|
9
10
|
from faker import Faker
|
|
10
11
|
|
|
11
12
|
from model2data.generate.faker import (
|
|
13
|
+
AsOf,
|
|
12
14
|
generate_column_values,
|
|
13
15
|
release_row_pools,
|
|
14
16
|
reset_duplicate_unique_columns,
|
|
@@ -68,6 +70,8 @@ def generate_data_from_dbml(
|
|
|
68
70
|
seed: Optional[int] = None,
|
|
69
71
|
row_overrides: Optional[Mapping[str, int]] = None,
|
|
70
72
|
locale: Optional[str] = None,
|
|
73
|
+
as_of: AsOf = None,
|
|
74
|
+
table_seeds: Optional[Mapping[str, int]] = None,
|
|
71
75
|
) -> dict[str, pd.DataFrame]:
|
|
72
76
|
"""
|
|
73
77
|
Generate synthetic datasets from parsed DBML definitions.
|
|
@@ -86,9 +90,28 @@ def generate_data_from_dbml(
|
|
|
86
90
|
holding one Belgian and one American address is the incoherence the row
|
|
87
91
|
pools exist to remove.
|
|
88
92
|
|
|
89
|
-
|
|
90
|
-
|
|
93
|
+
`as_of` is the date every generated date and timestamp is placed relative
|
|
94
|
+
to -- dates land in the two years up to it, timestamps in the year up to
|
|
95
|
+
midnight on it -- and defaults to today. Without it a seed reproduces only
|
|
96
|
+
for as long as the day lasts: the numbers come back identical and the dates
|
|
97
|
+
move, so a committed fixture churns and a saved project renders different
|
|
98
|
+
rows next month. Pass the day the run should look like it happened on and
|
|
99
|
+
the whole frame reproduces, on any later day.
|
|
100
|
+
|
|
101
|
+
`table_seeds` re-rolls individual tables without disturbing the rest. Each
|
|
102
|
+
table draws from its own RNG stream, derived from `(seed, table_name,
|
|
103
|
+
table_seeds[table_name])`, so bumping one table's entry changes that
|
|
104
|
+
table's rows and leaves every other table byte-identical. Children of a
|
|
105
|
+
re-rolled table keep pointing at rows that exist, because their foreign
|
|
106
|
+
keys are drawn from whatever their parent ended up holding. It needs a
|
|
107
|
+
`seed` to work off -- with none, every table is already different on every
|
|
108
|
+
run -- and unknown table names are an error rather than a silent no-op.
|
|
109
|
+
|
|
110
|
+
This function is deterministic if a seed is provided (and, with `as_of`,
|
|
111
|
+
on any day). It performs no filesystem I/O and returns pandas DataFrames.
|
|
91
112
|
"""
|
|
113
|
+
_validate_table_seeds(tables, table_seeds, seed)
|
|
114
|
+
|
|
92
115
|
# Locale first, then the seed: switching locale builds a new Faker, and the
|
|
93
116
|
# seed has to be the last word on the generator that actually runs.
|
|
94
117
|
set_locale(locale)
|
|
@@ -118,6 +141,26 @@ def generate_data_from_dbml(
|
|
|
118
141
|
table_def = tables[table_name]
|
|
119
142
|
row_count = _determine_row_count(table_def.name, base_rows, row_overrides)
|
|
120
143
|
|
|
144
|
+
# Give this table its own RNG stream before a single value of it is
|
|
145
|
+
# drawn. One stream for the whole run meant re-rolling a table
|
|
146
|
+
# re-rolled everything generated after it too -- fine for a one-shot
|
|
147
|
+
# CLI run, useless for a "regenerate just this table" button, which is
|
|
148
|
+
# exactly the thing users ask for once they like four tables out of
|
|
149
|
+
# five. Same locale-then-seed order as the run-level seeding above.
|
|
150
|
+
if seed is not None:
|
|
151
|
+
stream_seed = _table_stream_seed(
|
|
152
|
+
seed, table_name, table_seeds.get(table_name) if table_seeds else None
|
|
153
|
+
)
|
|
154
|
+
random.seed(stream_seed)
|
|
155
|
+
Faker.seed(stream_seed)
|
|
156
|
+
|
|
157
|
+
# A table's identities are its own. `release_row_pools` below already
|
|
158
|
+
# drops the pool keyed by this table's name; this also drops the
|
|
159
|
+
# un-named bucket the composite-key repair shares, so nothing a
|
|
160
|
+
# previous table left behind can reach this one's rows and make its
|
|
161
|
+
# stream depend on what came before it.
|
|
162
|
+
reset_row_pools()
|
|
163
|
+
|
|
121
164
|
data: dict[str, list] = {}
|
|
122
165
|
|
|
123
166
|
# A composite *primary* key's member columns are frequently declared
|
|
@@ -166,11 +209,16 @@ def generate_data_from_dbml(
|
|
|
166
209
|
ensure_unique=ensure_unique,
|
|
167
210
|
force_not_null=column.name in composite_pk_columns,
|
|
168
211
|
table_name=table_name,
|
|
212
|
+
as_of=as_of,
|
|
169
213
|
)
|
|
170
214
|
|
|
171
215
|
df = pd.DataFrame(data)
|
|
172
|
-
df = _resolve_self_referencing_fks(
|
|
173
|
-
|
|
216
|
+
df = _resolve_self_referencing_fks(
|
|
217
|
+
df, table_def, table_name, fk_lookup, row_count, as_of=as_of
|
|
218
|
+
)
|
|
219
|
+
df = _deduplicate_composite_keys(
|
|
220
|
+
df, table_def, table_name, fk_lookup, generated, as_of=as_of
|
|
221
|
+
)
|
|
174
222
|
|
|
175
223
|
# -----------------------------------------------------
|
|
176
224
|
# Second pass: attribute mirroring (non-FK refs)
|
|
@@ -250,6 +298,7 @@ def _deduplicate_composite_keys(
|
|
|
250
298
|
fk_lookup: dict[tuple[str, str], tuple[str, str]],
|
|
251
299
|
generated: dict[str, pd.DataFrame],
|
|
252
300
|
max_attempts: int = 20,
|
|
301
|
+
as_of: AsOf = None,
|
|
253
302
|
) -> pd.DataFrame:
|
|
254
303
|
"""
|
|
255
304
|
Regenerate colliding rows for any pk/unique composite key declared via an
|
|
@@ -300,7 +349,9 @@ def _deduplicate_composite_keys(
|
|
|
300
349
|
if col_name in fk_pools:
|
|
301
350
|
df.at[idx, col_name] = random.choice(fk_pools[col_name])
|
|
302
351
|
else:
|
|
303
|
-
df.at[idx, col_name] = generate_column_values(
|
|
352
|
+
df.at[idx, col_name] = generate_column_values(
|
|
353
|
+
col_def, row_count=1, as_of=as_of
|
|
354
|
+
)[0]
|
|
304
355
|
combo = tuple(df.at[idx, c] for c in key_columns)
|
|
305
356
|
attempts += 1
|
|
306
357
|
# Retry budget spent and still colliding: this row keeps a
|
|
@@ -325,6 +376,7 @@ def _resolve_self_referencing_fks(
|
|
|
325
376
|
table_name: str,
|
|
326
377
|
fk_lookup: dict[tuple[str, str], tuple[str, str]],
|
|
327
378
|
row_count: int,
|
|
379
|
+
as_of: AsOf = None,
|
|
328
380
|
) -> pd.DataFrame:
|
|
329
381
|
"""
|
|
330
382
|
Re-generate any FK column that references its own table (e.g. a
|
|
@@ -362,11 +414,56 @@ def _resolve_self_referencing_fks(
|
|
|
362
414
|
ensure_unique=ensure_unique,
|
|
363
415
|
force_not_null=column.name in composite_pk_columns,
|
|
364
416
|
table_name=table_name,
|
|
417
|
+
as_of=as_of,
|
|
365
418
|
)
|
|
366
419
|
|
|
367
420
|
return df
|
|
368
421
|
|
|
369
422
|
|
|
423
|
+
def _validate_table_seeds(
|
|
424
|
+
tables: dict[str, TableDef],
|
|
425
|
+
table_seeds: Optional[Mapping[str, int]],
|
|
426
|
+
seed: Optional[int],
|
|
427
|
+
) -> None:
|
|
428
|
+
"""Reject a `table_seeds` mapping that cannot do what it was asked to do.
|
|
429
|
+
|
|
430
|
+
Unlike `row_overrides`, which quietly ignores a name it does not know, an
|
|
431
|
+
unknown name here is always a mistake worth stopping for: the caller asked
|
|
432
|
+
for one table to be re-rolled and would otherwise get a run in which
|
|
433
|
+
nothing changed, with nothing said about why. The same goes for passing
|
|
434
|
+
`table_seeds` with no `seed` -- there is no stream to re-roll out of, and
|
|
435
|
+
every table is already different on every run.
|
|
436
|
+
"""
|
|
437
|
+
if not table_seeds:
|
|
438
|
+
return
|
|
439
|
+
|
|
440
|
+
unknown = sorted(name for name in table_seeds if name not in tables)
|
|
441
|
+
if unknown:
|
|
442
|
+
known = ", ".join(sorted(tables)) or "none"
|
|
443
|
+
label = "No table named" if len(unknown) == 1 else "No tables named"
|
|
444
|
+
named = ", ".join(repr(name) for name in unknown)
|
|
445
|
+
raise ValueError(f"{label} {named} in this schema. Tables: {known}.")
|
|
446
|
+
|
|
447
|
+
if seed is None:
|
|
448
|
+
raise ValueError(
|
|
449
|
+
"table_seeds needs a seed to re-roll a table out of: without one every "
|
|
450
|
+
"table is already generated afresh on every run."
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _table_stream_seed(seed: int, table_name: str, table_seed: Optional[int]) -> int:
|
|
455
|
+
"""The RNG seed one table draws from, derived from the run seed and its name.
|
|
456
|
+
|
|
457
|
+
Hashed with blake2b rather than Python's built-in `hash()`, which is salted
|
|
458
|
+
per interpreter process for strings: a "deterministic" seed built on it
|
|
459
|
+
would reproduce only within a single run of the program, which is the one
|
|
460
|
+
place determinism was never in doubt.
|
|
461
|
+
"""
|
|
462
|
+
payload = f"{seed}|{table_name}|{'' if table_seed is None else table_seed}"
|
|
463
|
+
digest = hashlib.blake2b(payload.encode("utf-8"), digest_size=8).digest()
|
|
464
|
+
return int.from_bytes(digest, "big")
|
|
465
|
+
|
|
466
|
+
|
|
370
467
|
def _determine_row_count(
|
|
371
468
|
table_name: str,
|
|
372
469
|
base_rows: int,
|
|
@@ -5,7 +5,7 @@ import re
|
|
|
5
5
|
import unicodedata
|
|
6
6
|
import uuid
|
|
7
7
|
from dataclasses import dataclass
|
|
8
|
-
from datetime import datetime, timedelta
|
|
8
|
+
from datetime import date, datetime, timedelta
|
|
9
9
|
from typing import Callable, Optional, Union
|
|
10
10
|
|
|
11
11
|
import pandas as pd
|
|
@@ -59,6 +59,47 @@ def set_locale(locale: Optional[str]) -> None:
|
|
|
59
59
|
_address_state.clear()
|
|
60
60
|
|
|
61
61
|
|
|
62
|
+
# ---------------------------------------------------------
|
|
63
|
+
# Date anchor
|
|
64
|
+
# ---------------------------------------------------------
|
|
65
|
+
# Every generated date and timestamp is placed relative to a single anchor
|
|
66
|
+
# date, which defaults to today. That default is what made a seeded run
|
|
67
|
+
# reproduce only for as long as the day lasted: re-run tomorrow, the same seed
|
|
68
|
+
# gave the same numbers and different dates, so a committed fixture churned and
|
|
69
|
+
# a shared demo drifted. Callers that need a run to reproduce across days pass
|
|
70
|
+
# `as_of` and pin the window instead.
|
|
71
|
+
AsOf = Union[date, datetime, None]
|
|
72
|
+
|
|
73
|
+
# How far back a `date` column's window reaches from the anchor.
|
|
74
|
+
_DATE_WINDOW_YEARS = 2
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _anchor_date(as_of: AsOf) -> date:
|
|
78
|
+
"""The date a run generates relative to: `as_of`, or today when it is None.
|
|
79
|
+
|
|
80
|
+
A `datetime` is narrowed to its day, so a caller who has a timestamp to
|
|
81
|
+
hand does not have to remember that only the date part is used.
|
|
82
|
+
"""
|
|
83
|
+
if as_of is None:
|
|
84
|
+
return date.today()
|
|
85
|
+
if isinstance(as_of, datetime):
|
|
86
|
+
return as_of.date()
|
|
87
|
+
return as_of
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _years_before(anchor: date, years: int) -> date:
|
|
91
|
+
"""`years` calendar years before `anchor`, moving 29 Feb back to 28 Feb.
|
|
92
|
+
|
|
93
|
+
Only 29 February has no counterpart in a non-leap year, and a whole run
|
|
94
|
+
failing on one day in four years is not a tradeoff worth taking for the
|
|
95
|
+
sake of an exact anniversary.
|
|
96
|
+
"""
|
|
97
|
+
try:
|
|
98
|
+
return anchor.replace(year=anchor.year - years)
|
|
99
|
+
except ValueError:
|
|
100
|
+
return anchor.replace(year=anchor.year - years, month=2, day=28)
|
|
101
|
+
|
|
102
|
+
|
|
62
103
|
# ---------------------------------------------------------
|
|
63
104
|
# Per-row identities
|
|
64
105
|
# ---------------------------------------------------------
|
|
@@ -471,11 +512,16 @@ def generate_column_values(
|
|
|
471
512
|
ensure_unique: bool = False,
|
|
472
513
|
force_not_null: bool = False,
|
|
473
514
|
table_name: Optional[str] = None,
|
|
515
|
+
as_of: AsOf = None,
|
|
474
516
|
) -> list:
|
|
475
517
|
"""
|
|
476
518
|
Generate synthetic values for a single column.
|
|
477
519
|
Respects FKs, uniqueness, and optional min/max hints in column notes.
|
|
478
520
|
|
|
521
|
+
`as_of` is the date every generated date and timestamp is placed relative
|
|
522
|
+
to, defaulting to today. Pass it to make a seeded run reproduce on any
|
|
523
|
+
later day rather than only on the day it first ran.
|
|
524
|
+
|
|
479
525
|
`force_not_null` lets a caller override the nullability pass below for a
|
|
480
526
|
column whose *individual* settings don't carry `not null`/`pk` but is
|
|
481
527
|
still never allowed to be null -- namely a composite primary key member
|
|
@@ -580,13 +626,20 @@ def generate_column_values(
|
|
|
580
626
|
# Dates
|
|
581
627
|
# -----------------------------------------------------
|
|
582
628
|
elif "date" in base_type and "time" not in base_type:
|
|
583
|
-
|
|
629
|
+
# Explicit endpoints rather than Faker's "-2y"/"today" shorthand: those
|
|
630
|
+
# strings are resolved against `date.today()` inside Faker, which is
|
|
631
|
+
# precisely the hidden dependency on the wall clock `as_of` removes.
|
|
632
|
+
anchor = _anchor_date(as_of)
|
|
633
|
+
values = [
|
|
634
|
+
fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
|
|
635
|
+
for _ in range(row_count)
|
|
636
|
+
]
|
|
584
637
|
|
|
585
638
|
elif "time" in base_type and "stamp" not in base_type:
|
|
586
639
|
values = [fake.time() for _ in range(row_count)]
|
|
587
640
|
|
|
588
641
|
elif any(key in base_type for key in ["timestamp", "datetime"]):
|
|
589
|
-
values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
|
|
642
|
+
values = [_random_datetime(as_of=as_of).isoformat(sep=" ") for _ in range(row_count)]
|
|
590
643
|
|
|
591
644
|
# -----------------------------------------------------
|
|
592
645
|
# Untyped / generic string columns: honour a type that names
|
|
@@ -693,17 +746,19 @@ def _suffixed(value: str, counter: int) -> str:
|
|
|
693
746
|
return f"{local}{counter}{at}{domain}"
|
|
694
747
|
|
|
695
748
|
|
|
696
|
-
def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
|
|
697
|
-
"""Pick a random timestamp in a window around
|
|
749
|
+
def _random_datetime(start_days: int = -365, end_days: int = 0, as_of: AsOf = None) -> datetime:
|
|
750
|
+
"""Pick a random timestamp in a window around the anchor, to whole seconds.
|
|
698
751
|
|
|
699
|
-
The window is anchored to midnight
|
|
700
|
-
random offset is a whole number of
|
|
701
|
-
its sub-second component leak straight
|
|
702
|
-
timestamp -- two runs with the same `--seed`
|
|
703
|
-
only in their microseconds, which quietly broke
|
|
704
|
-
`--seed` exists to provide.
|
|
752
|
+
The window is anchored to midnight of `as_of` (today when it is None)
|
|
753
|
+
rather than to a `datetime.now()`. The random offset is a whole number of
|
|
754
|
+
seconds, so anchoring on `now()` let its sub-second component leak straight
|
|
755
|
+
through into every generated timestamp -- two runs with the same `--seed`
|
|
756
|
+
produced values differing only in their microseconds, which quietly broke
|
|
757
|
+
the reproducibility `--seed` exists to provide. Midnight of an explicit
|
|
758
|
+
`as_of` extends that reproducibility past the end of the day.
|
|
705
759
|
"""
|
|
706
|
-
|
|
760
|
+
anchor = _anchor_date(as_of)
|
|
761
|
+
midnight = datetime(anchor.year, anchor.month, anchor.day)
|
|
707
762
|
start = midnight + timedelta(days=start_days)
|
|
708
763
|
end = midnight + timedelta(days=end_days)
|
|
709
764
|
random_second = random.randint(0, int((end - start).total_seconds()))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Author: JB Analytica
|
|
6
6
|
License-Expression: MIT
|
|
@@ -85,7 +85,10 @@ access required.
|
|
|
85
85
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
86
86
|
dependency order.
|
|
87
87
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
88
|
-
commit fixtures, safe to diff across CI runs.
|
|
88
|
+
commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
|
|
89
|
+
on, and the run reproduces on any later day rather than only on the day it first ran.
|
|
90
|
+
- **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
|
|
91
|
+
leaves every other table byte-identical, so you can keep the four tables that look right.
|
|
89
92
|
- **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
|
|
90
93
|
and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
|
|
91
94
|
A single `dbt build` loads, transforms, and tests the whole thing.
|
|
@@ -136,6 +139,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
136
139
|
|
|
137
140
|
This creates a `dbt_ecommerce/` folder with your data and dbt setup.
|
|
138
141
|
|
|
142
|
+
`--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
|
|
143
|
+
current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
|
|
144
|
+
anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
|
|
145
|
+
fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
|
|
146
|
+
re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
|
|
147
|
+
the country every generated person and address comes from:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
|
|
151
|
+
--as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
|
|
152
|
+
```
|
|
153
|
+
|
|
139
154
|
Run dbt to load, transform, and test the data:
|
|
140
155
|
|
|
141
156
|
```bash
|
|
@@ -274,7 +289,6 @@ wants to pick them up as a contribution:
|
|
|
274
289
|
parsed schema shape.
|
|
275
290
|
- Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
|
|
276
291
|
ready-to-uncomment `marts` schema/materialization config for this).
|
|
277
|
-
- Locale-aware generation (`--locale`) for non-English/US synthetic data.
|
|
278
292
|
|
|
279
293
|
See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
|
|
280
294
|
|
|
@@ -23,6 +23,7 @@ model2data/generate/faker.py
|
|
|
23
23
|
model2data/generate/relationships.py
|
|
24
24
|
model2data/parse/__init__.py
|
|
25
25
|
model2data/parse/dbml.py
|
|
26
|
+
tests/test_as_of_anchor.py
|
|
26
27
|
tests/test_cli.py
|
|
27
28
|
tests/test_coverage_gaps.py
|
|
28
29
|
tests/test_dbml_parser.py
|
|
@@ -34,4 +35,5 @@ tests/test_dbt_tests.py
|
|
|
34
35
|
tests/test_faker_name_inference.py
|
|
35
36
|
tests/test_generation.py
|
|
36
37
|
tests/test_release_stress.py
|
|
37
|
-
tests/test_row_identity.py
|
|
38
|
+
tests/test_row_identity.py
|
|
39
|
+
tests/test_table_seeds.py
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""A seed has to reproduce tomorrow, not only for the rest of today.
|
|
2
|
+
|
|
3
|
+
Dates and timestamps were generated relative to the wall clock, so two runs of
|
|
4
|
+
the same seed on different days agreed on every number and disagreed on every
|
|
5
|
+
date. `as_of` pins the anchor instead, and these tests hold both halves of that:
|
|
6
|
+
an explicit anchor survives the day changing underneath it, and the default
|
|
7
|
+
still follows today.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from datetime import date, datetime, timedelta
|
|
11
|
+
|
|
12
|
+
import pytest
|
|
13
|
+
|
|
14
|
+
import model2data.generate.faker as faker_module
|
|
15
|
+
from model2data.generate.core import generate_data_from_dbml
|
|
16
|
+
from model2data.generate.faker import _anchor_date, _years_before
|
|
17
|
+
from model2data.parse.dbml import ColumnDef, TableDef
|
|
18
|
+
|
|
19
|
+
ANCHOR = date(2024, 3, 15)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _dated_table() -> dict[str, TableDef]:
|
|
23
|
+
return {
|
|
24
|
+
"events": TableDef(
|
|
25
|
+
name="events",
|
|
26
|
+
columns=[
|
|
27
|
+
ColumnDef("id", "int", {"pk"}),
|
|
28
|
+
ColumnDef("happened_on", "date", {"not null"}),
|
|
29
|
+
ColumnDef("created_at", "timestamp", {"not null"}),
|
|
30
|
+
],
|
|
31
|
+
)
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _freeze_today(monkeypatch: pytest.MonkeyPatch, today: date) -> None:
|
|
36
|
+
"""Move the day the process thinks it is, the way the calendar would.
|
|
37
|
+
|
|
38
|
+
`date.today()` is the only clock generation reads, so replacing it with a
|
|
39
|
+
subclass that answers a fixed day is enough to run "tomorrow" inside a
|
|
40
|
+
single test.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
class _FrozenDate(date):
|
|
44
|
+
@classmethod
|
|
45
|
+
def today(cls) -> date:
|
|
46
|
+
return today
|
|
47
|
+
|
|
48
|
+
monkeypatch.setattr(faker_module, "date", _FrozenDate)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _frames(**kwargs) -> dict:
|
|
52
|
+
return generate_data_from_dbml(_dated_table(), [], base_rows=30, seed=7, **kwargs)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class TestAnchorHelpers:
|
|
56
|
+
def test_none_means_today(self, monkeypatch: pytest.MonkeyPatch):
|
|
57
|
+
_freeze_today(monkeypatch, date(2031, 8, 9))
|
|
58
|
+
assert _anchor_date(None) == date(2031, 8, 9)
|
|
59
|
+
|
|
60
|
+
def test_a_date_is_used_as_given(self):
|
|
61
|
+
assert _anchor_date(ANCHOR) == ANCHOR
|
|
62
|
+
|
|
63
|
+
def test_a_datetime_is_narrowed_to_its_day(self):
|
|
64
|
+
assert _anchor_date(datetime(2024, 3, 15, 23, 59, 59)) == ANCHOR
|
|
65
|
+
|
|
66
|
+
def test_two_years_back_is_the_same_day_of_the_year(self):
|
|
67
|
+
assert _years_before(date(2024, 3, 15), 2) == date(2022, 3, 15)
|
|
68
|
+
|
|
69
|
+
def test_the_29th_of_february_lands_on_the_28th(self):
|
|
70
|
+
"""The one day of the year with no counterpart two years earlier."""
|
|
71
|
+
assert _years_before(date(2024, 2, 29), 2) == date(2022, 2, 28)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class TestSeededRunsReproduceAcrossDays:
|
|
75
|
+
def test_an_explicit_anchor_survives_the_day_changing(self, monkeypatch: pytest.MonkeyPatch):
|
|
76
|
+
_freeze_today(monkeypatch, date(2024, 3, 15))
|
|
77
|
+
first = _frames(as_of=ANCHOR)["events"]
|
|
78
|
+
|
|
79
|
+
_freeze_today(monkeypatch, date(2026, 11, 2))
|
|
80
|
+
second = _frames(as_of=ANCHOR)["events"]
|
|
81
|
+
|
|
82
|
+
assert first.equals(second)
|
|
83
|
+
|
|
84
|
+
def test_without_an_anchor_the_dates_move_with_the_calendar(
|
|
85
|
+
self, monkeypatch: pytest.MonkeyPatch
|
|
86
|
+
):
|
|
87
|
+
"""The behaviour `as_of` exists to opt out of, pinned so it stays opt-out."""
|
|
88
|
+
_freeze_today(monkeypatch, date(2024, 3, 15))
|
|
89
|
+
first = _frames()["events"]
|
|
90
|
+
|
|
91
|
+
_freeze_today(monkeypatch, date(2026, 11, 2))
|
|
92
|
+
second = _frames()["events"]
|
|
93
|
+
|
|
94
|
+
assert not first["happened_on"].equals(second["happened_on"])
|
|
95
|
+
# Same stream, so the numbers underneath the dates are untouched.
|
|
96
|
+
assert first["id"].equals(second["id"])
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class TestTheWindowMoves:
|
|
100
|
+
def test_dates_fall_in_the_two_years_up_to_the_anchor(self):
|
|
101
|
+
dates = _frames(as_of=ANCHOR)["events"]["happened_on"]
|
|
102
|
+
|
|
103
|
+
assert dates.min() >= date(2022, 3, 15)
|
|
104
|
+
assert dates.max() <= ANCHOR
|
|
105
|
+
|
|
106
|
+
def test_a_different_anchor_shifts_the_window(self):
|
|
107
|
+
earlier = _frames(as_of=date(2019, 1, 1))["events"]["happened_on"]
|
|
108
|
+
later = _frames(as_of=ANCHOR)["events"]["happened_on"]
|
|
109
|
+
|
|
110
|
+
assert earlier.max() <= date(2019, 1, 1)
|
|
111
|
+
assert later.min() >= date(2022, 3, 15)
|
|
112
|
+
assert earlier.max() < later.min()
|
|
113
|
+
|
|
114
|
+
def test_timestamps_fall_in_the_year_up_to_midnight_on_the_anchor(self):
|
|
115
|
+
stamps = [
|
|
116
|
+
datetime.fromisoformat(value) for value in _frames(as_of=ANCHOR)["events"]["created_at"]
|
|
117
|
+
]
|
|
118
|
+
midnight = datetime(2024, 3, 15)
|
|
119
|
+
|
|
120
|
+
assert min(stamps) >= midnight - timedelta(days=365)
|
|
121
|
+
assert max(stamps) <= midnight
|
|
122
|
+
# Midnight, not `now()`: no sub-second component to churn a CSV.
|
|
123
|
+
assert all(stamp.microsecond == 0 for stamp in stamps)
|
|
124
|
+
|
|
125
|
+
def test_a_datetime_anchor_is_read_as_its_day(self):
|
|
126
|
+
from_date = _frames(as_of=ANCHOR)["events"]
|
|
127
|
+
from_datetime = _frames(as_of=datetime(2024, 3, 15, 16, 30, 45))["events"]
|
|
128
|
+
|
|
129
|
+
assert from_date.equals(from_datetime)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class TestTheDefaultIsUnchanged:
|
|
133
|
+
def test_no_anchor_still_means_today(self, monkeypatch: pytest.MonkeyPatch):
|
|
134
|
+
_freeze_today(monkeypatch, date(2024, 3, 15))
|
|
135
|
+
|
|
136
|
+
implicit = _frames()["events"]
|
|
137
|
+
explicit = _frames(as_of=ANCHOR)["events"]
|
|
138
|
+
|
|
139
|
+
assert implicit.equals(explicit)
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Direct tests of CLI main function for proper coverage."""
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
|
+
import re
|
|
4
5
|
|
|
5
6
|
import pandas as pd
|
|
6
7
|
from typer.testing import CliRunner
|
|
@@ -9,6 +10,14 @@ from model2data.cli import app
|
|
|
9
10
|
|
|
10
11
|
runner = CliRunner()
|
|
11
12
|
|
|
13
|
+
_ANSI = re.compile(r"\x1b\[[0-9;]*m")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _plain(output: str) -> str:
|
|
17
|
+
"""Typer's error box is coloured on CI, which splits `--flag` into styled
|
|
18
|
+
pieces; strip the escapes before looking for an option name in it."""
|
|
19
|
+
return _ANSI.sub("", output)
|
|
20
|
+
|
|
12
21
|
|
|
13
22
|
def test_cli_basic_generation(tmp_path):
|
|
14
23
|
"""Test basic CLI generation with direct invocation."""
|
|
@@ -832,3 +841,89 @@ def test_cli_rows_for_rejects_malformed_values(tmp_path):
|
|
|
832
841
|
assert "Expected TABLE=N" in _run_rows_for(tmp_path, "--rows-for", "users").output
|
|
833
842
|
assert "whole number" in _run_rows_for(tmp_path, "--rows-for", "users=lots").output
|
|
834
843
|
assert "at least 1" in _run_rows_for(tmp_path, "--rows-for", "users=0").output
|
|
844
|
+
|
|
845
|
+
|
|
846
|
+
# Same two tables as ROWS_FOR_SCHEMA, plus the date and timestamp columns
|
|
847
|
+
# --as-of has anything to say about.
|
|
848
|
+
DATED_SCHEMA = """
|
|
849
|
+
Table users {
|
|
850
|
+
id int [pk]
|
|
851
|
+
email email [unique]
|
|
852
|
+
signed_up date [not null]
|
|
853
|
+
}
|
|
854
|
+
Table orders {
|
|
855
|
+
id int [pk]
|
|
856
|
+
user_id int [not null]
|
|
857
|
+
placed_at timestamp [not null]
|
|
858
|
+
}
|
|
859
|
+
Ref: orders.user_id > users.id
|
|
860
|
+
"""
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
def _run_shop(tmp_path, name, *extra_args):
|
|
864
|
+
"""Generate the dated users/orders schema under a chosen project name."""
|
|
865
|
+
dbml_file = tmp_path / "shop.dbml"
|
|
866
|
+
dbml_file.write_text(DATED_SCHEMA)
|
|
867
|
+
cwd = os.getcwd()
|
|
868
|
+
os.chdir(tmp_path)
|
|
869
|
+
try:
|
|
870
|
+
return runner.invoke(
|
|
871
|
+
app,
|
|
872
|
+
["--file", str(dbml_file), "--rows", "30", "--name", name, *extra_args],
|
|
873
|
+
)
|
|
874
|
+
finally:
|
|
875
|
+
os.chdir(cwd)
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
def test_cli_as_of_anchors_generated_dates(tmp_path):
|
|
879
|
+
result = _run_shop(tmp_path, "anchored", "--seed", "42", "--as-of", "2024-03-15")
|
|
880
|
+
assert result.exit_code == 0, result.output
|
|
881
|
+
assert "Anchoring generated dates on: 2024-03-15" in result.output
|
|
882
|
+
|
|
883
|
+
orders = pd.read_csv(tmp_path / "dbt_anchored" / "seeds" / "raw" / "orders.csv")
|
|
884
|
+
placed = pd.to_datetime(orders["placed_at"].dropna())
|
|
885
|
+
assert placed.max() <= pd.Timestamp("2024-03-15")
|
|
886
|
+
assert placed.min() >= pd.Timestamp("2024-03-15") - pd.Timedelta(days=365)
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
def test_cli_as_of_rejects_a_date_it_cannot_read(tmp_path):
|
|
890
|
+
result = _run_shop(tmp_path, "bad_date", "--as-of", "the 15th")
|
|
891
|
+
assert result.exit_code != 0
|
|
892
|
+
assert "--as-of" in _plain(result.output)
|
|
893
|
+
|
|
894
|
+
|
|
895
|
+
def test_cli_table_seed_re_rolls_only_that_table(tmp_path):
|
|
896
|
+
common = ("--seed", "42", "--as-of", "2024-03-15")
|
|
897
|
+
assert _run_shop(tmp_path, "before", *common).exit_code == 0
|
|
898
|
+
result = _run_shop(tmp_path, "after", *common, "--table-seed", "orders=7")
|
|
899
|
+
assert result.exit_code == 0, result.output
|
|
900
|
+
assert "Re-rolling with a table seed of its own: orders=7" in result.output
|
|
901
|
+
|
|
902
|
+
def seed_csv(project, table):
|
|
903
|
+
return (tmp_path / f"dbt_{project}" / "seeds" / "raw" / f"{table}.csv").read_text()
|
|
904
|
+
|
|
905
|
+
assert seed_csv("before", "users") == seed_csv("after", "users")
|
|
906
|
+
assert seed_csv("before", "orders") != seed_csv("after", "orders")
|
|
907
|
+
|
|
908
|
+
|
|
909
|
+
def test_cli_table_seed_needs_a_seed_to_re_roll_out_of(tmp_path):
|
|
910
|
+
result = _run_shop(tmp_path, "no_seed", "--table-seed", "orders=7")
|
|
911
|
+
assert result.exit_code != 0
|
|
912
|
+
assert "--seed" in _plain(result.output)
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def test_cli_table_seed_rejects_an_unknown_table(tmp_path):
|
|
916
|
+
result = _run_shop(tmp_path, "typo", "--seed", "1", "--table-seed", "ordres=7")
|
|
917
|
+
assert result.exit_code != 0
|
|
918
|
+
assert "No table named 'ordres'" in result.output
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
def test_cli_table_seed_rejects_malformed_values(tmp_path):
|
|
922
|
+
assert (
|
|
923
|
+
"Expected TABLE=N"
|
|
924
|
+
in _run_shop(tmp_path, "m1", "--seed", "1", "--table-seed", "orders").output
|
|
925
|
+
)
|
|
926
|
+
assert (
|
|
927
|
+
"whole number"
|
|
928
|
+
in _run_shop(tmp_path, "m2", "--seed", "1", "--table-seed", "orders=x").output
|
|
929
|
+
)
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""Re-rolling one table must leave the others exactly where they were.
|
|
2
|
+
|
|
3
|
+
One RNG stream for the whole run meant a table's values depended on every table
|
|
4
|
+
generated before it, so "I like these customers, give me different orders" was
|
|
5
|
+
not a thing that could be asked. Each table now draws from its own stream,
|
|
6
|
+
derived from the run seed and its own name. These tests hold the property that
|
|
7
|
+
buys: change one table's entry in `table_seeds` and only that table -- plus the
|
|
8
|
+
foreign keys that have to follow it -- moves.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
import pytest
|
|
13
|
+
from pandas.testing import assert_frame_equal
|
|
14
|
+
|
|
15
|
+
from model2data.generate.core import _table_stream_seed, generate_data_from_dbml
|
|
16
|
+
from model2data.parse.dbml import ColumnDef, TableDef
|
|
17
|
+
|
|
18
|
+
# customers <- orders <- order_items, plus a products table connected to
|
|
19
|
+
# nothing, so the suite covers a parent, a middle table, a grandchild and a
|
|
20
|
+
# bystander in one schema.
|
|
21
|
+
REFS = [
|
|
22
|
+
{
|
|
23
|
+
"source_table": "orders",
|
|
24
|
+
"source_column": "customer_id",
|
|
25
|
+
"target_table": "customers",
|
|
26
|
+
"target_column": "id",
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"source_table": "order_items",
|
|
30
|
+
"source_column": "order_id",
|
|
31
|
+
"target_table": "orders",
|
|
32
|
+
"target_column": "id",
|
|
33
|
+
},
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _schema() -> dict[str, TableDef]:
|
|
38
|
+
return {
|
|
39
|
+
"customers": TableDef(
|
|
40
|
+
name="customers",
|
|
41
|
+
columns=[
|
|
42
|
+
ColumnDef("id", "int", {"pk"}),
|
|
43
|
+
ColumnDef("email", "varchar", {"not null"}),
|
|
44
|
+
],
|
|
45
|
+
),
|
|
46
|
+
"orders": TableDef(
|
|
47
|
+
name="orders",
|
|
48
|
+
columns=[
|
|
49
|
+
ColumnDef("id", "int", {"pk"}),
|
|
50
|
+
ColumnDef("customer_id", "int", {"not null"}),
|
|
51
|
+
ColumnDef("total", "numeric", {"not null"}),
|
|
52
|
+
],
|
|
53
|
+
),
|
|
54
|
+
"order_items": TableDef(
|
|
55
|
+
name="order_items",
|
|
56
|
+
columns=[
|
|
57
|
+
ColumnDef("id", "int", {"pk"}),
|
|
58
|
+
ColumnDef("order_id", "int", {"not null"}),
|
|
59
|
+
ColumnDef("quantity", "int", {"not null"}),
|
|
60
|
+
],
|
|
61
|
+
),
|
|
62
|
+
"products": TableDef(
|
|
63
|
+
name="products",
|
|
64
|
+
columns=[
|
|
65
|
+
ColumnDef("id", "int", {"pk"}),
|
|
66
|
+
ColumnDef("title", "varchar", {"not null"}),
|
|
67
|
+
],
|
|
68
|
+
),
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _generate(**kwargs) -> dict[str, pd.DataFrame]:
|
|
73
|
+
return generate_data_from_dbml(_schema(), REFS, base_rows=40, seed=11, **kwargs)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class TestReRollingOneTable:
|
|
77
|
+
def test_only_the_named_table_and_its_descendants_move(self):
|
|
78
|
+
baseline = _generate()
|
|
79
|
+
rolled = _generate(table_seeds={"orders": 7})
|
|
80
|
+
|
|
81
|
+
assert not rolled["orders"].equals(baseline["orders"]), (
|
|
82
|
+
"the whole point of the override is that this table changes"
|
|
83
|
+
)
|
|
84
|
+
assert_frame_equal(rolled["customers"], baseline["customers"])
|
|
85
|
+
assert_frame_equal(rolled["products"], baseline["products"])
|
|
86
|
+
|
|
87
|
+
def test_a_child_of_an_untouched_table_is_byte_identical(self):
|
|
88
|
+
"""Re-rolling `products` reaches nothing: `orders` and its child stay put.
|
|
89
|
+
|
|
90
|
+
`order_items` is the interesting one -- it is two hops downstream of a
|
|
91
|
+
table nobody touched, and under a single shared stream it would have
|
|
92
|
+
moved anyway simply for being generated later in the run.
|
|
93
|
+
"""
|
|
94
|
+
baseline = _generate()
|
|
95
|
+
rolled = _generate(table_seeds={"products": 5})
|
|
96
|
+
|
|
97
|
+
assert not rolled["products"].equals(baseline["products"])
|
|
98
|
+
assert_frame_equal(rolled["customers"], baseline["customers"])
|
|
99
|
+
assert_frame_equal(rolled["orders"], baseline["orders"])
|
|
100
|
+
assert_frame_equal(rolled["order_items"], baseline["order_items"])
|
|
101
|
+
|
|
102
|
+
def test_a_childs_own_columns_survive_its_parent_being_re_rolled(self):
|
|
103
|
+
"""Only the FK column of a child may follow its parent."""
|
|
104
|
+
baseline = _generate()
|
|
105
|
+
rolled = _generate(table_seeds={"orders": 7})
|
|
106
|
+
|
|
107
|
+
assert_frame_equal(
|
|
108
|
+
rolled["order_items"].drop(columns=["order_id"]),
|
|
109
|
+
baseline["order_items"].drop(columns=["order_id"]),
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
def test_referential_integrity_holds_against_the_new_parent(self):
|
|
113
|
+
rolled = _generate(table_seeds={"orders": 7})
|
|
114
|
+
|
|
115
|
+
assert set(rolled["orders"]["customer_id"]) <= set(rolled["customers"]["id"])
|
|
116
|
+
assert set(rolled["order_items"]["order_id"]) <= set(rolled["orders"]["id"])
|
|
117
|
+
|
|
118
|
+
def test_the_old_parent_rows_are_genuinely_gone(self):
|
|
119
|
+
"""A child that still pointed at the pre-roll ids would be dangling."""
|
|
120
|
+
baseline = _generate()
|
|
121
|
+
rolled = _generate(table_seeds={"orders": 7})
|
|
122
|
+
|
|
123
|
+
stale = set(baseline["orders"]["id"]) - set(rolled["orders"]["id"])
|
|
124
|
+
assert stale, "the re-rolled parent has to hand out at least some new ids"
|
|
125
|
+
assert not set(rolled["order_items"]["order_id"]) & stale
|
|
126
|
+
|
|
127
|
+
def test_two_tables_can_be_re_rolled_at_once(self):
|
|
128
|
+
baseline = _generate()
|
|
129
|
+
rolled = _generate(table_seeds={"customers": 2, "products": 5})
|
|
130
|
+
|
|
131
|
+
assert not rolled["customers"].equals(baseline["customers"])
|
|
132
|
+
assert not rolled["products"].equals(baseline["products"])
|
|
133
|
+
|
|
134
|
+
def test_re_rolling_is_itself_reproducible(self):
|
|
135
|
+
first = _generate(table_seeds={"orders": 7})
|
|
136
|
+
second = _generate(table_seeds={"orders": 7})
|
|
137
|
+
|
|
138
|
+
for name in first:
|
|
139
|
+
assert_frame_equal(first[name], second[name])
|
|
140
|
+
|
|
141
|
+
def test_a_different_override_value_gives_a_different_table(self):
|
|
142
|
+
seven = _generate(table_seeds={"orders": 7})
|
|
143
|
+
eight = _generate(table_seeds={"orders": 8})
|
|
144
|
+
|
|
145
|
+
assert not seven["orders"].equals(eight["orders"])
|
|
146
|
+
assert_frame_equal(seven["customers"], eight["customers"])
|
|
147
|
+
|
|
148
|
+
def test_an_empty_mapping_changes_nothing(self):
|
|
149
|
+
assert_frame_equal(_generate(table_seeds={})["orders"], _generate()["orders"])
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class TestValidation:
|
|
153
|
+
def test_an_unknown_table_name_is_an_error(self):
|
|
154
|
+
with pytest.raises(ValueError, match="No table named 'ordres'"):
|
|
155
|
+
_generate(table_seeds={"ordres": 7})
|
|
156
|
+
|
|
157
|
+
def test_the_message_lists_the_tables_that_do_exist(self):
|
|
158
|
+
with pytest.raises(ValueError, match="customers, order_items, orders, products"):
|
|
159
|
+
_generate(table_seeds={"nope": 1})
|
|
160
|
+
|
|
161
|
+
def test_several_unknown_names_are_reported_together(self):
|
|
162
|
+
with pytest.raises(ValueError, match="No tables named 'a', 'b'"):
|
|
163
|
+
_generate(table_seeds={"b": 1, "a": 2})
|
|
164
|
+
|
|
165
|
+
def test_it_needs_a_seed_to_re_roll_out_of(self):
|
|
166
|
+
with pytest.raises(ValueError, match="table_seeds needs a seed"):
|
|
167
|
+
generate_data_from_dbml(
|
|
168
|
+
_schema(), REFS, base_rows=10, seed=None, table_seeds={"orders": 7}
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
def test_no_seed_and_no_overrides_is_still_fine(self):
|
|
172
|
+
frames = generate_data_from_dbml(_schema(), REFS, base_rows=10, seed=None)
|
|
173
|
+
|
|
174
|
+
assert len(frames["orders"]) == 10
|
|
175
|
+
|
|
176
|
+
def test_validation_runs_before_anything_is_generated(self):
|
|
177
|
+
"""A typo should not cost a full generation pass first."""
|
|
178
|
+
with pytest.raises(ValueError):
|
|
179
|
+
generate_data_from_dbml(
|
|
180
|
+
_schema(), REFS, base_rows=2_000_000, seed=1, table_seeds={"typo": 1}
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
class TestTheDerivedSeed:
|
|
185
|
+
def test_it_is_stable_across_processes(self):
|
|
186
|
+
"""Pinned literally: `hash()` is salted per process and would not be."""
|
|
187
|
+
assert _table_stream_seed(11, "orders", None) == 11237587294246756286
|
|
188
|
+
assert _table_stream_seed(11, "orders", 7) == 9635424189123890928
|
|
189
|
+
|
|
190
|
+
def test_each_table_gets_a_different_stream(self):
|
|
191
|
+
assert _table_stream_seed(11, "orders", None) != _table_stream_seed(11, "customers", None)
|
|
192
|
+
|
|
193
|
+
def test_each_run_seed_gets_a_different_stream(self):
|
|
194
|
+
assert _table_stream_seed(11, "orders", None) != _table_stream_seed(12, "orders", None)
|
|
195
|
+
|
|
196
|
+
def test_no_override_is_not_the_same_as_an_override_of_zero(self):
|
|
197
|
+
assert _table_stream_seed(11, "orders", None) != _table_stream_seed(11, "orders", 0)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
class TestOtherOptionsStillCompose:
|
|
201
|
+
def test_row_counts_are_untouched_by_a_re_roll(self):
|
|
202
|
+
rolled = _generate(row_overrides={"orders": 120}, table_seeds={"orders": 7})
|
|
203
|
+
|
|
204
|
+
assert len(rolled["orders"]) == 120
|
|
205
|
+
assert len(rolled["customers"]) == 40
|
|
206
|
+
|
|
207
|
+
def test_a_pinned_anchor_and_a_re_roll_work_together(self):
|
|
208
|
+
from datetime import date
|
|
209
|
+
|
|
210
|
+
tables = {
|
|
211
|
+
"events": TableDef(
|
|
212
|
+
name="events",
|
|
213
|
+
columns=[
|
|
214
|
+
ColumnDef("id", "int", {"pk"}),
|
|
215
|
+
ColumnDef("happened_on", "date", {"not null"}),
|
|
216
|
+
],
|
|
217
|
+
)
|
|
218
|
+
}
|
|
219
|
+
kwargs = {"base_rows": 20, "seed": 4, "as_of": date(2024, 3, 15)}
|
|
220
|
+
|
|
221
|
+
first = generate_data_from_dbml(tables, [], table_seeds={"events": 1}, **kwargs)["events"]
|
|
222
|
+
second = generate_data_from_dbml(tables, [], table_seeds={"events": 1}, **kwargs)["events"]
|
|
223
|
+
|
|
224
|
+
assert_frame_equal(first, second)
|
|
225
|
+
assert first["happened_on"].max() <= date(2024, 3, 15)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/macros/generate_schema_name.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|