model2data 1.3.1__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {model2data-1.3.1/model2data.egg-info → model2data-1.4.0}/PKG-INFO +17 -3
  2. {model2data-1.3.1 → model2data-1.4.0}/README.md +28 -2
  3. {model2data-1.3.1 → model2data-1.4.0}/README_PYPI.md +16 -2
  4. {model2data-1.3.1 → model2data-1.4.0}/model2data/cli.py +82 -2
  5. {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/core.py +102 -5
  6. {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/faker.py +67 -12
  7. {model2data-1.3.1 → model2data-1.4.0/model2data.egg-info}/PKG-INFO +17 -3
  8. {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/SOURCES.txt +3 -1
  9. {model2data-1.3.1 → model2data-1.4.0}/pyproject.toml +1 -1
  10. model2data-1.4.0/tests/test_as_of_anchor.py +139 -0
  11. {model2data-1.3.1 → model2data-1.4.0}/tests/test_cli.py +95 -0
  12. model2data-1.4.0/tests/test_table_seeds.py +225 -0
  13. {model2data-1.3.1 → model2data-1.4.0}/LICENSE +0 -0
  14. {model2data-1.3.1 → model2data-1.4.0}/model2data/__init__.py +0 -0
  15. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/__init__.py +0 -0
  16. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/project.py +0 -0
  17. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  18. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  19. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  20. {model2data-1.3.1 → model2data-1.4.0}/model2data/dbt/tests.py +0 -0
  21. {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/__init__.py +0 -0
  22. {model2data-1.3.1 → model2data-1.4.0}/model2data/generate/relationships.py +0 -0
  23. {model2data-1.3.1 → model2data-1.4.0}/model2data/parse/__init__.py +0 -0
  24. {model2data-1.3.1 → model2data-1.4.0}/model2data/parse/dbml.py +0 -0
  25. {model2data-1.3.1 → model2data-1.4.0}/model2data/utils.py +0 -0
  26. {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/dependency_links.txt +0 -0
  27. {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/entry_points.txt +0 -0
  28. {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/requires.txt +0 -0
  29. {model2data-1.3.1 → model2data-1.4.0}/model2data.egg-info/top_level.txt +0 -0
  30. {model2data-1.3.1 → model2data-1.4.0}/setup.cfg +0 -0
  31. {model2data-1.3.1 → model2data-1.4.0}/tests/test_coverage_gaps.py +0 -0
  32. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbml_parser.py +0 -0
  33. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbml_parser_fuzz.py +0 -0
  34. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_integration.py +0 -0
  35. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_naming.py +0 -0
  36. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_project.py +0 -0
  37. {model2data-1.3.1 → model2data-1.4.0}/tests/test_dbt_tests.py +0 -0
  38. {model2data-1.3.1 → model2data-1.4.0}/tests/test_faker_name_inference.py +0 -0
  39. {model2data-1.3.1 → model2data-1.4.0}/tests/test_generation.py +0 -0
  40. {model2data-1.3.1 → model2data-1.4.0}/tests/test_release_stress.py +0 -0
  41. {model2data-1.3.1 → model2data-1.4.0}/tests/test_row_identity.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.3.1
3
+ Version: 1.4.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -85,7 +85,10 @@ access required.
85
85
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
86
86
  dependency order.
87
87
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
88
- commit fixtures, safe to diff across CI runs.
88
+ commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
89
+ on, and the run reproduces on any later day rather than only on the day it first ran.
90
+ - **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
91
+ leaves every other table byte-identical, so you can keep the four tables that look right.
89
92
  - **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
90
93
  and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
91
94
  A single `dbt build` loads, transforms, and tests the whole thing.
@@ -136,6 +139,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
136
139
 
137
140
  This creates a `dbt_ecommerce/` folder with your data and dbt setup.
138
141
 
142
+ `--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
143
+ current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
144
+ anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
145
+ fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
146
+ re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
147
+ the country every generated person and address comes from:
148
+
149
+ ```bash
150
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
151
+ --as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
152
+ ```
153
+
139
154
  Run dbt to load, transform, and test the data:
140
155
 
141
156
  ```bash
@@ -274,7 +289,6 @@ wants to pick them up as a contribution:
274
289
  parsed schema shape.
275
290
  - Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
276
291
  ready-to-uncomment `marts` schema/materialization config for this).
277
- - Locale-aware generation (`--locale`) for non-English/US synthetic data.
278
292
 
279
293
  See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
280
294
 
@@ -44,7 +44,10 @@ access required.
44
44
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
45
45
  dependency order.
46
46
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
47
- commit fixtures, safe to diff across CI runs.
47
+ commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
48
+ on, and the run reproduces on any later day rather than only on the day it first ran.
49
+ - **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
50
+ leaves every other table byte-identical, so you can keep the four tables that look right.
48
51
  - **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
49
52
  and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
50
53
  A single `dbt build` loads, transforms, and tests the whole thing.
@@ -138,6 +141,30 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
138
141
  --rows-for customers=50 --rows-for order_items=5000
139
142
  ```
140
143
 
144
+ `--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
145
+ current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
146
+ anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
147
+ fixture safe to commit:
148
+
149
+ ```bash
150
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --as-of 2026-01-31
151
+ ```
152
+
153
+ If one table comes out wrong and the rest looks right, `--table-seed` re-rolls just that table.
154
+ Every other table's seed CSV stays byte-identical, and children of the re-rolled table still
155
+ reference rows that exist, so there's nothing to re-check but the table you asked to change:
156
+
157
+ ```bash
158
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --table-seed orders=7
159
+ ```
160
+
161
+ `--locale` picks the country every generated person and address comes from (`en_US` by default);
162
+ it's a per-run setting, so a table can't end up holding one Belgian and one American address:
163
+
164
+ ```bash
165
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
166
+ ```
167
+
141
168
  Run dbt to load, transform, and test the data:
142
169
 
143
170
  ```bash
@@ -277,7 +304,6 @@ wants to pick them up as a contribution:
277
304
  parsed schema shape.
278
305
  - Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
279
306
  ready-to-uncomment `marts` schema/materialization config for this).
280
- - Locale-aware generation (`--locale`) for non-English/US synthetic data.
281
307
 
282
308
  See [CONTRIBUTING.md](CONTRIBUTING.md) if you'd like to work on any of these.
283
309
 
@@ -41,7 +41,10 @@ access required.
41
41
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
42
42
  dependency order.
43
43
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
44
- commit fixtures, safe to diff across CI runs.
44
+ commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
45
+ on, and the run reproduces on any later day rather than only on the day it first ran.
46
+ - **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
47
+ leaves every other table byte-identical, so you can keep the four tables that look right.
45
48
  - **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
46
49
  and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
47
50
  A single `dbt build` loads, transforms, and tests the whole thing.
@@ -92,6 +95,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
92
95
 
93
96
  This creates a `dbt_ecommerce/` folder with your data and dbt setup.
94
97
 
98
+ `--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
99
+ current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
100
+ anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
101
+ fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
102
+ re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
103
+ the country every generated person and address comes from:
104
+
105
+ ```bash
106
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
107
+ --as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
108
+ ```
109
+
95
110
  Run dbt to load, transform, and test the data:
96
111
 
97
112
  ```bash
@@ -230,7 +245,6 @@ wants to pick them up as a contribution:
230
245
  parsed schema shape.
231
246
  - Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
232
247
  ready-to-uncomment `marts` schema/materialization config for this).
233
- - Locale-aware generation (`--locale`) for non-English/US synthetic data.
234
248
 
235
249
  See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
236
250
 
@@ -1,5 +1,6 @@
1
1
  import random
2
2
  import shutil
3
+ from datetime import datetime
3
4
  from pathlib import Path
4
5
  from typing import Optional
5
6
 
@@ -75,6 +76,43 @@ def _parse_row_overrides(
75
76
  return overrides
76
77
 
77
78
 
79
+ def _parse_table_seeds(
80
+ raw: Optional[list[str]],
81
+ tables: dict,
82
+ ) -> dict[str, int]:
83
+ """Turn repeated `--table-seed TABLE=N` values into a {table: seed} mapping.
84
+
85
+ Same shape and the same loud failure on an unknown name as `--rows-for`:
86
+ the whole point of naming a table here is to change that table, so a typo
87
+ would otherwise produce a run in which nothing moved and nothing was said.
88
+ """
89
+ if not isinstance(raw, (list, tuple)):
90
+ return {}
91
+
92
+ table_seeds: dict[str, int] = {}
93
+ for item in raw:
94
+ table_name, separator, value = item.partition("=")
95
+ table_name = table_name.strip()
96
+ if not separator or not table_name:
97
+ raise typer.BadParameter(f"Expected TABLE=N, got {item!r}.", param_hint="--table-seed")
98
+
99
+ try:
100
+ table_seed = int(value)
101
+ except ValueError:
102
+ raise typer.BadParameter(
103
+ f"Seed for {table_name!r} must be a whole number, got {value!r}.",
104
+ param_hint="--table-seed",
105
+ ) from None
106
+ if table_name not in tables:
107
+ known = ", ".join(sorted(tables)) or "none"
108
+ raise typer.BadParameter(
109
+ f"No table named {table_name!r} in this schema. Tables: {known}.",
110
+ param_hint="--table-seed",
111
+ )
112
+ table_seeds[table_name] = table_seed
113
+ return table_seeds
114
+
115
+
78
116
  app = typer.Typer(
79
117
  help=(
80
118
  "model2data: Generate analytics-ready datasets from DBML models.\n\n"
@@ -107,8 +145,11 @@ def main(
107
145
  min=10,
108
146
  help="Number of rows to generate per table.",
109
147
  ),
110
- # noqa: B008 is only needed here (not on the other options) because a
111
- # repeatable option must be annotated with a mutable `list` type.
148
+ # noqa: B008 is needed on some options and not others because ruff waves a
149
+ # call through in a default only when the annotation is one of the types it
150
+ # knows to be immutable. `str`, `int`, `bool` and `Path` are on that list;
151
+ # the `list` a repeatable option must be annotated with, and `datetime`,
152
+ # are not -- neither is actually mutated here.
112
153
  rows_for: Optional[list[str]] = typer.Option( # noqa: B008
113
154
  None,
114
155
  "--rows-for",
@@ -126,6 +167,25 @@ def main(
126
167
  "Using the same seed will always produce identical datasets."
127
168
  ),
128
169
  ),
170
+ table_seed: Optional[list[str]] = typer.Option( # noqa: B008
171
+ None,
172
+ "--table-seed",
173
+ metavar="TABLE=N",
174
+ help=(
175
+ "Re-roll one table without disturbing the others, keeping --seed for the rest.\n"
176
+ "Repeatable, e.g. --table-seed orders=7. Requires --seed."
177
+ ),
178
+ ),
179
+ as_of: Optional[datetime] = typer.Option( # noqa: B008
180
+ None,
181
+ "--as-of",
182
+ formats=["%Y-%m-%d"],
183
+ metavar="YYYY-MM-DD",
184
+ help=(
185
+ "Date to anchor generated dates and timestamps on (default: today).\n"
186
+ "Pin it and a --seed run reproduces on any later day, not just the day it first ran."
187
+ ),
188
+ ),
129
189
  locale: Optional[str] = typer.Option(
130
190
  None,
131
191
  "--locale",
@@ -183,6 +243,14 @@ def main(
183
243
  Faker.seed(seed)
184
244
  typer.echo(f"🔁 Using deterministic seed: {seed}")
185
245
 
246
+ # Same reason as `_parse_row_overrides`'s isinstance guard: `main` is also
247
+ # called directly as a plain function, which leaves this holding its
248
+ # `OptionInfo` default rather than None. Anything that isn't a real
249
+ # datetime means "not supplied", i.e. anchor on today.
250
+ as_of = as_of if isinstance(as_of, datetime) else None
251
+ if as_of is not None:
252
+ typer.echo(f"📅 Anchoring generated dates on: {as_of.date()}")
253
+
186
254
  # -------------------------
187
255
  # Parse DBML (names untouched)
188
256
  # -------------------------
@@ -196,6 +264,16 @@ def main(
196
264
  # should not leave a half-scaffolded project behind for the next run to trip
197
265
  # over with a confusing "destination already exists".
198
266
  row_overrides = _parse_row_overrides(rows_for, tables)
267
+ table_seeds = _parse_table_seeds(table_seed, tables)
268
+ if table_seeds and seed is None:
269
+ raise typer.BadParameter(
270
+ "--table-seed re-rolls one table out of the run's seed, so there has to "
271
+ "be one. Add --seed.",
272
+ param_hint="--table-seed",
273
+ )
274
+ if table_seeds:
275
+ rolled = ", ".join(f"{name}={value}" for name, value in sorted(table_seeds.items()))
276
+ typer.echo(f"🎲 Re-rolling with a table seed of its own: {rolled}")
199
277
 
200
278
  project_name = normalize_identifier(name or file.stem)
201
279
  dest = Path.cwd() / f"dbt_{project_name}"
@@ -225,6 +303,8 @@ def main(
225
303
  seed=seed,
226
304
  row_overrides=row_overrides,
227
305
  locale=locale,
306
+ as_of=as_of,
307
+ table_seeds=table_seeds,
228
308
  )
229
309
 
230
310
  # -------------------------
@@ -1,5 +1,6 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import hashlib
3
4
  import random
4
5
  from collections import defaultdict, deque
5
6
  from collections.abc import Mapping
@@ -9,6 +10,7 @@ import pandas as pd
9
10
  from faker import Faker
10
11
 
11
12
  from model2data.generate.faker import (
13
+ AsOf,
12
14
  generate_column_values,
13
15
  release_row_pools,
14
16
  reset_duplicate_unique_columns,
@@ -68,6 +70,8 @@ def generate_data_from_dbml(
68
70
  seed: Optional[int] = None,
69
71
  row_overrides: Optional[Mapping[str, int]] = None,
70
72
  locale: Optional[str] = None,
73
+ as_of: AsOf = None,
74
+ table_seeds: Optional[Mapping[str, int]] = None,
71
75
  ) -> dict[str, pd.DataFrame]:
72
76
  """
73
77
  Generate synthetic datasets from parsed DBML definitions.
@@ -86,9 +90,28 @@ def generate_data_from_dbml(
86
90
  holding one Belgian and one American address is the incoherence the row
87
91
  pools exist to remove.
88
92
 
89
- This function is deterministic if a seed is provided.
90
- It performs no filesystem I/O and returns pandas DataFrames.
93
+ `as_of` is the date every generated date and timestamp is placed relative
94
+ to -- dates land in the two years up to it, timestamps in the year up to
95
+ midnight on it -- and defaults to today. Without it a seed reproduces only
96
+ for as long as the day lasts: the numbers come back identical and the dates
97
+ move, so a committed fixture churns and a saved project renders different
98
+ rows next month. Pass the day the run should look like it happened on and
99
+ the whole frame reproduces, on any later day.
100
+
101
+ `table_seeds` re-rolls individual tables without disturbing the rest. Each
102
+ table draws from its own RNG stream, derived from `(seed, table_name,
103
+ table_seeds[table_name])`, so bumping one table's entry changes that
104
+ table's rows and leaves every other table byte-identical. Children of a
105
+ re-rolled table keep pointing at rows that exist, because their foreign
106
+ keys are drawn from whatever their parent ended up holding. It needs a
107
+ `seed` to work off -- with none, every table is already different on every
108
+ run -- and unknown table names are an error rather than a silent no-op.
109
+
110
+ This function is deterministic if a seed is provided (and, with `as_of`,
111
+ on any day). It performs no filesystem I/O and returns pandas DataFrames.
91
112
  """
113
+ _validate_table_seeds(tables, table_seeds, seed)
114
+
92
115
  # Locale first, then the seed: switching locale builds a new Faker, and the
93
116
  # seed has to be the last word on the generator that actually runs.
94
117
  set_locale(locale)
@@ -118,6 +141,26 @@ def generate_data_from_dbml(
118
141
  table_def = tables[table_name]
119
142
  row_count = _determine_row_count(table_def.name, base_rows, row_overrides)
120
143
 
144
+ # Give this table its own RNG stream before a single value of it is
145
+ # drawn. One stream for the whole run meant re-rolling a table
146
+ # re-rolled everything generated after it too -- fine for a one-shot
147
+ # CLI run, useless for a "regenerate just this table" button, which is
148
+ # exactly the thing users ask for once they like four tables out of
149
+ # five. Same locale-then-seed order as the run-level seeding above.
150
+ if seed is not None:
151
+ stream_seed = _table_stream_seed(
152
+ seed, table_name, table_seeds.get(table_name) if table_seeds else None
153
+ )
154
+ random.seed(stream_seed)
155
+ Faker.seed(stream_seed)
156
+
157
+ # A table's identities are its own. `release_row_pools` below already
158
+ # drops the pool keyed by this table's name; this also drops the
159
+ # un-named bucket the composite-key repair shares, so nothing a
160
+ # previous table left behind can reach this one's rows and make its
161
+ # stream depend on what came before it.
162
+ reset_row_pools()
163
+
121
164
  data: dict[str, list] = {}
122
165
 
123
166
  # A composite *primary* key's member columns are frequently declared
@@ -166,11 +209,16 @@ def generate_data_from_dbml(
166
209
  ensure_unique=ensure_unique,
167
210
  force_not_null=column.name in composite_pk_columns,
168
211
  table_name=table_name,
212
+ as_of=as_of,
169
213
  )
170
214
 
171
215
  df = pd.DataFrame(data)
172
- df = _resolve_self_referencing_fks(df, table_def, table_name, fk_lookup, row_count)
173
- df = _deduplicate_composite_keys(df, table_def, table_name, fk_lookup, generated)
216
+ df = _resolve_self_referencing_fks(
217
+ df, table_def, table_name, fk_lookup, row_count, as_of=as_of
218
+ )
219
+ df = _deduplicate_composite_keys(
220
+ df, table_def, table_name, fk_lookup, generated, as_of=as_of
221
+ )
174
222
 
175
223
  # -----------------------------------------------------
176
224
  # Second pass: attribute mirroring (non-FK refs)
@@ -250,6 +298,7 @@ def _deduplicate_composite_keys(
250
298
  fk_lookup: dict[tuple[str, str], tuple[str, str]],
251
299
  generated: dict[str, pd.DataFrame],
252
300
  max_attempts: int = 20,
301
+ as_of: AsOf = None,
253
302
  ) -> pd.DataFrame:
254
303
  """
255
304
  Regenerate colliding rows for any pk/unique composite key declared via an
@@ -300,7 +349,9 @@ def _deduplicate_composite_keys(
300
349
  if col_name in fk_pools:
301
350
  df.at[idx, col_name] = random.choice(fk_pools[col_name])
302
351
  else:
303
- df.at[idx, col_name] = generate_column_values(col_def, row_count=1)[0]
352
+ df.at[idx, col_name] = generate_column_values(
353
+ col_def, row_count=1, as_of=as_of
354
+ )[0]
304
355
  combo = tuple(df.at[idx, c] for c in key_columns)
305
356
  attempts += 1
306
357
  # Retry budget spent and still colliding: this row keeps a
@@ -325,6 +376,7 @@ def _resolve_self_referencing_fks(
325
376
  table_name: str,
326
377
  fk_lookup: dict[tuple[str, str], tuple[str, str]],
327
378
  row_count: int,
379
+ as_of: AsOf = None,
328
380
  ) -> pd.DataFrame:
329
381
  """
330
382
  Re-generate any FK column that references its own table (e.g. a
@@ -362,11 +414,56 @@ def _resolve_self_referencing_fks(
362
414
  ensure_unique=ensure_unique,
363
415
  force_not_null=column.name in composite_pk_columns,
364
416
  table_name=table_name,
417
+ as_of=as_of,
365
418
  )
366
419
 
367
420
  return df
368
421
 
369
422
 
423
+ def _validate_table_seeds(
424
+ tables: dict[str, TableDef],
425
+ table_seeds: Optional[Mapping[str, int]],
426
+ seed: Optional[int],
427
+ ) -> None:
428
+ """Reject a `table_seeds` mapping that cannot do what it was asked to do.
429
+
430
+ Unlike `row_overrides`, which quietly ignores a name it does not know, an
431
+ unknown name here is always a mistake worth stopping for: the caller asked
432
+ for one table to be re-rolled and would otherwise get a run in which
433
+ nothing changed, with nothing said about why. The same goes for passing
434
+ `table_seeds` with no `seed` -- there is no stream to re-roll out of, and
435
+ every table is already different on every run.
436
+ """
437
+ if not table_seeds:
438
+ return
439
+
440
+ unknown = sorted(name for name in table_seeds if name not in tables)
441
+ if unknown:
442
+ known = ", ".join(sorted(tables)) or "none"
443
+ label = "No table named" if len(unknown) == 1 else "No tables named"
444
+ named = ", ".join(repr(name) for name in unknown)
445
+ raise ValueError(f"{label} {named} in this schema. Tables: {known}.")
446
+
447
+ if seed is None:
448
+ raise ValueError(
449
+ "table_seeds needs a seed to re-roll a table out of: without one every "
450
+ "table is already generated afresh on every run."
451
+ )
452
+
453
+
454
+ def _table_stream_seed(seed: int, table_name: str, table_seed: Optional[int]) -> int:
455
+ """The RNG seed one table draws from, derived from the run seed and its name.
456
+
457
+ Hashed with blake2b rather than Python's built-in `hash()`, which is salted
458
+ per interpreter process for strings: a "deterministic" seed built on it
459
+ would reproduce only within a single run of the program, which is the one
460
+ place determinism was never in doubt.
461
+ """
462
+ payload = f"{seed}|{table_name}|{'' if table_seed is None else table_seed}"
463
+ digest = hashlib.blake2b(payload.encode("utf-8"), digest_size=8).digest()
464
+ return int.from_bytes(digest, "big")
465
+
466
+
370
467
  def _determine_row_count(
371
468
  table_name: str,
372
469
  base_rows: int,
@@ -5,7 +5,7 @@ import re
5
5
  import unicodedata
6
6
  import uuid
7
7
  from dataclasses import dataclass
8
- from datetime import datetime, timedelta
8
+ from datetime import date, datetime, timedelta
9
9
  from typing import Callable, Optional, Union
10
10
 
11
11
  import pandas as pd
@@ -59,6 +59,47 @@ def set_locale(locale: Optional[str]) -> None:
59
59
  _address_state.clear()
60
60
 
61
61
 
62
+ # ---------------------------------------------------------
63
+ # Date anchor
64
+ # ---------------------------------------------------------
65
+ # Every generated date and timestamp is placed relative to a single anchor
66
+ # date, which defaults to today. That default is what made a seeded run
67
+ # reproduce only for as long as the day lasted: re-run tomorrow, the same seed
68
+ # gave the same numbers and different dates, so a committed fixture churned and
69
+ # a shared demo drifted. Callers that need a run to reproduce across days pass
70
+ # `as_of` and pin the window instead.
71
+ AsOf = Union[date, datetime, None]
72
+
73
+ # How far back a `date` column's window reaches from the anchor.
74
+ _DATE_WINDOW_YEARS = 2
75
+
76
+
77
+ def _anchor_date(as_of: AsOf) -> date:
78
+ """The date a run generates relative to: `as_of`, or today when it is None.
79
+
80
+ A `datetime` is narrowed to its day, so a caller who has a timestamp to
81
+ hand does not have to remember that only the date part is used.
82
+ """
83
+ if as_of is None:
84
+ return date.today()
85
+ if isinstance(as_of, datetime):
86
+ return as_of.date()
87
+ return as_of
88
+
89
+
90
+ def _years_before(anchor: date, years: int) -> date:
91
+ """`years` calendar years before `anchor`, moving 29 Feb back to 28 Feb.
92
+
93
+ Only 29 February has no counterpart in a non-leap year, and a whole run
94
+ failing on one day in four years is not a tradeoff worth taking for the
95
+ sake of an exact anniversary.
96
+ """
97
+ try:
98
+ return anchor.replace(year=anchor.year - years)
99
+ except ValueError:
100
+ return anchor.replace(year=anchor.year - years, month=2, day=28)
101
+
102
+
62
103
  # ---------------------------------------------------------
63
104
  # Per-row identities
64
105
  # ---------------------------------------------------------
@@ -471,11 +512,16 @@ def generate_column_values(
471
512
  ensure_unique: bool = False,
472
513
  force_not_null: bool = False,
473
514
  table_name: Optional[str] = None,
515
+ as_of: AsOf = None,
474
516
  ) -> list:
475
517
  """
476
518
  Generate synthetic values for a single column.
477
519
  Respects FKs, uniqueness, and optional min/max hints in column notes.
478
520
 
521
+ `as_of` is the date every generated date and timestamp is placed relative
522
+ to, defaulting to today. Pass it to make a seeded run reproduce on any
523
+ later day rather than only on the day it first ran.
524
+
479
525
  `force_not_null` lets a caller override the nullability pass below for a
480
526
  column whose *individual* settings don't carry `not null`/`pk` but is
481
527
  still never allowed to be null -- namely a composite primary key member
@@ -580,13 +626,20 @@ def generate_column_values(
580
626
  # Dates
581
627
  # -----------------------------------------------------
582
628
  elif "date" in base_type and "time" not in base_type:
583
- values = [fake.date_between(start_date="-2y", end_date="today") for _ in range(row_count)]
629
+ # Explicit endpoints rather than Faker's "-2y"/"today" shorthand: those
630
+ # strings are resolved against `date.today()` inside Faker, which is
631
+ # precisely the hidden dependency on the wall clock `as_of` removes.
632
+ anchor = _anchor_date(as_of)
633
+ values = [
634
+ fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
635
+ for _ in range(row_count)
636
+ ]
584
637
 
585
638
  elif "time" in base_type and "stamp" not in base_type:
586
639
  values = [fake.time() for _ in range(row_count)]
587
640
 
588
641
  elif any(key in base_type for key in ["timestamp", "datetime"]):
589
- values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
642
+ values = [_random_datetime(as_of=as_of).isoformat(sep=" ") for _ in range(row_count)]
590
643
 
591
644
  # -----------------------------------------------------
592
645
  # Untyped / generic string columns: honour a type that names
@@ -693,17 +746,19 @@ def _suffixed(value: str, counter: int) -> str:
693
746
  return f"{local}{counter}{at}{domain}"
694
747
 
695
748
 
696
- def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
697
- """Pick a random timestamp in a window around today, to whole seconds.
749
+ def _random_datetime(start_days: int = -365, end_days: int = 0, as_of: AsOf = None) -> datetime:
750
+ """Pick a random timestamp in a window around the anchor, to whole seconds.
698
751
 
699
- The window is anchored to midnight rather than `datetime.now()`. The
700
- random offset is a whole number of seconds, so anchoring on `now()` let
701
- its sub-second component leak straight through into every generated
702
- timestamp -- two runs with the same `--seed` produced values differing
703
- only in their microseconds, which quietly broke the reproducibility
704
- `--seed` exists to provide.
752
+ The window is anchored to midnight of `as_of` (today when it is None)
753
+ rather than to a `datetime.now()`. The random offset is a whole number of
754
+ seconds, so anchoring on `now()` let its sub-second component leak straight
755
+ through into every generated timestamp -- two runs with the same `--seed`
756
+ produced values differing only in their microseconds, which quietly broke
757
+ the reproducibility `--seed` exists to provide. Midnight of an explicit
758
+ `as_of` extends that reproducibility past the end of the day.
705
759
  """
706
- midnight = datetime.now().replace(hour=0, minute=0, second=0, microsecond=0)
760
+ anchor = _anchor_date(as_of)
761
+ midnight = datetime(anchor.year, anchor.month, anchor.day)
707
762
  start = midnight + timedelta(days=start_days)
708
763
  end = midnight + timedelta(days=end_days)
709
764
  random_second = random.randint(0, int((end - start).total_seconds()))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.3.1
3
+ Version: 1.4.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -85,7 +85,10 @@ access required.
85
85
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
86
86
  dependency order.
87
87
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
88
- commit fixtures, safe to diff across CI runs.
88
+ commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
89
+ on, and the run reproduces on any later day rather than only on the day it first ran.
90
+ - **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
91
+ leaves every other table byte-identical, so you can keep the four tables that look right.
89
92
  - **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
90
93
  and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
91
94
  A single `dbt build` loads, transforms, and tests the whole thing.
@@ -136,6 +139,18 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
136
139
 
137
140
  This creates a `dbt_ecommerce/` folder with your data and dbt setup.
138
141
 
142
+ `--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
143
+ current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
144
+ anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
145
+ fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
146
+ re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
147
+ the country every generated person and address comes from:
148
+
149
+ ```bash
150
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
151
+ --as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
152
+ ```
153
+
139
154
  Run dbt to load, transform, and test the data:
140
155
 
141
156
  ```bash
@@ -274,7 +289,6 @@ wants to pick them up as a contribution:
274
289
  parsed schema shape.
275
290
  - Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
276
291
  ready-to-uncomment `marts` schema/materialization config for this).
277
- - Locale-aware generation (`--locale`) for non-English/US synthetic data.
278
292
 
279
293
  See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
280
294
 
@@ -23,6 +23,7 @@ model2data/generate/faker.py
23
23
  model2data/generate/relationships.py
24
24
  model2data/parse/__init__.py
25
25
  model2data/parse/dbml.py
26
+ tests/test_as_of_anchor.py
26
27
  tests/test_cli.py
27
28
  tests/test_coverage_gaps.py
28
29
  tests/test_dbml_parser.py
@@ -34,4 +35,5 @@ tests/test_dbt_tests.py
34
35
  tests/test_faker_name_inference.py
35
36
  tests/test_generation.py
36
37
  tests/test_release_stress.py
37
- tests/test_row_identity.py
38
+ tests/test_row_identity.py
39
+ tests/test_table_seeds.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.3.1"
7
+ version = "1.4.0"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -0,0 +1,139 @@
1
+ """A seed has to reproduce tomorrow, not only for the rest of today.
2
+
3
+ Dates and timestamps were generated relative to the wall clock, so two runs of
4
+ the same seed on different days agreed on every number and disagreed on every
5
+ date. `as_of` pins the anchor instead, and these tests hold both halves of that:
6
+ an explicit anchor survives the day changing underneath it, and the default
7
+ still follows today.
8
+ """
9
+
10
+ from datetime import date, datetime, timedelta
11
+
12
+ import pytest
13
+
14
+ import model2data.generate.faker as faker_module
15
+ from model2data.generate.core import generate_data_from_dbml
16
+ from model2data.generate.faker import _anchor_date, _years_before
17
+ from model2data.parse.dbml import ColumnDef, TableDef
18
+
19
+ ANCHOR = date(2024, 3, 15)
20
+
21
+
22
+ def _dated_table() -> dict[str, TableDef]:
23
+ return {
24
+ "events": TableDef(
25
+ name="events",
26
+ columns=[
27
+ ColumnDef("id", "int", {"pk"}),
28
+ ColumnDef("happened_on", "date", {"not null"}),
29
+ ColumnDef("created_at", "timestamp", {"not null"}),
30
+ ],
31
+ )
32
+ }
33
+
34
+
35
+ def _freeze_today(monkeypatch: pytest.MonkeyPatch, today: date) -> None:
36
+ """Move the day the process thinks it is, the way the calendar would.
37
+
38
+ `date.today()` is the only clock generation reads, so replacing it with a
39
+ subclass that answers a fixed day is enough to run "tomorrow" inside a
40
+ single test.
41
+ """
42
+
43
+ class _FrozenDate(date):
44
+ @classmethod
45
+ def today(cls) -> date:
46
+ return today
47
+
48
+ monkeypatch.setattr(faker_module, "date", _FrozenDate)
49
+
50
+
51
+ def _frames(**kwargs) -> dict:
52
+ return generate_data_from_dbml(_dated_table(), [], base_rows=30, seed=7, **kwargs)
53
+
54
+
55
+ class TestAnchorHelpers:
56
+ def test_none_means_today(self, monkeypatch: pytest.MonkeyPatch):
57
+ _freeze_today(monkeypatch, date(2031, 8, 9))
58
+ assert _anchor_date(None) == date(2031, 8, 9)
59
+
60
+ def test_a_date_is_used_as_given(self):
61
+ assert _anchor_date(ANCHOR) == ANCHOR
62
+
63
+ def test_a_datetime_is_narrowed_to_its_day(self):
64
+ assert _anchor_date(datetime(2024, 3, 15, 23, 59, 59)) == ANCHOR
65
+
66
+ def test_two_years_back_is_the_same_day_of_the_year(self):
67
+ assert _years_before(date(2024, 3, 15), 2) == date(2022, 3, 15)
68
+
69
+ def test_the_29th_of_february_lands_on_the_28th(self):
70
+ """The one day of the year with no counterpart two years earlier."""
71
+ assert _years_before(date(2024, 2, 29), 2) == date(2022, 2, 28)
72
+
73
+
74
+ class TestSeededRunsReproduceAcrossDays:
75
+ def test_an_explicit_anchor_survives_the_day_changing(self, monkeypatch: pytest.MonkeyPatch):
76
+ _freeze_today(monkeypatch, date(2024, 3, 15))
77
+ first = _frames(as_of=ANCHOR)["events"]
78
+
79
+ _freeze_today(monkeypatch, date(2026, 11, 2))
80
+ second = _frames(as_of=ANCHOR)["events"]
81
+
82
+ assert first.equals(second)
83
+
84
+ def test_without_an_anchor_the_dates_move_with_the_calendar(
85
+ self, monkeypatch: pytest.MonkeyPatch
86
+ ):
87
+ """The behaviour `as_of` exists to opt out of, pinned so it stays opt-out."""
88
+ _freeze_today(monkeypatch, date(2024, 3, 15))
89
+ first = _frames()["events"]
90
+
91
+ _freeze_today(monkeypatch, date(2026, 11, 2))
92
+ second = _frames()["events"]
93
+
94
+ assert not first["happened_on"].equals(second["happened_on"])
95
+ # Same stream, so the numbers underneath the dates are untouched.
96
+ assert first["id"].equals(second["id"])
97
+
98
+
99
+ class TestTheWindowMoves:
100
+ def test_dates_fall_in_the_two_years_up_to_the_anchor(self):
101
+ dates = _frames(as_of=ANCHOR)["events"]["happened_on"]
102
+
103
+ assert dates.min() >= date(2022, 3, 15)
104
+ assert dates.max() <= ANCHOR
105
+
106
+ def test_a_different_anchor_shifts_the_window(self):
107
+ earlier = _frames(as_of=date(2019, 1, 1))["events"]["happened_on"]
108
+ later = _frames(as_of=ANCHOR)["events"]["happened_on"]
109
+
110
+ assert earlier.max() <= date(2019, 1, 1)
111
+ assert later.min() >= date(2022, 3, 15)
112
+ assert earlier.max() < later.min()
113
+
114
+ def test_timestamps_fall_in_the_year_up_to_midnight_on_the_anchor(self):
115
+ stamps = [
116
+ datetime.fromisoformat(value) for value in _frames(as_of=ANCHOR)["events"]["created_at"]
117
+ ]
118
+ midnight = datetime(2024, 3, 15)
119
+
120
+ assert min(stamps) >= midnight - timedelta(days=365)
121
+ assert max(stamps) <= midnight
122
+ # Midnight, not `now()`: no sub-second component to churn a CSV.
123
+ assert all(stamp.microsecond == 0 for stamp in stamps)
124
+
125
+ def test_a_datetime_anchor_is_read_as_its_day(self):
126
+ from_date = _frames(as_of=ANCHOR)["events"]
127
+ from_datetime = _frames(as_of=datetime(2024, 3, 15, 16, 30, 45))["events"]
128
+
129
+ assert from_date.equals(from_datetime)
130
+
131
+
132
+ class TestTheDefaultIsUnchanged:
133
+ def test_no_anchor_still_means_today(self, monkeypatch: pytest.MonkeyPatch):
134
+ _freeze_today(monkeypatch, date(2024, 3, 15))
135
+
136
+ implicit = _frames()["events"]
137
+ explicit = _frames(as_of=ANCHOR)["events"]
138
+
139
+ assert implicit.equals(explicit)
@@ -1,6 +1,7 @@
1
1
  """Direct tests of CLI main function for proper coverage."""
2
2
 
3
3
  import os
4
+ import re
4
5
 
5
6
  import pandas as pd
6
7
  from typer.testing import CliRunner
@@ -9,6 +10,14 @@ from model2data.cli import app
9
10
 
10
11
  runner = CliRunner()
11
12
 
13
+ _ANSI = re.compile(r"\x1b\[[0-9;]*m")
14
+
15
+
16
+ def _plain(output: str) -> str:
17
+ """Typer's error box is coloured on CI, which splits `--flag` into styled
18
+ pieces; strip the escapes before looking for an option name in it."""
19
+ return _ANSI.sub("", output)
20
+
12
21
 
13
22
  def test_cli_basic_generation(tmp_path):
14
23
  """Test basic CLI generation with direct invocation."""
@@ -832,3 +841,89 @@ def test_cli_rows_for_rejects_malformed_values(tmp_path):
832
841
  assert "Expected TABLE=N" in _run_rows_for(tmp_path, "--rows-for", "users").output
833
842
  assert "whole number" in _run_rows_for(tmp_path, "--rows-for", "users=lots").output
834
843
  assert "at least 1" in _run_rows_for(tmp_path, "--rows-for", "users=0").output
844
+
845
+
846
+ # Same two tables as ROWS_FOR_SCHEMA, plus the date and timestamp columns
847
+ # --as-of has anything to say about.
848
+ DATED_SCHEMA = """
849
+ Table users {
850
+ id int [pk]
851
+ email email [unique]
852
+ signed_up date [not null]
853
+ }
854
+ Table orders {
855
+ id int [pk]
856
+ user_id int [not null]
857
+ placed_at timestamp [not null]
858
+ }
859
+ Ref: orders.user_id > users.id
860
+ """
861
+
862
+
863
+ def _run_shop(tmp_path, name, *extra_args):
864
+ """Generate the dated users/orders schema under a chosen project name."""
865
+ dbml_file = tmp_path / "shop.dbml"
866
+ dbml_file.write_text(DATED_SCHEMA)
867
+ cwd = os.getcwd()
868
+ os.chdir(tmp_path)
869
+ try:
870
+ return runner.invoke(
871
+ app,
872
+ ["--file", str(dbml_file), "--rows", "30", "--name", name, *extra_args],
873
+ )
874
+ finally:
875
+ os.chdir(cwd)
876
+
877
+
878
+ def test_cli_as_of_anchors_generated_dates(tmp_path):
879
+ result = _run_shop(tmp_path, "anchored", "--seed", "42", "--as-of", "2024-03-15")
880
+ assert result.exit_code == 0, result.output
881
+ assert "Anchoring generated dates on: 2024-03-15" in result.output
882
+
883
+ orders = pd.read_csv(tmp_path / "dbt_anchored" / "seeds" / "raw" / "orders.csv")
884
+ placed = pd.to_datetime(orders["placed_at"].dropna())
885
+ assert placed.max() <= pd.Timestamp("2024-03-15")
886
+ assert placed.min() >= pd.Timestamp("2024-03-15") - pd.Timedelta(days=365)
887
+
888
+
889
+ def test_cli_as_of_rejects_a_date_it_cannot_read(tmp_path):
890
+ result = _run_shop(tmp_path, "bad_date", "--as-of", "the 15th")
891
+ assert result.exit_code != 0
892
+ assert "--as-of" in _plain(result.output)
893
+
894
+
895
+ def test_cli_table_seed_re_rolls_only_that_table(tmp_path):
896
+ common = ("--seed", "42", "--as-of", "2024-03-15")
897
+ assert _run_shop(tmp_path, "before", *common).exit_code == 0
898
+ result = _run_shop(tmp_path, "after", *common, "--table-seed", "orders=7")
899
+ assert result.exit_code == 0, result.output
900
+ assert "Re-rolling with a table seed of its own: orders=7" in result.output
901
+
902
+ def seed_csv(project, table):
903
+ return (tmp_path / f"dbt_{project}" / "seeds" / "raw" / f"{table}.csv").read_text()
904
+
905
+ assert seed_csv("before", "users") == seed_csv("after", "users")
906
+ assert seed_csv("before", "orders") != seed_csv("after", "orders")
907
+
908
+
909
+ def test_cli_table_seed_needs_a_seed_to_re_roll_out_of(tmp_path):
910
+ result = _run_shop(tmp_path, "no_seed", "--table-seed", "orders=7")
911
+ assert result.exit_code != 0
912
+ assert "--seed" in _plain(result.output)
913
+
914
+
915
+ def test_cli_table_seed_rejects_an_unknown_table(tmp_path):
916
+ result = _run_shop(tmp_path, "typo", "--seed", "1", "--table-seed", "ordres=7")
917
+ assert result.exit_code != 0
918
+ assert "No table named 'ordres'" in result.output
919
+
920
+
921
+ def test_cli_table_seed_rejects_malformed_values(tmp_path):
922
+ assert (
923
+ "Expected TABLE=N"
924
+ in _run_shop(tmp_path, "m1", "--seed", "1", "--table-seed", "orders").output
925
+ )
926
+ assert (
927
+ "whole number"
928
+ in _run_shop(tmp_path, "m2", "--seed", "1", "--table-seed", "orders=x").output
929
+ )
@@ -0,0 +1,225 @@
1
+ """Re-rolling one table must leave the others exactly where they were.
2
+
3
+ One RNG stream for the whole run meant a table's values depended on every table
4
+ generated before it, so "I like these customers, give me different orders" was
5
+ not a thing that could be asked. Each table now draws from its own stream,
6
+ derived from the run seed and its own name. These tests hold the property that
7
+ buys: change one table's entry in `table_seeds` and only that table -- plus the
8
+ foreign keys that have to follow it -- moves.
9
+ """
10
+
11
+ import pandas as pd
12
+ import pytest
13
+ from pandas.testing import assert_frame_equal
14
+
15
+ from model2data.generate.core import _table_stream_seed, generate_data_from_dbml
16
+ from model2data.parse.dbml import ColumnDef, TableDef
17
+
18
+ # customers <- orders <- order_items, plus a products table connected to
19
+ # nothing, so the suite covers a parent, a middle table, a grandchild and a
20
+ # bystander in one schema.
21
+ REFS = [
22
+ {
23
+ "source_table": "orders",
24
+ "source_column": "customer_id",
25
+ "target_table": "customers",
26
+ "target_column": "id",
27
+ },
28
+ {
29
+ "source_table": "order_items",
30
+ "source_column": "order_id",
31
+ "target_table": "orders",
32
+ "target_column": "id",
33
+ },
34
+ ]
35
+
36
+
37
+ def _schema() -> dict[str, TableDef]:
38
+ return {
39
+ "customers": TableDef(
40
+ name="customers",
41
+ columns=[
42
+ ColumnDef("id", "int", {"pk"}),
43
+ ColumnDef("email", "varchar", {"not null"}),
44
+ ],
45
+ ),
46
+ "orders": TableDef(
47
+ name="orders",
48
+ columns=[
49
+ ColumnDef("id", "int", {"pk"}),
50
+ ColumnDef("customer_id", "int", {"not null"}),
51
+ ColumnDef("total", "numeric", {"not null"}),
52
+ ],
53
+ ),
54
+ "order_items": TableDef(
55
+ name="order_items",
56
+ columns=[
57
+ ColumnDef("id", "int", {"pk"}),
58
+ ColumnDef("order_id", "int", {"not null"}),
59
+ ColumnDef("quantity", "int", {"not null"}),
60
+ ],
61
+ ),
62
+ "products": TableDef(
63
+ name="products",
64
+ columns=[
65
+ ColumnDef("id", "int", {"pk"}),
66
+ ColumnDef("title", "varchar", {"not null"}),
67
+ ],
68
+ ),
69
+ }
70
+
71
+
72
+ def _generate(**kwargs) -> dict[str, pd.DataFrame]:
73
+ return generate_data_from_dbml(_schema(), REFS, base_rows=40, seed=11, **kwargs)
74
+
75
+
76
+ class TestReRollingOneTable:
77
+ def test_only_the_named_table_and_its_descendants_move(self):
78
+ baseline = _generate()
79
+ rolled = _generate(table_seeds={"orders": 7})
80
+
81
+ assert not rolled["orders"].equals(baseline["orders"]), (
82
+ "the whole point of the override is that this table changes"
83
+ )
84
+ assert_frame_equal(rolled["customers"], baseline["customers"])
85
+ assert_frame_equal(rolled["products"], baseline["products"])
86
+
87
+ def test_a_child_of_an_untouched_table_is_byte_identical(self):
88
+ """Re-rolling `products` reaches nothing: `orders` and its child stay put.
89
+
90
+ `order_items` is the interesting one -- it is two hops downstream of a
91
+ table nobody touched, and under a single shared stream it would have
92
+ moved anyway simply for being generated later in the run.
93
+ """
94
+ baseline = _generate()
95
+ rolled = _generate(table_seeds={"products": 5})
96
+
97
+ assert not rolled["products"].equals(baseline["products"])
98
+ assert_frame_equal(rolled["customers"], baseline["customers"])
99
+ assert_frame_equal(rolled["orders"], baseline["orders"])
100
+ assert_frame_equal(rolled["order_items"], baseline["order_items"])
101
+
102
+ def test_a_childs_own_columns_survive_its_parent_being_re_rolled(self):
103
+ """Only the FK column of a child may follow its parent."""
104
+ baseline = _generate()
105
+ rolled = _generate(table_seeds={"orders": 7})
106
+
107
+ assert_frame_equal(
108
+ rolled["order_items"].drop(columns=["order_id"]),
109
+ baseline["order_items"].drop(columns=["order_id"]),
110
+ )
111
+
112
+ def test_referential_integrity_holds_against_the_new_parent(self):
113
+ rolled = _generate(table_seeds={"orders": 7})
114
+
115
+ assert set(rolled["orders"]["customer_id"]) <= set(rolled["customers"]["id"])
116
+ assert set(rolled["order_items"]["order_id"]) <= set(rolled["orders"]["id"])
117
+
118
+ def test_the_old_parent_rows_are_genuinely_gone(self):
119
+ """A child that still pointed at the pre-roll ids would be dangling."""
120
+ baseline = _generate()
121
+ rolled = _generate(table_seeds={"orders": 7})
122
+
123
+ stale = set(baseline["orders"]["id"]) - set(rolled["orders"]["id"])
124
+ assert stale, "the re-rolled parent has to hand out at least some new ids"
125
+ assert not set(rolled["order_items"]["order_id"]) & stale
126
+
127
+ def test_two_tables_can_be_re_rolled_at_once(self):
128
+ baseline = _generate()
129
+ rolled = _generate(table_seeds={"customers": 2, "products": 5})
130
+
131
+ assert not rolled["customers"].equals(baseline["customers"])
132
+ assert not rolled["products"].equals(baseline["products"])
133
+
134
+ def test_re_rolling_is_itself_reproducible(self):
135
+ first = _generate(table_seeds={"orders": 7})
136
+ second = _generate(table_seeds={"orders": 7})
137
+
138
+ for name in first:
139
+ assert_frame_equal(first[name], second[name])
140
+
141
+ def test_a_different_override_value_gives_a_different_table(self):
142
+ seven = _generate(table_seeds={"orders": 7})
143
+ eight = _generate(table_seeds={"orders": 8})
144
+
145
+ assert not seven["orders"].equals(eight["orders"])
146
+ assert_frame_equal(seven["customers"], eight["customers"])
147
+
148
+ def test_an_empty_mapping_changes_nothing(self):
149
+ assert_frame_equal(_generate(table_seeds={})["orders"], _generate()["orders"])
150
+
151
+
152
+ class TestValidation:
153
+ def test_an_unknown_table_name_is_an_error(self):
154
+ with pytest.raises(ValueError, match="No table named 'ordres'"):
155
+ _generate(table_seeds={"ordres": 7})
156
+
157
+ def test_the_message_lists_the_tables_that_do_exist(self):
158
+ with pytest.raises(ValueError, match="customers, order_items, orders, products"):
159
+ _generate(table_seeds={"nope": 1})
160
+
161
+ def test_several_unknown_names_are_reported_together(self):
162
+ with pytest.raises(ValueError, match="No tables named 'a', 'b'"):
163
+ _generate(table_seeds={"b": 1, "a": 2})
164
+
165
+ def test_it_needs_a_seed_to_re_roll_out_of(self):
166
+ with pytest.raises(ValueError, match="table_seeds needs a seed"):
167
+ generate_data_from_dbml(
168
+ _schema(), REFS, base_rows=10, seed=None, table_seeds={"orders": 7}
169
+ )
170
+
171
+ def test_no_seed_and_no_overrides_is_still_fine(self):
172
+ frames = generate_data_from_dbml(_schema(), REFS, base_rows=10, seed=None)
173
+
174
+ assert len(frames["orders"]) == 10
175
+
176
+ def test_validation_runs_before_anything_is_generated(self):
177
+ """A typo should not cost a full generation pass first."""
178
+ with pytest.raises(ValueError):
179
+ generate_data_from_dbml(
180
+ _schema(), REFS, base_rows=2_000_000, seed=1, table_seeds={"typo": 1}
181
+ )
182
+
183
+
184
+ class TestTheDerivedSeed:
185
+ def test_it_is_stable_across_processes(self):
186
+ """Pinned literally: `hash()` is salted per process and would not be."""
187
+ assert _table_stream_seed(11, "orders", None) == 11237587294246756286
188
+ assert _table_stream_seed(11, "orders", 7) == 9635424189123890928
189
+
190
+ def test_each_table_gets_a_different_stream(self):
191
+ assert _table_stream_seed(11, "orders", None) != _table_stream_seed(11, "customers", None)
192
+
193
+ def test_each_run_seed_gets_a_different_stream(self):
194
+ assert _table_stream_seed(11, "orders", None) != _table_stream_seed(12, "orders", None)
195
+
196
+ def test_no_override_is_not_the_same_as_an_override_of_zero(self):
197
+ assert _table_stream_seed(11, "orders", None) != _table_stream_seed(11, "orders", 0)
198
+
199
+
200
+ class TestOtherOptionsStillCompose:
201
+ def test_row_counts_are_untouched_by_a_re_roll(self):
202
+ rolled = _generate(row_overrides={"orders": 120}, table_seeds={"orders": 7})
203
+
204
+ assert len(rolled["orders"]) == 120
205
+ assert len(rolled["customers"]) == 40
206
+
207
+ def test_a_pinned_anchor_and_a_re_roll_work_together(self):
208
+ from datetime import date
209
+
210
+ tables = {
211
+ "events": TableDef(
212
+ name="events",
213
+ columns=[
214
+ ColumnDef("id", "int", {"pk"}),
215
+ ColumnDef("happened_on", "date", {"not null"}),
216
+ ],
217
+ )
218
+ }
219
+ kwargs = {"base_rows": 20, "seed": 4, "as_of": date(2024, 3, 15)}
220
+
221
+ first = generate_data_from_dbml(tables, [], table_seeds={"events": 1}, **kwargs)["events"]
222
+ second = generate_data_from_dbml(tables, [], table_seeds={"events": 1}, **kwargs)["events"]
223
+
224
+ assert_frame_equal(first, second)
225
+ assert first["happened_on"].max() <= date(2024, 3, 15)
File without changes
File without changes