model2data 1.4.0__tar.gz → 1.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.4.0/model2data.egg-info → model2data-1.6.0}/PKG-INFO +1 -1
- {model2data-1.4.0 → model2data-1.6.0}/README.md +67 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/cli.py +80 -10
- {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/core.py +52 -3
- {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/faker.py +184 -19
- model2data-1.6.0/model2data/generate/hints.py +229 -0
- model2data-1.6.0/model2data/generate/options.py +80 -0
- model2data-1.6.0/model2data/generate/timeline.py +454 -0
- {model2data-1.4.0 → model2data-1.6.0/model2data.egg-info}/PKG-INFO +1 -1
- {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/SOURCES.txt +8 -1
- {model2data-1.4.0 → model2data-1.6.0}/pyproject.toml +1 -1
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_cli.py +22 -0
- model2data-1.6.0/tests/test_column_time_hints.py +241 -0
- model2data-1.6.0/tests/test_options.py +52 -0
- model2data-1.6.0/tests/test_shaping.py +493 -0
- model2data-1.6.0/tests/test_timeline.py +421 -0
- {model2data-1.4.0 → model2data-1.6.0}/LICENSE +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/README_PYPI.md +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/__init__.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/__init__.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/project.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/tests.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/__init__.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/relationships.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/parse/__init__.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/parse/dbml.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data/utils.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/setup.cfg +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_as_of_anchor.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbml_parser.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_integration.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_naming.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_project.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_tests.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_faker_name_inference.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_generation.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_release_stress.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_row_identity.py +0 -0
- {model2data-1.4.0 → model2data-1.6.0}/tests/test_table_seeds.py +0 -0
|
@@ -165,6 +165,73 @@ it's a per-run setting, so a table can't end up holding one Belgian and one Amer
|
|
|
165
165
|
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
|
|
166
166
|
```
|
|
167
167
|
|
|
168
|
+
### Shape when things happen
|
|
169
|
+
|
|
170
|
+
By default, every date and timestamp is drawn uniformly across its window. `--business-hours`,
|
|
171
|
+
`--growth`, and `--seasonality` shape that instead — weekdays and working hours, a trend across
|
|
172
|
+
the window, and an annual cycle peaking in Q4:
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
|
|
176
|
+
--business-hours --growth 0.5 --seasonality 0.3
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
Within a row, created/updated/deleted-style columns are ordered automatically — `updated_at` never
|
|
180
|
+
lands before its own `created_at` — under any profile, uniform included. A column whose name
|
|
181
|
+
doesn't say what it depends on can say so explicitly with an `after` note:
|
|
182
|
+
|
|
183
|
+
```dbml
|
|
184
|
+
shipped_at timestamp [note: '{"after": "ordered_at"}']
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
The flags above shape every date and timestamp column the same way, run-wide. `business_hours`,
|
|
188
|
+
`growth`, and `seasonality` column note hints override that for one column at a time — the whole
|
|
189
|
+
point being a run can be uniform everywhere except the one column that needs shaping, or shaped
|
|
190
|
+
everywhere except the one column that shouldn't be:
|
|
191
|
+
|
|
192
|
+
```dbml
|
|
193
|
+
Table orders {
|
|
194
|
+
id int [pk]
|
|
195
|
+
created_at timestamp [note: '{"business_hours": true, "growth": 0.4}']
|
|
196
|
+
refunded_at timestamp [note: '{"growth": 0}']
|
|
197
|
+
}
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Here `created_at` gets business hours and growth even on an otherwise-uniform run, while
|
|
201
|
+
`refunded_at` stays flat even under `--growth 0.5` — each hint only replaces the fields it names,
|
|
202
|
+
so a partial hint like `{"growth": 0}` leaves that column's `business_hours`/`seasonality` at
|
|
203
|
+
whatever the run-level flags set.
|
|
204
|
+
|
|
205
|
+
### Shape how the data is spread
|
|
206
|
+
|
|
207
|
+
By default every parent row is equally likely to be picked for a child row, and every column
|
|
208
|
+
gets the same generic null rate, value spread, and true/false split. `--skew` changes the first
|
|
209
|
+
part: `0.0` is that uniform default, `1.0` means a handful of parents hold most of the children —
|
|
210
|
+
"a fifth of the customers place most of the orders":
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --skew 0.8
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Column note hints shape the rest, per column:
|
|
217
|
+
|
|
218
|
+
```dbml
|
|
219
|
+
Table orders {
|
|
220
|
+
id int [pk]
|
|
221
|
+
customer_id int [ref: > customers.id, note: '{"skew": 0.9}']
|
|
222
|
+
status order_status [note: '{"weights": {"delivered": 20, "cancelled": 2}}']
|
|
223
|
+
is_paid boolean [note: '{"true_rate": 0.9}']
|
|
224
|
+
discount_code varchar [note: '{"null_rate": 0.8}']
|
|
225
|
+
shipping_city varchar [note: '{"distinct": 12}']
|
|
226
|
+
}
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
`skew` on a foreign key overrides `--skew` for just that column. `weights` biases an enum column
|
|
230
|
+
toward the values named (unnamed values still appear, at weight 1). `true_rate` is the fraction of
|
|
231
|
+
non-null rows a boolean column comes back `true`. `null_rate` replaces the column's default null
|
|
232
|
+
fraction outright. `distinct` draws the column's values from a fixed-size pool instead of a fresh
|
|
233
|
+
value per row — a `shipping_city` most warehouses only ever see a handful of.
|
|
234
|
+
|
|
168
235
|
Run dbt to load, transform, and test the data:
|
|
169
236
|
|
|
170
237
|
```bash
|
|
@@ -24,6 +24,7 @@ from model2data.generate.faker import (
|
|
|
24
24
|
get_unmapped_columns,
|
|
25
25
|
reset_stats,
|
|
26
26
|
)
|
|
27
|
+
from model2data.generate.options import TimeProfile
|
|
27
28
|
from model2data.parse.dbml import get_parse_warnings, parse_dbml
|
|
28
29
|
from model2data.utils import normalize_identifier
|
|
29
30
|
|
|
@@ -186,6 +187,43 @@ def main(
|
|
|
186
187
|
"Pin it and a --seed run reproduces on any later day, not just the day it first ran."
|
|
187
188
|
),
|
|
188
189
|
),
|
|
190
|
+
business_hours: bool = typer.Option(
|
|
191
|
+
False,
|
|
192
|
+
"--business-hours",
|
|
193
|
+
help=(
|
|
194
|
+
"Weight generated timestamps toward weekdays and working hours,\n"
|
|
195
|
+
"instead of spreading them evenly over every hour of every day."
|
|
196
|
+
),
|
|
197
|
+
),
|
|
198
|
+
growth: float = typer.Option(
|
|
199
|
+
0.0,
|
|
200
|
+
"--growth",
|
|
201
|
+
min=-1.0,
|
|
202
|
+
help=(
|
|
203
|
+
"Relative change in activity across the generated window: 0.5 means the end\n"
|
|
204
|
+
"is half again as busy as the start, -0.3 means it tailed off. Default: flat."
|
|
205
|
+
),
|
|
206
|
+
),
|
|
207
|
+
seasonality: float = typer.Option(
|
|
208
|
+
0.0,
|
|
209
|
+
"--seasonality",
|
|
210
|
+
min=0.0,
|
|
211
|
+
max=1.0,
|
|
212
|
+
help=(
|
|
213
|
+
"Strength of an annual cycle in generated timestamps, 0 (none) to 1,\n"
|
|
214
|
+
"peaking in the fourth quarter."
|
|
215
|
+
),
|
|
216
|
+
),
|
|
217
|
+
skew: float = typer.Option(
|
|
218
|
+
0.0,
|
|
219
|
+
"--skew",
|
|
220
|
+
min=0.0,
|
|
221
|
+
max=1.0,
|
|
222
|
+
help=(
|
|
223
|
+
"How unevenly child rows are spread over their parents: 0 (every parent equally\n"
|
|
224
|
+
"likely, the default) to 1 (a few parents hold most of the children)."
|
|
225
|
+
),
|
|
226
|
+
),
|
|
189
227
|
locale: Optional[str] = typer.Option(
|
|
190
228
|
None,
|
|
191
229
|
"--locale",
|
|
@@ -251,6 +289,28 @@ def main(
|
|
|
251
289
|
if as_of is not None:
|
|
252
290
|
typer.echo(f"📅 Anchoring generated dates on: {as_of.date()}")
|
|
253
291
|
|
|
292
|
+
# Same guard again for the shaping options: a direct call leaves them as
|
|
293
|
+
# `OptionInfo` objects, which means "not supplied", i.e. the uniform draw.
|
|
294
|
+
time_profile = TimeProfile(
|
|
295
|
+
business_hours=business_hours if isinstance(business_hours, bool) else False,
|
|
296
|
+
growth=growth if isinstance(growth, (int, float)) else 0.0,
|
|
297
|
+
seasonality=seasonality if isinstance(seasonality, (int, float)) else 0.0,
|
|
298
|
+
)
|
|
299
|
+
skew = skew if isinstance(skew, (int, float)) else 0.0
|
|
300
|
+
if not time_profile.is_uniform:
|
|
301
|
+
shaped = [
|
|
302
|
+
label
|
|
303
|
+
for label, active in (
|
|
304
|
+
("business hours", time_profile.business_hours),
|
|
305
|
+
(f"growth {time_profile.growth:+.0%}", time_profile.growth != 0.0),
|
|
306
|
+
(f"seasonality {time_profile.seasonality:.0%}", time_profile.seasonality != 0.0),
|
|
307
|
+
)
|
|
308
|
+
if active
|
|
309
|
+
]
|
|
310
|
+
typer.echo(f"🕒 Shaping timestamps: {', '.join(shaped)}")
|
|
311
|
+
if skew:
|
|
312
|
+
typer.echo(f"📈 Skewing child rows over their parents: {skew:.0%}")
|
|
313
|
+
|
|
254
314
|
# -------------------------
|
|
255
315
|
# Parse DBML (names untouched)
|
|
256
316
|
# -------------------------
|
|
@@ -296,16 +356,26 @@ def main(
|
|
|
296
356
|
# -------------------------
|
|
297
357
|
typer.echo("🧮 Generating synthetic datasets from DBML definitions...")
|
|
298
358
|
reset_stats()
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
359
|
+
try:
|
|
360
|
+
generated_tables = generate_data_from_dbml(
|
|
361
|
+
tables=tables,
|
|
362
|
+
refs=refs,
|
|
363
|
+
base_rows=rows,
|
|
364
|
+
seed=seed,
|
|
365
|
+
row_overrides=row_overrides,
|
|
366
|
+
locale=locale,
|
|
367
|
+
as_of=as_of,
|
|
368
|
+
table_seeds=table_seeds,
|
|
369
|
+
time_profile=time_profile,
|
|
370
|
+
skew=skew,
|
|
371
|
+
)
|
|
372
|
+
except ValueError as exc:
|
|
373
|
+
# A bad column-note hint (an unknown enum value in `weights`, `distinct`
|
|
374
|
+
# on a foreign key...) is a mistake in the schema, not a bug -- it
|
|
375
|
+
# deserves the same one-line, non-traceback treatment as every other
|
|
376
|
+
# validation error this command already reports.
|
|
377
|
+
typer.echo(f"❌ {exc}")
|
|
378
|
+
raise typer.Exit(1) from None
|
|
309
379
|
|
|
310
380
|
# -------------------------
|
|
311
381
|
# Write dbt seeds (normalized names)
|
|
@@ -17,10 +17,13 @@ from model2data.generate.faker import (
|
|
|
17
17
|
reset_row_pools,
|
|
18
18
|
set_locale,
|
|
19
19
|
)
|
|
20
|
+
from model2data.generate.hints import validate_hints
|
|
21
|
+
from model2data.generate.options import UNIFORM, TimeProfile, validate_skew
|
|
20
22
|
from model2data.generate.relationships import (
|
|
21
23
|
build_fk_lookup,
|
|
22
24
|
classify_refs,
|
|
23
25
|
)
|
|
26
|
+
from model2data.generate.timeline import order_row_times
|
|
24
27
|
from model2data.parse.dbml import TableDef
|
|
25
28
|
|
|
26
29
|
# Tables the most recent generate_data_from_dbml() call found stuck in an
|
|
@@ -72,6 +75,8 @@ def generate_data_from_dbml(
|
|
|
72
75
|
locale: Optional[str] = None,
|
|
73
76
|
as_of: AsOf = None,
|
|
74
77
|
table_seeds: Optional[Mapping[str, int]] = None,
|
|
78
|
+
time_profile: Optional[TimeProfile] = None,
|
|
79
|
+
skew: float = 0.0,
|
|
75
80
|
) -> dict[str, pd.DataFrame]:
|
|
76
81
|
"""
|
|
77
82
|
Generate synthetic datasets from parsed DBML definitions.
|
|
@@ -107,10 +112,23 @@ def generate_data_from_dbml(
|
|
|
107
112
|
`seed` to work off -- with none, every table is already different on every
|
|
108
113
|
run -- and unknown table names are an error rather than a silent no-op.
|
|
109
114
|
|
|
115
|
+
`time_profile` shapes *when* generated timestamps and dates fall: toward
|
|
116
|
+
business hours and weekdays, along a growth trend, with a seasonal peak.
|
|
117
|
+
See `TimeProfile`. None is the uniform profile, which draws every second of
|
|
118
|
+
the window with equal probability the way earlier releases did.
|
|
119
|
+
|
|
120
|
+
`skew` is how unevenly a child table's rows are spread over its parents,
|
|
121
|
+
from `0.0` (every parent equally likely, the earlier behaviour) to `1.0` (a
|
|
122
|
+
few parents hold most of the children). A column can override it with a
|
|
123
|
+
`{"skew": ...}` hint in its note.
|
|
124
|
+
|
|
110
125
|
This function is deterministic if a seed is provided (and, with `as_of`,
|
|
111
126
|
on any day). It performs no filesystem I/O and returns pandas DataFrames.
|
|
112
127
|
"""
|
|
113
128
|
_validate_table_seeds(tables, table_seeds, seed)
|
|
129
|
+
validate_hints(tables, refs)
|
|
130
|
+
profile = time_profile or UNIFORM
|
|
131
|
+
skew = validate_skew(skew)
|
|
114
132
|
|
|
115
133
|
# Locale first, then the seed: switching locale builds a new Faker, and the
|
|
116
134
|
# seed has to be the last word on the generator that actually runs.
|
|
@@ -210,14 +228,35 @@ def generate_data_from_dbml(
|
|
|
210
228
|
force_not_null=column.name in composite_pk_columns,
|
|
211
229
|
table_name=table_name,
|
|
212
230
|
as_of=as_of,
|
|
231
|
+
time_profile=profile,
|
|
232
|
+
skew=skew,
|
|
213
233
|
)
|
|
214
234
|
|
|
215
235
|
df = pd.DataFrame(data)
|
|
236
|
+
# Ordering runs before FK resolution/dedup so a self-ref repair or a
|
|
237
|
+
# composite-key retry regenerates a temporal column's value into a
|
|
238
|
+
# frame that already respects created/updated/closed ordering, rather
|
|
239
|
+
# than one where only the untouched columns do.
|
|
240
|
+
df = order_row_times(df, table_def, as_of=as_of)
|
|
216
241
|
df = _resolve_self_referencing_fks(
|
|
217
|
-
df,
|
|
242
|
+
df,
|
|
243
|
+
table_def,
|
|
244
|
+
table_name,
|
|
245
|
+
fk_lookup,
|
|
246
|
+
row_count,
|
|
247
|
+
as_of=as_of,
|
|
248
|
+
time_profile=profile,
|
|
249
|
+
skew=skew,
|
|
218
250
|
)
|
|
219
251
|
df = _deduplicate_composite_keys(
|
|
220
|
-
df,
|
|
252
|
+
df,
|
|
253
|
+
table_def,
|
|
254
|
+
table_name,
|
|
255
|
+
fk_lookup,
|
|
256
|
+
generated,
|
|
257
|
+
as_of=as_of,
|
|
258
|
+
time_profile=profile,
|
|
259
|
+
skew=skew,
|
|
221
260
|
)
|
|
222
261
|
|
|
223
262
|
# -----------------------------------------------------
|
|
@@ -299,6 +338,8 @@ def _deduplicate_composite_keys(
|
|
|
299
338
|
generated: dict[str, pd.DataFrame],
|
|
300
339
|
max_attempts: int = 20,
|
|
301
340
|
as_of: AsOf = None,
|
|
341
|
+
time_profile: Optional[TimeProfile] = None,
|
|
342
|
+
skew: float = 0.0,
|
|
302
343
|
) -> pd.DataFrame:
|
|
303
344
|
"""
|
|
304
345
|
Regenerate colliding rows for any pk/unique composite key declared via an
|
|
@@ -350,7 +391,11 @@ def _deduplicate_composite_keys(
|
|
|
350
391
|
df.at[idx, col_name] = random.choice(fk_pools[col_name])
|
|
351
392
|
else:
|
|
352
393
|
df.at[idx, col_name] = generate_column_values(
|
|
353
|
-
col_def,
|
|
394
|
+
col_def,
|
|
395
|
+
row_count=1,
|
|
396
|
+
as_of=as_of,
|
|
397
|
+
time_profile=time_profile,
|
|
398
|
+
skew=skew,
|
|
354
399
|
)[0]
|
|
355
400
|
combo = tuple(df.at[idx, c] for c in key_columns)
|
|
356
401
|
attempts += 1
|
|
@@ -377,6 +422,8 @@ def _resolve_self_referencing_fks(
|
|
|
377
422
|
fk_lookup: dict[tuple[str, str], tuple[str, str]],
|
|
378
423
|
row_count: int,
|
|
379
424
|
as_of: AsOf = None,
|
|
425
|
+
time_profile: Optional[TimeProfile] = None,
|
|
426
|
+
skew: float = 0.0,
|
|
380
427
|
) -> pd.DataFrame:
|
|
381
428
|
"""
|
|
382
429
|
Re-generate any FK column that references its own table (e.g. a
|
|
@@ -415,6 +462,8 @@ def _resolve_self_referencing_fks(
|
|
|
415
462
|
force_not_null=column.name in composite_pk_columns,
|
|
416
463
|
table_name=table_name,
|
|
417
464
|
as_of=as_of,
|
|
465
|
+
time_profile=time_profile,
|
|
466
|
+
skew=skew,
|
|
418
467
|
)
|
|
419
468
|
|
|
420
469
|
return df
|
|
@@ -1,16 +1,19 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import math
|
|
3
4
|
import random
|
|
4
5
|
import re
|
|
5
6
|
import unicodedata
|
|
6
7
|
import uuid
|
|
7
|
-
from dataclasses import dataclass
|
|
8
|
+
from dataclasses import dataclass, replace
|
|
8
9
|
from datetime import date, datetime, timedelta
|
|
9
10
|
from typing import Callable, Optional, Union
|
|
10
11
|
|
|
11
12
|
import pandas as pd
|
|
12
13
|
from faker import Faker
|
|
13
14
|
|
|
15
|
+
from model2data.generate.options import UNIFORM, TimeProfile
|
|
16
|
+
from model2data.generate.timeline import weighted_dates, weighted_timestamps
|
|
14
17
|
from model2data.parse.dbml import ColumnDef
|
|
15
18
|
|
|
16
19
|
# ---------------------------------------------------------
|
|
@@ -502,6 +505,68 @@ def _infer_by_type(base_type: str) -> Optional[_Provider]:
|
|
|
502
505
|
return lambda: fake.format(base_type)
|
|
503
506
|
|
|
504
507
|
|
|
508
|
+
def _column_time_profile(
|
|
509
|
+
column: ColumnDef, time_profile: Optional[TimeProfile]
|
|
510
|
+
) -> Optional[TimeProfile]:
|
|
511
|
+
"""The run-level `time_profile` with this column's note hints applied on top.
|
|
512
|
+
|
|
513
|
+
Mirrors the `skew` override on the FK branch, one level up: a note doesn't
|
|
514
|
+
replace the run's profile, it patches only the fields it names (via
|
|
515
|
+
`dataclasses.replace`), so `{"growth": 0}` flattens one column of a
|
|
516
|
+
growing run while `business_hours`/`seasonality` stay exactly what the
|
|
517
|
+
run set. `validate_hints` has already confirmed the column is a date or
|
|
518
|
+
timestamp column and that each hint present is well-typed, so this does
|
|
519
|
+
no validation of its own -- a column with no hint gets `time_profile`
|
|
520
|
+
back unchanged, including a bare `None`, so the uniform path stays
|
|
521
|
+
untouched.
|
|
522
|
+
"""
|
|
523
|
+
note = column.note or {}
|
|
524
|
+
overrides = {
|
|
525
|
+
key: note[key] for key in ("business_hours", "growth", "seasonality") if key in note
|
|
526
|
+
}
|
|
527
|
+
if not overrides:
|
|
528
|
+
return time_profile
|
|
529
|
+
base = time_profile if time_profile is not None else UNIFORM
|
|
530
|
+
return replace(base, **overrides)
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _generate_dates(row_count: int, as_of: AsOf, time_profile: Optional[TimeProfile]) -> list:
|
|
534
|
+
"""The date branch's values: shaped by `time_profile` when it isn't uniform.
|
|
535
|
+
|
|
536
|
+
A bare `-> list` return, matching `generate_column_values` itself: `values`
|
|
537
|
+
there is reassigned by every branch of one big if/elif chain (int, float,
|
|
538
|
+
uuid, this one...) and then written into by the nullability pass below with
|
|
539
|
+
an arbitrary `column.default`, so none of those branches can commit to a
|
|
540
|
+
concrete element type without the type checker flagging the later write as
|
|
541
|
+
unsound. Keeping `weighted_dates`'s real `list[date]` return type contained
|
|
542
|
+
to this one call, instead of leaking it into `values`, is what lets
|
|
543
|
+
`weighted_dates` itself stay properly typed for its own callers and tests.
|
|
544
|
+
"""
|
|
545
|
+
anchor = _anchor_date(as_of)
|
|
546
|
+
if time_profile is not None and not time_profile.is_uniform:
|
|
547
|
+
return weighted_dates(
|
|
548
|
+
row_count, time_profile, start=_years_before(anchor, _DATE_WINDOW_YEARS), end=anchor
|
|
549
|
+
)
|
|
550
|
+
# Explicit endpoints rather than Faker's "-2y"/"today" shorthand: those
|
|
551
|
+
# strings are resolved against `date.today()` inside Faker, which is
|
|
552
|
+
# precisely the hidden dependency on the wall clock `as_of` removes.
|
|
553
|
+
return [
|
|
554
|
+
fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
|
|
555
|
+
for _ in range(row_count)
|
|
556
|
+
]
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _generate_timestamps(row_count: int, as_of: AsOf, time_profile: Optional[TimeProfile]) -> list:
|
|
560
|
+
"""The timestamp branch's values: shaped by `time_profile` when it isn't uniform.
|
|
561
|
+
|
|
562
|
+
See `_generate_dates` for why this returns a bare `list` rather than
|
|
563
|
+
`weighted_timestamps`'s own `list[str]`.
|
|
564
|
+
"""
|
|
565
|
+
if time_profile is not None and not time_profile.is_uniform:
|
|
566
|
+
return weighted_timestamps(row_count, time_profile, anchor=_anchor_date(as_of))
|
|
567
|
+
return [_random_datetime(as_of=as_of).isoformat(sep=" ") for _ in range(row_count)]
|
|
568
|
+
|
|
569
|
+
|
|
505
570
|
# ---------------------------------------------------------
|
|
506
571
|
# Public API
|
|
507
572
|
# ---------------------------------------------------------
|
|
@@ -513,11 +578,19 @@ def generate_column_values(
|
|
|
513
578
|
force_not_null: bool = False,
|
|
514
579
|
table_name: Optional[str] = None,
|
|
515
580
|
as_of: AsOf = None,
|
|
581
|
+
time_profile: Optional[TimeProfile] = None,
|
|
582
|
+
skew: float = 0.0,
|
|
516
583
|
) -> list:
|
|
517
584
|
"""
|
|
518
585
|
Generate synthetic values for a single column.
|
|
519
586
|
Respects FKs, uniqueness, and optional min/max hints in column notes.
|
|
520
587
|
|
|
588
|
+
`time_profile` is read by the date and timestamp branches, `skew` by the
|
|
589
|
+
foreign-key branch; both default to the uniform behaviour of earlier
|
|
590
|
+
releases. They are accepted here rather than in a separate pass so that
|
|
591
|
+
every path that draws a value -- the main pass, the self-referencing FK
|
|
592
|
+
repair, the composite-key retry -- draws it the same way.
|
|
593
|
+
|
|
521
594
|
`as_of` is the date every generated date and timestamp is placed relative
|
|
522
595
|
to, defaulting to today. Pass it to make a seeded run reproduce on any
|
|
523
596
|
later day rather than only on the day it first ran.
|
|
@@ -534,12 +607,57 @@ def generate_column_values(
|
|
|
534
607
|
# (an `id` on each of two tables) stay distinguishable in the report.
|
|
535
608
|
unique_label = f"{table_name}.{column.name}" if table_name else column.name
|
|
536
609
|
|
|
610
|
+
# A `distinct` hint means "draw from a small pool", which is orthogonal to
|
|
611
|
+
# every type-specific branch below: generate the pool through this same
|
|
612
|
+
# function (so a pooled `city` still reads the row-identity pool, a pooled
|
|
613
|
+
# `int` still respects its own min/max), then repeat pool entries to fill
|
|
614
|
+
# row_count. validate_hints has already refused this on an FK/pk/unique/
|
|
615
|
+
# enum column, so there is no interaction with those branches to worry
|
|
616
|
+
# about. Handled before anything else so every other branch stays exactly
|
|
617
|
+
# what it was for a column with no `distinct` hint.
|
|
618
|
+
column_note = column.note or {}
|
|
619
|
+
distinct = column_note.get("distinct")
|
|
620
|
+
if distinct is not None:
|
|
621
|
+
pool_note = {key: value for key, value in column_note.items() if key != "distinct"}
|
|
622
|
+
pool = generate_column_values(
|
|
623
|
+
column=replace(column, note=pool_note or None),
|
|
624
|
+
row_count=distinct,
|
|
625
|
+
fk_series=None,
|
|
626
|
+
ensure_unique=False,
|
|
627
|
+
force_not_null=True,
|
|
628
|
+
table_name=table_name,
|
|
629
|
+
as_of=as_of,
|
|
630
|
+
time_profile=time_profile,
|
|
631
|
+
skew=skew,
|
|
632
|
+
)
|
|
633
|
+
values = random.choices(pool, k=row_count)
|
|
634
|
+
if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
|
|
635
|
+
_null_out(values, _null_fraction_for(column, row_count), column.default, row_count)
|
|
636
|
+
return values
|
|
637
|
+
|
|
537
638
|
if column.enum_values:
|
|
538
|
-
|
|
639
|
+
note = column.note or {}
|
|
640
|
+
weights = note.get("weights")
|
|
641
|
+
null_rate_hint = "null_rate" in note
|
|
642
|
+
if weights is None and not null_rate_hint:
|
|
643
|
+
return [random.choice(column.enum_values) for _ in range(row_count)]
|
|
644
|
+
|
|
645
|
+
if weights is not None:
|
|
646
|
+
# Values the hint doesn't mention default to weight 1, so naming
|
|
647
|
+
# only the ones that matter (`{"delivered": 20}`) doesn't silently
|
|
648
|
+
# drop the rest of the enum.
|
|
649
|
+
enum_weights = [float(weights.get(value, 1)) for value in column.enum_values]
|
|
650
|
+
values = random.choices(column.enum_values, weights=enum_weights, k=row_count)
|
|
651
|
+
else:
|
|
652
|
+
values = [random.choice(column.enum_values) for _ in range(row_count)]
|
|
653
|
+
|
|
654
|
+
if null_rate_hint and not force_not_null:
|
|
655
|
+
_null_out(values, note["null_rate"], column.default, row_count)
|
|
656
|
+
return values
|
|
539
657
|
|
|
540
658
|
dtype = column.data_type.lower()
|
|
541
659
|
base_type = dtype.split("(")[0].strip()
|
|
542
|
-
values
|
|
660
|
+
values = []
|
|
543
661
|
|
|
544
662
|
# Extract min/max from note if present
|
|
545
663
|
min_val = None
|
|
@@ -556,7 +674,34 @@ def generate_column_values(
|
|
|
556
674
|
# other branch here already respects `not null`/`pk` via the
|
|
557
675
|
# nullability pass below.
|
|
558
676
|
fk_values = fk_series.tolist()
|
|
559
|
-
|
|
677
|
+
effective_skew = column.note.get("skew") if column.note else None
|
|
678
|
+
if effective_skew is None:
|
|
679
|
+
effective_skew = skew
|
|
680
|
+
|
|
681
|
+
if effective_skew == 0.0:
|
|
682
|
+
# Unchanged from every release before skew existed.
|
|
683
|
+
values = [random.choice(fk_values) for _ in range(row_count)]
|
|
684
|
+
else:
|
|
685
|
+
# Distinct parents, in their original order, shuffled so *which*
|
|
686
|
+
# ones end up popular is randomized under the seed rather than
|
|
687
|
+
# always being the first ones inserted.
|
|
688
|
+
seen: set = set()
|
|
689
|
+
parents = []
|
|
690
|
+
for value in fk_values:
|
|
691
|
+
if value not in seen:
|
|
692
|
+
seen.add(value)
|
|
693
|
+
parents.append(value)
|
|
694
|
+
random.shuffle(parents)
|
|
695
|
+
|
|
696
|
+
# Geometric decay by rank: w_i = exp(-lam * i / n), lam = 8 *
|
|
697
|
+
# skew**2. At skew 0.8 (lam=5.12) the top 20% of parents hold
|
|
698
|
+
# roughly 65-75% of the children; at skew 1.0 (lam=8) they hold
|
|
699
|
+
# roughly 80%; at skew 0 every parent is equally likely, handled
|
|
700
|
+
# above. Both ranges are pinned in tests/test_shaping.py.
|
|
701
|
+
n = len(parents)
|
|
702
|
+
lam = 8 * effective_skew**2
|
|
703
|
+
weights = [math.exp(-lam * i / n) for i in range(n)]
|
|
704
|
+
values = random.choices(parents, weights=weights, k=row_count)
|
|
560
705
|
|
|
561
706
|
# -----------------------------------------------------
|
|
562
707
|
# UUIDs / hashes
|
|
@@ -620,26 +765,23 @@ def generate_column_values(
|
|
|
620
765
|
# Booleans
|
|
621
766
|
# -----------------------------------------------------
|
|
622
767
|
elif "boolean" in base_type or "bool" in base_type:
|
|
623
|
-
|
|
768
|
+
true_rate = column.note.get("true_rate") if column.note else None
|
|
769
|
+
if true_rate is None:
|
|
770
|
+
values = [random.choice([True, False]) for _ in range(row_count)]
|
|
771
|
+
else:
|
|
772
|
+
values = [random.random() < true_rate for _ in range(row_count)]
|
|
624
773
|
|
|
625
774
|
# -----------------------------------------------------
|
|
626
775
|
# Dates
|
|
627
776
|
# -----------------------------------------------------
|
|
628
777
|
elif "date" in base_type and "time" not in base_type:
|
|
629
|
-
|
|
630
|
-
# strings are resolved against `date.today()` inside Faker, which is
|
|
631
|
-
# precisely the hidden dependency on the wall clock `as_of` removes.
|
|
632
|
-
anchor = _anchor_date(as_of)
|
|
633
|
-
values = [
|
|
634
|
-
fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
|
|
635
|
-
for _ in range(row_count)
|
|
636
|
-
]
|
|
778
|
+
values = _generate_dates(row_count, as_of, _column_time_profile(column, time_profile))
|
|
637
779
|
|
|
638
780
|
elif "time" in base_type and "stamp" not in base_type:
|
|
639
781
|
values = [fake.time() for _ in range(row_count)]
|
|
640
782
|
|
|
641
783
|
elif any(key in base_type for key in ["timestamp", "datetime"]):
|
|
642
|
-
values =
|
|
784
|
+
values = _generate_timestamps(row_count, as_of, _column_time_profile(column, time_profile))
|
|
643
785
|
|
|
644
786
|
# -----------------------------------------------------
|
|
645
787
|
# Untyped / generic string columns: honour a type that names
|
|
@@ -674,11 +816,7 @@ def generate_column_values(
|
|
|
674
816
|
# Nullability
|
|
675
817
|
# -----------------------------------------------------
|
|
676
818
|
if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
|
|
677
|
-
|
|
678
|
-
sample_size = int(row_count * null_fraction)
|
|
679
|
-
if sample_size:
|
|
680
|
-
for idx in random.sample(range(row_count), k=sample_size):
|
|
681
|
-
values[idx] = column.default
|
|
819
|
+
_null_out(values, _null_fraction_for(column, row_count), column.default, row_count)
|
|
682
820
|
|
|
683
821
|
return values
|
|
684
822
|
|
|
@@ -686,6 +824,33 @@ def generate_column_values(
|
|
|
686
824
|
# ---------------------------------------------------------
|
|
687
825
|
# Internal helpers
|
|
688
826
|
# ---------------------------------------------------------
|
|
827
|
+
def _null_fraction_for(column: ColumnDef, row_count: int) -> float:
|
|
828
|
+
"""The fraction of `row_count` rows this column should turn null.
|
|
829
|
+
|
|
830
|
+
A `null_rate` hint replaces the default outright; with none, the same
|
|
831
|
+
formula that has applied since before hints existed -- up to a fifth of
|
|
832
|
+
the rows, tapering off for very small tables so a 3-row lookup table
|
|
833
|
+
doesn't lose a third of itself to nulls.
|
|
834
|
+
"""
|
|
835
|
+
note_null_rate = column.note.get("null_rate") if column.note else None
|
|
836
|
+
if note_null_rate is not None:
|
|
837
|
+
return note_null_rate
|
|
838
|
+
return max(0, min(0.2, 1 - (row_count / (row_count + 50))))
|
|
839
|
+
|
|
840
|
+
|
|
841
|
+
def _null_out(values: list, fraction: float, default: object, row_count: int) -> None:
|
|
842
|
+
"""Overwrite `fraction` of `values`, in place, with `default`.
|
|
843
|
+
|
|
844
|
+
A fixed count drawn once via `random.sample` rather than a per-row coin
|
|
845
|
+
flip, so the null count is exactly `round(row_count * fraction)` instead
|
|
846
|
+
of only approximately so.
|
|
847
|
+
"""
|
|
848
|
+
sample_size = int(row_count * fraction)
|
|
849
|
+
if sample_size:
|
|
850
|
+
for idx in random.sample(range(row_count), k=sample_size):
|
|
851
|
+
values[idx] = default
|
|
852
|
+
|
|
853
|
+
|
|
689
854
|
def _deduplicate(
|
|
690
855
|
values: list,
|
|
691
856
|
generator: Callable[[], object],
|