model2data 1.4.0__tar.gz → 1.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {model2data-1.4.0/model2data.egg-info → model2data-1.6.0}/PKG-INFO +1 -1
  2. {model2data-1.4.0 → model2data-1.6.0}/README.md +67 -0
  3. {model2data-1.4.0 → model2data-1.6.0}/model2data/cli.py +80 -10
  4. {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/core.py +52 -3
  5. {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/faker.py +184 -19
  6. model2data-1.6.0/model2data/generate/hints.py +229 -0
  7. model2data-1.6.0/model2data/generate/options.py +80 -0
  8. model2data-1.6.0/model2data/generate/timeline.py +454 -0
  9. {model2data-1.4.0 → model2data-1.6.0/model2data.egg-info}/PKG-INFO +1 -1
  10. {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/SOURCES.txt +8 -1
  11. {model2data-1.4.0 → model2data-1.6.0}/pyproject.toml +1 -1
  12. {model2data-1.4.0 → model2data-1.6.0}/tests/test_cli.py +22 -0
  13. model2data-1.6.0/tests/test_column_time_hints.py +241 -0
  14. model2data-1.6.0/tests/test_options.py +52 -0
  15. model2data-1.6.0/tests/test_shaping.py +493 -0
  16. model2data-1.6.0/tests/test_timeline.py +421 -0
  17. {model2data-1.4.0 → model2data-1.6.0}/LICENSE +0 -0
  18. {model2data-1.4.0 → model2data-1.6.0}/README_PYPI.md +0 -0
  19. {model2data-1.4.0 → model2data-1.6.0}/model2data/__init__.py +0 -0
  20. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/__init__.py +0 -0
  21. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/project.py +0 -0
  22. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  23. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  24. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  25. {model2data-1.4.0 → model2data-1.6.0}/model2data/dbt/tests.py +0 -0
  26. {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/__init__.py +0 -0
  27. {model2data-1.4.0 → model2data-1.6.0}/model2data/generate/relationships.py +0 -0
  28. {model2data-1.4.0 → model2data-1.6.0}/model2data/parse/__init__.py +0 -0
  29. {model2data-1.4.0 → model2data-1.6.0}/model2data/parse/dbml.py +0 -0
  30. {model2data-1.4.0 → model2data-1.6.0}/model2data/utils.py +0 -0
  31. {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/dependency_links.txt +0 -0
  32. {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/entry_points.txt +0 -0
  33. {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/requires.txt +0 -0
  34. {model2data-1.4.0 → model2data-1.6.0}/model2data.egg-info/top_level.txt +0 -0
  35. {model2data-1.4.0 → model2data-1.6.0}/setup.cfg +0 -0
  36. {model2data-1.4.0 → model2data-1.6.0}/tests/test_as_of_anchor.py +0 -0
  37. {model2data-1.4.0 → model2data-1.6.0}/tests/test_coverage_gaps.py +0 -0
  38. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbml_parser.py +0 -0
  39. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbml_parser_fuzz.py +0 -0
  40. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_integration.py +0 -0
  41. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_naming.py +0 -0
  42. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_project.py +0 -0
  43. {model2data-1.4.0 → model2data-1.6.0}/tests/test_dbt_tests.py +0 -0
  44. {model2data-1.4.0 → model2data-1.6.0}/tests/test_faker_name_inference.py +0 -0
  45. {model2data-1.4.0 → model2data-1.6.0}/tests/test_generation.py +0 -0
  46. {model2data-1.4.0 → model2data-1.6.0}/tests/test_release_stress.py +0 -0
  47. {model2data-1.4.0 → model2data-1.6.0}/tests/test_row_identity.py +0 -0
  48. {model2data-1.4.0 → model2data-1.6.0}/tests/test_table_seeds.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.4.0
3
+ Version: 1.6.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -165,6 +165,73 @@ it's a per-run setting, so a table can't end up holding one Belgian and one Amer
165
165
  model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
166
166
  ```
167
167
 
168
+ ### Shape when things happen
169
+
170
+ By default, every date and timestamp is drawn uniformly across its window. `--business-hours`,
171
+ `--growth`, and `--seasonality` shape that instead — weekdays and working hours, a trend across
172
+ the window, and an annual cycle peaking in Q4:
173
+
174
+ ```bash
175
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
176
+ --business-hours --growth 0.5 --seasonality 0.3
177
+ ```
178
+
179
+ Within a row, created/updated/deleted-style columns are ordered automatically — `updated_at` never
180
+ lands before its own `created_at` — under any profile, uniform included. A column whose name
181
+ doesn't say what it depends on can say so explicitly with an `after` note:
182
+
183
+ ```dbml
184
+ shipped_at timestamp [note: '{"after": "ordered_at"}']
185
+ ```
186
+
187
+ The flags above shape every date and timestamp column the same way, run-wide. `business_hours`,
188
+ `growth`, and `seasonality` column note hints override that for one column at a time — the whole
189
+ point being a run can be uniform everywhere except the one column that needs shaping, or shaped
190
+ everywhere except the one column that shouldn't be:
191
+
192
+ ```dbml
193
+ Table orders {
194
+ id int [pk]
195
+ created_at timestamp [note: '{"business_hours": true, "growth": 0.4}']
196
+ refunded_at timestamp [note: '{"growth": 0}']
197
+ }
198
+ ```
199
+
200
+ Here `created_at` gets business hours and growth even on an otherwise-uniform run, while
201
+ `refunded_at` stays flat even under `--growth 0.5` — each hint only replaces the fields it names,
202
+ so a partial hint like `{"growth": 0}` leaves that column's `business_hours`/`seasonality` at
203
+ whatever the run-level flags set.
204
+
205
+ ### Shape how the data is spread
206
+
207
+ By default every parent row is equally likely to be picked for a child row, and every column
208
+ gets the same generic null rate, value spread, and true/false split. `--skew` changes the first
209
+ part: `0.0` is that uniform default, `1.0` means a handful of parents hold most of the children —
210
+ "a fifth of the customers place most of the orders":
211
+
212
+ ```bash
213
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --skew 0.8
214
+ ```
215
+
216
+ Column note hints shape the rest, per column:
217
+
218
+ ```dbml
219
+ Table orders {
220
+ id int [pk]
221
+ customer_id int [ref: > customers.id, note: '{"skew": 0.9}']
222
+ status order_status [note: '{"weights": {"delivered": 20, "cancelled": 2}}']
223
+ is_paid boolean [note: '{"true_rate": 0.9}']
224
+ discount_code varchar [note: '{"null_rate": 0.8}']
225
+ shipping_city varchar [note: '{"distinct": 12}']
226
+ }
227
+ ```
228
+
229
+ `skew` on a foreign key overrides `--skew` for just that column. `weights` biases an enum column
230
+ toward the values named (unnamed values still appear, at weight 1). `true_rate` is the fraction of
231
+ non-null rows a boolean column comes back `true`. `null_rate` replaces the column's default null
232
+ fraction outright. `distinct` draws the column's values from a fixed-size pool instead of a fresh
233
+ value per row — a `shipping_city` most warehouses only ever see a handful of.
234
+
168
235
  Run dbt to load, transform, and test the data:
169
236
 
170
237
  ```bash
@@ -24,6 +24,7 @@ from model2data.generate.faker import (
24
24
  get_unmapped_columns,
25
25
  reset_stats,
26
26
  )
27
+ from model2data.generate.options import TimeProfile
27
28
  from model2data.parse.dbml import get_parse_warnings, parse_dbml
28
29
  from model2data.utils import normalize_identifier
29
30
 
@@ -186,6 +187,43 @@ def main(
186
187
  "Pin it and a --seed run reproduces on any later day, not just the day it first ran."
187
188
  ),
188
189
  ),
190
+ business_hours: bool = typer.Option(
191
+ False,
192
+ "--business-hours",
193
+ help=(
194
+ "Weight generated timestamps toward weekdays and working hours,\n"
195
+ "instead of spreading them evenly over every hour of every day."
196
+ ),
197
+ ),
198
+ growth: float = typer.Option(
199
+ 0.0,
200
+ "--growth",
201
+ min=-1.0,
202
+ help=(
203
+ "Relative change in activity across the generated window: 0.5 means the end\n"
204
+ "is half again as busy as the start, -0.3 means it tailed off. Default: flat."
205
+ ),
206
+ ),
207
+ seasonality: float = typer.Option(
208
+ 0.0,
209
+ "--seasonality",
210
+ min=0.0,
211
+ max=1.0,
212
+ help=(
213
+ "Strength of an annual cycle in generated timestamps, 0 (none) to 1,\n"
214
+ "peaking in the fourth quarter."
215
+ ),
216
+ ),
217
+ skew: float = typer.Option(
218
+ 0.0,
219
+ "--skew",
220
+ min=0.0,
221
+ max=1.0,
222
+ help=(
223
+ "How unevenly child rows are spread over their parents: 0 (every parent equally\n"
224
+ "likely, the default) to 1 (a few parents hold most of the children)."
225
+ ),
226
+ ),
189
227
  locale: Optional[str] = typer.Option(
190
228
  None,
191
229
  "--locale",
@@ -251,6 +289,28 @@ def main(
251
289
  if as_of is not None:
252
290
  typer.echo(f"📅 Anchoring generated dates on: {as_of.date()}")
253
291
 
292
+ # Same guard again for the shaping options: a direct call leaves them as
293
+ # `OptionInfo` objects, which means "not supplied", i.e. the uniform draw.
294
+ time_profile = TimeProfile(
295
+ business_hours=business_hours if isinstance(business_hours, bool) else False,
296
+ growth=growth if isinstance(growth, (int, float)) else 0.0,
297
+ seasonality=seasonality if isinstance(seasonality, (int, float)) else 0.0,
298
+ )
299
+ skew = skew if isinstance(skew, (int, float)) else 0.0
300
+ if not time_profile.is_uniform:
301
+ shaped = [
302
+ label
303
+ for label, active in (
304
+ ("business hours", time_profile.business_hours),
305
+ (f"growth {time_profile.growth:+.0%}", time_profile.growth != 0.0),
306
+ (f"seasonality {time_profile.seasonality:.0%}", time_profile.seasonality != 0.0),
307
+ )
308
+ if active
309
+ ]
310
+ typer.echo(f"🕒 Shaping timestamps: {', '.join(shaped)}")
311
+ if skew:
312
+ typer.echo(f"📈 Skewing child rows over their parents: {skew:.0%}")
313
+
254
314
  # -------------------------
255
315
  # Parse DBML (names untouched)
256
316
  # -------------------------
@@ -296,16 +356,26 @@ def main(
296
356
  # -------------------------
297
357
  typer.echo("🧮 Generating synthetic datasets from DBML definitions...")
298
358
  reset_stats()
299
- generated_tables = generate_data_from_dbml(
300
- tables=tables,
301
- refs=refs,
302
- base_rows=rows,
303
- seed=seed,
304
- row_overrides=row_overrides,
305
- locale=locale,
306
- as_of=as_of,
307
- table_seeds=table_seeds,
308
- )
359
+ try:
360
+ generated_tables = generate_data_from_dbml(
361
+ tables=tables,
362
+ refs=refs,
363
+ base_rows=rows,
364
+ seed=seed,
365
+ row_overrides=row_overrides,
366
+ locale=locale,
367
+ as_of=as_of,
368
+ table_seeds=table_seeds,
369
+ time_profile=time_profile,
370
+ skew=skew,
371
+ )
372
+ except ValueError as exc:
373
+ # A bad column-note hint (an unknown enum value in `weights`, `distinct`
374
+ # on a foreign key...) is a mistake in the schema, not a bug -- it
375
+ # deserves the same one-line, non-traceback treatment as every other
376
+ # validation error this command already reports.
377
+ typer.echo(f"❌ {exc}")
378
+ raise typer.Exit(1) from None
309
379
 
310
380
  # -------------------------
311
381
  # Write dbt seeds (normalized names)
@@ -17,10 +17,13 @@ from model2data.generate.faker import (
17
17
  reset_row_pools,
18
18
  set_locale,
19
19
  )
20
+ from model2data.generate.hints import validate_hints
21
+ from model2data.generate.options import UNIFORM, TimeProfile, validate_skew
20
22
  from model2data.generate.relationships import (
21
23
  build_fk_lookup,
22
24
  classify_refs,
23
25
  )
26
+ from model2data.generate.timeline import order_row_times
24
27
  from model2data.parse.dbml import TableDef
25
28
 
26
29
  # Tables the most recent generate_data_from_dbml() call found stuck in an
@@ -72,6 +75,8 @@ def generate_data_from_dbml(
72
75
  locale: Optional[str] = None,
73
76
  as_of: AsOf = None,
74
77
  table_seeds: Optional[Mapping[str, int]] = None,
78
+ time_profile: Optional[TimeProfile] = None,
79
+ skew: float = 0.0,
75
80
  ) -> dict[str, pd.DataFrame]:
76
81
  """
77
82
  Generate synthetic datasets from parsed DBML definitions.
@@ -107,10 +112,23 @@ def generate_data_from_dbml(
107
112
  `seed` to work off -- with none, every table is already different on every
108
113
  run -- and unknown table names are an error rather than a silent no-op.
109
114
 
115
+ `time_profile` shapes *when* generated timestamps and dates fall: toward
116
+ business hours and weekdays, along a growth trend, with a seasonal peak.
117
+ See `TimeProfile`. None is the uniform profile, which draws every second of
118
+ the window with equal probability the way earlier releases did.
119
+
120
+ `skew` is how unevenly a child table's rows are spread over its parents,
121
+ from `0.0` (every parent equally likely, the earlier behaviour) to `1.0` (a
122
+ few parents hold most of the children). A column can override it with a
123
+ `{"skew": ...}` hint in its note.
124
+
110
125
  This function is deterministic if a seed is provided (and, with `as_of`,
111
126
  on any day). It performs no filesystem I/O and returns pandas DataFrames.
112
127
  """
113
128
  _validate_table_seeds(tables, table_seeds, seed)
129
+ validate_hints(tables, refs)
130
+ profile = time_profile or UNIFORM
131
+ skew = validate_skew(skew)
114
132
 
115
133
  # Locale first, then the seed: switching locale builds a new Faker, and the
116
134
  # seed has to be the last word on the generator that actually runs.
@@ -210,14 +228,35 @@ def generate_data_from_dbml(
210
228
  force_not_null=column.name in composite_pk_columns,
211
229
  table_name=table_name,
212
230
  as_of=as_of,
231
+ time_profile=profile,
232
+ skew=skew,
213
233
  )
214
234
 
215
235
  df = pd.DataFrame(data)
236
+ # Ordering runs before FK resolution/dedup so a self-ref repair or a
237
+ # composite-key retry regenerates a temporal column's value into a
238
+ # frame that already respects created/updated/closed ordering, rather
239
+ # than one where only the untouched columns do.
240
+ df = order_row_times(df, table_def, as_of=as_of)
216
241
  df = _resolve_self_referencing_fks(
217
- df, table_def, table_name, fk_lookup, row_count, as_of=as_of
242
+ df,
243
+ table_def,
244
+ table_name,
245
+ fk_lookup,
246
+ row_count,
247
+ as_of=as_of,
248
+ time_profile=profile,
249
+ skew=skew,
218
250
  )
219
251
  df = _deduplicate_composite_keys(
220
- df, table_def, table_name, fk_lookup, generated, as_of=as_of
252
+ df,
253
+ table_def,
254
+ table_name,
255
+ fk_lookup,
256
+ generated,
257
+ as_of=as_of,
258
+ time_profile=profile,
259
+ skew=skew,
221
260
  )
222
261
 
223
262
  # -----------------------------------------------------
@@ -299,6 +338,8 @@ def _deduplicate_composite_keys(
299
338
  generated: dict[str, pd.DataFrame],
300
339
  max_attempts: int = 20,
301
340
  as_of: AsOf = None,
341
+ time_profile: Optional[TimeProfile] = None,
342
+ skew: float = 0.0,
302
343
  ) -> pd.DataFrame:
303
344
  """
304
345
  Regenerate colliding rows for any pk/unique composite key declared via an
@@ -350,7 +391,11 @@ def _deduplicate_composite_keys(
350
391
  df.at[idx, col_name] = random.choice(fk_pools[col_name])
351
392
  else:
352
393
  df.at[idx, col_name] = generate_column_values(
353
- col_def, row_count=1, as_of=as_of
394
+ col_def,
395
+ row_count=1,
396
+ as_of=as_of,
397
+ time_profile=time_profile,
398
+ skew=skew,
354
399
  )[0]
355
400
  combo = tuple(df.at[idx, c] for c in key_columns)
356
401
  attempts += 1
@@ -377,6 +422,8 @@ def _resolve_self_referencing_fks(
377
422
  fk_lookup: dict[tuple[str, str], tuple[str, str]],
378
423
  row_count: int,
379
424
  as_of: AsOf = None,
425
+ time_profile: Optional[TimeProfile] = None,
426
+ skew: float = 0.0,
380
427
  ) -> pd.DataFrame:
381
428
  """
382
429
  Re-generate any FK column that references its own table (e.g. a
@@ -415,6 +462,8 @@ def _resolve_self_referencing_fks(
415
462
  force_not_null=column.name in composite_pk_columns,
416
463
  table_name=table_name,
417
464
  as_of=as_of,
465
+ time_profile=time_profile,
466
+ skew=skew,
418
467
  )
419
468
 
420
469
  return df
@@ -1,16 +1,19 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import math
3
4
  import random
4
5
  import re
5
6
  import unicodedata
6
7
  import uuid
7
- from dataclasses import dataclass
8
+ from dataclasses import dataclass, replace
8
9
  from datetime import date, datetime, timedelta
9
10
  from typing import Callable, Optional, Union
10
11
 
11
12
  import pandas as pd
12
13
  from faker import Faker
13
14
 
15
+ from model2data.generate.options import UNIFORM, TimeProfile
16
+ from model2data.generate.timeline import weighted_dates, weighted_timestamps
14
17
  from model2data.parse.dbml import ColumnDef
15
18
 
16
19
  # ---------------------------------------------------------
@@ -502,6 +505,68 @@ def _infer_by_type(base_type: str) -> Optional[_Provider]:
502
505
  return lambda: fake.format(base_type)
503
506
 
504
507
 
508
+ def _column_time_profile(
509
+ column: ColumnDef, time_profile: Optional[TimeProfile]
510
+ ) -> Optional[TimeProfile]:
511
+ """The run-level `time_profile` with this column's note hints applied on top.
512
+
513
+ Mirrors the `skew` override on the FK branch, one level up: a note doesn't
514
+ replace the run's profile, it patches only the fields it names (via
515
+ `dataclasses.replace`), so `{"growth": 0}` flattens one column of a
516
+ growing run while `business_hours`/`seasonality` stay exactly what the
517
+ run set. `validate_hints` has already confirmed the column is a date or
518
+ timestamp column and that each hint present is well-typed, so this does
519
+ no validation of its own -- a column with no hint gets `time_profile`
520
+ back unchanged, including a bare `None`, so the uniform path stays
521
+ untouched.
522
+ """
523
+ note = column.note or {}
524
+ overrides = {
525
+ key: note[key] for key in ("business_hours", "growth", "seasonality") if key in note
526
+ }
527
+ if not overrides:
528
+ return time_profile
529
+ base = time_profile if time_profile is not None else UNIFORM
530
+ return replace(base, **overrides)
531
+
532
+
533
+ def _generate_dates(row_count: int, as_of: AsOf, time_profile: Optional[TimeProfile]) -> list:
534
+ """The date branch's values: shaped by `time_profile` when it isn't uniform.
535
+
536
+ A bare `-> list` return, matching `generate_column_values` itself: `values`
537
+ there is reassigned by every branch of one big if/elif chain (int, float,
538
+ uuid, this one...) and then written into by the nullability pass below with
539
+ an arbitrary `column.default`, so none of those branches can commit to a
540
+ concrete element type without the type checker flagging the later write as
541
+ unsound. Keeping `weighted_dates`'s real `list[date]` return type contained
542
+ to this one call, instead of leaking it into `values`, is what lets
543
+ `weighted_dates` itself stay properly typed for its own callers and tests.
544
+ """
545
+ anchor = _anchor_date(as_of)
546
+ if time_profile is not None and not time_profile.is_uniform:
547
+ return weighted_dates(
548
+ row_count, time_profile, start=_years_before(anchor, _DATE_WINDOW_YEARS), end=anchor
549
+ )
550
+ # Explicit endpoints rather than Faker's "-2y"/"today" shorthand: those
551
+ # strings are resolved against `date.today()` inside Faker, which is
552
+ # precisely the hidden dependency on the wall clock `as_of` removes.
553
+ return [
554
+ fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
555
+ for _ in range(row_count)
556
+ ]
557
+
558
+
559
+ def _generate_timestamps(row_count: int, as_of: AsOf, time_profile: Optional[TimeProfile]) -> list:
560
+ """The timestamp branch's values: shaped by `time_profile` when it isn't uniform.
561
+
562
+ See `_generate_dates` for why this returns a bare `list` rather than
563
+ `weighted_timestamps`'s own `list[str]`.
564
+ """
565
+ if time_profile is not None and not time_profile.is_uniform:
566
+ return weighted_timestamps(row_count, time_profile, anchor=_anchor_date(as_of))
567
+ return [_random_datetime(as_of=as_of).isoformat(sep=" ") for _ in range(row_count)]
568
+
569
+
505
570
  # ---------------------------------------------------------
506
571
  # Public API
507
572
  # ---------------------------------------------------------
@@ -513,11 +578,19 @@ def generate_column_values(
513
578
  force_not_null: bool = False,
514
579
  table_name: Optional[str] = None,
515
580
  as_of: AsOf = None,
581
+ time_profile: Optional[TimeProfile] = None,
582
+ skew: float = 0.0,
516
583
  ) -> list:
517
584
  """
518
585
  Generate synthetic values for a single column.
519
586
  Respects FKs, uniqueness, and optional min/max hints in column notes.
520
587
 
588
+ `time_profile` is read by the date and timestamp branches, `skew` by the
589
+ foreign-key branch; both default to the uniform behaviour of earlier
590
+ releases. They are accepted here rather than in a separate pass so that
591
+ every path that draws a value -- the main pass, the self-referencing FK
592
+ repair, the composite-key retry -- draws it the same way.
593
+
521
594
  `as_of` is the date every generated date and timestamp is placed relative
522
595
  to, defaulting to today. Pass it to make a seeded run reproduce on any
523
596
  later day rather than only on the day it first ran.
@@ -534,12 +607,57 @@ def generate_column_values(
534
607
  # (an `id` on each of two tables) stay distinguishable in the report.
535
608
  unique_label = f"{table_name}.{column.name}" if table_name else column.name
536
609
 
610
+ # A `distinct` hint means "draw from a small pool", which is orthogonal to
611
+ # every type-specific branch below: generate the pool through this same
612
+ # function (so a pooled `city` still reads the row-identity pool, a pooled
613
+ # `int` still respects its own min/max), then repeat pool entries to fill
614
+ # row_count. validate_hints has already refused this on an FK/pk/unique/
615
+ # enum column, so there is no interaction with those branches to worry
616
+ # about. Handled before anything else so every other branch stays exactly
617
+ # what it was for a column with no `distinct` hint.
618
+ column_note = column.note or {}
619
+ distinct = column_note.get("distinct")
620
+ if distinct is not None:
621
+ pool_note = {key: value for key, value in column_note.items() if key != "distinct"}
622
+ pool = generate_column_values(
623
+ column=replace(column, note=pool_note or None),
624
+ row_count=distinct,
625
+ fk_series=None,
626
+ ensure_unique=False,
627
+ force_not_null=True,
628
+ table_name=table_name,
629
+ as_of=as_of,
630
+ time_profile=time_profile,
631
+ skew=skew,
632
+ )
633
+ values = random.choices(pool, k=row_count)
634
+ if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
635
+ _null_out(values, _null_fraction_for(column, row_count), column.default, row_count)
636
+ return values
637
+
537
638
  if column.enum_values:
538
- return [random.choice(column.enum_values) for _ in range(row_count)]
639
+ note = column.note or {}
640
+ weights = note.get("weights")
641
+ null_rate_hint = "null_rate" in note
642
+ if weights is None and not null_rate_hint:
643
+ return [random.choice(column.enum_values) for _ in range(row_count)]
644
+
645
+ if weights is not None:
646
+ # Values the hint doesn't mention default to weight 1, so naming
647
+ # only the ones that matter (`{"delivered": 20}`) doesn't silently
648
+ # drop the rest of the enum.
649
+ enum_weights = [float(weights.get(value, 1)) for value in column.enum_values]
650
+ values = random.choices(column.enum_values, weights=enum_weights, k=row_count)
651
+ else:
652
+ values = [random.choice(column.enum_values) for _ in range(row_count)]
653
+
654
+ if null_rate_hint and not force_not_null:
655
+ _null_out(values, note["null_rate"], column.default, row_count)
656
+ return values
539
657
 
540
658
  dtype = column.data_type.lower()
541
659
  base_type = dtype.split("(")[0].strip()
542
- values: list = []
660
+ values = []
543
661
 
544
662
  # Extract min/max from note if present
545
663
  min_val = None
@@ -556,7 +674,34 @@ def generate_column_values(
556
674
  # other branch here already respects `not null`/`pk` via the
557
675
  # nullability pass below.
558
676
  fk_values = fk_series.tolist()
559
- values = [random.choice(fk_values) for _ in range(row_count)]
677
+ effective_skew = column.note.get("skew") if column.note else None
678
+ if effective_skew is None:
679
+ effective_skew = skew
680
+
681
+ if effective_skew == 0.0:
682
+ # Unchanged from every release before skew existed.
683
+ values = [random.choice(fk_values) for _ in range(row_count)]
684
+ else:
685
+ # Distinct parents, in their original order, shuffled so *which*
686
+ # ones end up popular is randomized under the seed rather than
687
+ # always being the first ones inserted.
688
+ seen: set = set()
689
+ parents = []
690
+ for value in fk_values:
691
+ if value not in seen:
692
+ seen.add(value)
693
+ parents.append(value)
694
+ random.shuffle(parents)
695
+
696
+ # Geometric decay by rank: w_i = exp(-lam * i / n), lam = 8 *
697
+ # skew**2. At skew 0.8 (lam=5.12) the top 20% of parents hold
698
+ # roughly 65-75% of the children; at skew 1.0 (lam=8) they hold
699
+ # roughly 80%; at skew 0 every parent is equally likely, handled
700
+ # above. Both ranges are pinned in tests/test_shaping.py.
701
+ n = len(parents)
702
+ lam = 8 * effective_skew**2
703
+ weights = [math.exp(-lam * i / n) for i in range(n)]
704
+ values = random.choices(parents, weights=weights, k=row_count)
560
705
 
561
706
  # -----------------------------------------------------
562
707
  # UUIDs / hashes
@@ -620,26 +765,23 @@ def generate_column_values(
620
765
  # Booleans
621
766
  # -----------------------------------------------------
622
767
  elif "boolean" in base_type or "bool" in base_type:
623
- values = [random.choice([True, False]) for _ in range(row_count)]
768
+ true_rate = column.note.get("true_rate") if column.note else None
769
+ if true_rate is None:
770
+ values = [random.choice([True, False]) for _ in range(row_count)]
771
+ else:
772
+ values = [random.random() < true_rate for _ in range(row_count)]
624
773
 
625
774
  # -----------------------------------------------------
626
775
  # Dates
627
776
  # -----------------------------------------------------
628
777
  elif "date" in base_type and "time" not in base_type:
629
- # Explicit endpoints rather than Faker's "-2y"/"today" shorthand: those
630
- # strings are resolved against `date.today()` inside Faker, which is
631
- # precisely the hidden dependency on the wall clock `as_of` removes.
632
- anchor = _anchor_date(as_of)
633
- values = [
634
- fake.date_between(start_date=_years_before(anchor, _DATE_WINDOW_YEARS), end_date=anchor)
635
- for _ in range(row_count)
636
- ]
778
+ values = _generate_dates(row_count, as_of, _column_time_profile(column, time_profile))
637
779
 
638
780
  elif "time" in base_type and "stamp" not in base_type:
639
781
  values = [fake.time() for _ in range(row_count)]
640
782
 
641
783
  elif any(key in base_type for key in ["timestamp", "datetime"]):
642
- values = [_random_datetime(as_of=as_of).isoformat(sep=" ") for _ in range(row_count)]
784
+ values = _generate_timestamps(row_count, as_of, _column_time_profile(column, time_profile))
643
785
 
644
786
  # -----------------------------------------------------
645
787
  # Untyped / generic string columns: honour a type that names
@@ -674,11 +816,7 @@ def generate_column_values(
674
816
  # Nullability
675
817
  # -----------------------------------------------------
676
818
  if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
677
- null_fraction = max(0, min(0.2, 1 - (row_count / (row_count + 50))))
678
- sample_size = int(row_count * null_fraction)
679
- if sample_size:
680
- for idx in random.sample(range(row_count), k=sample_size):
681
- values[idx] = column.default
819
+ _null_out(values, _null_fraction_for(column, row_count), column.default, row_count)
682
820
 
683
821
  return values
684
822
 
@@ -686,6 +824,33 @@ def generate_column_values(
686
824
  # ---------------------------------------------------------
687
825
  # Internal helpers
688
826
  # ---------------------------------------------------------
827
+ def _null_fraction_for(column: ColumnDef, row_count: int) -> float:
828
+ """The fraction of `row_count` rows this column should turn null.
829
+
830
+ A `null_rate` hint replaces the default outright; with none, the same
831
+ formula that has applied since before hints existed -- up to a fifth of
832
+ the rows, tapering off for very small tables so a 3-row lookup table
833
+ doesn't lose a third of itself to nulls.
834
+ """
835
+ note_null_rate = column.note.get("null_rate") if column.note else None
836
+ if note_null_rate is not None:
837
+ return note_null_rate
838
+ return max(0, min(0.2, 1 - (row_count / (row_count + 50))))
839
+
840
+
841
+ def _null_out(values: list, fraction: float, default: object, row_count: int) -> None:
842
+ """Overwrite `fraction` of `values`, in place, with `default`.
843
+
844
+ A fixed count drawn once via `random.sample` rather than a per-row coin
845
+ flip, so the null count is exactly `round(row_count * fraction)` instead
846
+ of only approximately so.
847
+ """
848
+ sample_size = int(row_count * fraction)
849
+ if sample_size:
850
+ for idx in random.sample(range(row_count), k=sample_size):
851
+ values[idx] = default
852
+
853
+
689
854
  def _deduplicate(
690
855
  values: list,
691
856
  generator: Callable[[], object],