model2data 1.2.0__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. {model2data-1.2.0/model2data.egg-info → model2data-1.3.0}/PKG-INFO +1 -1
  2. model2data-1.3.0/model2data/__init__.py +12 -0
  3. {model2data-1.2.0 → model2data-1.3.0}/model2data/cli.py +11 -0
  4. {model2data-1.2.0 → model2data-1.3.0}/model2data/generate/core.py +18 -2
  5. {model2data-1.2.0 → model2data-1.3.0}/model2data/generate/faker.py +325 -21
  6. {model2data-1.2.0 → model2data-1.3.0/model2data.egg-info}/PKG-INFO +1 -1
  7. {model2data-1.2.0 → model2data-1.3.0}/model2data.egg-info/SOURCES.txt +2 -1
  8. {model2data-1.2.0 → model2data-1.3.0}/pyproject.toml +1 -1
  9. model2data-1.3.0/tests/test_row_identity.py +372 -0
  10. model2data-1.2.0/model2data/__init__.py +0 -3
  11. {model2data-1.2.0 → model2data-1.3.0}/LICENSE +0 -0
  12. {model2data-1.2.0 → model2data-1.3.0}/README.md +0 -0
  13. {model2data-1.2.0 → model2data-1.3.0}/README_PYPI.md +0 -0
  14. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/__init__.py +0 -0
  15. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/project.py +0 -0
  16. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  17. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  18. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  19. {model2data-1.2.0 → model2data-1.3.0}/model2data/dbt/tests.py +0 -0
  20. {model2data-1.2.0 → model2data-1.3.0}/model2data/generate/__init__.py +0 -0
  21. {model2data-1.2.0 → model2data-1.3.0}/model2data/generate/relationships.py +0 -0
  22. {model2data-1.2.0 → model2data-1.3.0}/model2data/parse/__init__.py +0 -0
  23. {model2data-1.2.0 → model2data-1.3.0}/model2data/parse/dbml.py +0 -0
  24. {model2data-1.2.0 → model2data-1.3.0}/model2data/utils.py +0 -0
  25. {model2data-1.2.0 → model2data-1.3.0}/model2data.egg-info/dependency_links.txt +0 -0
  26. {model2data-1.2.0 → model2data-1.3.0}/model2data.egg-info/entry_points.txt +0 -0
  27. {model2data-1.2.0 → model2data-1.3.0}/model2data.egg-info/requires.txt +0 -0
  28. {model2data-1.2.0 → model2data-1.3.0}/model2data.egg-info/top_level.txt +0 -0
  29. {model2data-1.2.0 → model2data-1.3.0}/setup.cfg +0 -0
  30. {model2data-1.2.0 → model2data-1.3.0}/tests/test_cli.py +0 -0
  31. {model2data-1.2.0 → model2data-1.3.0}/tests/test_coverage_gaps.py +0 -0
  32. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbml_parser.py +0 -0
  33. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbml_parser_fuzz.py +0 -0
  34. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbt_integration.py +0 -0
  35. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbt_naming.py +0 -0
  36. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbt_project.py +0 -0
  37. {model2data-1.2.0 → model2data-1.3.0}/tests/test_dbt_tests.py +0 -0
  38. {model2data-1.2.0 → model2data-1.3.0}/tests/test_faker_name_inference.py +0 -0
  39. {model2data-1.2.0 → model2data-1.3.0}/tests/test_generation.py +0 -0
  40. {model2data-1.2.0 → model2data-1.3.0}/tests/test_release_stress.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.2.0
3
+ Version: 1.3.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -0,0 +1,12 @@
1
+ """Model2Data: Generate analytics-ready datasets from DBML models."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ # Read the installed distribution's version rather than repeating it here.
7
+ # The literal that used to live in this file said 0.1.1 for every release up
8
+ # to and including 1.2.0: nothing reads `__version__`, so nothing caught it
9
+ # drifting. Deriving it means it cannot drift again.
10
+ __version__ = version("model2data")
11
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
12
+ __version__ = "0.0.0.dev0"
@@ -18,6 +18,7 @@ from model2data.generate.core import (
18
18
  get_unresolved_composite_keys,
19
19
  )
20
20
  from model2data.generate.faker import (
21
+ DEFAULT_LOCALE,
21
22
  get_duplicate_unique_columns,
22
23
  get_unmapped_columns,
23
24
  reset_stats,
@@ -125,6 +126,15 @@ def main(
125
126
  "Using the same seed will always produce identical datasets."
126
127
  ),
127
128
  ),
129
+ locale: Optional[str] = typer.Option(
130
+ None,
131
+ "--locale",
132
+ help=(
133
+ f"Faker locale for generated people and addresses "
134
+ f"(default: {DEFAULT_LOCALE}).\n"
135
+ "Examples: en_GB, nl_BE, fr_FR, de_DE."
136
+ ),
137
+ ),
128
138
  name: Optional[str] = typer.Option(
129
139
  None,
130
140
  "--name",
@@ -214,6 +224,7 @@ def main(
214
224
  base_rows=rows,
215
225
  seed=seed,
216
226
  row_overrides=row_overrides,
227
+ locale=locale,
217
228
  )
218
229
 
219
230
  # -------------------------
@@ -10,7 +10,10 @@ from faker import Faker
10
10
 
11
11
  from model2data.generate.faker import (
12
12
  generate_column_values,
13
+ release_row_pools,
13
14
  reset_duplicate_unique_columns,
15
+ reset_row_pools,
16
+ set_locale,
14
17
  )
15
18
  from model2data.generate.relationships import (
16
19
  build_fk_lookup,
@@ -18,8 +21,6 @@ from model2data.generate.relationships import (
18
21
  )
19
22
  from model2data.parse.dbml import TableDef
20
23
 
21
- fake = Faker()
22
-
23
24
  # Tables the most recent generate_data_from_dbml() call found stuck in an
24
25
  # unresolved FK cycle (never reached indegree 0 during the topological
25
26
  # sort). Exposed out-of-band, mirroring generate.faker's
@@ -66,6 +67,7 @@ def generate_data_from_dbml(
66
67
  base_rows: int = 100,
67
68
  seed: Optional[int] = None,
68
69
  row_overrides: Optional[Mapping[str, int]] = None,
70
+ locale: Optional[str] = None,
69
71
  ) -> dict[str, pd.DataFrame]:
70
72
  """
71
73
  Generate synthetic datasets from parsed DBML definitions.
@@ -78,9 +80,19 @@ def generate_data_from_dbml(
78
80
  present in `row_overrides` fall back to `base_rows`; unknown names are
79
81
  ignored.
80
82
 
83
+ `locale` picks the Faker locale every generated person and address is drawn
84
+ from -- `"nl_BE"`, `"fr_FR"`, `"en_GB"` -- defaulting to `DEFAULT_LOCALE`.
85
+ It is a per-run setting rather than a per-column one on purpose: a table
86
+ holding one Belgian and one American address is the incoherence the row
87
+ pools exist to remove.
88
+
81
89
  This function is deterministic if a seed is provided.
82
90
  It performs no filesystem I/O and returns pandas DataFrames.
83
91
  """
92
+ # Locale first, then the seed: switching locale builds a new Faker, and the
93
+ # seed has to be the last word on the generator that actually runs.
94
+ set_locale(locale)
95
+
84
96
  if seed is not None:
85
97
  random.seed(seed)
86
98
  Faker.seed(seed)
@@ -88,6 +100,7 @@ def generate_data_from_dbml(
88
100
  reset_cycle_state()
89
101
  reset_dedup_state()
90
102
  reset_duplicate_unique_columns()
103
+ reset_row_pools()
91
104
 
92
105
  # ---------------------------------------------------------
93
106
  # Classify references
@@ -193,6 +206,9 @@ def generate_data_from_dbml(
193
206
 
194
207
  df = _coerce_integer_dtypes(df, table_def)
195
208
  generated[table_name] = df
209
+ # This table is finished: nothing will read its people or addresses
210
+ # again, and on a million-row table they are worth tens of megabytes.
211
+ release_row_pools(table_name)
196
212
 
197
213
  return generated
198
214
 
@@ -3,15 +3,258 @@ from __future__ import annotations
3
3
  import random
4
4
  import re
5
5
  import uuid
6
+ from dataclasses import dataclass
6
7
  from datetime import datetime, timedelta
7
- from typing import Callable, Optional
8
+ from typing import Callable, Optional, Union
8
9
 
9
10
  import pandas as pd
10
11
  from faker import Faker
11
12
 
12
13
  from model2data.parse.dbml import ColumnDef
13
14
 
14
- fake = Faker()
15
+ # ---------------------------------------------------------
16
+ # Locale
17
+ # ---------------------------------------------------------
18
+ # The locale every generated person and address comes from until a caller says
19
+ # otherwise. Named rather than left implicit because locale is on its way to
20
+ # being a first-class option in the CLI and the studio, and this is the seam it
21
+ # plugs into: one instance, swapped in one place.
22
+ DEFAULT_LOCALE = "en_US"
23
+
24
+ _locale = DEFAULT_LOCALE
25
+ fake = Faker(DEFAULT_LOCALE)
26
+
27
+
28
+ def current_locale() -> str:
29
+ """The locale generation is currently drawing from."""
30
+ return _locale
31
+
32
+
33
+ def set_locale(locale: Optional[str]) -> None:
34
+ """Point generation at a locale, or back at the default when given None.
35
+
36
+ Rebinding the module-level `fake` reaches every provider here, because each
37
+ one looks the name up when it runs rather than capturing it. The row pools
38
+ are dropped on the way through: a Belgian address sitting next to an
39
+ American one in the same table is precisely the incoherence they exist to
40
+ prevent.
41
+ """
42
+ global fake, _locale
43
+ target = locale or DEFAULT_LOCALE
44
+ if target == _locale:
45
+ return
46
+ try:
47
+ fake = Faker(target)
48
+ except (AttributeError, ValueError) as exc:
49
+ # Faker's own message for a bad locale names the attribute it failed to
50
+ # find, which reads like an internal error rather than a typo in a flag.
51
+ raise ValueError(
52
+ f"Unknown locale {target!r}. Use a Faker locale name such as "
53
+ f"'en_US', 'en_GB', 'nl_BE' or 'fr_FR'."
54
+ ) from exc
55
+ _locale = target
56
+ _resolve_locale()
57
+ _person_state.clear()
58
+ _address_state.clear()
59
+
60
+
61
+ # ---------------------------------------------------------
62
+ # Per-row identities
63
+ # ---------------------------------------------------------
64
+ # Columns are generated one at a time, so nothing connected the `first_name`,
65
+ # `last_name` and `email` of a single row: each drew from Faker independently
66
+ # and one row described three different people. Anyone who points a BI tool at
67
+ # the output sees it immediately, which makes it a credibility problem rather
68
+ # than a cosmetic one.
69
+ #
70
+ # The fix is a per-table pool of identities, one per row index. A column whose
71
+ # name (or declared type) means "a person's email" reads row i's identity
72
+ # instead of rolling its own, so every person-shaped column in a row agrees.
73
+ #
74
+ # Addresses get the same treatment, one pool along. What that can and cannot
75
+ # promise is worth being precise about: every component now comes from the same
76
+ # locale, so a row reads as one country with one set of conventions instead of
77
+ # "Brussels, Texas, 3000, Japan". It is not real geography -- Faker does not
78
+ # pair a city with its state or its postcode even inside a locale, so the
79
+ # postcode is a plausible postcode for that country rather than that city's.
80
+ # Closing that last gap needs a reference table of real combinations, not a
81
+ # cleverer arrangement of Faker calls.
82
+ #
83
+ # Both classes carry only what is drawn and compute the rest, and both use
84
+ # slots. A million-row `person` table is a normal request: storing five strings
85
+ # per row in a dict-backed object costs hundreds of megabytes, storing three
86
+ # slotted references costs tens, and the derived strings then exist only for
87
+ # the columns a schema actually declares.
88
+ @dataclass(frozen=True, slots=True)
89
+ class _Person:
90
+ first_name: str
91
+ last_name: str
92
+ email_domain: str
93
+
94
+ @property
95
+ def full_name(self) -> str:
96
+ return f"{self.first_name} {self.last_name}"
97
+
98
+ @property
99
+ def user_name(self) -> str:
100
+ return f"{_slug(self.first_name)[:1]}{_slug(self.last_name)}"
101
+
102
+ @property
103
+ def email(self) -> str:
104
+ return f"{_slug(self.first_name)}.{_slug(self.last_name)}@{self.email_domain}"
105
+
106
+
107
+ @dataclass(frozen=True, slots=True)
108
+ class _Address:
109
+ street: str
110
+ city: str
111
+ state: str
112
+ country: str
113
+ postcode: str
114
+
115
+ @property
116
+ def full(self) -> str:
117
+ """The row's own components, not a separate `fake.address()` draw.
118
+
119
+ Composing it here rather than calling Faker again is the whole point:
120
+ an `address` column has to agree with the `city` column beside it.
121
+ """
122
+ lines = [self.street, f"{self.postcode} {self.city}".strip(), self.country]
123
+ return ", ".join(line for line in lines if line)
124
+
125
+
126
+ class _FromRow:
127
+ """Marker for a provider that reads row i's person or address.
128
+
129
+ A sentinel rather than a callable so `_NAME_PATTERNS` can stay a single
130
+ ordered list: splitting these patterns into lists of their own would quietly
131
+ reorder them against the rest, and that order is load-bearing (see the
132
+ comment on `_NAME_PATTERNS`).
133
+ """
134
+
135
+ __slots__ = ("pool", "field")
136
+
137
+ def __init__(self, pool: str, field: str) -> None:
138
+ self.pool = pool
139
+ self.field = field
140
+
141
+
142
+ _Provider = Union[Callable[[], object], _FromRow]
143
+
144
+ # One pool per table, keyed by table name, grown on demand and released as soon
145
+ # as that table's frame is finished.
146
+ _person_state: dict[str, list[_Person]] = {}
147
+ _address_state: dict[str, list[_Address]] = {}
148
+
149
+
150
+ def _slug(value: str) -> str:
151
+ """Reduce a name to something that can sit inside an email or a username."""
152
+ return re.sub(r"[^a-z0-9]+", "", value.lower()) or "user"
153
+
154
+
155
+ # Resolved once per locale, not once per row. Both of these are constant for a
156
+ # locale, and a million-row table makes the difference stark: `current_country`
157
+ # would be recomputed a million times for an answer that never changes, and
158
+ # probing for an administrative-unit provider means catching AttributeError --
159
+ # on a locale that has none, four raised exceptions per row.
160
+ _country_name: str = ""
161
+ _state_provider: Optional[str] = None
162
+
163
+
164
+ def _first_provider(*names: str) -> Optional[str]:
165
+ """The first of these provider names this locale actually has, else None.
166
+
167
+ Locales disagree about what exists: `state` is American, `province` is
168
+ Belgian, and plenty of countries have no administrative unit worth naming.
169
+ None is the honest answer there -- better than inventing a region the
170
+ country does not have, purely so a column looks full.
171
+ """
172
+ for name in names:
173
+ try:
174
+ fake.format(name)
175
+ except (AttributeError, TypeError, ValueError):
176
+ continue
177
+ return name
178
+ return None
179
+
180
+
181
+ def _resolve_locale() -> None:
182
+ """Cache the per-locale constants the address pool reads on every row."""
183
+ global _country_name, _state_provider
184
+ # `current_country` is the locale's own country, and it is what stops a US
185
+ # street from landing in Japan. Locales too generic to have one (plain "en")
186
+ # fall back to a single country picked once, so at least every row agrees.
187
+ if _first_provider("current_country"):
188
+ _country_name = str(fake.current_country())
189
+ else:
190
+ _country_name = fake.country()
191
+ _state_provider = _first_provider("state", "province", "administrative_unit", "region")
192
+
193
+
194
+ def _new_person() -> _Person:
195
+ return _Person(
196
+ first_name=fake.first_name(),
197
+ last_name=fake.last_name(),
198
+ email_domain=fake.free_email_domain(),
199
+ )
200
+
201
+
202
+ def _new_address() -> _Address:
203
+ return _Address(
204
+ street=fake.street_address(),
205
+ city=fake.city(),
206
+ state=str(fake.format(_state_provider)) if _state_provider else "",
207
+ country=_country_name,
208
+ postcode=fake.postcode(),
209
+ )
210
+
211
+
212
+ def reset_row_pools() -> None:
213
+ """Drop every table's person and address pool.
214
+
215
+ Called per run by generate_data_from_dbml alongside the other per-run state.
216
+ Without it a second run in the same process reuses the first run's people,
217
+ which looks harmless right up until a seeded run stops reproducing.
218
+ """
219
+ _person_state.clear()
220
+ _address_state.clear()
221
+
222
+
223
+ def release_row_pools(table_name: str) -> None:
224
+ """Drop one table's pools once its frame is finished.
225
+
226
+ A pool only has to outlive the columns of its own table. Holding every
227
+ table's pool until the end of the run means a schema of twenty million-row
228
+ tables carries twenty million identities nothing will read again; releasing
229
+ per table bounds the cost to the largest single table instead of the sum.
230
+ """
231
+ _person_state.pop(table_name, None)
232
+ _address_state.pop(table_name, None)
233
+
234
+
235
+ def _row_pool(pool: str, table_name: Optional[str], row_count: int) -> list:
236
+ """Row i's person or address for this table, created and cached as needed.
237
+
238
+ Keyed by table so two tables of people hold two different populations, and
239
+ grown rather than rebuilt so every column of the same table sees the same
240
+ row i. Callers with no table name (core's composite-key repair, which
241
+ regenerates a single cell) share one bucket; that path only ever touches key
242
+ columns, never person or address ones.
243
+ """
244
+ key = table_name or ""
245
+ # Written out per pool rather than shared behind a generic helper: the two
246
+ # loops are three lines each, and pairing the right factory with the right
247
+ # pool is exactly the thing a reader (and a type checker) wants to see.
248
+ if pool == "person":
249
+ people = _person_state.setdefault(key, [])
250
+ while len(people) < row_count:
251
+ people.append(_new_person())
252
+ return people
253
+
254
+ addresses = _address_state.setdefault(key, [])
255
+ while len(addresses) < row_count:
256
+ addresses.append(_new_address())
257
+ return addresses
15
258
 
16
259
 
17
260
  # ---------------------------------------------------------
@@ -21,25 +264,25 @@ fake = Faker()
21
264
  # "first_name" before any looser pattern gets a chance. There is
22
265
  # deliberately no generic "name" pattern, since "product_name" or
23
266
  # "company_name" would otherwise be filled with a person's name.
24
- _NAME_PATTERNS: list[tuple[str, Callable[[], object]]] = [
25
- ("first_name", lambda: fake.first_name()),
26
- ("last_name", lambda: fake.last_name()),
27
- ("full_name", lambda: fake.name()),
28
- ("user_name", lambda: fake.user_name()),
29
- ("username", lambda: fake.user_name()),
267
+ _NAME_PATTERNS: list[tuple[str, _Provider]] = [
268
+ ("first_name", _FromRow("person", "first_name")),
269
+ ("last_name", _FromRow("person", "last_name")),
270
+ ("full_name", _FromRow("person", "full_name")),
271
+ ("user_name", _FromRow("person", "user_name")),
272
+ ("username", _FromRow("person", "user_name")),
30
273
  ("password", lambda: fake.password()),
31
- ("email", lambda: fake.email()),
274
+ ("email", _FromRow("person", "email")),
32
275
  ("phone", lambda: fake.phone_number()),
33
276
  ("mobile", lambda: fake.phone_number()),
34
277
  ("fax", lambda: fake.phone_number()),
35
- ("street", lambda: fake.street_address()),
36
- ("address", lambda: fake.address().replace("\n", ", ")),
37
- ("city", lambda: fake.city()),
38
- ("province", lambda: fake.state()),
39
- ("state", lambda: fake.state()),
40
- ("country", lambda: fake.country()),
41
- ("zip", lambda: fake.postcode()),
42
- ("postal", lambda: fake.postcode()),
278
+ ("street", _FromRow("address", "street")),
279
+ ("address", _FromRow("address", "full")),
280
+ ("city", _FromRow("address", "city")),
281
+ ("province", _FromRow("address", "state")),
282
+ ("state", _FromRow("address", "state")),
283
+ ("country", _FromRow("address", "country")),
284
+ ("zip", _FromRow("address", "postcode")),
285
+ ("postal", _FromRow("address", "postcode")),
43
286
  ("homepage", lambda: fake.url()),
44
287
  ("website", lambda: fake.url()),
45
288
  ("url", lambda: fake.url()),
@@ -141,7 +384,7 @@ def get_duplicate_unique_columns() -> list[str]:
141
384
  return list(_duplicate_unique_columns)
142
385
 
143
386
 
144
- def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
387
+ def _infer_by_name(column_name: str) -> Optional[_Provider]:
145
388
  normalized = re.sub(r"[^a-z0-9]+", "_", column_name.lower())
146
389
  padded = f"_{normalized}_"
147
390
  for pattern, generator in _NAME_PATTERNS:
@@ -156,8 +399,27 @@ def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
156
399
  # declaration in any schema would stop generating emails.
157
400
  _SQL_TYPES_SHADOWING_A_PROVIDER = frozenset({"text", "json", "jsonb", "xml", "binary", "year"})
158
401
 
159
-
160
- def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
402
+ # Declared types that name a person or address field. `contact email` says
403
+ # exactly what a column called `email` says, so it has to reach the same row --
404
+ # otherwise row-level coherence has a second door it does not cover.
405
+ _ROW_TYPE_FIELDS = {
406
+ "first_name": ("person", "first_name"),
407
+ "last_name": ("person", "last_name"),
408
+ "name": ("person", "full_name"),
409
+ "user_name": ("person", "user_name"),
410
+ "username": ("person", "user_name"),
411
+ "email": ("person", "email"),
412
+ "address": ("address", "full"),
413
+ "street_address": ("address", "street"),
414
+ "city": ("address", "city"),
415
+ "state": ("address", "state"),
416
+ "province": ("address", "state"),
417
+ "country": ("address", "country"),
418
+ "postcode": ("address", "postcode"),
419
+ }
420
+
421
+
422
+ def _infer_by_type(base_type: str) -> Optional[_Provider]:
161
423
  """The provider a column's declared type names, if it names one deliberately.
162
424
 
163
425
  `sku ean13` and `home_state state` are the user saying which generator they
@@ -169,6 +431,9 @@ def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
169
431
  """
170
432
  if base_type in _SQL_TYPES_SHADOWING_A_PROVIDER:
171
433
  return None
434
+ row_field = _ROW_TYPE_FIELDS.get(base_type)
435
+ if row_field is not None:
436
+ return _FromRow(*row_field)
172
437
  try:
173
438
  fake.format(base_type)
174
439
  except (AttributeError, TypeError):
@@ -311,7 +576,12 @@ def generate_column_values(
311
576
  # -----------------------------------------------------
312
577
  else:
313
578
  generator = _infer_by_type(base_type) or _infer_by_name(column.name)
314
- if generator is not None:
579
+ if isinstance(generator, _FromRow):
580
+ rows = _row_pool(generator.pool, table_name, row_count)
581
+ values = [getattr(rows[index], generator.field) for index in range(row_count)]
582
+ if ensure_unique:
583
+ values = _deduplicate_identity(values)
584
+ elif generator is not None:
315
585
  values = [generator() for _ in range(row_count)]
316
586
  values = (
317
587
  _deduplicate(values, generator, column_name=unique_label)
@@ -375,6 +645,35 @@ def _deduplicate(
375
645
  return result
376
646
 
377
647
 
648
+ def _deduplicate_identity(values: list) -> list:
649
+ """Make identity-derived values unique without swapping the person.
650
+
651
+ `_deduplicate` resolves a collision by calling the generator again, which
652
+ for an identity column would hand row i a different person's email and undo
653
+ the coherence this module just established. Suffixing keeps the row's
654
+ identity and disambiguates only the value, the way a real system issues
655
+ `jane.doe2@...` once `jane.doe@...` is taken. It also always succeeds, so
656
+ an identity column never lands in `_duplicate_unique_columns`.
657
+ """
658
+ seen: set = set()
659
+ result = []
660
+ for value in values:
661
+ candidate = value
662
+ counter = 1
663
+ while candidate in seen:
664
+ counter += 1
665
+ candidate = _suffixed(str(value), counter)
666
+ seen.add(candidate)
667
+ result.append(candidate)
668
+ return result
669
+
670
+
671
+ def _suffixed(value: str, counter: int) -> str:
672
+ """Append a disambiguating number, before the @ when the value is an email."""
673
+ local, at, domain = value.partition("@")
674
+ return f"{local}{counter}{at}{domain}"
675
+
676
+
378
677
  def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
379
678
  """Pick a random timestamp in a window around today, to whole seconds.
380
679
 
@@ -390,3 +689,8 @@ def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
390
689
  end = midnight + timedelta(days=end_days)
391
690
  random_second = random.randint(0, int((end - start).total_seconds()))
392
691
  return start + timedelta(seconds=random_second)
692
+
693
+
694
+ # The default locale's constants, resolved at import so the first generation
695
+ # does not pay for them and `set_locale`'s early return stays correct.
696
+ _resolve_locale()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.2.0
3
+ Version: 1.3.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -33,4 +33,5 @@ tests/test_dbt_project.py
33
33
  tests/test_dbt_tests.py
34
34
  tests/test_faker_name_inference.py
35
35
  tests/test_generation.py
36
- tests/test_release_stress.py
36
+ tests/test_release_stress.py
37
+ tests/test_row_identity.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.2.0"
7
+ version = "1.3.0"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -0,0 +1,372 @@
1
+ """Person-shaped columns in one row have to describe one person.
2
+
3
+ Columns are generated independently, one at a time, so before the identity
4
+ pool a single row's first_name, last_name and email came from three unrelated
5
+ Faker draws. These tests pin the row-level agreement, and the two properties it
6
+ must not cost: reproducibility under a seed, and uniqueness where it is asked
7
+ for.
8
+ """
9
+
10
+ import re
11
+
12
+ import pytest
13
+ from faker import Faker
14
+
15
+ import model2data.generate.faker as faker_module
16
+ from model2data.generate.core import generate_data_from_dbml
17
+ from model2data.generate.faker import (
18
+ DEFAULT_LOCALE,
19
+ _slug,
20
+ current_locale,
21
+ generate_column_values,
22
+ reset_row_pools,
23
+ set_locale,
24
+ )
25
+ from model2data.parse.dbml import ColumnDef, TableDef
26
+
27
+
28
+ def _person_table(name: str = "customers", columns: list[ColumnDef] | None = None) -> TableDef:
29
+ return TableDef(
30
+ name=name,
31
+ columns=columns
32
+ or [
33
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
34
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
35
+ ColumnDef(name="full_name", data_type="varchar", settings={"not null"}),
36
+ ColumnDef(name="user_name", data_type="varchar", settings={"not null"}),
37
+ ColumnDef(name="email", data_type="varchar", settings={"not null"}),
38
+ ],
39
+ )
40
+
41
+
42
+ def _assert_row_is_one_person(row) -> None:
43
+ first, last = row["first_name"], row["last_name"]
44
+ assert row["full_name"] == f"{first} {last}"
45
+ assert row["email"].startswith(f"{_slug(first)}.{_slug(last)}@")
46
+ assert row["user_name"] == f"{_slug(first)[:1]}{_slug(last)}"
47
+
48
+
49
+ class TestRowIdentityCoherence:
50
+ def test_person_columns_in_a_row_describe_one_person(self):
51
+ frames = generate_data_from_dbml(
52
+ tables={"customers": _person_table()}, refs=[], base_rows=25, seed=1
53
+ )
54
+ df = frames["customers"]
55
+ assert len(df) == 25
56
+ for _, row in df.iterrows():
57
+ _assert_row_is_one_person(row)
58
+
59
+ def test_two_tables_draw_from_different_populations(self):
60
+ frames = generate_data_from_dbml(
61
+ tables={
62
+ "customers": _person_table("customers"),
63
+ "employees": _person_table("employees"),
64
+ },
65
+ refs=[],
66
+ base_rows=30,
67
+ seed=2,
68
+ )
69
+ # Asserted on the keying rather than on the values. "These two sets of
70
+ # emails do not overlap" is true here, but only by a probability -- the
71
+ # same shape of luck-based assertion that made the password test fail
72
+ # the first time the RNG moved.
73
+ customers_pool = faker_module._row_pool("person", "customers", 30)
74
+ employees_pool = faker_module._row_pool("person", "employees", 30)
75
+ assert customers_pool is not employees_pool
76
+
77
+ # And the values follow from that: two tables of people are two groups
78
+ # of people, not one pool handed out twice.
79
+ assert list(frames["customers"]["email"]) != list(frames["employees"]["email"])
80
+
81
+ def test_same_seed_reproduces_the_same_people(self):
82
+ def run():
83
+ return generate_data_from_dbml(
84
+ tables={"customers": _person_table()}, refs=[], base_rows=15, seed=7
85
+ )["customers"]
86
+
87
+ first_run, second_run = run(), run()
88
+ assert list(first_run["email"]) == list(second_run["email"])
89
+ assert list(first_run["full_name"]) == list(second_run["full_name"])
90
+
91
+ def test_unique_email_stays_unique_without_swapping_the_person(self):
92
+ # A tiny name space forces collisions, so the suffixing path actually
93
+ # runs rather than being skipped because every value happened to differ.
94
+ table = _person_table(
95
+ columns=[
96
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
97
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
98
+ ColumnDef(name="email", data_type="varchar", settings={"not null", "unique"}),
99
+ ],
100
+ )
101
+ df = generate_data_from_dbml(tables={"customers": table}, refs=[], base_rows=400, seed=3)[
102
+ "customers"
103
+ ]
104
+
105
+ assert df["email"].is_unique
106
+ for _, row in df.iterrows():
107
+ # The suffix may sit before the @, but the row's own name is still
108
+ # what the address is built from.
109
+ local = row["email"].split("@")[0]
110
+ assert re.fullmatch(
111
+ rf"{re.escape(_slug(row['first_name']))}\.{re.escape(_slug(row['last_name']))}\d*",
112
+ local,
113
+ )
114
+
115
+ def test_declared_email_type_also_reaches_the_identity(self):
116
+ table = TableDef(
117
+ name="contacts",
118
+ columns=[
119
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
120
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
121
+ # Declared as a provider rather than named `email`.
122
+ ColumnDef(name="contact", data_type="email", settings={"not null"}),
123
+ ],
124
+ )
125
+ df = generate_data_from_dbml(tables={"contacts": table}, refs=[], base_rows=20, seed=4)[
126
+ "contacts"
127
+ ]
128
+ for _, row in df.iterrows():
129
+ assert row["contact"].startswith(
130
+ f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@"
131
+ )
132
+
133
+ def test_email_declared_as_text_is_still_an_email(self):
134
+ # `email text` means the SQL type; the name inference must still win,
135
+ # and still go through the identity.
136
+ table = TableDef(
137
+ name="people",
138
+ columns=[
139
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
140
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
141
+ ColumnDef(name="email", data_type="text", settings={"not null"}),
142
+ ],
143
+ )
144
+ df = generate_data_from_dbml(tables={"people": table}, refs=[], base_rows=10, seed=5)[
145
+ "people"
146
+ ]
147
+ for _, row in df.iterrows():
148
+ assert row["email"].startswith(f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@")
149
+
150
+ def test_password_column_is_not_identity_derived(self):
151
+ # `password` sits between `username` and `email` in the pattern list, so
152
+ # it is the entry most at risk of having been swept into the identity
153
+ # work. Asserted against the pattern table rather than the output: an
154
+ # earlier version of this test looked for "@" in the generated password,
155
+ # which passed only by luck -- Faker's passwords include punctuation, "@"
156
+ # among it, so the test failed the first time the RNG moved.
157
+ assert isinstance(faker_module._infer_by_name("email"), faker_module._FromRow)
158
+ assert not isinstance(faker_module._infer_by_name("password"), faker_module._FromRow)
159
+
160
+ table = TableDef(
161
+ name="users",
162
+ columns=[
163
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
164
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
165
+ ColumnDef(name="email", data_type="varchar", settings={"not null"}),
166
+ ColumnDef(name="password", data_type="varchar", settings={"not null"}),
167
+ ],
168
+ )
169
+ df = generate_data_from_dbml(tables={"users": table}, refs=[], base_rows=20, seed=21)[
170
+ "users"
171
+ ]
172
+ assert df["password"].nunique() == 20
173
+ for _, row in df.iterrows():
174
+ # Exact comparison, not a substring heuristic: the password must not
175
+ # be one of the identity's own values.
176
+ assert row["password"] not in {row["first_name"], row["last_name"], row["email"]}
177
+
178
+ def test_non_person_columns_are_untouched(self):
179
+ reset_row_pools()
180
+ cities = generate_column_values(
181
+ ColumnDef(name="city", data_type="varchar", settings={"not null"}),
182
+ row_count=10,
183
+ table_name="places",
184
+ )
185
+ assert len(cities) == 10
186
+ assert all(isinstance(city, str) and city for city in cities)
187
+
188
+
189
+ @pytest.fixture(autouse=True)
190
+ def _restore_locale():
191
+ """Locale is process-wide state, so a test that changes it must not leak."""
192
+ yield
193
+ set_locale(None)
194
+ # set_locale returns early when the locale is already the default, which
195
+ # would leave the cached per-locale constants behind for any test that
196
+ # stubbed `fake` rather than switching locale. Re-resolve unconditionally.
197
+ faker_module._resolve_locale()
198
+
199
+
200
+ def _address_table(name: str = "sites") -> TableDef:
201
+ return TableDef(
202
+ name=name,
203
+ columns=[
204
+ ColumnDef(name="street", data_type="varchar", settings={"not null"}),
205
+ ColumnDef(name="address", data_type="varchar", settings={"not null"}),
206
+ ColumnDef(name="city", data_type="varchar", settings={"not null"}),
207
+ ColumnDef(name="state", data_type="varchar", settings={"not null"}),
208
+ ColumnDef(name="country", data_type="varchar", settings={"not null"}),
209
+ ColumnDef(name="zip", data_type="varchar", settings={"not null"}),
210
+ ],
211
+ )
212
+
213
+
214
+ class TestAddressCoherence:
215
+ def test_address_column_is_built_from_the_row_beside_it(self):
216
+ df = generate_data_from_dbml(
217
+ tables={"sites": _address_table()}, refs=[], base_rows=20, seed=11
218
+ )["sites"]
219
+ for _, row in df.iterrows():
220
+ # The composed address has to be this row's street, city and
221
+ # postcode -- not a separate draw that happens to look like one.
222
+ assert row["street"] in row["address"]
223
+ assert row["city"] in row["address"]
224
+ assert row["zip"] in row["address"]
225
+
226
+ def test_every_row_is_in_one_country(self):
227
+ df = generate_data_from_dbml(
228
+ tables={"sites": _address_table()}, refs=[], base_rows=30, seed=12
229
+ )["sites"]
230
+ # The old behaviour drew country independently, so a US street could sit
231
+ # in Japan. One locale means one country, on every row.
232
+ assert set(df["country"]) == {"United States"}
233
+
234
+ def test_declared_city_type_agrees_with_the_row(self):
235
+ table = TableDef(
236
+ name="offices",
237
+ columns=[
238
+ ColumnDef(name="address", data_type="varchar", settings={"not null"}),
239
+ # Declared as a provider rather than named `city`.
240
+ ColumnDef(name="located_in", data_type="city", settings={"not null"}),
241
+ ],
242
+ )
243
+ df = generate_data_from_dbml(tables={"offices": table}, refs=[], base_rows=15, seed=13)[
244
+ "offices"
245
+ ]
246
+ for _, row in df.iterrows():
247
+ assert row["located_in"] in row["address"]
248
+
249
+
250
+ class TestLocale:
251
+ def test_default_locale_is_the_documented_one(self):
252
+ assert current_locale() == DEFAULT_LOCALE
253
+
254
+ def test_locale_changes_the_country_and_the_addresses(self):
255
+ def country_and_cities(locale):
256
+ df = generate_data_from_dbml(
257
+ tables={"sites": _address_table()},
258
+ refs=[],
259
+ base_rows=20,
260
+ seed=14,
261
+ locale=locale,
262
+ )["sites"]
263
+ return set(df["country"]), set(df["city"])
264
+
265
+ us_country, us_cities = country_and_cities(None)
266
+ be_country, be_cities = country_and_cities("nl_BE")
267
+
268
+ assert us_country == {"United States"}
269
+ assert be_country == {"Belgium"}
270
+ assert not us_cities & be_cities
271
+
272
+ def test_locale_applies_to_people_too(self):
273
+ df = generate_data_from_dbml(
274
+ tables={"customers": _person_table()},
275
+ refs=[],
276
+ base_rows=20,
277
+ seed=15,
278
+ locale="fr_FR",
279
+ )["customers"]
280
+ # Still coherent, just French: the pool swap must not break the
281
+ # first.last@domain derivation.
282
+ for _, row in df.iterrows():
283
+ _assert_row_is_one_person(row)
284
+
285
+ def test_locale_is_restored_between_runs(self):
286
+ generate_data_from_dbml(
287
+ tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16, locale="nl_BE"
288
+ )
289
+ assert current_locale() == "nl_BE"
290
+ # Passing nothing means the default, not "whatever the last run used".
291
+ generate_data_from_dbml(tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16)
292
+ assert current_locale() == DEFAULT_LOCALE
293
+
294
+ def test_unknown_locale_says_so_plainly(self):
295
+ with pytest.raises(ValueError, match="Unknown locale 'zz_ZZ'"):
296
+ generate_data_from_dbml(
297
+ tables={"sites": _address_table()}, refs=[], base_rows=2, locale="zz_ZZ"
298
+ )
299
+
300
+
301
+ class TestPoolLifetime:
302
+ def test_pools_are_released_once_a_table_is_finished(self):
303
+ # A million-row table is a normal request; holding its identities for
304
+ # the rest of the run is what makes that unaffordable.
305
+ generate_data_from_dbml(
306
+ tables={"customers": _person_table(), "sites": _address_table()},
307
+ refs=[],
308
+ base_rows=50,
309
+ seed=17,
310
+ )
311
+ assert faker_module._person_state == {}
312
+ assert faker_module._address_state == {}
313
+
314
+ def test_pool_rows_stay_small(self):
315
+ # Slots rather than a per-instance __dict__. Worth a test because the
316
+ # difference is ~6x per row (344 bytes vs 56), the derived fields are
317
+ # properties precisely so they are not stored, and a million-row person
318
+ # table is a normal request -- an innocent-looking field added here
319
+ # would cost hundreds of megabytes on one.
320
+ assert not hasattr(faker_module._new_person(), "__dict__")
321
+ assert not hasattr(faker_module._new_address(), "__dict__")
322
+
323
+
324
+ class _LocaleMissing:
325
+ """A Faker with named providers removed, standing in for a thinner locale.
326
+
327
+ Every locale Faker actually ships has `current_country` and at least one
328
+ administrative unit, so the fallbacks for a locale without them cannot be
329
+ reached by naming a real one -- but they still decide what lands in a
330
+ column, and the docstrings make a claim about what they do. This stands in
331
+ for the locale that would reach them.
332
+ """
333
+
334
+ def __init__(self, inner, missing):
335
+ self._inner = inner
336
+ self._missing = set(missing)
337
+
338
+ def format(self, name, *args, **kwargs):
339
+ if name in self._missing:
340
+ raise AttributeError(name)
341
+ return self._inner.format(name, *args, **kwargs)
342
+
343
+ def __getattr__(self, name):
344
+ if name in self._missing:
345
+ raise AttributeError(name)
346
+ return getattr(self._inner, name)
347
+
348
+
349
+ class TestThinLocales:
350
+ def test_no_administrative_unit_leaves_state_empty(self, monkeypatch):
351
+ stub = _LocaleMissing(
352
+ Faker("en_US"), {"state", "province", "administrative_unit", "region"}
353
+ )
354
+ monkeypatch.setattr(faker_module, "fake", stub)
355
+ faker_module._resolve_locale()
356
+
357
+ address = faker_module._new_address()
358
+ # Empty is the honest answer for a country with no region worth naming.
359
+ # Inventing one just to fill the column would be worse data, not better.
360
+ assert address.state == ""
361
+ assert address.city and address.street and address.country
362
+
363
+ def test_no_current_country_still_puts_every_row_in_one_country(self, monkeypatch):
364
+ stub = _LocaleMissing(Faker("en_US"), {"current_country"})
365
+ monkeypatch.setattr(faker_module, "fake", stub)
366
+ faker_module._resolve_locale()
367
+
368
+ countries = {faker_module._new_address().country for _ in range(10)}
369
+ # The point of the fallback is agreement, not accuracy: one country
370
+ # picked once beats a different random country on every row.
371
+ assert len(countries) == 1
372
+ assert countries != {""}
@@ -1,3 +0,0 @@
1
- """Model2Data: Generate analytics-ready datasets from DBML models."""
2
-
3
- __version__ = "0.1.1"
File without changes
File without changes
File without changes
File without changes
File without changes