model2data 1.2.0__tar.gz → 1.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. {model2data-1.2.0/model2data.egg-info → model2data-1.3.1}/PKG-INFO +1 -1
  2. model2data-1.3.1/model2data/__init__.py +12 -0
  3. {model2data-1.2.0 → model2data-1.3.1}/model2data/cli.py +11 -0
  4. {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/core.py +18 -2
  5. {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/faker.py +344 -21
  6. {model2data-1.2.0 → model2data-1.3.1/model2data.egg-info}/PKG-INFO +1 -1
  7. {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/SOURCES.txt +2 -1
  8. {model2data-1.2.0 → model2data-1.3.1}/pyproject.toml +1 -1
  9. model2data-1.3.1/tests/test_row_identity.py +414 -0
  10. model2data-1.2.0/model2data/__init__.py +0 -3
  11. {model2data-1.2.0 → model2data-1.3.1}/LICENSE +0 -0
  12. {model2data-1.2.0 → model2data-1.3.1}/README.md +0 -0
  13. {model2data-1.2.0 → model2data-1.3.1}/README_PYPI.md +0 -0
  14. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/__init__.py +0 -0
  15. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/project.py +0 -0
  16. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  17. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  18. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  19. {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/tests.py +0 -0
  20. {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/__init__.py +0 -0
  21. {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/relationships.py +0 -0
  22. {model2data-1.2.0 → model2data-1.3.1}/model2data/parse/__init__.py +0 -0
  23. {model2data-1.2.0 → model2data-1.3.1}/model2data/parse/dbml.py +0 -0
  24. {model2data-1.2.0 → model2data-1.3.1}/model2data/utils.py +0 -0
  25. {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/dependency_links.txt +0 -0
  26. {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/entry_points.txt +0 -0
  27. {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/requires.txt +0 -0
  28. {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/top_level.txt +0 -0
  29. {model2data-1.2.0 → model2data-1.3.1}/setup.cfg +0 -0
  30. {model2data-1.2.0 → model2data-1.3.1}/tests/test_cli.py +0 -0
  31. {model2data-1.2.0 → model2data-1.3.1}/tests/test_coverage_gaps.py +0 -0
  32. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbml_parser.py +0 -0
  33. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbml_parser_fuzz.py +0 -0
  34. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_integration.py +0 -0
  35. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_naming.py +0 -0
  36. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_project.py +0 -0
  37. {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_tests.py +0 -0
  38. {model2data-1.2.0 → model2data-1.3.1}/tests/test_faker_name_inference.py +0 -0
  39. {model2data-1.2.0 → model2data-1.3.1}/tests/test_generation.py +0 -0
  40. {model2data-1.2.0 → model2data-1.3.1}/tests/test_release_stress.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.2.0
3
+ Version: 1.3.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -0,0 +1,12 @@
1
+ """Model2Data: Generate analytics-ready datasets from DBML models."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ # Read the installed distribution's version rather than repeating it here.
7
+ # The literal that used to live in this file said 0.1.1 for every release up
8
+ # to and including 1.2.0: nothing reads `__version__`, so nothing caught it
9
+ # drifting. Deriving it means it cannot drift again.
10
+ __version__ = version("model2data")
11
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
12
+ __version__ = "0.0.0.dev0"
@@ -18,6 +18,7 @@ from model2data.generate.core import (
18
18
  get_unresolved_composite_keys,
19
19
  )
20
20
  from model2data.generate.faker import (
21
+ DEFAULT_LOCALE,
21
22
  get_duplicate_unique_columns,
22
23
  get_unmapped_columns,
23
24
  reset_stats,
@@ -125,6 +126,15 @@ def main(
125
126
  "Using the same seed will always produce identical datasets."
126
127
  ),
127
128
  ),
129
+ locale: Optional[str] = typer.Option(
130
+ None,
131
+ "--locale",
132
+ help=(
133
+ f"Faker locale for generated people and addresses "
134
+ f"(default: {DEFAULT_LOCALE}).\n"
135
+ "Examples: en_GB, nl_BE, fr_FR, de_DE."
136
+ ),
137
+ ),
128
138
  name: Optional[str] = typer.Option(
129
139
  None,
130
140
  "--name",
@@ -214,6 +224,7 @@ def main(
214
224
  base_rows=rows,
215
225
  seed=seed,
216
226
  row_overrides=row_overrides,
227
+ locale=locale,
217
228
  )
218
229
 
219
230
  # -------------------------
@@ -10,7 +10,10 @@ from faker import Faker
10
10
 
11
11
  from model2data.generate.faker import (
12
12
  generate_column_values,
13
+ release_row_pools,
13
14
  reset_duplicate_unique_columns,
15
+ reset_row_pools,
16
+ set_locale,
14
17
  )
15
18
  from model2data.generate.relationships import (
16
19
  build_fk_lookup,
@@ -18,8 +21,6 @@ from model2data.generate.relationships import (
18
21
  )
19
22
  from model2data.parse.dbml import TableDef
20
23
 
21
- fake = Faker()
22
-
23
24
  # Tables the most recent generate_data_from_dbml() call found stuck in an
24
25
  # unresolved FK cycle (never reached indegree 0 during the topological
25
26
  # sort). Exposed out-of-band, mirroring generate.faker's
@@ -66,6 +67,7 @@ def generate_data_from_dbml(
66
67
  base_rows: int = 100,
67
68
  seed: Optional[int] = None,
68
69
  row_overrides: Optional[Mapping[str, int]] = None,
70
+ locale: Optional[str] = None,
69
71
  ) -> dict[str, pd.DataFrame]:
70
72
  """
71
73
  Generate synthetic datasets from parsed DBML definitions.
@@ -78,9 +80,19 @@ def generate_data_from_dbml(
78
80
  present in `row_overrides` fall back to `base_rows`; unknown names are
79
81
  ignored.
80
82
 
83
+ `locale` picks the Faker locale every generated person and address is drawn
84
+ from -- `"nl_BE"`, `"fr_FR"`, `"en_GB"` -- defaulting to `DEFAULT_LOCALE`.
85
+ It is a per-run setting rather than a per-column one on purpose: a table
86
+ holding one Belgian and one American address is the incoherence the row
87
+ pools exist to remove.
88
+
81
89
  This function is deterministic if a seed is provided.
82
90
  It performs no filesystem I/O and returns pandas DataFrames.
83
91
  """
92
+ # Locale first, then the seed: switching locale builds a new Faker, and the
93
+ # seed has to be the last word on the generator that actually runs.
94
+ set_locale(locale)
95
+
84
96
  if seed is not None:
85
97
  random.seed(seed)
86
98
  Faker.seed(seed)
@@ -88,6 +100,7 @@ def generate_data_from_dbml(
88
100
  reset_cycle_state()
89
101
  reset_dedup_state()
90
102
  reset_duplicate_unique_columns()
103
+ reset_row_pools()
91
104
 
92
105
  # ---------------------------------------------------------
93
106
  # Classify references
@@ -193,6 +206,9 @@ def generate_data_from_dbml(
193
206
 
194
207
  df = _coerce_integer_dtypes(df, table_def)
195
208
  generated[table_name] = df
209
+ # This table is finished: nothing will read its people or addresses
210
+ # again, and on a million-row table they are worth tens of megabytes.
211
+ release_row_pools(table_name)
196
212
 
197
213
  return generated
198
214
 
@@ -2,16 +2,278 @@ from __future__ import annotations
2
2
 
3
3
  import random
4
4
  import re
5
+ import unicodedata
5
6
  import uuid
7
+ from dataclasses import dataclass
6
8
  from datetime import datetime, timedelta
7
- from typing import Callable, Optional
9
+ from typing import Callable, Optional, Union
8
10
 
9
11
  import pandas as pd
10
12
  from faker import Faker
11
13
 
12
14
  from model2data.parse.dbml import ColumnDef
13
15
 
14
- fake = Faker()
16
+ # ---------------------------------------------------------
17
+ # Locale
18
+ # ---------------------------------------------------------
19
+ # The locale every generated person and address comes from until a caller says
20
+ # otherwise. Named rather than left implicit because locale is on its way to
21
+ # being a first-class option in the CLI and the studio, and this is the seam it
22
+ # plugs into: one instance, swapped in one place.
23
+ DEFAULT_LOCALE = "en_US"
24
+
25
+ _locale = DEFAULT_LOCALE
26
+ fake = Faker(DEFAULT_LOCALE)
27
+
28
+
29
+ def current_locale() -> str:
30
+ """The locale generation is currently drawing from."""
31
+ return _locale
32
+
33
+
34
+ def set_locale(locale: Optional[str]) -> None:
35
+ """Point generation at a locale, or back at the default when given None.
36
+
37
+ Rebinding the module-level `fake` reaches every provider here, because each
38
+ one looks the name up when it runs rather than capturing it. The row pools
39
+ are dropped on the way through: a Belgian address sitting next to an
40
+ American one in the same table is precisely the incoherence they exist to
41
+ prevent.
42
+ """
43
+ global fake, _locale
44
+ target = locale or DEFAULT_LOCALE
45
+ if target == _locale:
46
+ return
47
+ try:
48
+ fake = Faker(target)
49
+ except (AttributeError, ValueError) as exc:
50
+ # Faker's own message for a bad locale names the attribute it failed to
51
+ # find, which reads like an internal error rather than a typo in a flag.
52
+ raise ValueError(
53
+ f"Unknown locale {target!r}. Use a Faker locale name such as "
54
+ f"'en_US', 'en_GB', 'nl_BE' or 'fr_FR'."
55
+ ) from exc
56
+ _locale = target
57
+ _resolve_locale()
58
+ _person_state.clear()
59
+ _address_state.clear()
60
+
61
+
62
+ # ---------------------------------------------------------
63
+ # Per-row identities
64
+ # ---------------------------------------------------------
65
+ # Columns are generated one at a time, so nothing connected the `first_name`,
66
+ # `last_name` and `email` of a single row: each drew from Faker independently
67
+ # and one row described three different people. Anyone who points a BI tool at
68
+ # the output sees it immediately, which makes it a credibility problem rather
69
+ # than a cosmetic one.
70
+ #
71
+ # The fix is a per-table pool of identities, one per row index. A column whose
72
+ # name (or declared type) means "a person's email" reads row i's identity
73
+ # instead of rolling its own, so every person-shaped column in a row agrees.
74
+ #
75
+ # Addresses get the same treatment, one pool along. What that can and cannot
76
+ # promise is worth being precise about: every component now comes from the same
77
+ # locale, so a row reads as one country with one set of conventions instead of
78
+ # "Brussels, Texas, 3000, Japan". It is not real geography -- Faker does not
79
+ # pair a city with its state or its postcode even inside a locale, so the
80
+ # postcode is a plausible postcode for that country rather than that city's.
81
+ # Closing that last gap needs a reference table of real combinations, not a
82
+ # cleverer arrangement of Faker calls.
83
+ #
84
+ # Both classes carry only what is drawn and compute the rest, and both use
85
+ # slots. A million-row `person` table is a normal request: storing five strings
86
+ # per row in a dict-backed object costs hundreds of megabytes, storing three
87
+ # slotted references costs tens, and the derived strings then exist only for
88
+ # the columns a schema actually declares.
89
+ @dataclass(frozen=True, slots=True)
90
+ class _Person:
91
+ first_name: str
92
+ last_name: str
93
+ email_domain: str
94
+
95
+ @property
96
+ def full_name(self) -> str:
97
+ return f"{self.first_name} {self.last_name}"
98
+
99
+ @property
100
+ def user_name(self) -> str:
101
+ return f"{_slug(self.first_name)[:1]}{_slug(self.last_name)}"
102
+
103
+ @property
104
+ def email(self) -> str:
105
+ return f"{_slug(self.first_name)}.{_slug(self.last_name)}@{self.email_domain}"
106
+
107
+
108
+ @dataclass(frozen=True, slots=True)
109
+ class _Address:
110
+ street: str
111
+ city: str
112
+ state: str
113
+ country: str
114
+ postcode: str
115
+
116
+ @property
117
+ def full(self) -> str:
118
+ """The row's own components, not a separate `fake.address()` draw.
119
+
120
+ Composing it here rather than calling Faker again is the whole point:
121
+ an `address` column has to agree with the `city` column beside it.
122
+ """
123
+ lines = [self.street, f"{self.postcode} {self.city}".strip(), self.country]
124
+ return ", ".join(line for line in lines if line)
125
+
126
+
127
+ class _FromRow:
128
+ """Marker for a provider that reads row i's person or address.
129
+
130
+ A sentinel rather than a callable so `_NAME_PATTERNS` can stay a single
131
+ ordered list: splitting these patterns into lists of their own would quietly
132
+ reorder them against the rest, and that order is load-bearing (see the
133
+ comment on `_NAME_PATTERNS`).
134
+ """
135
+
136
+ __slots__ = ("pool", "field")
137
+
138
+ def __init__(self, pool: str, field: str) -> None:
139
+ self.pool = pool
140
+ self.field = field
141
+
142
+
143
+ _Provider = Union[Callable[[], object], _FromRow]
144
+
145
+ # One pool per table, keyed by table name, grown on demand and released as soon
146
+ # as that table's frame is finished.
147
+ _person_state: dict[str, list[_Person]] = {}
148
+ _address_state: dict[str, list[_Address]] = {}
149
+
150
+
151
+ # Letters that NFKD does not take apart, because they are their own letters
152
+ # rather than a base plus an accent. Without these, "ø" and "ß" would simply
153
+ # vanish along with the accents.
154
+ _UNDECOMPOSED_LETTERS = str.maketrans(
155
+ {"ø": "o", "æ": "ae", "œ": "oe", "ß": "ss", "ł": "l", "đ": "d", "ð": "d", "þ": "th", "ı": "i"}
156
+ )
157
+
158
+
159
+ def _slug(value: str) -> str:
160
+ """Reduce a name to something that can sit inside an email or a username.
161
+
162
+ Accents are folded, not dropped. Stripping them outright turned `Aimée` into
163
+ `aime` and `Müller` into `mller` -- not that person's name, and conspicuously
164
+ broken in exactly the European locales the locale option exists to serve.
165
+ NFKD splits most accented letters into a base letter plus a combining mark,
166
+ which encoding to ASCII then discards; the letters that do not decompose are
167
+ mapped first.
168
+ """
169
+ folded = value.lower().translate(_UNDECOMPOSED_LETTERS)
170
+ ascii_only = unicodedata.normalize("NFKD", folded).encode("ascii", "ignore").decode("ascii")
171
+ return re.sub(r"[^a-z0-9]+", "", ascii_only) or "user"
172
+
173
+
174
+ # Resolved once per locale, not once per row. Both of these are constant for a
175
+ # locale, and a million-row table makes the difference stark: `current_country`
176
+ # would be recomputed a million times for an answer that never changes, and
177
+ # probing for an administrative-unit provider means catching AttributeError --
178
+ # on a locale that has none, four raised exceptions per row.
179
+ _country_name: str = ""
180
+ _state_provider: Optional[str] = None
181
+
182
+
183
+ def _first_provider(*names: str) -> Optional[str]:
184
+ """The first of these provider names this locale actually has, else None.
185
+
186
+ Locales disagree about what exists: `state` is American, `province` is
187
+ Belgian, and plenty of countries have no administrative unit worth naming.
188
+ None is the honest answer there -- better than inventing a region the
189
+ country does not have, purely so a column looks full.
190
+ """
191
+ for name in names:
192
+ try:
193
+ fake.format(name)
194
+ except (AttributeError, TypeError, ValueError):
195
+ continue
196
+ return name
197
+ return None
198
+
199
+
200
+ def _resolve_locale() -> None:
201
+ """Cache the per-locale constants the address pool reads on every row."""
202
+ global _country_name, _state_provider
203
+ # `current_country` is the locale's own country, and it is what stops a US
204
+ # street from landing in Japan. Locales too generic to have one (plain "en")
205
+ # fall back to a single country picked once, so at least every row agrees.
206
+ if _first_provider("current_country"):
207
+ _country_name = str(fake.current_country())
208
+ else:
209
+ _country_name = fake.country()
210
+ _state_provider = _first_provider("state", "province", "administrative_unit", "region")
211
+
212
+
213
+ def _new_person() -> _Person:
214
+ return _Person(
215
+ first_name=fake.first_name(),
216
+ last_name=fake.last_name(),
217
+ email_domain=fake.free_email_domain(),
218
+ )
219
+
220
+
221
+ def _new_address() -> _Address:
222
+ return _Address(
223
+ street=fake.street_address(),
224
+ city=fake.city(),
225
+ state=str(fake.format(_state_provider)) if _state_provider else "",
226
+ country=_country_name,
227
+ postcode=fake.postcode(),
228
+ )
229
+
230
+
231
+ def reset_row_pools() -> None:
232
+ """Drop every table's person and address pool.
233
+
234
+ Called per run by generate_data_from_dbml alongside the other per-run state.
235
+ Without it a second run in the same process reuses the first run's people,
236
+ which looks harmless right up until a seeded run stops reproducing.
237
+ """
238
+ _person_state.clear()
239
+ _address_state.clear()
240
+
241
+
242
+ def release_row_pools(table_name: str) -> None:
243
+ """Drop one table's pools once its frame is finished.
244
+
245
+ A pool only has to outlive the columns of its own table. Holding every
246
+ table's pool until the end of the run means a schema of twenty million-row
247
+ tables carries twenty million identities nothing will read again; releasing
248
+ per table bounds the cost to the largest single table instead of the sum.
249
+ """
250
+ _person_state.pop(table_name, None)
251
+ _address_state.pop(table_name, None)
252
+
253
+
254
+ def _row_pool(pool: str, table_name: Optional[str], row_count: int) -> list:
255
+ """Row i's person or address for this table, created and cached as needed.
256
+
257
+ Keyed by table so two tables of people hold two different populations, and
258
+ grown rather than rebuilt so every column of the same table sees the same
259
+ row i. Callers with no table name (core's composite-key repair, which
260
+ regenerates a single cell) share one bucket; that path only ever touches key
261
+ columns, never person or address ones.
262
+ """
263
+ key = table_name or ""
264
+ # Written out per pool rather than shared behind a generic helper: the two
265
+ # loops are three lines each, and pairing the right factory with the right
266
+ # pool is exactly the thing a reader (and a type checker) wants to see.
267
+ if pool == "person":
268
+ people = _person_state.setdefault(key, [])
269
+ while len(people) < row_count:
270
+ people.append(_new_person())
271
+ return people
272
+
273
+ addresses = _address_state.setdefault(key, [])
274
+ while len(addresses) < row_count:
275
+ addresses.append(_new_address())
276
+ return addresses
15
277
 
16
278
 
17
279
  # ---------------------------------------------------------
@@ -21,25 +283,25 @@ fake = Faker()
21
283
  # "first_name" before any looser pattern gets a chance. There is
22
284
  # deliberately no generic "name" pattern, since "product_name" or
23
285
  # "company_name" would otherwise be filled with a person's name.
24
- _NAME_PATTERNS: list[tuple[str, Callable[[], object]]] = [
25
- ("first_name", lambda: fake.first_name()),
26
- ("last_name", lambda: fake.last_name()),
27
- ("full_name", lambda: fake.name()),
28
- ("user_name", lambda: fake.user_name()),
29
- ("username", lambda: fake.user_name()),
286
+ _NAME_PATTERNS: list[tuple[str, _Provider]] = [
287
+ ("first_name", _FromRow("person", "first_name")),
288
+ ("last_name", _FromRow("person", "last_name")),
289
+ ("full_name", _FromRow("person", "full_name")),
290
+ ("user_name", _FromRow("person", "user_name")),
291
+ ("username", _FromRow("person", "user_name")),
30
292
  ("password", lambda: fake.password()),
31
- ("email", lambda: fake.email()),
293
+ ("email", _FromRow("person", "email")),
32
294
  ("phone", lambda: fake.phone_number()),
33
295
  ("mobile", lambda: fake.phone_number()),
34
296
  ("fax", lambda: fake.phone_number()),
35
- ("street", lambda: fake.street_address()),
36
- ("address", lambda: fake.address().replace("\n", ", ")),
37
- ("city", lambda: fake.city()),
38
- ("province", lambda: fake.state()),
39
- ("state", lambda: fake.state()),
40
- ("country", lambda: fake.country()),
41
- ("zip", lambda: fake.postcode()),
42
- ("postal", lambda: fake.postcode()),
297
+ ("street", _FromRow("address", "street")),
298
+ ("address", _FromRow("address", "full")),
299
+ ("city", _FromRow("address", "city")),
300
+ ("province", _FromRow("address", "state")),
301
+ ("state", _FromRow("address", "state")),
302
+ ("country", _FromRow("address", "country")),
303
+ ("zip", _FromRow("address", "postcode")),
304
+ ("postal", _FromRow("address", "postcode")),
43
305
  ("homepage", lambda: fake.url()),
44
306
  ("website", lambda: fake.url()),
45
307
  ("url", lambda: fake.url()),
@@ -141,7 +403,7 @@ def get_duplicate_unique_columns() -> list[str]:
141
403
  return list(_duplicate_unique_columns)
142
404
 
143
405
 
144
- def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
406
+ def _infer_by_name(column_name: str) -> Optional[_Provider]:
145
407
  normalized = re.sub(r"[^a-z0-9]+", "_", column_name.lower())
146
408
  padded = f"_{normalized}_"
147
409
  for pattern, generator in _NAME_PATTERNS:
@@ -156,8 +418,27 @@ def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
156
418
  # declaration in any schema would stop generating emails.
157
419
  _SQL_TYPES_SHADOWING_A_PROVIDER = frozenset({"text", "json", "jsonb", "xml", "binary", "year"})
158
420
 
159
-
160
- def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
421
+ # Declared types that name a person or address field. `contact email` says
422
+ # exactly what a column called `email` says, so it has to reach the same row --
423
+ # otherwise row-level coherence has a second door it does not cover.
424
+ _ROW_TYPE_FIELDS = {
425
+ "first_name": ("person", "first_name"),
426
+ "last_name": ("person", "last_name"),
427
+ "name": ("person", "full_name"),
428
+ "user_name": ("person", "user_name"),
429
+ "username": ("person", "user_name"),
430
+ "email": ("person", "email"),
431
+ "address": ("address", "full"),
432
+ "street_address": ("address", "street"),
433
+ "city": ("address", "city"),
434
+ "state": ("address", "state"),
435
+ "province": ("address", "state"),
436
+ "country": ("address", "country"),
437
+ "postcode": ("address", "postcode"),
438
+ }
439
+
440
+
441
+ def _infer_by_type(base_type: str) -> Optional[_Provider]:
161
442
  """The provider a column's declared type names, if it names one deliberately.
162
443
 
163
444
  `sku ean13` and `home_state state` are the user saying which generator they
@@ -169,6 +450,9 @@ def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
169
450
  """
170
451
  if base_type in _SQL_TYPES_SHADOWING_A_PROVIDER:
171
452
  return None
453
+ row_field = _ROW_TYPE_FIELDS.get(base_type)
454
+ if row_field is not None:
455
+ return _FromRow(*row_field)
172
456
  try:
173
457
  fake.format(base_type)
174
458
  except (AttributeError, TypeError):
@@ -311,7 +595,12 @@ def generate_column_values(
311
595
  # -----------------------------------------------------
312
596
  else:
313
597
  generator = _infer_by_type(base_type) or _infer_by_name(column.name)
314
- if generator is not None:
598
+ if isinstance(generator, _FromRow):
599
+ rows = _row_pool(generator.pool, table_name, row_count)
600
+ values = [getattr(rows[index], generator.field) for index in range(row_count)]
601
+ if ensure_unique:
602
+ values = _deduplicate_identity(values)
603
+ elif generator is not None:
315
604
  values = [generator() for _ in range(row_count)]
316
605
  values = (
317
606
  _deduplicate(values, generator, column_name=unique_label)
@@ -375,6 +664,35 @@ def _deduplicate(
375
664
  return result
376
665
 
377
666
 
667
+ def _deduplicate_identity(values: list) -> list:
668
+ """Make identity-derived values unique without swapping the person.
669
+
670
+ `_deduplicate` resolves a collision by calling the generator again, which
671
+ for an identity column would hand row i a different person's email and undo
672
+ the coherence this module just established. Suffixing keeps the row's
673
+ identity and disambiguates only the value, the way a real system issues
674
+ `jane.doe2@...` once `jane.doe@...` is taken. It also always succeeds, so
675
+ an identity column never lands in `_duplicate_unique_columns`.
676
+ """
677
+ seen: set = set()
678
+ result = []
679
+ for value in values:
680
+ candidate = value
681
+ counter = 1
682
+ while candidate in seen:
683
+ counter += 1
684
+ candidate = _suffixed(str(value), counter)
685
+ seen.add(candidate)
686
+ result.append(candidate)
687
+ return result
688
+
689
+
690
+ def _suffixed(value: str, counter: int) -> str:
691
+ """Append a disambiguating number, before the @ when the value is an email."""
692
+ local, at, domain = value.partition("@")
693
+ return f"{local}{counter}{at}{domain}"
694
+
695
+
378
696
  def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
379
697
  """Pick a random timestamp in a window around today, to whole seconds.
380
698
 
@@ -390,3 +708,8 @@ def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
390
708
  end = midnight + timedelta(days=end_days)
391
709
  random_second = random.randint(0, int((end - start).total_seconds()))
392
710
  return start + timedelta(seconds=random_second)
711
+
712
+
713
+ # The default locale's constants, resolved at import so the first generation
714
+ # does not pay for them and `set_locale`'s early return stays correct.
715
+ _resolve_locale()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.2.0
3
+ Version: 1.3.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -33,4 +33,5 @@ tests/test_dbt_project.py
33
33
  tests/test_dbt_tests.py
34
34
  tests/test_faker_name_inference.py
35
35
  tests/test_generation.py
36
- tests/test_release_stress.py
36
+ tests/test_release_stress.py
37
+ tests/test_row_identity.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.2.0"
7
+ version = "1.3.1"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -0,0 +1,414 @@
1
+ """Person-shaped columns in one row have to describe one person.
2
+
3
+ Columns are generated independently, one at a time, so before the identity
4
+ pool a single row's first_name, last_name and email came from three unrelated
5
+ Faker draws. These tests pin the row-level agreement, and the two properties it
6
+ must not cost: reproducibility under a seed, and uniqueness where it is asked
7
+ for.
8
+ """
9
+
10
+ import re
11
+
12
+ import pytest
13
+ from faker import Faker
14
+
15
+ import model2data.generate.faker as faker_module
16
+ from model2data.generate.core import generate_data_from_dbml
17
+ from model2data.generate.faker import (
18
+ DEFAULT_LOCALE,
19
+ _slug,
20
+ current_locale,
21
+ generate_column_values,
22
+ reset_row_pools,
23
+ set_locale,
24
+ )
25
+ from model2data.parse.dbml import ColumnDef, TableDef
26
+
27
+
28
+ def _person_table(name: str = "customers", columns: list[ColumnDef] | None = None) -> TableDef:
29
+ return TableDef(
30
+ name=name,
31
+ columns=columns
32
+ or [
33
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
34
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
35
+ ColumnDef(name="full_name", data_type="varchar", settings={"not null"}),
36
+ ColumnDef(name="user_name", data_type="varchar", settings={"not null"}),
37
+ ColumnDef(name="email", data_type="varchar", settings={"not null"}),
38
+ ],
39
+ )
40
+
41
+
42
+ def _assert_row_is_one_person(row) -> None:
43
+ first, last = row["first_name"], row["last_name"]
44
+ assert row["full_name"] == f"{first} {last}"
45
+ assert row["email"].startswith(f"{_slug(first)}.{_slug(last)}@")
46
+ assert row["user_name"] == f"{_slug(first)[:1]}{_slug(last)}"
47
+
48
+
49
+ class TestRowIdentityCoherence:
50
+ def test_person_columns_in_a_row_describe_one_person(self):
51
+ frames = generate_data_from_dbml(
52
+ tables={"customers": _person_table()}, refs=[], base_rows=25, seed=1
53
+ )
54
+ df = frames["customers"]
55
+ assert len(df) == 25
56
+ for _, row in df.iterrows():
57
+ _assert_row_is_one_person(row)
58
+
59
+ def test_two_tables_draw_from_different_populations(self):
60
+ frames = generate_data_from_dbml(
61
+ tables={
62
+ "customers": _person_table("customers"),
63
+ "employees": _person_table("employees"),
64
+ },
65
+ refs=[],
66
+ base_rows=30,
67
+ seed=2,
68
+ )
69
+ # Asserted on the keying rather than on the values. "These two sets of
70
+ # emails do not overlap" is true here, but only by a probability -- the
71
+ # same shape of luck-based assertion that made the password test fail
72
+ # the first time the RNG moved.
73
+ customers_pool = faker_module._row_pool("person", "customers", 30)
74
+ employees_pool = faker_module._row_pool("person", "employees", 30)
75
+ assert customers_pool is not employees_pool
76
+
77
+ # And the values follow from that: two tables of people are two groups
78
+ # of people, not one pool handed out twice.
79
+ assert list(frames["customers"]["email"]) != list(frames["employees"]["email"])
80
+
81
+ def test_same_seed_reproduces_the_same_people(self):
82
+ def run():
83
+ return generate_data_from_dbml(
84
+ tables={"customers": _person_table()}, refs=[], base_rows=15, seed=7
85
+ )["customers"]
86
+
87
+ first_run, second_run = run(), run()
88
+ assert list(first_run["email"]) == list(second_run["email"])
89
+ assert list(first_run["full_name"]) == list(second_run["full_name"])
90
+
91
+ def test_unique_email_stays_unique_without_swapping_the_person(self):
92
+ # A tiny name space forces collisions, so the suffixing path actually
93
+ # runs rather than being skipped because every value happened to differ.
94
+ table = _person_table(
95
+ columns=[
96
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
97
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
98
+ ColumnDef(name="email", data_type="varchar", settings={"not null", "unique"}),
99
+ ],
100
+ )
101
+ df = generate_data_from_dbml(tables={"customers": table}, refs=[], base_rows=400, seed=3)[
102
+ "customers"
103
+ ]
104
+
105
+ assert df["email"].is_unique
106
+ for _, row in df.iterrows():
107
+ # The suffix may sit before the @, but the row's own name is still
108
+ # what the address is built from.
109
+ local = row["email"].split("@")[0]
110
+ assert re.fullmatch(
111
+ rf"{re.escape(_slug(row['first_name']))}\.{re.escape(_slug(row['last_name']))}\d*",
112
+ local,
113
+ )
114
+
115
+ def test_declared_email_type_also_reaches_the_identity(self):
116
+ table = TableDef(
117
+ name="contacts",
118
+ columns=[
119
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
120
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
121
+ # Declared as a provider rather than named `email`.
122
+ ColumnDef(name="contact", data_type="email", settings={"not null"}),
123
+ ],
124
+ )
125
+ df = generate_data_from_dbml(tables={"contacts": table}, refs=[], base_rows=20, seed=4)[
126
+ "contacts"
127
+ ]
128
+ for _, row in df.iterrows():
129
+ assert row["contact"].startswith(
130
+ f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@"
131
+ )
132
+
133
+ def test_email_declared_as_text_is_still_an_email(self):
134
+ # `email text` means the SQL type; the name inference must still win,
135
+ # and still go through the identity.
136
+ table = TableDef(
137
+ name="people",
138
+ columns=[
139
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
140
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
141
+ ColumnDef(name="email", data_type="text", settings={"not null"}),
142
+ ],
143
+ )
144
+ df = generate_data_from_dbml(tables={"people": table}, refs=[], base_rows=10, seed=5)[
145
+ "people"
146
+ ]
147
+ for _, row in df.iterrows():
148
+ assert row["email"].startswith(f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@")
149
+
150
+ def test_password_column_is_not_identity_derived(self):
151
+ # `password` sits between `username` and `email` in the pattern list, so
152
+ # it is the entry most at risk of having been swept into the identity
153
+ # work. Asserted against the pattern table rather than the output: an
154
+ # earlier version of this test looked for "@" in the generated password,
155
+ # which passed only by luck -- Faker's passwords include punctuation, "@"
156
+ # among it, so the test failed the first time the RNG moved.
157
+ assert isinstance(faker_module._infer_by_name("email"), faker_module._FromRow)
158
+ assert not isinstance(faker_module._infer_by_name("password"), faker_module._FromRow)
159
+
160
+ table = TableDef(
161
+ name="users",
162
+ columns=[
163
+ ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
164
+ ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
165
+ ColumnDef(name="email", data_type="varchar", settings={"not null"}),
166
+ ColumnDef(name="password", data_type="varchar", settings={"not null"}),
167
+ ],
168
+ )
169
+ df = generate_data_from_dbml(tables={"users": table}, refs=[], base_rows=20, seed=21)[
170
+ "users"
171
+ ]
172
+ assert df["password"].nunique() == 20
173
+ for _, row in df.iterrows():
174
+ # Exact comparison, not a substring heuristic: the password must not
175
+ # be one of the identity's own values.
176
+ assert row["password"] not in {row["first_name"], row["last_name"], row["email"]}
177
+
178
+ def test_non_person_columns_are_untouched(self):
179
+ reset_row_pools()
180
+ cities = generate_column_values(
181
+ ColumnDef(name="city", data_type="varchar", settings={"not null"}),
182
+ row_count=10,
183
+ table_name="places",
184
+ )
185
+ assert len(cities) == 10
186
+ assert all(isinstance(city, str) and city for city in cities)
187
+
188
+
189
+ @pytest.fixture(autouse=True)
190
+ def _restore_locale():
191
+ """Locale is process-wide state, so a test that changes it must not leak."""
192
+ yield
193
+ set_locale(None)
194
+ # set_locale returns early when the locale is already the default, which
195
+ # would leave the cached per-locale constants behind for any test that
196
+ # stubbed `fake` rather than switching locale. Re-resolve unconditionally.
197
+ faker_module._resolve_locale()
198
+
199
+
200
+ def _address_table(name: str = "sites") -> TableDef:
201
+ return TableDef(
202
+ name=name,
203
+ columns=[
204
+ ColumnDef(name="street", data_type="varchar", settings={"not null"}),
205
+ ColumnDef(name="address", data_type="varchar", settings={"not null"}),
206
+ ColumnDef(name="city", data_type="varchar", settings={"not null"}),
207
+ ColumnDef(name="state", data_type="varchar", settings={"not null"}),
208
+ ColumnDef(name="country", data_type="varchar", settings={"not null"}),
209
+ ColumnDef(name="zip", data_type="varchar", settings={"not null"}),
210
+ ],
211
+ )
212
+
213
+
214
+ class TestAddressCoherence:
215
+ def test_address_column_is_built_from_the_row_beside_it(self):
216
+ df = generate_data_from_dbml(
217
+ tables={"sites": _address_table()}, refs=[], base_rows=20, seed=11
218
+ )["sites"]
219
+ for _, row in df.iterrows():
220
+ # The composed address has to be this row's street, city and
221
+ # postcode -- not a separate draw that happens to look like one.
222
+ assert row["street"] in row["address"]
223
+ assert row["city"] in row["address"]
224
+ assert row["zip"] in row["address"]
225
+
226
+ def test_every_row_is_in_one_country(self):
227
+ df = generate_data_from_dbml(
228
+ tables={"sites": _address_table()}, refs=[], base_rows=30, seed=12
229
+ )["sites"]
230
+ # The old behaviour drew country independently, so a US street could sit
231
+ # in Japan. One locale means one country, on every row.
232
+ assert set(df["country"]) == {"United States"}
233
+
234
+ def test_declared_city_type_agrees_with_the_row(self):
235
+ table = TableDef(
236
+ name="offices",
237
+ columns=[
238
+ ColumnDef(name="address", data_type="varchar", settings={"not null"}),
239
+ # Declared as a provider rather than named `city`.
240
+ ColumnDef(name="located_in", data_type="city", settings={"not null"}),
241
+ ],
242
+ )
243
+ df = generate_data_from_dbml(tables={"offices": table}, refs=[], base_rows=15, seed=13)[
244
+ "offices"
245
+ ]
246
+ for _, row in df.iterrows():
247
+ assert row["located_in"] in row["address"]
248
+
249
+
250
+ class TestLocale:
251
+ def test_default_locale_is_the_documented_one(self):
252
+ assert current_locale() == DEFAULT_LOCALE
253
+
254
+ def test_locale_changes_the_country_and_the_addresses(self):
255
+ def country_and_cities(locale):
256
+ df = generate_data_from_dbml(
257
+ tables={"sites": _address_table()},
258
+ refs=[],
259
+ base_rows=20,
260
+ seed=14,
261
+ locale=locale,
262
+ )["sites"]
263
+ return set(df["country"]), set(df["city"])
264
+
265
+ us_country, us_cities = country_and_cities(None)
266
+ be_country, be_cities = country_and_cities("nl_BE")
267
+
268
+ assert us_country == {"United States"}
269
+ assert be_country == {"Belgium"}
270
+ assert not us_cities & be_cities
271
+
272
+ def test_locale_applies_to_people_too(self):
273
+ df = generate_data_from_dbml(
274
+ tables={"customers": _person_table()},
275
+ refs=[],
276
+ base_rows=20,
277
+ seed=15,
278
+ locale="fr_FR",
279
+ )["customers"]
280
+ # Still coherent, just French: the pool swap must not break the
281
+ # first.last@domain derivation.
282
+ for _, row in df.iterrows():
283
+ _assert_row_is_one_person(row)
284
+
285
+ def test_locale_is_restored_between_runs(self):
286
+ generate_data_from_dbml(
287
+ tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16, locale="nl_BE"
288
+ )
289
+ assert current_locale() == "nl_BE"
290
+ # Passing nothing means the default, not "whatever the last run used".
291
+ generate_data_from_dbml(tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16)
292
+ assert current_locale() == DEFAULT_LOCALE
293
+
294
+ def test_unknown_locale_says_so_plainly(self):
295
+ with pytest.raises(ValueError, match="Unknown locale 'zz_ZZ'"):
296
+ generate_data_from_dbml(
297
+ tables={"sites": _address_table()}, refs=[], base_rows=2, locale="zz_ZZ"
298
+ )
299
+
300
+
301
+ class TestPoolLifetime:
302
+ def test_pools_are_released_once_a_table_is_finished(self):
303
+ # A million-row table is a normal request; holding its identities for
304
+ # the rest of the run is what makes that unaffordable.
305
+ generate_data_from_dbml(
306
+ tables={"customers": _person_table(), "sites": _address_table()},
307
+ refs=[],
308
+ base_rows=50,
309
+ seed=17,
310
+ )
311
+ assert faker_module._person_state == {}
312
+ assert faker_module._address_state == {}
313
+
314
+ def test_pool_rows_stay_small(self):
315
+ # Slots rather than a per-instance __dict__. Worth a test because the
316
+ # difference is ~6x per row (344 bytes vs 56), the derived fields are
317
+ # properties precisely so they are not stored, and a million-row person
318
+ # table is a normal request -- an innocent-looking field added here
319
+ # would cost hundreds of megabytes on one.
320
+ assert not hasattr(faker_module._new_person(), "__dict__")
321
+ assert not hasattr(faker_module._new_address(), "__dict__")
322
+
323
+
324
+ class _LocaleMissing:
325
+ """A Faker with named providers removed, standing in for a thinner locale.
326
+
327
+ Every locale Faker actually ships has `current_country` and at least one
328
+ administrative unit, so the fallbacks for a locale without them cannot be
329
+ reached by naming a real one -- but they still decide what lands in a
330
+ column, and the docstrings make a claim about what they do. This stands in
331
+ for the locale that would reach them.
332
+ """
333
+
334
+ def __init__(self, inner, missing):
335
+ self._inner = inner
336
+ self._missing = set(missing)
337
+
338
+ def format(self, name, *args, **kwargs):
339
+ if name in self._missing:
340
+ raise AttributeError(name)
341
+ return self._inner.format(name, *args, **kwargs)
342
+
343
+ def __getattr__(self, name):
344
+ if name in self._missing:
345
+ raise AttributeError(name)
346
+ return getattr(self._inner, name)
347
+
348
+
349
+ class TestThinLocales:
350
+ def test_no_administrative_unit_leaves_state_empty(self, monkeypatch):
351
+ stub = _LocaleMissing(
352
+ Faker("en_US"), {"state", "province", "administrative_unit", "region"}
353
+ )
354
+ monkeypatch.setattr(faker_module, "fake", stub)
355
+ faker_module._resolve_locale()
356
+
357
+ address = faker_module._new_address()
358
+ # Empty is the honest answer for a country with no region worth naming.
359
+ # Inventing one just to fill the column would be worse data, not better.
360
+ assert address.state == ""
361
+ assert address.city and address.street and address.country
362
+
363
+ def test_no_current_country_still_puts_every_row_in_one_country(self, monkeypatch):
364
+ stub = _LocaleMissing(Faker("en_US"), {"current_country"})
365
+ monkeypatch.setattr(faker_module, "fake", stub)
366
+ faker_module._resolve_locale()
367
+
368
+ countries = {faker_module._new_address().country for _ in range(10)}
369
+ # The point of the fallback is agreement, not accuracy: one country
370
+ # picked once beats a different random country on every row.
371
+ assert len(countries) == 1
372
+ assert countries != {""}
373
+
374
+
375
+ class TestNameFolding:
376
+ def test_accents_are_folded_not_dropped(self):
377
+ # Dropping them turned Aimée into "aime" and Müller into "mller" -- not
378
+ # that person's name, and most visible in exactly the European locales
379
+ # the locale option exists to serve.
380
+ assert faker_module._slug("Aimée") == "aimee"
381
+ assert faker_module._slug("Müller") == "muller"
382
+ assert faker_module._slug("Björn") == "bjorn"
383
+
384
+ def test_letters_that_do_not_decompose_are_mapped(self):
385
+ # NFKD leaves these intact because they are their own letters, not a
386
+ # base plus an accent, so they would vanish with the combining marks.
387
+ assert faker_module._slug("Søren") == "soren"
388
+ assert faker_module._slug("Weiß") == "weiss"
389
+ assert faker_module._slug("Łukasz") == "lukasz"
390
+ assert faker_module._slug("Æther") == "aether"
391
+
392
+ def test_punctuation_and_spacing_still_go(self):
393
+ assert faker_module._slug("O'Brien") == "obrien"
394
+ assert faker_module._slug("Van Der Berg") == "vanderberg"
395
+
396
+ def test_a_name_with_no_latin_letters_falls_back(self):
397
+ # Documented limitation rather than a target: deriving an address from a
398
+ # CJK name needs romanization, and Faker exposes no romanized first/last
399
+ # pair to build one from. Such a locale gets "user", disambiguated by the
400
+ # unique suffixing, which is meaningless but at least stable and unique.
401
+ assert faker_module._slug("日本") == "user"
402
+
403
+ def test_generated_emails_are_ascii_in_an_accented_locale(self):
404
+ df = generate_data_from_dbml(
405
+ tables={"customers": _person_table()},
406
+ refs=[],
407
+ base_rows=40,
408
+ seed=31,
409
+ locale="fr_FR",
410
+ )["customers"]
411
+ for _, row in df.iterrows():
412
+ _assert_row_is_one_person(row)
413
+ # An address has to be usable as an address, whatever the name says.
414
+ row["email"].encode("ascii")
@@ -1,3 +0,0 @@
1
- """Model2Data: Generate analytics-ready datasets from DBML models."""
2
-
3
- __version__ = "0.1.1"
File without changes
File without changes
File without changes
File without changes
File without changes