model2data 1.2.0__tar.gz → 1.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.2.0/model2data.egg-info → model2data-1.3.1}/PKG-INFO +1 -1
- model2data-1.3.1/model2data/__init__.py +12 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/cli.py +11 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/core.py +18 -2
- {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/faker.py +344 -21
- {model2data-1.2.0 → model2data-1.3.1/model2data.egg-info}/PKG-INFO +1 -1
- {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/SOURCES.txt +2 -1
- {model2data-1.2.0 → model2data-1.3.1}/pyproject.toml +1 -1
- model2data-1.3.1/tests/test_row_identity.py +414 -0
- model2data-1.2.0/model2data/__init__.py +0 -3
- {model2data-1.2.0 → model2data-1.3.1}/LICENSE +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/README.md +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/README_PYPI.md +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/__init__.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/project.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/tests.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/__init__.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/generate/relationships.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/parse/__init__.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/parse/dbml.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data/utils.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/setup.cfg +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_cli.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbml_parser.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_integration.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_naming.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_project.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_dbt_tests.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_faker_name_inference.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_generation.py +0 -0
- {model2data-1.2.0 → model2data-1.3.1}/tests/test_release_stress.py +0 -0
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Model2Data: Generate analytics-ready datasets from DBML models."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
# Read the installed distribution's version rather than repeating it here.
|
|
7
|
+
# The literal that used to live in this file said 0.1.1 for every release up
|
|
8
|
+
# to and including 1.2.0: nothing reads `__version__`, so nothing caught it
|
|
9
|
+
# drifting. Deriving it means it cannot drift again.
|
|
10
|
+
__version__ = version("model2data")
|
|
11
|
+
except PackageNotFoundError: # pragma: no cover - running from a source tree
|
|
12
|
+
__version__ = "0.0.0.dev0"
|
|
@@ -18,6 +18,7 @@ from model2data.generate.core import (
|
|
|
18
18
|
get_unresolved_composite_keys,
|
|
19
19
|
)
|
|
20
20
|
from model2data.generate.faker import (
|
|
21
|
+
DEFAULT_LOCALE,
|
|
21
22
|
get_duplicate_unique_columns,
|
|
22
23
|
get_unmapped_columns,
|
|
23
24
|
reset_stats,
|
|
@@ -125,6 +126,15 @@ def main(
|
|
|
125
126
|
"Using the same seed will always produce identical datasets."
|
|
126
127
|
),
|
|
127
128
|
),
|
|
129
|
+
locale: Optional[str] = typer.Option(
|
|
130
|
+
None,
|
|
131
|
+
"--locale",
|
|
132
|
+
help=(
|
|
133
|
+
f"Faker locale for generated people and addresses "
|
|
134
|
+
f"(default: {DEFAULT_LOCALE}).\n"
|
|
135
|
+
"Examples: en_GB, nl_BE, fr_FR, de_DE."
|
|
136
|
+
),
|
|
137
|
+
),
|
|
128
138
|
name: Optional[str] = typer.Option(
|
|
129
139
|
None,
|
|
130
140
|
"--name",
|
|
@@ -214,6 +224,7 @@ def main(
|
|
|
214
224
|
base_rows=rows,
|
|
215
225
|
seed=seed,
|
|
216
226
|
row_overrides=row_overrides,
|
|
227
|
+
locale=locale,
|
|
217
228
|
)
|
|
218
229
|
|
|
219
230
|
# -------------------------
|
|
@@ -10,7 +10,10 @@ from faker import Faker
|
|
|
10
10
|
|
|
11
11
|
from model2data.generate.faker import (
|
|
12
12
|
generate_column_values,
|
|
13
|
+
release_row_pools,
|
|
13
14
|
reset_duplicate_unique_columns,
|
|
15
|
+
reset_row_pools,
|
|
16
|
+
set_locale,
|
|
14
17
|
)
|
|
15
18
|
from model2data.generate.relationships import (
|
|
16
19
|
build_fk_lookup,
|
|
@@ -18,8 +21,6 @@ from model2data.generate.relationships import (
|
|
|
18
21
|
)
|
|
19
22
|
from model2data.parse.dbml import TableDef
|
|
20
23
|
|
|
21
|
-
fake = Faker()
|
|
22
|
-
|
|
23
24
|
# Tables the most recent generate_data_from_dbml() call found stuck in an
|
|
24
25
|
# unresolved FK cycle (never reached indegree 0 during the topological
|
|
25
26
|
# sort). Exposed out-of-band, mirroring generate.faker's
|
|
@@ -66,6 +67,7 @@ def generate_data_from_dbml(
|
|
|
66
67
|
base_rows: int = 100,
|
|
67
68
|
seed: Optional[int] = None,
|
|
68
69
|
row_overrides: Optional[Mapping[str, int]] = None,
|
|
70
|
+
locale: Optional[str] = None,
|
|
69
71
|
) -> dict[str, pd.DataFrame]:
|
|
70
72
|
"""
|
|
71
73
|
Generate synthetic datasets from parsed DBML definitions.
|
|
@@ -78,9 +80,19 @@ def generate_data_from_dbml(
|
|
|
78
80
|
present in `row_overrides` fall back to `base_rows`; unknown names are
|
|
79
81
|
ignored.
|
|
80
82
|
|
|
83
|
+
`locale` picks the Faker locale every generated person and address is drawn
|
|
84
|
+
from -- `"nl_BE"`, `"fr_FR"`, `"en_GB"` -- defaulting to `DEFAULT_LOCALE`.
|
|
85
|
+
It is a per-run setting rather than a per-column one on purpose: a table
|
|
86
|
+
holding one Belgian and one American address is the incoherence the row
|
|
87
|
+
pools exist to remove.
|
|
88
|
+
|
|
81
89
|
This function is deterministic if a seed is provided.
|
|
82
90
|
It performs no filesystem I/O and returns pandas DataFrames.
|
|
83
91
|
"""
|
|
92
|
+
# Locale first, then the seed: switching locale builds a new Faker, and the
|
|
93
|
+
# seed has to be the last word on the generator that actually runs.
|
|
94
|
+
set_locale(locale)
|
|
95
|
+
|
|
84
96
|
if seed is not None:
|
|
85
97
|
random.seed(seed)
|
|
86
98
|
Faker.seed(seed)
|
|
@@ -88,6 +100,7 @@ def generate_data_from_dbml(
|
|
|
88
100
|
reset_cycle_state()
|
|
89
101
|
reset_dedup_state()
|
|
90
102
|
reset_duplicate_unique_columns()
|
|
103
|
+
reset_row_pools()
|
|
91
104
|
|
|
92
105
|
# ---------------------------------------------------------
|
|
93
106
|
# Classify references
|
|
@@ -193,6 +206,9 @@ def generate_data_from_dbml(
|
|
|
193
206
|
|
|
194
207
|
df = _coerce_integer_dtypes(df, table_def)
|
|
195
208
|
generated[table_name] = df
|
|
209
|
+
# This table is finished: nothing will read its people or addresses
|
|
210
|
+
# again, and on a million-row table they are worth tens of megabytes.
|
|
211
|
+
release_row_pools(table_name)
|
|
196
212
|
|
|
197
213
|
return generated
|
|
198
214
|
|
|
@@ -2,16 +2,278 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import random
|
|
4
4
|
import re
|
|
5
|
+
import unicodedata
|
|
5
6
|
import uuid
|
|
7
|
+
from dataclasses import dataclass
|
|
6
8
|
from datetime import datetime, timedelta
|
|
7
|
-
from typing import Callable, Optional
|
|
9
|
+
from typing import Callable, Optional, Union
|
|
8
10
|
|
|
9
11
|
import pandas as pd
|
|
10
12
|
from faker import Faker
|
|
11
13
|
|
|
12
14
|
from model2data.parse.dbml import ColumnDef
|
|
13
15
|
|
|
14
|
-
|
|
16
|
+
# ---------------------------------------------------------
|
|
17
|
+
# Locale
|
|
18
|
+
# ---------------------------------------------------------
|
|
19
|
+
# The locale every generated person and address comes from until a caller says
|
|
20
|
+
# otherwise. Named rather than left implicit because locale is on its way to
|
|
21
|
+
# being a first-class option in the CLI and the studio, and this is the seam it
|
|
22
|
+
# plugs into: one instance, swapped in one place.
|
|
23
|
+
DEFAULT_LOCALE = "en_US"
|
|
24
|
+
|
|
25
|
+
_locale = DEFAULT_LOCALE
|
|
26
|
+
fake = Faker(DEFAULT_LOCALE)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def current_locale() -> str:
|
|
30
|
+
"""The locale generation is currently drawing from."""
|
|
31
|
+
return _locale
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def set_locale(locale: Optional[str]) -> None:
|
|
35
|
+
"""Point generation at a locale, or back at the default when given None.
|
|
36
|
+
|
|
37
|
+
Rebinding the module-level `fake` reaches every provider here, because each
|
|
38
|
+
one looks the name up when it runs rather than capturing it. The row pools
|
|
39
|
+
are dropped on the way through: a Belgian address sitting next to an
|
|
40
|
+
American one in the same table is precisely the incoherence they exist to
|
|
41
|
+
prevent.
|
|
42
|
+
"""
|
|
43
|
+
global fake, _locale
|
|
44
|
+
target = locale or DEFAULT_LOCALE
|
|
45
|
+
if target == _locale:
|
|
46
|
+
return
|
|
47
|
+
try:
|
|
48
|
+
fake = Faker(target)
|
|
49
|
+
except (AttributeError, ValueError) as exc:
|
|
50
|
+
# Faker's own message for a bad locale names the attribute it failed to
|
|
51
|
+
# find, which reads like an internal error rather than a typo in a flag.
|
|
52
|
+
raise ValueError(
|
|
53
|
+
f"Unknown locale {target!r}. Use a Faker locale name such as "
|
|
54
|
+
f"'en_US', 'en_GB', 'nl_BE' or 'fr_FR'."
|
|
55
|
+
) from exc
|
|
56
|
+
_locale = target
|
|
57
|
+
_resolve_locale()
|
|
58
|
+
_person_state.clear()
|
|
59
|
+
_address_state.clear()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# ---------------------------------------------------------
|
|
63
|
+
# Per-row identities
|
|
64
|
+
# ---------------------------------------------------------
|
|
65
|
+
# Columns are generated one at a time, so nothing connected the `first_name`,
|
|
66
|
+
# `last_name` and `email` of a single row: each drew from Faker independently
|
|
67
|
+
# and one row described three different people. Anyone who points a BI tool at
|
|
68
|
+
# the output sees it immediately, which makes it a credibility problem rather
|
|
69
|
+
# than a cosmetic one.
|
|
70
|
+
#
|
|
71
|
+
# The fix is a per-table pool of identities, one per row index. A column whose
|
|
72
|
+
# name (or declared type) means "a person's email" reads row i's identity
|
|
73
|
+
# instead of rolling its own, so every person-shaped column in a row agrees.
|
|
74
|
+
#
|
|
75
|
+
# Addresses get the same treatment, one pool along. What that can and cannot
|
|
76
|
+
# promise is worth being precise about: every component now comes from the same
|
|
77
|
+
# locale, so a row reads as one country with one set of conventions instead of
|
|
78
|
+
# "Brussels, Texas, 3000, Japan". It is not real geography -- Faker does not
|
|
79
|
+
# pair a city with its state or its postcode even inside a locale, so the
|
|
80
|
+
# postcode is a plausible postcode for that country rather than that city's.
|
|
81
|
+
# Closing that last gap needs a reference table of real combinations, not a
|
|
82
|
+
# cleverer arrangement of Faker calls.
|
|
83
|
+
#
|
|
84
|
+
# Both classes carry only what is drawn and compute the rest, and both use
|
|
85
|
+
# slots. A million-row `person` table is a normal request: storing five strings
|
|
86
|
+
# per row in a dict-backed object costs hundreds of megabytes, storing three
|
|
87
|
+
# slotted references costs tens, and the derived strings then exist only for
|
|
88
|
+
# the columns a schema actually declares.
|
|
89
|
+
@dataclass(frozen=True, slots=True)
|
|
90
|
+
class _Person:
|
|
91
|
+
first_name: str
|
|
92
|
+
last_name: str
|
|
93
|
+
email_domain: str
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def full_name(self) -> str:
|
|
97
|
+
return f"{self.first_name} {self.last_name}"
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def user_name(self) -> str:
|
|
101
|
+
return f"{_slug(self.first_name)[:1]}{_slug(self.last_name)}"
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def email(self) -> str:
|
|
105
|
+
return f"{_slug(self.first_name)}.{_slug(self.last_name)}@{self.email_domain}"
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass(frozen=True, slots=True)
|
|
109
|
+
class _Address:
|
|
110
|
+
street: str
|
|
111
|
+
city: str
|
|
112
|
+
state: str
|
|
113
|
+
country: str
|
|
114
|
+
postcode: str
|
|
115
|
+
|
|
116
|
+
@property
|
|
117
|
+
def full(self) -> str:
|
|
118
|
+
"""The row's own components, not a separate `fake.address()` draw.
|
|
119
|
+
|
|
120
|
+
Composing it here rather than calling Faker again is the whole point:
|
|
121
|
+
an `address` column has to agree with the `city` column beside it.
|
|
122
|
+
"""
|
|
123
|
+
lines = [self.street, f"{self.postcode} {self.city}".strip(), self.country]
|
|
124
|
+
return ", ".join(line for line in lines if line)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class _FromRow:
|
|
128
|
+
"""Marker for a provider that reads row i's person or address.
|
|
129
|
+
|
|
130
|
+
A sentinel rather than a callable so `_NAME_PATTERNS` can stay a single
|
|
131
|
+
ordered list: splitting these patterns into lists of their own would quietly
|
|
132
|
+
reorder them against the rest, and that order is load-bearing (see the
|
|
133
|
+
comment on `_NAME_PATTERNS`).
|
|
134
|
+
"""
|
|
135
|
+
|
|
136
|
+
__slots__ = ("pool", "field")
|
|
137
|
+
|
|
138
|
+
def __init__(self, pool: str, field: str) -> None:
|
|
139
|
+
self.pool = pool
|
|
140
|
+
self.field = field
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
_Provider = Union[Callable[[], object], _FromRow]
|
|
144
|
+
|
|
145
|
+
# One pool per table, keyed by table name, grown on demand and released as soon
|
|
146
|
+
# as that table's frame is finished.
|
|
147
|
+
_person_state: dict[str, list[_Person]] = {}
|
|
148
|
+
_address_state: dict[str, list[_Address]] = {}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
# Letters that NFKD does not take apart, because they are their own letters
|
|
152
|
+
# rather than a base plus an accent. Without these, "ø" and "ß" would simply
|
|
153
|
+
# vanish along with the accents.
|
|
154
|
+
_UNDECOMPOSED_LETTERS = str.maketrans(
|
|
155
|
+
{"ø": "o", "æ": "ae", "œ": "oe", "ß": "ss", "ł": "l", "đ": "d", "ð": "d", "þ": "th", "ı": "i"}
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _slug(value: str) -> str:
|
|
160
|
+
"""Reduce a name to something that can sit inside an email or a username.
|
|
161
|
+
|
|
162
|
+
Accents are folded, not dropped. Stripping them outright turned `Aimée` into
|
|
163
|
+
`aime` and `Müller` into `mller` -- not that person's name, and conspicuously
|
|
164
|
+
broken in exactly the European locales the locale option exists to serve.
|
|
165
|
+
NFKD splits most accented letters into a base letter plus a combining mark,
|
|
166
|
+
which encoding to ASCII then discards; the letters that do not decompose are
|
|
167
|
+
mapped first.
|
|
168
|
+
"""
|
|
169
|
+
folded = value.lower().translate(_UNDECOMPOSED_LETTERS)
|
|
170
|
+
ascii_only = unicodedata.normalize("NFKD", folded).encode("ascii", "ignore").decode("ascii")
|
|
171
|
+
return re.sub(r"[^a-z0-9]+", "", ascii_only) or "user"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# Resolved once per locale, not once per row. Both of these are constant for a
|
|
175
|
+
# locale, and a million-row table makes the difference stark: `current_country`
|
|
176
|
+
# would be recomputed a million times for an answer that never changes, and
|
|
177
|
+
# probing for an administrative-unit provider means catching AttributeError --
|
|
178
|
+
# on a locale that has none, four raised exceptions per row.
|
|
179
|
+
_country_name: str = ""
|
|
180
|
+
_state_provider: Optional[str] = None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _first_provider(*names: str) -> Optional[str]:
|
|
184
|
+
"""The first of these provider names this locale actually has, else None.
|
|
185
|
+
|
|
186
|
+
Locales disagree about what exists: `state` is American, `province` is
|
|
187
|
+
Belgian, and plenty of countries have no administrative unit worth naming.
|
|
188
|
+
None is the honest answer there -- better than inventing a region the
|
|
189
|
+
country does not have, purely so a column looks full.
|
|
190
|
+
"""
|
|
191
|
+
for name in names:
|
|
192
|
+
try:
|
|
193
|
+
fake.format(name)
|
|
194
|
+
except (AttributeError, TypeError, ValueError):
|
|
195
|
+
continue
|
|
196
|
+
return name
|
|
197
|
+
return None
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _resolve_locale() -> None:
|
|
201
|
+
"""Cache the per-locale constants the address pool reads on every row."""
|
|
202
|
+
global _country_name, _state_provider
|
|
203
|
+
# `current_country` is the locale's own country, and it is what stops a US
|
|
204
|
+
# street from landing in Japan. Locales too generic to have one (plain "en")
|
|
205
|
+
# fall back to a single country picked once, so at least every row agrees.
|
|
206
|
+
if _first_provider("current_country"):
|
|
207
|
+
_country_name = str(fake.current_country())
|
|
208
|
+
else:
|
|
209
|
+
_country_name = fake.country()
|
|
210
|
+
_state_provider = _first_provider("state", "province", "administrative_unit", "region")
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _new_person() -> _Person:
|
|
214
|
+
return _Person(
|
|
215
|
+
first_name=fake.first_name(),
|
|
216
|
+
last_name=fake.last_name(),
|
|
217
|
+
email_domain=fake.free_email_domain(),
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _new_address() -> _Address:
|
|
222
|
+
return _Address(
|
|
223
|
+
street=fake.street_address(),
|
|
224
|
+
city=fake.city(),
|
|
225
|
+
state=str(fake.format(_state_provider)) if _state_provider else "",
|
|
226
|
+
country=_country_name,
|
|
227
|
+
postcode=fake.postcode(),
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def reset_row_pools() -> None:
|
|
232
|
+
"""Drop every table's person and address pool.
|
|
233
|
+
|
|
234
|
+
Called per run by generate_data_from_dbml alongside the other per-run state.
|
|
235
|
+
Without it a second run in the same process reuses the first run's people,
|
|
236
|
+
which looks harmless right up until a seeded run stops reproducing.
|
|
237
|
+
"""
|
|
238
|
+
_person_state.clear()
|
|
239
|
+
_address_state.clear()
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def release_row_pools(table_name: str) -> None:
|
|
243
|
+
"""Drop one table's pools once its frame is finished.
|
|
244
|
+
|
|
245
|
+
A pool only has to outlive the columns of its own table. Holding every
|
|
246
|
+
table's pool until the end of the run means a schema of twenty million-row
|
|
247
|
+
tables carries twenty million identities nothing will read again; releasing
|
|
248
|
+
per table bounds the cost to the largest single table instead of the sum.
|
|
249
|
+
"""
|
|
250
|
+
_person_state.pop(table_name, None)
|
|
251
|
+
_address_state.pop(table_name, None)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _row_pool(pool: str, table_name: Optional[str], row_count: int) -> list:
|
|
255
|
+
"""Row i's person or address for this table, created and cached as needed.
|
|
256
|
+
|
|
257
|
+
Keyed by table so two tables of people hold two different populations, and
|
|
258
|
+
grown rather than rebuilt so every column of the same table sees the same
|
|
259
|
+
row i. Callers with no table name (core's composite-key repair, which
|
|
260
|
+
regenerates a single cell) share one bucket; that path only ever touches key
|
|
261
|
+
columns, never person or address ones.
|
|
262
|
+
"""
|
|
263
|
+
key = table_name or ""
|
|
264
|
+
# Written out per pool rather than shared behind a generic helper: the two
|
|
265
|
+
# loops are three lines each, and pairing the right factory with the right
|
|
266
|
+
# pool is exactly the thing a reader (and a type checker) wants to see.
|
|
267
|
+
if pool == "person":
|
|
268
|
+
people = _person_state.setdefault(key, [])
|
|
269
|
+
while len(people) < row_count:
|
|
270
|
+
people.append(_new_person())
|
|
271
|
+
return people
|
|
272
|
+
|
|
273
|
+
addresses = _address_state.setdefault(key, [])
|
|
274
|
+
while len(addresses) < row_count:
|
|
275
|
+
addresses.append(_new_address())
|
|
276
|
+
return addresses
|
|
15
277
|
|
|
16
278
|
|
|
17
279
|
# ---------------------------------------------------------
|
|
@@ -21,25 +283,25 @@ fake = Faker()
|
|
|
21
283
|
# "first_name" before any looser pattern gets a chance. There is
|
|
22
284
|
# deliberately no generic "name" pattern, since "product_name" or
|
|
23
285
|
# "company_name" would otherwise be filled with a person's name.
|
|
24
|
-
_NAME_PATTERNS: list[tuple[str,
|
|
25
|
-
("first_name",
|
|
26
|
-
("last_name",
|
|
27
|
-
("full_name",
|
|
28
|
-
("user_name",
|
|
29
|
-
("username",
|
|
286
|
+
_NAME_PATTERNS: list[tuple[str, _Provider]] = [
|
|
287
|
+
("first_name", _FromRow("person", "first_name")),
|
|
288
|
+
("last_name", _FromRow("person", "last_name")),
|
|
289
|
+
("full_name", _FromRow("person", "full_name")),
|
|
290
|
+
("user_name", _FromRow("person", "user_name")),
|
|
291
|
+
("username", _FromRow("person", "user_name")),
|
|
30
292
|
("password", lambda: fake.password()),
|
|
31
|
-
("email",
|
|
293
|
+
("email", _FromRow("person", "email")),
|
|
32
294
|
("phone", lambda: fake.phone_number()),
|
|
33
295
|
("mobile", lambda: fake.phone_number()),
|
|
34
296
|
("fax", lambda: fake.phone_number()),
|
|
35
|
-
("street",
|
|
36
|
-
("address",
|
|
37
|
-
("city",
|
|
38
|
-
("province",
|
|
39
|
-
("state",
|
|
40
|
-
("country",
|
|
41
|
-
("zip",
|
|
42
|
-
("postal",
|
|
297
|
+
("street", _FromRow("address", "street")),
|
|
298
|
+
("address", _FromRow("address", "full")),
|
|
299
|
+
("city", _FromRow("address", "city")),
|
|
300
|
+
("province", _FromRow("address", "state")),
|
|
301
|
+
("state", _FromRow("address", "state")),
|
|
302
|
+
("country", _FromRow("address", "country")),
|
|
303
|
+
("zip", _FromRow("address", "postcode")),
|
|
304
|
+
("postal", _FromRow("address", "postcode")),
|
|
43
305
|
("homepage", lambda: fake.url()),
|
|
44
306
|
("website", lambda: fake.url()),
|
|
45
307
|
("url", lambda: fake.url()),
|
|
@@ -141,7 +403,7 @@ def get_duplicate_unique_columns() -> list[str]:
|
|
|
141
403
|
return list(_duplicate_unique_columns)
|
|
142
404
|
|
|
143
405
|
|
|
144
|
-
def _infer_by_name(column_name: str) -> Optional[
|
|
406
|
+
def _infer_by_name(column_name: str) -> Optional[_Provider]:
|
|
145
407
|
normalized = re.sub(r"[^a-z0-9]+", "_", column_name.lower())
|
|
146
408
|
padded = f"_{normalized}_"
|
|
147
409
|
for pattern, generator in _NAME_PATTERNS:
|
|
@@ -156,8 +418,27 @@ def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
|
|
|
156
418
|
# declaration in any schema would stop generating emails.
|
|
157
419
|
_SQL_TYPES_SHADOWING_A_PROVIDER = frozenset({"text", "json", "jsonb", "xml", "binary", "year"})
|
|
158
420
|
|
|
159
|
-
|
|
160
|
-
|
|
421
|
+
# Declared types that name a person or address field. `contact email` says
|
|
422
|
+
# exactly what a column called `email` says, so it has to reach the same row --
|
|
423
|
+
# otherwise row-level coherence has a second door it does not cover.
|
|
424
|
+
_ROW_TYPE_FIELDS = {
|
|
425
|
+
"first_name": ("person", "first_name"),
|
|
426
|
+
"last_name": ("person", "last_name"),
|
|
427
|
+
"name": ("person", "full_name"),
|
|
428
|
+
"user_name": ("person", "user_name"),
|
|
429
|
+
"username": ("person", "user_name"),
|
|
430
|
+
"email": ("person", "email"),
|
|
431
|
+
"address": ("address", "full"),
|
|
432
|
+
"street_address": ("address", "street"),
|
|
433
|
+
"city": ("address", "city"),
|
|
434
|
+
"state": ("address", "state"),
|
|
435
|
+
"province": ("address", "state"),
|
|
436
|
+
"country": ("address", "country"),
|
|
437
|
+
"postcode": ("address", "postcode"),
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _infer_by_type(base_type: str) -> Optional[_Provider]:
|
|
161
442
|
"""The provider a column's declared type names, if it names one deliberately.
|
|
162
443
|
|
|
163
444
|
`sku ean13` and `home_state state` are the user saying which generator they
|
|
@@ -169,6 +450,9 @@ def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
|
|
|
169
450
|
"""
|
|
170
451
|
if base_type in _SQL_TYPES_SHADOWING_A_PROVIDER:
|
|
171
452
|
return None
|
|
453
|
+
row_field = _ROW_TYPE_FIELDS.get(base_type)
|
|
454
|
+
if row_field is not None:
|
|
455
|
+
return _FromRow(*row_field)
|
|
172
456
|
try:
|
|
173
457
|
fake.format(base_type)
|
|
174
458
|
except (AttributeError, TypeError):
|
|
@@ -311,7 +595,12 @@ def generate_column_values(
|
|
|
311
595
|
# -----------------------------------------------------
|
|
312
596
|
else:
|
|
313
597
|
generator = _infer_by_type(base_type) or _infer_by_name(column.name)
|
|
314
|
-
if generator
|
|
598
|
+
if isinstance(generator, _FromRow):
|
|
599
|
+
rows = _row_pool(generator.pool, table_name, row_count)
|
|
600
|
+
values = [getattr(rows[index], generator.field) for index in range(row_count)]
|
|
601
|
+
if ensure_unique:
|
|
602
|
+
values = _deduplicate_identity(values)
|
|
603
|
+
elif generator is not None:
|
|
315
604
|
values = [generator() for _ in range(row_count)]
|
|
316
605
|
values = (
|
|
317
606
|
_deduplicate(values, generator, column_name=unique_label)
|
|
@@ -375,6 +664,35 @@ def _deduplicate(
|
|
|
375
664
|
return result
|
|
376
665
|
|
|
377
666
|
|
|
667
|
+
def _deduplicate_identity(values: list) -> list:
|
|
668
|
+
"""Make identity-derived values unique without swapping the person.
|
|
669
|
+
|
|
670
|
+
`_deduplicate` resolves a collision by calling the generator again, which
|
|
671
|
+
for an identity column would hand row i a different person's email and undo
|
|
672
|
+
the coherence this module just established. Suffixing keeps the row's
|
|
673
|
+
identity and disambiguates only the value, the way a real system issues
|
|
674
|
+
`jane.doe2@...` once `jane.doe@...` is taken. It also always succeeds, so
|
|
675
|
+
an identity column never lands in `_duplicate_unique_columns`.
|
|
676
|
+
"""
|
|
677
|
+
seen: set = set()
|
|
678
|
+
result = []
|
|
679
|
+
for value in values:
|
|
680
|
+
candidate = value
|
|
681
|
+
counter = 1
|
|
682
|
+
while candidate in seen:
|
|
683
|
+
counter += 1
|
|
684
|
+
candidate = _suffixed(str(value), counter)
|
|
685
|
+
seen.add(candidate)
|
|
686
|
+
result.append(candidate)
|
|
687
|
+
return result
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _suffixed(value: str, counter: int) -> str:
|
|
691
|
+
"""Append a disambiguating number, before the @ when the value is an email."""
|
|
692
|
+
local, at, domain = value.partition("@")
|
|
693
|
+
return f"{local}{counter}{at}{domain}"
|
|
694
|
+
|
|
695
|
+
|
|
378
696
|
def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
|
|
379
697
|
"""Pick a random timestamp in a window around today, to whole seconds.
|
|
380
698
|
|
|
@@ -390,3 +708,8 @@ def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
|
|
|
390
708
|
end = midnight + timedelta(days=end_days)
|
|
391
709
|
random_second = random.randint(0, int((end - start).total_seconds()))
|
|
392
710
|
return start + timedelta(seconds=random_second)
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
# The default locale's constants, resolved at import so the first generation
|
|
714
|
+
# does not pay for them and `set_locale`'s early return stays correct.
|
|
715
|
+
_resolve_locale()
|
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
"""Person-shaped columns in one row have to describe one person.
|
|
2
|
+
|
|
3
|
+
Columns are generated independently, one at a time, so before the identity
|
|
4
|
+
pool a single row's first_name, last_name and email came from three unrelated
|
|
5
|
+
Faker draws. These tests pin the row-level agreement, and the two properties it
|
|
6
|
+
must not cost: reproducibility under a seed, and uniqueness where it is asked
|
|
7
|
+
for.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
import pytest
|
|
13
|
+
from faker import Faker
|
|
14
|
+
|
|
15
|
+
import model2data.generate.faker as faker_module
|
|
16
|
+
from model2data.generate.core import generate_data_from_dbml
|
|
17
|
+
from model2data.generate.faker import (
|
|
18
|
+
DEFAULT_LOCALE,
|
|
19
|
+
_slug,
|
|
20
|
+
current_locale,
|
|
21
|
+
generate_column_values,
|
|
22
|
+
reset_row_pools,
|
|
23
|
+
set_locale,
|
|
24
|
+
)
|
|
25
|
+
from model2data.parse.dbml import ColumnDef, TableDef
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _person_table(name: str = "customers", columns: list[ColumnDef] | None = None) -> TableDef:
|
|
29
|
+
return TableDef(
|
|
30
|
+
name=name,
|
|
31
|
+
columns=columns
|
|
32
|
+
or [
|
|
33
|
+
ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
|
|
34
|
+
ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
|
|
35
|
+
ColumnDef(name="full_name", data_type="varchar", settings={"not null"}),
|
|
36
|
+
ColumnDef(name="user_name", data_type="varchar", settings={"not null"}),
|
|
37
|
+
ColumnDef(name="email", data_type="varchar", settings={"not null"}),
|
|
38
|
+
],
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _assert_row_is_one_person(row) -> None:
|
|
43
|
+
first, last = row["first_name"], row["last_name"]
|
|
44
|
+
assert row["full_name"] == f"{first} {last}"
|
|
45
|
+
assert row["email"].startswith(f"{_slug(first)}.{_slug(last)}@")
|
|
46
|
+
assert row["user_name"] == f"{_slug(first)[:1]}{_slug(last)}"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class TestRowIdentityCoherence:
|
|
50
|
+
def test_person_columns_in_a_row_describe_one_person(self):
|
|
51
|
+
frames = generate_data_from_dbml(
|
|
52
|
+
tables={"customers": _person_table()}, refs=[], base_rows=25, seed=1
|
|
53
|
+
)
|
|
54
|
+
df = frames["customers"]
|
|
55
|
+
assert len(df) == 25
|
|
56
|
+
for _, row in df.iterrows():
|
|
57
|
+
_assert_row_is_one_person(row)
|
|
58
|
+
|
|
59
|
+
def test_two_tables_draw_from_different_populations(self):
|
|
60
|
+
frames = generate_data_from_dbml(
|
|
61
|
+
tables={
|
|
62
|
+
"customers": _person_table("customers"),
|
|
63
|
+
"employees": _person_table("employees"),
|
|
64
|
+
},
|
|
65
|
+
refs=[],
|
|
66
|
+
base_rows=30,
|
|
67
|
+
seed=2,
|
|
68
|
+
)
|
|
69
|
+
# Asserted on the keying rather than on the values. "These two sets of
|
|
70
|
+
# emails do not overlap" is true here, but only by a probability -- the
|
|
71
|
+
# same shape of luck-based assertion that made the password test fail
|
|
72
|
+
# the first time the RNG moved.
|
|
73
|
+
customers_pool = faker_module._row_pool("person", "customers", 30)
|
|
74
|
+
employees_pool = faker_module._row_pool("person", "employees", 30)
|
|
75
|
+
assert customers_pool is not employees_pool
|
|
76
|
+
|
|
77
|
+
# And the values follow from that: two tables of people are two groups
|
|
78
|
+
# of people, not one pool handed out twice.
|
|
79
|
+
assert list(frames["customers"]["email"]) != list(frames["employees"]["email"])
|
|
80
|
+
|
|
81
|
+
def test_same_seed_reproduces_the_same_people(self):
|
|
82
|
+
def run():
|
|
83
|
+
return generate_data_from_dbml(
|
|
84
|
+
tables={"customers": _person_table()}, refs=[], base_rows=15, seed=7
|
|
85
|
+
)["customers"]
|
|
86
|
+
|
|
87
|
+
first_run, second_run = run(), run()
|
|
88
|
+
assert list(first_run["email"]) == list(second_run["email"])
|
|
89
|
+
assert list(first_run["full_name"]) == list(second_run["full_name"])
|
|
90
|
+
|
|
91
|
+
def test_unique_email_stays_unique_without_swapping_the_person(self):
|
|
92
|
+
# A tiny name space forces collisions, so the suffixing path actually
|
|
93
|
+
# runs rather than being skipped because every value happened to differ.
|
|
94
|
+
table = _person_table(
|
|
95
|
+
columns=[
|
|
96
|
+
ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
|
|
97
|
+
ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
|
|
98
|
+
ColumnDef(name="email", data_type="varchar", settings={"not null", "unique"}),
|
|
99
|
+
],
|
|
100
|
+
)
|
|
101
|
+
df = generate_data_from_dbml(tables={"customers": table}, refs=[], base_rows=400, seed=3)[
|
|
102
|
+
"customers"
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
assert df["email"].is_unique
|
|
106
|
+
for _, row in df.iterrows():
|
|
107
|
+
# The suffix may sit before the @, but the row's own name is still
|
|
108
|
+
# what the address is built from.
|
|
109
|
+
local = row["email"].split("@")[0]
|
|
110
|
+
assert re.fullmatch(
|
|
111
|
+
rf"{re.escape(_slug(row['first_name']))}\.{re.escape(_slug(row['last_name']))}\d*",
|
|
112
|
+
local,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
def test_declared_email_type_also_reaches_the_identity(self):
|
|
116
|
+
table = TableDef(
|
|
117
|
+
name="contacts",
|
|
118
|
+
columns=[
|
|
119
|
+
ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
|
|
120
|
+
ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
|
|
121
|
+
# Declared as a provider rather than named `email`.
|
|
122
|
+
ColumnDef(name="contact", data_type="email", settings={"not null"}),
|
|
123
|
+
],
|
|
124
|
+
)
|
|
125
|
+
df = generate_data_from_dbml(tables={"contacts": table}, refs=[], base_rows=20, seed=4)[
|
|
126
|
+
"contacts"
|
|
127
|
+
]
|
|
128
|
+
for _, row in df.iterrows():
|
|
129
|
+
assert row["contact"].startswith(
|
|
130
|
+
f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
def test_email_declared_as_text_is_still_an_email(self):
|
|
134
|
+
# `email text` means the SQL type; the name inference must still win,
|
|
135
|
+
# and still go through the identity.
|
|
136
|
+
table = TableDef(
|
|
137
|
+
name="people",
|
|
138
|
+
columns=[
|
|
139
|
+
ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
|
|
140
|
+
ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
|
|
141
|
+
ColumnDef(name="email", data_type="text", settings={"not null"}),
|
|
142
|
+
],
|
|
143
|
+
)
|
|
144
|
+
df = generate_data_from_dbml(tables={"people": table}, refs=[], base_rows=10, seed=5)[
|
|
145
|
+
"people"
|
|
146
|
+
]
|
|
147
|
+
for _, row in df.iterrows():
|
|
148
|
+
assert row["email"].startswith(f"{_slug(row['first_name'])}.{_slug(row['last_name'])}@")
|
|
149
|
+
|
|
150
|
+
def test_password_column_is_not_identity_derived(self):
|
|
151
|
+
# `password` sits between `username` and `email` in the pattern list, so
|
|
152
|
+
# it is the entry most at risk of having been swept into the identity
|
|
153
|
+
# work. Asserted against the pattern table rather than the output: an
|
|
154
|
+
# earlier version of this test looked for "@" in the generated password,
|
|
155
|
+
# which passed only by luck -- Faker's passwords include punctuation, "@"
|
|
156
|
+
# among it, so the test failed the first time the RNG moved.
|
|
157
|
+
assert isinstance(faker_module._infer_by_name("email"), faker_module._FromRow)
|
|
158
|
+
assert not isinstance(faker_module._infer_by_name("password"), faker_module._FromRow)
|
|
159
|
+
|
|
160
|
+
table = TableDef(
|
|
161
|
+
name="users",
|
|
162
|
+
columns=[
|
|
163
|
+
ColumnDef(name="first_name", data_type="varchar", settings={"not null"}),
|
|
164
|
+
ColumnDef(name="last_name", data_type="varchar", settings={"not null"}),
|
|
165
|
+
ColumnDef(name="email", data_type="varchar", settings={"not null"}),
|
|
166
|
+
ColumnDef(name="password", data_type="varchar", settings={"not null"}),
|
|
167
|
+
],
|
|
168
|
+
)
|
|
169
|
+
df = generate_data_from_dbml(tables={"users": table}, refs=[], base_rows=20, seed=21)[
|
|
170
|
+
"users"
|
|
171
|
+
]
|
|
172
|
+
assert df["password"].nunique() == 20
|
|
173
|
+
for _, row in df.iterrows():
|
|
174
|
+
# Exact comparison, not a substring heuristic: the password must not
|
|
175
|
+
# be one of the identity's own values.
|
|
176
|
+
assert row["password"] not in {row["first_name"], row["last_name"], row["email"]}
|
|
177
|
+
|
|
178
|
+
def test_non_person_columns_are_untouched(self):
|
|
179
|
+
reset_row_pools()
|
|
180
|
+
cities = generate_column_values(
|
|
181
|
+
ColumnDef(name="city", data_type="varchar", settings={"not null"}),
|
|
182
|
+
row_count=10,
|
|
183
|
+
table_name="places",
|
|
184
|
+
)
|
|
185
|
+
assert len(cities) == 10
|
|
186
|
+
assert all(isinstance(city, str) and city for city in cities)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
@pytest.fixture(autouse=True)
|
|
190
|
+
def _restore_locale():
|
|
191
|
+
"""Locale is process-wide state, so a test that changes it must not leak."""
|
|
192
|
+
yield
|
|
193
|
+
set_locale(None)
|
|
194
|
+
# set_locale returns early when the locale is already the default, which
|
|
195
|
+
# would leave the cached per-locale constants behind for any test that
|
|
196
|
+
# stubbed `fake` rather than switching locale. Re-resolve unconditionally.
|
|
197
|
+
faker_module._resolve_locale()
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _address_table(name: str = "sites") -> TableDef:
|
|
201
|
+
return TableDef(
|
|
202
|
+
name=name,
|
|
203
|
+
columns=[
|
|
204
|
+
ColumnDef(name="street", data_type="varchar", settings={"not null"}),
|
|
205
|
+
ColumnDef(name="address", data_type="varchar", settings={"not null"}),
|
|
206
|
+
ColumnDef(name="city", data_type="varchar", settings={"not null"}),
|
|
207
|
+
ColumnDef(name="state", data_type="varchar", settings={"not null"}),
|
|
208
|
+
ColumnDef(name="country", data_type="varchar", settings={"not null"}),
|
|
209
|
+
ColumnDef(name="zip", data_type="varchar", settings={"not null"}),
|
|
210
|
+
],
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class TestAddressCoherence:
|
|
215
|
+
def test_address_column_is_built_from_the_row_beside_it(self):
|
|
216
|
+
df = generate_data_from_dbml(
|
|
217
|
+
tables={"sites": _address_table()}, refs=[], base_rows=20, seed=11
|
|
218
|
+
)["sites"]
|
|
219
|
+
for _, row in df.iterrows():
|
|
220
|
+
# The composed address has to be this row's street, city and
|
|
221
|
+
# postcode -- not a separate draw that happens to look like one.
|
|
222
|
+
assert row["street"] in row["address"]
|
|
223
|
+
assert row["city"] in row["address"]
|
|
224
|
+
assert row["zip"] in row["address"]
|
|
225
|
+
|
|
226
|
+
def test_every_row_is_in_one_country(self):
|
|
227
|
+
df = generate_data_from_dbml(
|
|
228
|
+
tables={"sites": _address_table()}, refs=[], base_rows=30, seed=12
|
|
229
|
+
)["sites"]
|
|
230
|
+
# The old behaviour drew country independently, so a US street could sit
|
|
231
|
+
# in Japan. One locale means one country, on every row.
|
|
232
|
+
assert set(df["country"]) == {"United States"}
|
|
233
|
+
|
|
234
|
+
def test_declared_city_type_agrees_with_the_row(self):
|
|
235
|
+
table = TableDef(
|
|
236
|
+
name="offices",
|
|
237
|
+
columns=[
|
|
238
|
+
ColumnDef(name="address", data_type="varchar", settings={"not null"}),
|
|
239
|
+
# Declared as a provider rather than named `city`.
|
|
240
|
+
ColumnDef(name="located_in", data_type="city", settings={"not null"}),
|
|
241
|
+
],
|
|
242
|
+
)
|
|
243
|
+
df = generate_data_from_dbml(tables={"offices": table}, refs=[], base_rows=15, seed=13)[
|
|
244
|
+
"offices"
|
|
245
|
+
]
|
|
246
|
+
for _, row in df.iterrows():
|
|
247
|
+
assert row["located_in"] in row["address"]
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
class TestLocale:
|
|
251
|
+
def test_default_locale_is_the_documented_one(self):
|
|
252
|
+
assert current_locale() == DEFAULT_LOCALE
|
|
253
|
+
|
|
254
|
+
def test_locale_changes_the_country_and_the_addresses(self):
|
|
255
|
+
def country_and_cities(locale):
|
|
256
|
+
df = generate_data_from_dbml(
|
|
257
|
+
tables={"sites": _address_table()},
|
|
258
|
+
refs=[],
|
|
259
|
+
base_rows=20,
|
|
260
|
+
seed=14,
|
|
261
|
+
locale=locale,
|
|
262
|
+
)["sites"]
|
|
263
|
+
return set(df["country"]), set(df["city"])
|
|
264
|
+
|
|
265
|
+
us_country, us_cities = country_and_cities(None)
|
|
266
|
+
be_country, be_cities = country_and_cities("nl_BE")
|
|
267
|
+
|
|
268
|
+
assert us_country == {"United States"}
|
|
269
|
+
assert be_country == {"Belgium"}
|
|
270
|
+
assert not us_cities & be_cities
|
|
271
|
+
|
|
272
|
+
def test_locale_applies_to_people_too(self):
|
|
273
|
+
df = generate_data_from_dbml(
|
|
274
|
+
tables={"customers": _person_table()},
|
|
275
|
+
refs=[],
|
|
276
|
+
base_rows=20,
|
|
277
|
+
seed=15,
|
|
278
|
+
locale="fr_FR",
|
|
279
|
+
)["customers"]
|
|
280
|
+
# Still coherent, just French: the pool swap must not break the
|
|
281
|
+
# first.last@domain derivation.
|
|
282
|
+
for _, row in df.iterrows():
|
|
283
|
+
_assert_row_is_one_person(row)
|
|
284
|
+
|
|
285
|
+
def test_locale_is_restored_between_runs(self):
|
|
286
|
+
generate_data_from_dbml(
|
|
287
|
+
tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16, locale="nl_BE"
|
|
288
|
+
)
|
|
289
|
+
assert current_locale() == "nl_BE"
|
|
290
|
+
# Passing nothing means the default, not "whatever the last run used".
|
|
291
|
+
generate_data_from_dbml(tables={"sites": _address_table()}, refs=[], base_rows=2, seed=16)
|
|
292
|
+
assert current_locale() == DEFAULT_LOCALE
|
|
293
|
+
|
|
294
|
+
def test_unknown_locale_says_so_plainly(self):
|
|
295
|
+
with pytest.raises(ValueError, match="Unknown locale 'zz_ZZ'"):
|
|
296
|
+
generate_data_from_dbml(
|
|
297
|
+
tables={"sites": _address_table()}, refs=[], base_rows=2, locale="zz_ZZ"
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
class TestPoolLifetime:
|
|
302
|
+
def test_pools_are_released_once_a_table_is_finished(self):
|
|
303
|
+
# A million-row table is a normal request; holding its identities for
|
|
304
|
+
# the rest of the run is what makes that unaffordable.
|
|
305
|
+
generate_data_from_dbml(
|
|
306
|
+
tables={"customers": _person_table(), "sites": _address_table()},
|
|
307
|
+
refs=[],
|
|
308
|
+
base_rows=50,
|
|
309
|
+
seed=17,
|
|
310
|
+
)
|
|
311
|
+
assert faker_module._person_state == {}
|
|
312
|
+
assert faker_module._address_state == {}
|
|
313
|
+
|
|
314
|
+
def test_pool_rows_stay_small(self):
|
|
315
|
+
# Slots rather than a per-instance __dict__. Worth a test because the
|
|
316
|
+
# difference is ~6x per row (344 bytes vs 56), the derived fields are
|
|
317
|
+
# properties precisely so they are not stored, and a million-row person
|
|
318
|
+
# table is a normal request -- an innocent-looking field added here
|
|
319
|
+
# would cost hundreds of megabytes on one.
|
|
320
|
+
assert not hasattr(faker_module._new_person(), "__dict__")
|
|
321
|
+
assert not hasattr(faker_module._new_address(), "__dict__")
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
class _LocaleMissing:
|
|
325
|
+
"""A Faker with named providers removed, standing in for a thinner locale.
|
|
326
|
+
|
|
327
|
+
Every locale Faker actually ships has `current_country` and at least one
|
|
328
|
+
administrative unit, so the fallbacks for a locale without them cannot be
|
|
329
|
+
reached by naming a real one -- but they still decide what lands in a
|
|
330
|
+
column, and the docstrings make a claim about what they do. This stands in
|
|
331
|
+
for the locale that would reach them.
|
|
332
|
+
"""
|
|
333
|
+
|
|
334
|
+
def __init__(self, inner, missing):
|
|
335
|
+
self._inner = inner
|
|
336
|
+
self._missing = set(missing)
|
|
337
|
+
|
|
338
|
+
def format(self, name, *args, **kwargs):
|
|
339
|
+
if name in self._missing:
|
|
340
|
+
raise AttributeError(name)
|
|
341
|
+
return self._inner.format(name, *args, **kwargs)
|
|
342
|
+
|
|
343
|
+
def __getattr__(self, name):
|
|
344
|
+
if name in self._missing:
|
|
345
|
+
raise AttributeError(name)
|
|
346
|
+
return getattr(self._inner, name)
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
class TestThinLocales:
|
|
350
|
+
def test_no_administrative_unit_leaves_state_empty(self, monkeypatch):
|
|
351
|
+
stub = _LocaleMissing(
|
|
352
|
+
Faker("en_US"), {"state", "province", "administrative_unit", "region"}
|
|
353
|
+
)
|
|
354
|
+
monkeypatch.setattr(faker_module, "fake", stub)
|
|
355
|
+
faker_module._resolve_locale()
|
|
356
|
+
|
|
357
|
+
address = faker_module._new_address()
|
|
358
|
+
# Empty is the honest answer for a country with no region worth naming.
|
|
359
|
+
# Inventing one just to fill the column would be worse data, not better.
|
|
360
|
+
assert address.state == ""
|
|
361
|
+
assert address.city and address.street and address.country
|
|
362
|
+
|
|
363
|
+
def test_no_current_country_still_puts_every_row_in_one_country(self, monkeypatch):
|
|
364
|
+
stub = _LocaleMissing(Faker("en_US"), {"current_country"})
|
|
365
|
+
monkeypatch.setattr(faker_module, "fake", stub)
|
|
366
|
+
faker_module._resolve_locale()
|
|
367
|
+
|
|
368
|
+
countries = {faker_module._new_address().country for _ in range(10)}
|
|
369
|
+
# The point of the fallback is agreement, not accuracy: one country
|
|
370
|
+
# picked once beats a different random country on every row.
|
|
371
|
+
assert len(countries) == 1
|
|
372
|
+
assert countries != {""}
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
class TestNameFolding:
|
|
376
|
+
def test_accents_are_folded_not_dropped(self):
|
|
377
|
+
# Dropping them turned Aimée into "aime" and Müller into "mller" -- not
|
|
378
|
+
# that person's name, and most visible in exactly the European locales
|
|
379
|
+
# the locale option exists to serve.
|
|
380
|
+
assert faker_module._slug("Aimée") == "aimee"
|
|
381
|
+
assert faker_module._slug("Müller") == "muller"
|
|
382
|
+
assert faker_module._slug("Björn") == "bjorn"
|
|
383
|
+
|
|
384
|
+
def test_letters_that_do_not_decompose_are_mapped(self):
|
|
385
|
+
# NFKD leaves these intact because they are their own letters, not a
|
|
386
|
+
# base plus an accent, so they would vanish with the combining marks.
|
|
387
|
+
assert faker_module._slug("Søren") == "soren"
|
|
388
|
+
assert faker_module._slug("Weiß") == "weiss"
|
|
389
|
+
assert faker_module._slug("Łukasz") == "lukasz"
|
|
390
|
+
assert faker_module._slug("Æther") == "aether"
|
|
391
|
+
|
|
392
|
+
def test_punctuation_and_spacing_still_go(self):
|
|
393
|
+
assert faker_module._slug("O'Brien") == "obrien"
|
|
394
|
+
assert faker_module._slug("Van Der Berg") == "vanderberg"
|
|
395
|
+
|
|
396
|
+
def test_a_name_with_no_latin_letters_falls_back(self):
|
|
397
|
+
# Documented limitation rather than a target: deriving an address from a
|
|
398
|
+
# CJK name needs romanization, and Faker exposes no romanized first/last
|
|
399
|
+
# pair to build one from. Such a locale gets "user", disambiguated by the
|
|
400
|
+
# unique suffixing, which is meaningless but at least stable and unique.
|
|
401
|
+
assert faker_module._slug("日本") == "user"
|
|
402
|
+
|
|
403
|
+
def test_generated_emails_are_ascii_in_an_accented_locale(self):
|
|
404
|
+
df = generate_data_from_dbml(
|
|
405
|
+
tables={"customers": _person_table()},
|
|
406
|
+
refs=[],
|
|
407
|
+
base_rows=40,
|
|
408
|
+
seed=31,
|
|
409
|
+
locale="fr_FR",
|
|
410
|
+
)["customers"]
|
|
411
|
+
for _, row in df.iterrows():
|
|
412
|
+
_assert_row_is_one_person(row)
|
|
413
|
+
# An address has to be usable as an address, whatever the name says.
|
|
414
|
+
row["email"].encode("ascii")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{model2data-1.2.0 → model2data-1.3.1}/model2data/dbt/templates/macros/generate_schema_name.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|