model2data 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.1.0/model2data.egg-info → model2data-1.2.0}/PKG-INFO +1 -1
- {model2data-1.1.0 → model2data-1.2.0}/README.md +5 -3
- {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/faker.py +34 -7
- {model2data-1.1.0 → model2data-1.2.0/model2data.egg-info}/PKG-INFO +1 -1
- {model2data-1.1.0 → model2data-1.2.0}/pyproject.toml +1 -1
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_faker_name_inference.py +67 -0
- {model2data-1.1.0 → model2data-1.2.0}/LICENSE +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/README_PYPI.md +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/cli.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/project.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/tests.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/core.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/relationships.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/parse/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/parse/dbml.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data/utils.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/SOURCES.txt +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/setup.cfg +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_cli.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbml_parser.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_integration.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_naming.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_project.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_tests.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_generation.py +0 -0
- {model2data-1.1.0 → model2data-1.2.0}/tests/test_release_stress.py +0 -0
|
@@ -39,7 +39,8 @@ access required.
|
|
|
39
39
|
- **Privacy-safe.** Nothing but a schema definition goes in; nothing but synthetic data comes out.
|
|
40
40
|
- **Realistic, not random.** Column names are matched against ~35 common patterns — `email`,
|
|
41
41
|
`first_name`, `city`, `phone`, `company`, ... — so a column called `email` gets real-looking
|
|
42
|
-
emails, not `Lorem ipsum` text.
|
|
42
|
+
emails, not `Lorem ipsum` text. Type a column with any Faker provider (`billing_country state`,
|
|
43
|
+
`sku ean13`) to pick its generator outright when the name is wrong for the data.
|
|
43
44
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
44
45
|
dependency order.
|
|
45
46
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
@@ -97,8 +98,9 @@ flowchart LR
|
|
|
97
98
|
|
|
98
99
|
1. **Parse.** Reads tables, columns, types, and `Ref` relationships from a DBML file.
|
|
99
100
|
2. **Generate.** Produces synthetic values per column — typed generation for known SQL types
|
|
100
|
-
(int, date, timestamp, ...),
|
|
101
|
-
`city`, ...), foreign keys
|
|
101
|
+
(int, date, timestamp, ...), then a Faker provider named as the type (`sku ean13`), then
|
|
102
|
+
name-aware inference for everything else (`email`, `phone`, `city`, ...), foreign keys
|
|
103
|
+
resolved against already-generated parent rows.
|
|
102
104
|
3. **Scaffold.** Writes a complete dbt project around that data: CSV seeds, staging models that
|
|
103
105
|
`ref()` those seeds, `not_null`/`unique`/`relationships` tests, `accepted_values` tests for
|
|
104
106
|
DBML `Enum`-typed columns, singular SQL tests for composite primary/unique keys, table and
|
|
@@ -150,6 +150,33 @@ def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
|
|
|
150
150
|
return None
|
|
151
151
|
|
|
152
152
|
|
|
153
|
+
# Type names that are ordinary SQL types first and Faker providers only by
|
|
154
|
+
# coincidence. Someone typing `email text` means the SQL type, so these must
|
|
155
|
+
# not count as a deliberate choice of provider -- otherwise the commonest
|
|
156
|
+
# declaration in any schema would stop generating emails.
|
|
157
|
+
_SQL_TYPES_SHADOWING_A_PROVIDER = frozenset({"text", "json", "jsonb", "xml", "binary", "year"})
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
|
|
161
|
+
"""The provider a column's declared type names, if it names one deliberately.
|
|
162
|
+
|
|
163
|
+
`sku ean13` and `home_state state` are the user saying which generator they
|
|
164
|
+
want, in the only place DBML gives them to say it. That has to outrank the
|
|
165
|
+
guess made from the column's name: a name pattern is inferred, a type is
|
|
166
|
+
declared, and `first_name email` silently generating first names -- the
|
|
167
|
+
type having no effect whatsoever -- is the single most confusing thing this
|
|
168
|
+
module did.
|
|
169
|
+
"""
|
|
170
|
+
if base_type in _SQL_TYPES_SHADOWING_A_PROVIDER:
|
|
171
|
+
return None
|
|
172
|
+
try:
|
|
173
|
+
fake.format(base_type)
|
|
174
|
+
except (AttributeError, TypeError):
|
|
175
|
+
# Not a provider, or one that needs arguments: nothing was declared.
|
|
176
|
+
return None
|
|
177
|
+
return lambda: fake.format(base_type)
|
|
178
|
+
|
|
179
|
+
|
|
153
180
|
# ---------------------------------------------------------
|
|
154
181
|
# Public API
|
|
155
182
|
# ---------------------------------------------------------
|
|
@@ -278,16 +305,16 @@ def generate_column_values(
|
|
|
278
305
|
values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
|
|
279
306
|
|
|
280
307
|
# -----------------------------------------------------
|
|
281
|
-
# Untyped / generic string columns:
|
|
282
|
-
#
|
|
283
|
-
#
|
|
308
|
+
# Untyped / generic string columns: honour a type that names
|
|
309
|
+
# a Faker provider (`sku ean13`), then infer intent from the
|
|
310
|
+
# column name (email, city, phone...), then a generic value.
|
|
284
311
|
# -----------------------------------------------------
|
|
285
312
|
else:
|
|
286
|
-
|
|
287
|
-
if
|
|
288
|
-
values = [
|
|
313
|
+
generator = _infer_by_type(base_type) or _infer_by_name(column.name)
|
|
314
|
+
if generator is not None:
|
|
315
|
+
values = [generator() for _ in range(row_count)]
|
|
289
316
|
values = (
|
|
290
|
-
_deduplicate(values,
|
|
317
|
+
_deduplicate(values, generator, column_name=unique_label)
|
|
291
318
|
if ensure_unique
|
|
292
319
|
else values
|
|
293
320
|
)
|
|
@@ -82,6 +82,73 @@ class TestNameInference:
|
|
|
82
82
|
assert result == ["x", "x", "x"]
|
|
83
83
|
|
|
84
84
|
|
|
85
|
+
class TestTypeBeatsName:
|
|
86
|
+
"""A declared Faker provider type outranks the guess made from the name.
|
|
87
|
+
|
|
88
|
+
DBML has one place to say which generator a column should use -- its type --
|
|
89
|
+
and until this held, a recognised column name silently overruled it:
|
|
90
|
+
`home_state state` generated countries and `first_name email` generated
|
|
91
|
+
first names, with the declared type having no effect at all.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
def test_a_declared_provider_overrules_the_name_pattern(self):
|
|
95
|
+
values = generate_column_values(
|
|
96
|
+
ColumnDef(name="first_name", data_type="email", settings={"not null"}),
|
|
97
|
+
row_count=20,
|
|
98
|
+
)
|
|
99
|
+
assert all(EMAIL_RE.match(v) for v in values)
|
|
100
|
+
|
|
101
|
+
def test_a_column_can_be_typed_against_its_own_name(self):
|
|
102
|
+
# The case the feature exists for: a state column that isn't called one.
|
|
103
|
+
# Without this the `country` in its name won and it generated countries.
|
|
104
|
+
from faker.providers.address.en_US import Provider as UsAddress
|
|
105
|
+
|
|
106
|
+
values = generate_column_values(
|
|
107
|
+
ColumnDef(name="billing_country", data_type="state", settings={"not null"}),
|
|
108
|
+
row_count=40,
|
|
109
|
+
)
|
|
110
|
+
assert set(values) <= set(UsAddress.states)
|
|
111
|
+
|
|
112
|
+
def test_a_plain_sql_type_does_not_count_as_a_declaration(self):
|
|
113
|
+
# `text` and `json` are Faker providers by coincidence; a column typed
|
|
114
|
+
# that way means the SQL type, and its name must still be inferred.
|
|
115
|
+
for sql_type in ("text", "json", "varchar", "char(64)"):
|
|
116
|
+
values = generate_column_values(
|
|
117
|
+
ColumnDef(name="email", data_type=sql_type, settings={"not null"}),
|
|
118
|
+
row_count=10,
|
|
119
|
+
)
|
|
120
|
+
assert all(EMAIL_RE.match(v) for v in values), sql_type
|
|
121
|
+
|
|
122
|
+
def test_an_unknown_type_still_falls_back_to_the_name(self):
|
|
123
|
+
reset_stats()
|
|
124
|
+
values = generate_column_values(
|
|
125
|
+
ColumnDef(name="city", data_type="weird_custom_type", settings={"not null"}),
|
|
126
|
+
row_count=10,
|
|
127
|
+
)
|
|
128
|
+
assert all(isinstance(value, str) and value for value in values)
|
|
129
|
+
assert get_unmapped_columns() == []
|
|
130
|
+
|
|
131
|
+
def test_a_declared_provider_is_still_deduplicated_when_unique(self):
|
|
132
|
+
values = generate_column_values(
|
|
133
|
+
ColumnDef(name="first_name", data_type="email", settings={"not null", "unique"}),
|
|
134
|
+
row_count=50,
|
|
135
|
+
ensure_unique=True,
|
|
136
|
+
)
|
|
137
|
+
assert len(values) == len(set(values)) == 50
|
|
138
|
+
|
|
139
|
+
def test_structured_types_are_untouched_by_this(self):
|
|
140
|
+
# `date` and `boolean` are providers too, but they are decided long
|
|
141
|
+
# before either inference runs, and must keep their real Python types.
|
|
142
|
+
dates = generate_column_values(
|
|
143
|
+
ColumnDef(name="signup_date", data_type="date", settings={"not null"}), row_count=5
|
|
144
|
+
)
|
|
145
|
+
assert all(hasattr(value, "year") for value in dates)
|
|
146
|
+
flags = generate_column_values(
|
|
147
|
+
ColumnDef(name="is_active", data_type="boolean", settings={"not null"}), row_count=5
|
|
148
|
+
)
|
|
149
|
+
assert set(flags) <= {True, False}
|
|
150
|
+
|
|
151
|
+
|
|
85
152
|
class TestEnumGeneration:
|
|
86
153
|
def test_enum_values_only_ever_come_from_the_enum_set(self):
|
|
87
154
|
allowed = {"active", "inactive", "pending"}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/macros/generate_schema_name.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|