model2data 1.7.0__tar.gz → 1.7.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.7.0/model2data.egg-info → model2data-1.7.3}/PKG-INFO +32 -1
- {model2data-1.7.0 → model2data-1.7.3}/README.md +35 -1
- {model2data-1.7.0 → model2data-1.7.3}/README_PYPI.md +30 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/core.py +41 -4
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/faker.py +60 -2
- model2data-1.7.3/model2data/generate/relationships.py +95 -0
- {model2data-1.7.0 → model2data-1.7.3/model2data.egg-info}/PKG-INFO +32 -1
- {model2data-1.7.0 → model2data-1.7.3}/model2data.egg-info/SOURCES.txt +1 -0
- {model2data-1.7.0 → model2data-1.7.3}/pyproject.toml +2 -1
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbt_tests.py +38 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_generation.py +165 -0
- model2data-1.7.3/tests/test_lone_country.py +132 -0
- model2data-1.7.0/model2data/generate/relationships.py +0 -52
- {model2data-1.7.0 → model2data-1.7.3}/LICENSE +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/__init__.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/cli.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/__init__.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/project.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/tests.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/__init__.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/hints.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/options.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/generate/timeline.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/parse/__init__.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/parse/dbml.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data/utils.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/setup.cfg +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_as_of_anchor.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_cli.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_column_time_hints.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbml_parser.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbt_integration.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbt_naming.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_dbt_project.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_distributions.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_faker_name_inference.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_options.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_release_stress.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_row_identity.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_shaping.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_table_seeds.py +0 -0
- {model2data-1.7.0 → model2data-1.7.3}/tests/test_timeline.py +0 -0
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 1.7.
|
|
3
|
+
Version: 1.7.3
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Author: JB Analytica
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/JB-Analytica/model2data
|
|
8
|
+
Project-URL: Studio, https://studio.jbanalytica.com
|
|
8
9
|
Project-URL: Repository, https://github.com/JB-Analytica/model2data
|
|
9
10
|
Project-URL: Issues, https://github.com/JB-Analytica/model2data/issues
|
|
10
11
|
Project-URL: Changelog, https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md
|
|
@@ -63,6 +64,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
63
64
|
cd dbt_ecommerce && dbt build
|
|
64
65
|
```
|
|
65
66
|
|
|
67
|
+
> **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
|
|
68
|
+
> engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
|
|
69
|
+
> column will generate before you generate it, then export CSVs or a runnable dbt project.
|
|
70
|
+
> Nothing to install, free to start.
|
|
71
|
+
|
|
66
72
|
That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
|
|
67
73
|
DuckDB — from a schema file, in seconds:
|
|
68
74
|
|
|
@@ -118,6 +124,29 @@ access required.
|
|
|
118
124
|
|
|
119
125
|
---
|
|
120
126
|
|
|
127
|
+
## model2data studio — the same engine, in the browser
|
|
128
|
+
|
|
129
|
+
[**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
|
|
130
|
+
web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
|
|
131
|
+
model2data, and it's the better fit while a schema is still being designed:
|
|
132
|
+
|
|
133
|
+
- **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
|
|
134
|
+
entity diagram redraws as you type. Click a column to trace what actually joins to it.
|
|
135
|
+
- **See what you'll get before you generate.** Every column shows an example of the value it will
|
|
136
|
+
produce, and columns nothing recognises are marked — so placeholder data is visible rather than
|
|
137
|
+
silent.
|
|
138
|
+
- **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
|
|
139
|
+
seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
|
|
140
|
+
you previewed.
|
|
141
|
+
- **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
|
|
142
|
+
Confluence or wiki page.
|
|
143
|
+
|
|
144
|
+
Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
|
|
145
|
+
The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
|
|
146
|
+
model gets designed and shown.
|
|
147
|
+
|
|
148
|
+
---
|
|
149
|
+
|
|
121
150
|
## Installation
|
|
122
151
|
|
|
123
152
|
```bash
|
|
@@ -323,4 +352,6 @@ MIT License. See LICENSE for details.
|
|
|
323
352
|
<br>
|
|
324
353
|
Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
|
|
325
354
|
Data & Analytics Engineering · Data Platform Architecture · Modern BI.
|
|
355
|
+
<br>
|
|
356
|
+
Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
|
|
326
357
|
</p>
|
|
@@ -26,6 +26,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
26
26
|
cd dbt_ecommerce && dbt build
|
|
27
27
|
```
|
|
28
28
|
|
|
29
|
+
> **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
|
|
30
|
+
> engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
|
|
31
|
+
> column will generate before you generate it, then export CSVs or a runnable dbt project.
|
|
32
|
+
> Nothing to install, free to start.
|
|
33
|
+
|
|
29
34
|
---
|
|
30
35
|
|
|
31
36
|
## Why this exists
|
|
@@ -112,6 +117,30 @@ flowchart LR
|
|
|
112
117
|
|
|
113
118
|
---
|
|
114
119
|
|
|
120
|
+
## model2data studio — the same engine, in the browser
|
|
121
|
+
|
|
122
|
+
[**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
|
|
123
|
+
web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
|
|
124
|
+
model2data, and it's the better fit while a schema is still being designed:
|
|
125
|
+
|
|
126
|
+
- **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
|
|
127
|
+
entity diagram redraws as you type. Click a column to trace what actually joins to it.
|
|
128
|
+
- **See what you'll get before you generate.** Every column shows an example of the value it will
|
|
129
|
+
produce, and columns nothing recognises are marked — so placeholder data is visible rather than
|
|
130
|
+
silent.
|
|
131
|
+
- **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
|
|
132
|
+
seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
|
|
133
|
+
you previewed.
|
|
134
|
+
- **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
|
|
135
|
+
Confluence or wiki page.
|
|
136
|
+
|
|
137
|
+
Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com)
|
|
138
|
+
([plans and pricing](https://www.jbanalytica.com/model2data/pricing/)).
|
|
139
|
+
The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
|
|
140
|
+
model gets designed and shown.
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
115
144
|
## Installation
|
|
116
145
|
|
|
117
146
|
```bash
|
|
@@ -159,7 +188,10 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --table-seed orde
|
|
|
159
188
|
```
|
|
160
189
|
|
|
161
190
|
`--locale` picks the country every generated person and address comes from (`en_US` by default);
|
|
162
|
-
it's a per-run setting, so a table can't end up holding one Belgian and one American address
|
|
191
|
+
it's a per-run setting, so a table can't end up holding one Belgian and one American address. A
|
|
192
|
+
`country` column that sits beside a `city`/`street`/`state`/`postcode` column always agrees with
|
|
193
|
+
that place; a `country` column with none of those beside it isn't describing anyone's address, so
|
|
194
|
+
it reads as an international mix instead, with the locale's own country the most common:
|
|
163
195
|
|
|
164
196
|
```bash
|
|
165
197
|
model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
|
|
@@ -425,4 +457,6 @@ MIT License. See LICENSE for details.
|
|
|
425
457
|
<br>
|
|
426
458
|
Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
|
|
427
459
|
Data & Analytics Engineering · Data Platform Architecture · Modern BI.
|
|
460
|
+
<br>
|
|
461
|
+
Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
|
|
428
462
|
</p>
|
|
@@ -19,6 +19,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
19
19
|
cd dbt_ecommerce && dbt build
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
+
> **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
|
|
23
|
+
> engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
|
|
24
|
+
> column will generate before you generate it, then export CSVs or a runnable dbt project.
|
|
25
|
+
> Nothing to install, free to start.
|
|
26
|
+
|
|
22
27
|
That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
|
|
23
28
|
DuckDB — from a schema file, in seconds:
|
|
24
29
|
|
|
@@ -74,6 +79,29 @@ access required.
|
|
|
74
79
|
|
|
75
80
|
---
|
|
76
81
|
|
|
82
|
+
## model2data studio — the same engine, in the browser
|
|
83
|
+
|
|
84
|
+
[**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
|
|
85
|
+
web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
|
|
86
|
+
model2data, and it's the better fit while a schema is still being designed:
|
|
87
|
+
|
|
88
|
+
- **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
|
|
89
|
+
entity diagram redraws as you type. Click a column to trace what actually joins to it.
|
|
90
|
+
- **See what you'll get before you generate.** Every column shows an example of the value it will
|
|
91
|
+
produce, and columns nothing recognises are marked — so placeholder data is visible rather than
|
|
92
|
+
silent.
|
|
93
|
+
- **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
|
|
94
|
+
seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
|
|
95
|
+
you previewed.
|
|
96
|
+
- **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
|
|
97
|
+
Confluence or wiki page.
|
|
98
|
+
|
|
99
|
+
Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
|
|
100
|
+
The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
|
|
101
|
+
model gets designed and shown.
|
|
102
|
+
|
|
103
|
+
---
|
|
104
|
+
|
|
77
105
|
## Installation
|
|
78
106
|
|
|
79
107
|
```bash
|
|
@@ -279,4 +307,6 @@ MIT License. See LICENSE for details.
|
|
|
279
307
|
<br>
|
|
280
308
|
Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
|
|
281
309
|
Data & Analytics Engineering · Data Platform Architecture · Modern BI.
|
|
310
|
+
<br>
|
|
311
|
+
Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
|
|
282
312
|
</p>
|
|
@@ -15,6 +15,7 @@ from model2data.generate.faker import (
|
|
|
15
15
|
release_row_pools,
|
|
16
16
|
reset_duplicate_unique_columns,
|
|
17
17
|
reset_row_pools,
|
|
18
|
+
resolve_address_pool_field,
|
|
18
19
|
set_locale,
|
|
19
20
|
)
|
|
20
21
|
from model2data.generate.hints import validate_hints
|
|
@@ -201,6 +202,12 @@ def generate_data_from_dbml(
|
|
|
201
202
|
for column_name in key.get("columns") or []
|
|
202
203
|
}
|
|
203
204
|
|
|
205
|
+
# A table's own shape, worked out before a single value is drawn: does
|
|
206
|
+
# this table have a `country` column with no `city`/`street`/`state`/
|
|
207
|
+
# `postcode` beside it to keep coherent with. See
|
|
208
|
+
# _lone_country_columns.
|
|
209
|
+
lone_country_columns = _lone_country_columns(table_def)
|
|
210
|
+
|
|
204
211
|
# -----------------------
|
|
205
212
|
# First pass: columns + FKs
|
|
206
213
|
# -----------------------
|
|
@@ -230,6 +237,7 @@ def generate_data_from_dbml(
|
|
|
230
237
|
as_of=as_of,
|
|
231
238
|
time_profile=profile,
|
|
232
239
|
skew=skew,
|
|
240
|
+
lone_country=column.name in lone_country_columns,
|
|
233
241
|
)
|
|
234
242
|
|
|
235
243
|
df = pd.DataFrame(data)
|
|
@@ -275,19 +283,23 @@ def generate_data_from_dbml(
|
|
|
275
283
|
continue
|
|
276
284
|
|
|
277
285
|
# find FK linking child → parent
|
|
278
|
-
|
|
286
|
+
fk_ref = next(
|
|
279
287
|
(
|
|
280
|
-
r
|
|
288
|
+
r
|
|
281
289
|
for r in fk_refs
|
|
282
290
|
if r["source_table"] == table_name and r["target_table"] == parent_table
|
|
283
291
|
),
|
|
284
292
|
None,
|
|
285
293
|
)
|
|
294
|
+
if fk_ref is None:
|
|
295
|
+
continue
|
|
286
296
|
|
|
287
|
-
|
|
297
|
+
fk_column = fk_ref["source_column"]
|
|
298
|
+
parent_key = fk_ref["target_column"]
|
|
299
|
+
if fk_column not in df.columns or parent_key not in parent_df.columns:
|
|
288
300
|
continue
|
|
289
301
|
|
|
290
|
-
lookup = parent_df.groupby(
|
|
302
|
+
lookup = parent_df.groupby(parent_key)[parent_column].first().to_dict()
|
|
291
303
|
|
|
292
304
|
df[child_column] = df[fk_column].map(lookup)
|
|
293
305
|
|
|
@@ -303,6 +315,31 @@ def generate_data_from_dbml(
|
|
|
303
315
|
# ---------------------------------------------------------
|
|
304
316
|
# Internal helpers
|
|
305
317
|
# ---------------------------------------------------------
|
|
318
|
+
def _lone_country_columns(table_def: TableDef) -> set[str]:
|
|
319
|
+
"""Names of this table's *lone* country columns.
|
|
320
|
+
|
|
321
|
+
A `country` column reads as the locale's own country on every row when it
|
|
322
|
+
sits beside a `city`/`street`/`state`/`postcode` column -- together they
|
|
323
|
+
describe one place, and the country has to agree with the rest of it.
|
|
324
|
+
Alone, repeating that same country on every row reads as a single-country
|
|
325
|
+
customer base rather than an international one, so
|
|
326
|
+
`generate_column_values` draws it from a home-heavy mix instead (see
|
|
327
|
+
`faker._HOME_COUNTRY_SHARE`). `country` columns don't count as company
|
|
328
|
+
for each other -- only a *different* address-pool field does.
|
|
329
|
+
"""
|
|
330
|
+
address_fields = {
|
|
331
|
+
column.name: field
|
|
332
|
+
for column in table_def.columns
|
|
333
|
+
for field in [resolve_address_pool_field(column)]
|
|
334
|
+
if field is not None
|
|
335
|
+
}
|
|
336
|
+
country_columns = {name for name, field in address_fields.items() if field == "country"}
|
|
337
|
+
if not country_columns:
|
|
338
|
+
return set()
|
|
339
|
+
has_place_column = any(field != "country" for field in address_fields.values())
|
|
340
|
+
return set() if has_place_column else country_columns
|
|
341
|
+
|
|
342
|
+
|
|
306
343
|
def _coerce_integer_dtypes(df: pd.DataFrame, table_def: TableDef) -> pd.DataFrame:
|
|
307
344
|
"""
|
|
308
345
|
Cast int/bigint/smallint-typed columns to pandas' nullable "Int64" dtype.
|
|
@@ -505,6 +505,42 @@ def _infer_by_type(base_type: str) -> Optional[_Provider]:
|
|
|
505
505
|
return lambda: fake.format(base_type)
|
|
506
506
|
|
|
507
507
|
|
|
508
|
+
def resolve_address_pool_field(column: ColumnDef) -> Optional[str]:
|
|
509
|
+
"""The address-pool field (`street`, `full`, `city`, `state`, `postcode`,
|
|
510
|
+
`country`) this column would draw from, if any -- else None.
|
|
511
|
+
|
|
512
|
+
Same declared-type-then-name precedence `generate_column_values`'s own
|
|
513
|
+
untyped-column branch uses (`_infer_by_type` before `_infer_by_name`),
|
|
514
|
+
narrowed to the address pool. Exposed for `generate.core` to tell a
|
|
515
|
+
*lone* country column -- the only address-pool-shaped column in its
|
|
516
|
+
table -- from one that sits beside a `city`/`street`/`state`/`postcode`
|
|
517
|
+
column, before a single value of the table has been generated.
|
|
518
|
+
|
|
519
|
+
Restricted to columns that would actually reach that branch: an enum
|
|
520
|
+
column, or one whose type is a structured (int/date/uuid/...) type, never
|
|
521
|
+
gets there in `generate_column_values` -- and `_infer_by_type` probes an
|
|
522
|
+
unrecognized type by actually calling it (`fake.format(base_type)`), so
|
|
523
|
+
running it over a `date` or `int` column here would consume real draws
|
|
524
|
+
from the shared RNG and shift every value generated after it, breaking
|
|
525
|
+
reproducibility for reasons invisible to whoever hits it.
|
|
526
|
+
"""
|
|
527
|
+
if column.enum_values or not is_free_text_type(column.data_type):
|
|
528
|
+
return None
|
|
529
|
+
base_type = column.data_type.lower().split("(")[0].strip()
|
|
530
|
+
generator = _infer_by_type(base_type) or _infer_by_name(column.name)
|
|
531
|
+
if isinstance(generator, _FromRow) and generator.pool == "address":
|
|
532
|
+
return generator.field
|
|
533
|
+
return None
|
|
534
|
+
|
|
535
|
+
|
|
536
|
+
# A lone country column (see resolve_address_pool_field's caller) mixes the
|
|
537
|
+
# locale's own country in with the rest of the world rather than repeating it
|
|
538
|
+
# on every row. 0.6 is a default, not a claim about any real market -- a
|
|
539
|
+
# business selling internationally still has a home market, and most of its
|
|
540
|
+
# rows are plausibly it, but "most" is not "all".
|
|
541
|
+
_HOME_COUNTRY_SHARE = 0.6
|
|
542
|
+
|
|
543
|
+
|
|
508
544
|
def _column_time_profile(
|
|
509
545
|
column: ColumnDef, time_profile: Optional[TimeProfile]
|
|
510
546
|
) -> Optional[TimeProfile]:
|
|
@@ -580,6 +616,7 @@ def generate_column_values(
|
|
|
580
616
|
as_of: AsOf = None,
|
|
581
617
|
time_profile: Optional[TimeProfile] = None,
|
|
582
618
|
skew: float = 0.0,
|
|
619
|
+
lone_country: bool = False,
|
|
583
620
|
) -> list:
|
|
584
621
|
"""
|
|
585
622
|
Generate synthetic values for a single column.
|
|
@@ -591,6 +628,16 @@ def generate_column_values(
|
|
|
591
628
|
every path that draws a value -- the main pass, the self-referencing FK
|
|
592
629
|
repair, the composite-key retry -- draws it the same way.
|
|
593
630
|
|
|
631
|
+
`lone_country` tells the address-pool branch this column is the *only*
|
|
632
|
+
address-shaped column in its table (see
|
|
633
|
+
`generate.core._lone_country_columns`). A `country` column that sits
|
|
634
|
+
beside a `city`/`street`/`state`/`postcode` column still reads that
|
|
635
|
+
place's own country, byte-identical to earlier releases; a lone one
|
|
636
|
+
instead draws a home-heavy mix of the locale's country and the wider
|
|
637
|
+
world, since "Belgium" on every row of a customers table with no other
|
|
638
|
+
address column reads as a single-country customer base rather than an
|
|
639
|
+
international one.
|
|
640
|
+
|
|
594
641
|
`as_of` is the date every generated date and timestamp is placed relative
|
|
595
642
|
to, defaulting to today. Pass it to make a seeded run reproduce on any
|
|
596
643
|
later day rather than only on the day it first ran.
|
|
@@ -629,6 +676,7 @@ def generate_column_values(
|
|
|
629
676
|
as_of=as_of,
|
|
630
677
|
time_profile=time_profile,
|
|
631
678
|
skew=skew,
|
|
679
|
+
lone_country=lone_country,
|
|
632
680
|
)
|
|
633
681
|
values = random.choices(pool, k=row_count)
|
|
634
682
|
if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
|
|
@@ -827,8 +875,18 @@ def generate_column_values(
|
|
|
827
875
|
else:
|
|
828
876
|
generator = _infer_by_type(base_type) or _infer_by_name(column.name)
|
|
829
877
|
if isinstance(generator, _FromRow):
|
|
830
|
-
|
|
831
|
-
|
|
878
|
+
if lone_country and generator.pool == "address" and generator.field == "country":
|
|
879
|
+
# No sibling city/street/state/postcode column to keep this
|
|
880
|
+
# one coherent with, so it isn't "this row's place" at all --
|
|
881
|
+
# draw a home-heavy mix instead of repeating the locale's own
|
|
882
|
+
# country on every row.
|
|
883
|
+
values = [
|
|
884
|
+
_country_name if random.random() < _HOME_COUNTRY_SHARE else fake.country()
|
|
885
|
+
for _ in range(row_count)
|
|
886
|
+
]
|
|
887
|
+
else:
|
|
888
|
+
rows = _row_pool(generator.pool, table_name, row_count)
|
|
889
|
+
values = [getattr(rows[index], generator.field) for index in range(row_count)]
|
|
832
890
|
if ensure_unique:
|
|
833
891
|
values = _deduplicate_identity(values)
|
|
834
892
|
elif generator is not None:
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
from typing import Dict, List, Optional, Tuple
|
|
2
|
+
|
|
3
|
+
from model2data.parse.dbml import ColumnDef, TableDef
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# ---------------------------------------------------------
|
|
7
|
+
# Public API
|
|
8
|
+
# ---------------------------------------------------------
|
|
9
|
+
def classify_refs(
|
|
10
|
+
tables: Dict[str, TableDef],
|
|
11
|
+
refs: List[Dict],
|
|
12
|
+
) -> Tuple[List[Dict], List[Dict]]:
|
|
13
|
+
"""
|
|
14
|
+
Classify references into:
|
|
15
|
+
- fk_refs: Foreign keys (target column is a key of its table)
|
|
16
|
+
- attribute_refs: Non-FK dependencies (mirroring parent attributes)
|
|
17
|
+
|
|
18
|
+
A Ref onto a primary key, or a column named "id", is always a foreign key.
|
|
19
|
+
A Ref onto a unique column is one too -- dbt projects commonly declare
|
|
20
|
+
their keys with `unique` + `not_null` tests rather than a primary key
|
|
21
|
+
constraint -- unless the child already has a primary-key FK to the same
|
|
22
|
+
parent. Then it stays an attribute ref and is mirrored through that FK, so
|
|
23
|
+
`orders.customer_email > customers.email` next to
|
|
24
|
+
`orders.customer_id > customers.id` keeps the email the customer's own.
|
|
25
|
+
"""
|
|
26
|
+
kinds = [_key_kind(tables, ref) for ref in refs]
|
|
27
|
+
pk_pairs = {
|
|
28
|
+
(ref["source_table"], ref["target_table"])
|
|
29
|
+
for ref, kind in zip(refs, kinds, strict=True)
|
|
30
|
+
if kind == "pk"
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
fk_refs = []
|
|
34
|
+
attribute_refs = []
|
|
35
|
+
for ref, kind in zip(refs, kinds, strict=True):
|
|
36
|
+
pair = (ref["source_table"], ref["target_table"])
|
|
37
|
+
if kind == "pk" or (kind == "unique" and pair not in pk_pairs):
|
|
38
|
+
fk_refs.append(ref)
|
|
39
|
+
else:
|
|
40
|
+
attribute_refs.append(ref)
|
|
41
|
+
|
|
42
|
+
return fk_refs, attribute_refs
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _key_kind(tables: Dict[str, TableDef], ref: Dict) -> Optional[str]:
|
|
46
|
+
"""Whether the column a Ref points at is a "pk", a "unique" key, or neither (None)."""
|
|
47
|
+
table = tables.get(ref["target_table"])
|
|
48
|
+
if table is None:
|
|
49
|
+
return None
|
|
50
|
+
column = next((c for c in table.columns if c.name == ref["target_column"]), None)
|
|
51
|
+
if column is None:
|
|
52
|
+
return None
|
|
53
|
+
if _is_primary_key(table, column):
|
|
54
|
+
return "pk"
|
|
55
|
+
if _is_unique(table, column):
|
|
56
|
+
return "unique"
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _single_column_keys(table: TableDef, key_type: str) -> set:
|
|
61
|
+
return {
|
|
62
|
+
key["columns"][0]
|
|
63
|
+
for key in table.composite_keys
|
|
64
|
+
if key["type"] == key_type and len(key["columns"]) == 1
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _is_primary_key(table: TableDef, column: ColumnDef) -> bool:
|
|
69
|
+
return (
|
|
70
|
+
"pk" in column.settings
|
|
71
|
+
or "primary key" in column.settings
|
|
72
|
+
or column.name.lower() == "id"
|
|
73
|
+
or column.name in _single_column_keys(table, "pk")
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _is_unique(table: TableDef, column: ColumnDef) -> bool:
|
|
78
|
+
return "unique" in column.settings or column.name in _single_column_keys(table, "unique")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
|
|
82
|
+
"""
|
|
83
|
+
Build a lookup dictionary for FK relationships.
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
{
|
|
87
|
+
(child_table, child_column): (parent_table, parent_column)
|
|
88
|
+
}
|
|
89
|
+
"""
|
|
90
|
+
lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
|
|
91
|
+
for ref in fk_refs:
|
|
92
|
+
key = (ref["source_table"], ref["source_column"])
|
|
93
|
+
value = (ref["target_table"], ref["target_column"])
|
|
94
|
+
lookup[key] = value
|
|
95
|
+
return lookup
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 1.7.
|
|
3
|
+
Version: 1.7.3
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Author: JB Analytica
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/JB-Analytica/model2data
|
|
8
|
+
Project-URL: Studio, https://studio.jbanalytica.com
|
|
8
9
|
Project-URL: Repository, https://github.com/JB-Analytica/model2data
|
|
9
10
|
Project-URL: Issues, https://github.com/JB-Analytica/model2data/issues
|
|
10
11
|
Project-URL: Changelog, https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md
|
|
@@ -63,6 +64,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
|
|
|
63
64
|
cd dbt_ecommerce && dbt build
|
|
64
65
|
```
|
|
65
66
|
|
|
67
|
+
> **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
|
|
68
|
+
> engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
|
|
69
|
+
> column will generate before you generate it, then export CSVs or a runnable dbt project.
|
|
70
|
+
> Nothing to install, free to start.
|
|
71
|
+
|
|
66
72
|
That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
|
|
67
73
|
DuckDB — from a schema file, in seconds:
|
|
68
74
|
|
|
@@ -118,6 +124,29 @@ access required.
|
|
|
118
124
|
|
|
119
125
|
---
|
|
120
126
|
|
|
127
|
+
## model2data studio — the same engine, in the browser
|
|
128
|
+
|
|
129
|
+
[**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
|
|
130
|
+
web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
|
|
131
|
+
model2data, and it's the better fit while a schema is still being designed:
|
|
132
|
+
|
|
133
|
+
- **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
|
|
134
|
+
entity diagram redraws as you type. Click a column to trace what actually joins to it.
|
|
135
|
+
- **See what you'll get before you generate.** Every column shows an example of the value it will
|
|
136
|
+
produce, and columns nothing recognises are marked — so placeholder data is visible rather than
|
|
137
|
+
silent.
|
|
138
|
+
- **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
|
|
139
|
+
seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
|
|
140
|
+
you previewed.
|
|
141
|
+
- **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
|
|
142
|
+
Confluence or wiki page.
|
|
143
|
+
|
|
144
|
+
Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
|
|
145
|
+
The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
|
|
146
|
+
model gets designed and shown.
|
|
147
|
+
|
|
148
|
+
---
|
|
149
|
+
|
|
121
150
|
## Installation
|
|
122
151
|
|
|
123
152
|
```bash
|
|
@@ -323,4 +352,6 @@ MIT License. See LICENSE for details.
|
|
|
323
352
|
<br>
|
|
324
353
|
Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
|
|
325
354
|
Data & Analytics Engineering · Data Platform Architecture · Modern BI.
|
|
355
|
+
<br>
|
|
356
|
+
Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
|
|
326
357
|
</p>
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "model2data"
|
|
7
|
-
version = "1.7.
|
|
7
|
+
version = "1.7.3"
|
|
8
8
|
description = "Generate analytics-ready datasets from DBML models"
|
|
9
9
|
readme = "README_PYPI.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -60,6 +60,7 @@ dev = [
|
|
|
60
60
|
|
|
61
61
|
[project.urls]
|
|
62
62
|
Homepage = "https://github.com/JB-Analytica/model2data"
|
|
63
|
+
Studio = "https://studio.jbanalytica.com"
|
|
63
64
|
Repository = "https://github.com/JB-Analytica/model2data"
|
|
64
65
|
Issues = "https://github.com/JB-Analytica/model2data/issues"
|
|
65
66
|
Changelog = "https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md"
|
|
@@ -638,3 +638,41 @@ def test_generic_tests_nest_parameters_under_arguments(tmp_path):
|
|
|
638
638
|
assert set(relationships) == {"arguments"}, "params must live under `arguments:`"
|
|
639
639
|
assert relationships["arguments"]["to"] == "ref('stg_customers')"
|
|
640
640
|
assert relationships["arguments"]["field"] == "id"
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def test_relationships_test_emitted_for_ref_onto_unique_non_pk_column(tmp_path):
|
|
644
|
+
"""A key declared with `unique` + `not null` rather than `pk` is still a
|
|
645
|
+
foreign-key target: the generator draws the child from the parent's
|
|
646
|
+
values, so the project gets the relationships test that proves it.
|
|
647
|
+
"""
|
|
648
|
+
tables = {
|
|
649
|
+
"customers": TableDef(
|
|
650
|
+
name="customers",
|
|
651
|
+
columns=[ColumnDef("customer_id", "int", {"not null", "unique"})],
|
|
652
|
+
),
|
|
653
|
+
"orders": TableDef(
|
|
654
|
+
name="orders",
|
|
655
|
+
columns=[
|
|
656
|
+
ColumnDef("order_id", "int", {"not null", "unique"}),
|
|
657
|
+
ColumnDef("customer_id", "int", {"not null"}),
|
|
658
|
+
],
|
|
659
|
+
),
|
|
660
|
+
}
|
|
661
|
+
refs = [
|
|
662
|
+
{
|
|
663
|
+
"source_table": "orders",
|
|
664
|
+
"source_column": "customer_id",
|
|
665
|
+
"target_table": "customers",
|
|
666
|
+
"target_column": "customer_id",
|
|
667
|
+
}
|
|
668
|
+
]
|
|
669
|
+
generate_dbt_yml(tmp_path, tables, refs, source_name="shop")
|
|
670
|
+
|
|
671
|
+
stg_yaml = yaml.safe_load((tmp_path / "models" / "staging" / "stg_orders.yml").read_text())
|
|
672
|
+
orders_col = next(c for c in stg_yaml["models"][0]["columns"] if c["name"] == "customer_id")
|
|
673
|
+
relationship_test = next(
|
|
674
|
+
t["relationships"]
|
|
675
|
+
for t in orders_col["tests"]
|
|
676
|
+
if isinstance(t, dict) and "relationships" in t
|
|
677
|
+
)
|
|
678
|
+
assert relationship_test["arguments"]["field"] == "customer_id"
|
|
@@ -958,3 +958,168 @@ def test_row_overrides_are_deterministic_with_a_seed():
|
|
|
958
958
|
|
|
959
959
|
assert first["users"].equals(second["users"])
|
|
960
960
|
assert first["orders"].equals(second["orders"])
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
def _customers_and_orders_on_a_unique_key() -> tuple[dict, list[dict]]:
|
|
964
|
+
# The shape a dbt import produces when the key is declared with
|
|
965
|
+
# `unique` + `not_null` tests rather than a primary key constraint.
|
|
966
|
+
tables = {
|
|
967
|
+
"customers": TableDef(
|
|
968
|
+
name="customers",
|
|
969
|
+
columns=[
|
|
970
|
+
ColumnDef("customer_id", "int", {"not null", "unique"}),
|
|
971
|
+
ColumnDef("name", "varchar"),
|
|
972
|
+
],
|
|
973
|
+
),
|
|
974
|
+
"orders": TableDef(
|
|
975
|
+
name="orders",
|
|
976
|
+
columns=[
|
|
977
|
+
ColumnDef("order_id", "int", {"not null", "unique"}),
|
|
978
|
+
ColumnDef("customer_id", "int", {"not null"}),
|
|
979
|
+
],
|
|
980
|
+
),
|
|
981
|
+
}
|
|
982
|
+
refs = [
|
|
983
|
+
{
|
|
984
|
+
"source_table": "orders",
|
|
985
|
+
"source_column": "customer_id",
|
|
986
|
+
"target_table": "customers",
|
|
987
|
+
"target_column": "customer_id",
|
|
988
|
+
}
|
|
989
|
+
]
|
|
990
|
+
return tables, refs
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
def test_ref_onto_a_unique_non_pk_column_is_a_foreign_key():
|
|
994
|
+
# At the default 100 rows the column's own integer range is wider than the
|
|
995
|
+
# parent's keys, so a ref treated as unrelated data fails the dbt
|
|
996
|
+
# relationships test on about half of all seeds.
|
|
997
|
+
tables, refs = _customers_and_orders_on_a_unique_key()
|
|
998
|
+
for seed in range(42, 62):
|
|
999
|
+
data = generate_data_from_dbml(tables, refs, base_rows=100, seed=seed)
|
|
1000
|
+
parent_keys = set(data["customers"]["customer_id"])
|
|
1001
|
+
assert set(data["orders"]["customer_id"]) <= parent_keys, seed
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
def test_ref_onto_a_unique_index_column_is_a_foreign_key():
|
|
1005
|
+
tables, refs = _customers_and_orders_on_a_unique_key()
|
|
1006
|
+
tables["customers"].columns[0].settings = {"not null"}
|
|
1007
|
+
tables["customers"].composite_keys = [{"columns": ["customer_id"], "type": "unique"}]
|
|
1008
|
+
for seed in range(42, 47):
|
|
1009
|
+
data = generate_data_from_dbml(tables, refs, base_rows=100, seed=seed)
|
|
1010
|
+
parent_keys = set(data["customers"]["customer_id"])
|
|
1011
|
+
assert set(data["orders"]["customer_id"]) <= parent_keys, seed
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def test_ref_onto_a_unique_column_beside_a_pk_fk_is_still_mirrored():
|
|
1015
|
+
tables = {
|
|
1016
|
+
"customers": TableDef(
|
|
1017
|
+
name="customers",
|
|
1018
|
+
columns=[
|
|
1019
|
+
ColumnDef("id", "int", {"pk"}),
|
|
1020
|
+
ColumnDef("email", "varchar", {"unique"}),
|
|
1021
|
+
],
|
|
1022
|
+
),
|
|
1023
|
+
"orders": TableDef(
|
|
1024
|
+
name="orders",
|
|
1025
|
+
columns=[
|
|
1026
|
+
ColumnDef("id", "int", {"pk"}),
|
|
1027
|
+
ColumnDef("customer_id", "int"),
|
|
1028
|
+
ColumnDef("customer_email", "varchar"),
|
|
1029
|
+
],
|
|
1030
|
+
),
|
|
1031
|
+
}
|
|
1032
|
+
refs = [
|
|
1033
|
+
{
|
|
1034
|
+
"source_table": "orders",
|
|
1035
|
+
"source_column": "customer_id",
|
|
1036
|
+
"target_table": "customers",
|
|
1037
|
+
"target_column": "id",
|
|
1038
|
+
},
|
|
1039
|
+
{
|
|
1040
|
+
"source_table": "orders",
|
|
1041
|
+
"source_column": "customer_email",
|
|
1042
|
+
"target_table": "customers",
|
|
1043
|
+
"target_column": "email",
|
|
1044
|
+
},
|
|
1045
|
+
]
|
|
1046
|
+
data = generate_data_from_dbml(tables, refs, base_rows=50, seed=7)
|
|
1047
|
+
email_of = data["customers"].set_index("id")["email"]
|
|
1048
|
+
orders = data["orders"]
|
|
1049
|
+
assert orders["customer_email"].equals(orders["customer_id"].map(email_of))
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def test_mirroring_goes_through_a_foreign_key_onto_a_non_id_key():
|
|
1053
|
+
tables, refs = _customers_and_orders_on_a_unique_key()
|
|
1054
|
+
tables["orders"].columns.append(ColumnDef("customer_name", "varchar"))
|
|
1055
|
+
refs.append(
|
|
1056
|
+
{
|
|
1057
|
+
"source_table": "orders",
|
|
1058
|
+
"source_column": "customer_name",
|
|
1059
|
+
"target_table": "customers",
|
|
1060
|
+
"target_column": "name",
|
|
1061
|
+
}
|
|
1062
|
+
)
|
|
1063
|
+
data = generate_data_from_dbml(tables, refs, base_rows=50, seed=7)
|
|
1064
|
+
name_of = data["customers"].set_index("customer_id")["name"]
|
|
1065
|
+
orders = data["orders"]
|
|
1066
|
+
assert orders["customer_name"].equals(orders["customer_id"].map(name_of))
|
|
1067
|
+
|
|
1068
|
+
|
|
1069
|
+
def test_mirroring_is_skipped_when_the_fk_column_is_not_in_the_child_table():
|
|
1070
|
+
# A Ref naming a child column the table doesn't declare still classifies
|
|
1071
|
+
# as a foreign key; the mirror that would ride on it has nothing to read.
|
|
1072
|
+
tables = {
|
|
1073
|
+
"customers": TableDef(
|
|
1074
|
+
name="customers",
|
|
1075
|
+
columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("name", "varchar")],
|
|
1076
|
+
),
|
|
1077
|
+
"orders": TableDef(
|
|
1078
|
+
name="orders",
|
|
1079
|
+
columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("customer_name", "varchar")],
|
|
1080
|
+
),
|
|
1081
|
+
}
|
|
1082
|
+
refs = [
|
|
1083
|
+
{
|
|
1084
|
+
"source_table": "orders",
|
|
1085
|
+
"source_column": "customer_id",
|
|
1086
|
+
"target_table": "customers",
|
|
1087
|
+
"target_column": "id",
|
|
1088
|
+
},
|
|
1089
|
+
{
|
|
1090
|
+
"source_table": "orders",
|
|
1091
|
+
"source_column": "customer_name",
|
|
1092
|
+
"target_table": "customers",
|
|
1093
|
+
"target_column": "name",
|
|
1094
|
+
},
|
|
1095
|
+
]
|
|
1096
|
+
data = generate_data_from_dbml(tables, refs, base_rows=20, seed=3)
|
|
1097
|
+
orders = data["orders"]
|
|
1098
|
+
assert "customer_id" not in orders.columns
|
|
1099
|
+
assert not orders["customer_name"].isin(data["customers"]["name"]).all()
|
|
1100
|
+
|
|
1101
|
+
|
|
1102
|
+
def test_attribute_ref_without_fk_is_skipped_when_the_parent_is_generated_first():
|
|
1103
|
+
# Tables with no FK between them are generated in name order, so here the
|
|
1104
|
+
# parent ("accounts") already exists when the child's mirror pass runs and
|
|
1105
|
+
# it is the missing FK, not the missing parent, that skips the mirror.
|
|
1106
|
+
tables = {
|
|
1107
|
+
"accounts": TableDef(
|
|
1108
|
+
name="accounts",
|
|
1109
|
+
columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("name", "varchar")],
|
|
1110
|
+
),
|
|
1111
|
+
"orders": TableDef(
|
|
1112
|
+
name="orders",
|
|
1113
|
+
columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("account_name", "varchar")],
|
|
1114
|
+
),
|
|
1115
|
+
}
|
|
1116
|
+
refs = [
|
|
1117
|
+
{
|
|
1118
|
+
"source_table": "orders",
|
|
1119
|
+
"source_column": "account_name",
|
|
1120
|
+
"target_table": "accounts",
|
|
1121
|
+
"target_column": "name",
|
|
1122
|
+
}
|
|
1123
|
+
]
|
|
1124
|
+
data = generate_data_from_dbml(tables, refs, base_rows=20, seed=2)
|
|
1125
|
+
assert not data["orders"]["account_name"].isin(data["accounts"]["name"]).all()
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""A lone `country` column reads as an international mix, not one repeated
|
|
2
|
+
country.
|
|
3
|
+
|
|
4
|
+
Since 1.3.0 every row draws one address from a per-table pool, and that
|
|
5
|
+
pool's country is always the locale's own -- right when a sibling
|
|
6
|
+
city/street/state/postcode column needs it to agree, wrong when `country` is
|
|
7
|
+
the only address-shaped column in the table. These tests pin: the mix (a
|
|
8
|
+
majority-but-not-all home-country share, several distinct countries), that a
|
|
9
|
+
sibling place column switches it back off, that a declared `country` type
|
|
10
|
+
behaves like the name, and that `distinct`/`null_rate`/determinism/locale
|
|
11
|
+
keep working the same way they do everywhere else.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from model2data.generate.core import generate_data_from_dbml
|
|
15
|
+
from model2data.parse.dbml import ColumnDef, TableDef
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _customers_table(*extra_columns: ColumnDef) -> TableDef:
|
|
19
|
+
return TableDef(
|
|
20
|
+
name="customers",
|
|
21
|
+
columns=[
|
|
22
|
+
ColumnDef(name="id", data_type="int", settings={"pk"}),
|
|
23
|
+
ColumnDef(name="country", data_type="varchar", settings={"not null"}),
|
|
24
|
+
*extra_columns,
|
|
25
|
+
],
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TestLoneCountryColumn:
|
|
30
|
+
def test_lone_country_column_reads_as_an_international_mix(self):
|
|
31
|
+
df = generate_data_from_dbml(
|
|
32
|
+
tables={"customers": _customers_table()},
|
|
33
|
+
refs=[],
|
|
34
|
+
base_rows=500,
|
|
35
|
+
seed=1,
|
|
36
|
+
locale="nl_BE",
|
|
37
|
+
)["customers"]
|
|
38
|
+
|
|
39
|
+
counts = df["country"].value_counts(normalize=True)
|
|
40
|
+
assert len(counts) >= 5
|
|
41
|
+
|
|
42
|
+
home_share = counts.get("Belgium", 0.0)
|
|
43
|
+
assert 0.5 <= home_share <= 0.7
|
|
44
|
+
assert counts.idxmax() == "Belgium"
|
|
45
|
+
|
|
46
|
+
def test_a_sibling_place_column_keeps_the_country_single(self):
|
|
47
|
+
df = generate_data_from_dbml(
|
|
48
|
+
tables={
|
|
49
|
+
"customers": _customers_table(
|
|
50
|
+
ColumnDef(name="city", data_type="varchar", settings={"not null"})
|
|
51
|
+
)
|
|
52
|
+
},
|
|
53
|
+
refs=[],
|
|
54
|
+
base_rows=200,
|
|
55
|
+
seed=2,
|
|
56
|
+
locale="nl_BE",
|
|
57
|
+
)["customers"]
|
|
58
|
+
|
|
59
|
+
assert set(df["country"]) == {"Belgium"}
|
|
60
|
+
|
|
61
|
+
def test_declared_country_type_behaves_like_the_name(self):
|
|
62
|
+
df = generate_data_from_dbml(
|
|
63
|
+
tables={
|
|
64
|
+
"customers": TableDef(
|
|
65
|
+
name="customers",
|
|
66
|
+
columns=[
|
|
67
|
+
ColumnDef(name="id", data_type="int", settings={"pk"}),
|
|
68
|
+
# Named generically; the *declared type* is what
|
|
69
|
+
# says "country".
|
|
70
|
+
ColumnDef(name="hq", data_type="country", settings={"not null"}),
|
|
71
|
+
],
|
|
72
|
+
)
|
|
73
|
+
},
|
|
74
|
+
refs=[],
|
|
75
|
+
base_rows=500,
|
|
76
|
+
seed=3,
|
|
77
|
+
locale="nl_BE",
|
|
78
|
+
)["customers"]
|
|
79
|
+
|
|
80
|
+
counts = df["hq"].value_counts(normalize=True)
|
|
81
|
+
assert len(counts) >= 5
|
|
82
|
+
assert 0.5 <= counts.get("Belgium", 0.0) <= 0.7
|
|
83
|
+
|
|
84
|
+
def test_distinct_hint_still_bounds_the_pool(self):
|
|
85
|
+
df = generate_data_from_dbml(
|
|
86
|
+
tables={
|
|
87
|
+
"customers": TableDef(
|
|
88
|
+
name="customers",
|
|
89
|
+
columns=[
|
|
90
|
+
ColumnDef(name="id", data_type="int", settings={"pk"}),
|
|
91
|
+
ColumnDef(
|
|
92
|
+
name="country",
|
|
93
|
+
data_type="varchar",
|
|
94
|
+
settings={"not null"},
|
|
95
|
+
note={"distinct": 3},
|
|
96
|
+
),
|
|
97
|
+
],
|
|
98
|
+
)
|
|
99
|
+
},
|
|
100
|
+
refs=[],
|
|
101
|
+
base_rows=300,
|
|
102
|
+
seed=4,
|
|
103
|
+
locale="nl_BE",
|
|
104
|
+
)["customers"]
|
|
105
|
+
|
|
106
|
+
assert df["country"].nunique() <= 3
|
|
107
|
+
|
|
108
|
+
def test_same_seed_reproduces_the_same_mix(self):
|
|
109
|
+
def run():
|
|
110
|
+
return generate_data_from_dbml(
|
|
111
|
+
tables={"customers": _customers_table()},
|
|
112
|
+
refs=[],
|
|
113
|
+
base_rows=200,
|
|
114
|
+
seed=5,
|
|
115
|
+
locale="nl_BE",
|
|
116
|
+
)["customers"]
|
|
117
|
+
|
|
118
|
+
first_run, second_run = run(), run()
|
|
119
|
+
assert list(first_run["country"]) == list(second_run["country"])
|
|
120
|
+
|
|
121
|
+
def test_default_locale_also_produces_a_mix(self):
|
|
122
|
+
df = generate_data_from_dbml(
|
|
123
|
+
tables={"customers": _customers_table()},
|
|
124
|
+
refs=[],
|
|
125
|
+
base_rows=500,
|
|
126
|
+
seed=6,
|
|
127
|
+
)["customers"]
|
|
128
|
+
|
|
129
|
+
counts = df["country"].value_counts(normalize=True)
|
|
130
|
+
assert len(counts) >= 5
|
|
131
|
+
assert 0.5 <= counts.get("United States", 0.0) <= 0.7
|
|
132
|
+
assert counts.idxmax() == "United States"
|
|
@@ -1,52 +0,0 @@
|
|
|
1
|
-
from typing import Dict, List, Tuple
|
|
2
|
-
|
|
3
|
-
from model2data.parse.dbml import TableDef
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
# ---------------------------------------------------------
|
|
7
|
-
# Public API
|
|
8
|
-
# ---------------------------------------------------------
|
|
9
|
-
def classify_refs(
|
|
10
|
-
tables: Dict[str, TableDef],
|
|
11
|
-
refs: List[Dict],
|
|
12
|
-
) -> Tuple[List[Dict], List[Dict]]:
|
|
13
|
-
"""
|
|
14
|
-
Classify references into:
|
|
15
|
-
- fk_refs: Foreign keys (target column looks like a PK)
|
|
16
|
-
- attribute_refs: Non-FK dependencies (mirroring parent attributes)
|
|
17
|
-
"""
|
|
18
|
-
fk_refs = []
|
|
19
|
-
attribute_refs = []
|
|
20
|
-
|
|
21
|
-
for ref in refs:
|
|
22
|
-
target_table = tables.get(ref["target_table"])
|
|
23
|
-
target_col = None
|
|
24
|
-
if target_table:
|
|
25
|
-
target_col = next(
|
|
26
|
-
(c for c in target_table.columns if c.name == ref["target_column"]), None
|
|
27
|
-
)
|
|
28
|
-
|
|
29
|
-
# FK if target column is a primary key or named "id"
|
|
30
|
-
if target_col and ("pk" in target_col.settings or target_col.name.lower() == "id"):
|
|
31
|
-
fk_refs.append(ref)
|
|
32
|
-
else:
|
|
33
|
-
attribute_refs.append(ref)
|
|
34
|
-
|
|
35
|
-
return fk_refs, attribute_refs
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
|
|
39
|
-
"""
|
|
40
|
-
Build a lookup dictionary for FK relationships.
|
|
41
|
-
|
|
42
|
-
Returns:
|
|
43
|
-
{
|
|
44
|
-
(child_table, child_column): (parent_table, parent_column)
|
|
45
|
-
}
|
|
46
|
-
"""
|
|
47
|
-
lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
|
|
48
|
-
for ref in fk_refs:
|
|
49
|
-
key = (ref["source_table"], ref["source_column"])
|
|
50
|
-
value = (ref["target_table"], ref["target_column"])
|
|
51
|
-
lookup[key] = value
|
|
52
|
-
return lookup
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{model2data-1.7.0 → model2data-1.7.3}/model2data/dbt/templates/macros/generate_schema_name.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|