model2data 1.7.1__tar.gz → 1.7.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {model2data-1.7.1/model2data.egg-info → model2data-1.7.3}/PKG-INFO +32 -1
  2. {model2data-1.7.1 → model2data-1.7.3}/README.md +31 -0
  3. {model2data-1.7.1 → model2data-1.7.3}/README_PYPI.md +30 -0
  4. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/core.py +8 -4
  5. model2data-1.7.3/model2data/generate/relationships.py +95 -0
  6. {model2data-1.7.1 → model2data-1.7.3/model2data.egg-info}/PKG-INFO +32 -1
  7. {model2data-1.7.1 → model2data-1.7.3}/pyproject.toml +2 -1
  8. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbt_tests.py +38 -0
  9. {model2data-1.7.1 → model2data-1.7.3}/tests/test_generation.py +165 -0
  10. model2data-1.7.1/model2data/generate/relationships.py +0 -52
  11. {model2data-1.7.1 → model2data-1.7.3}/LICENSE +0 -0
  12. {model2data-1.7.1 → model2data-1.7.3}/model2data/__init__.py +0 -0
  13. {model2data-1.7.1 → model2data-1.7.3}/model2data/cli.py +0 -0
  14. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/__init__.py +0 -0
  15. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/project.py +0 -0
  16. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  17. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  18. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  19. {model2data-1.7.1 → model2data-1.7.3}/model2data/dbt/tests.py +0 -0
  20. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/__init__.py +0 -0
  21. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/faker.py +0 -0
  22. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/hints.py +0 -0
  23. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/options.py +0 -0
  24. {model2data-1.7.1 → model2data-1.7.3}/model2data/generate/timeline.py +0 -0
  25. {model2data-1.7.1 → model2data-1.7.3}/model2data/parse/__init__.py +0 -0
  26. {model2data-1.7.1 → model2data-1.7.3}/model2data/parse/dbml.py +0 -0
  27. {model2data-1.7.1 → model2data-1.7.3}/model2data/utils.py +0 -0
  28. {model2data-1.7.1 → model2data-1.7.3}/model2data.egg-info/SOURCES.txt +0 -0
  29. {model2data-1.7.1 → model2data-1.7.3}/model2data.egg-info/dependency_links.txt +0 -0
  30. {model2data-1.7.1 → model2data-1.7.3}/model2data.egg-info/entry_points.txt +0 -0
  31. {model2data-1.7.1 → model2data-1.7.3}/model2data.egg-info/requires.txt +0 -0
  32. {model2data-1.7.1 → model2data-1.7.3}/model2data.egg-info/top_level.txt +0 -0
  33. {model2data-1.7.1 → model2data-1.7.3}/setup.cfg +0 -0
  34. {model2data-1.7.1 → model2data-1.7.3}/tests/test_as_of_anchor.py +0 -0
  35. {model2data-1.7.1 → model2data-1.7.3}/tests/test_cli.py +0 -0
  36. {model2data-1.7.1 → model2data-1.7.3}/tests/test_column_time_hints.py +0 -0
  37. {model2data-1.7.1 → model2data-1.7.3}/tests/test_coverage_gaps.py +0 -0
  38. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbml_parser.py +0 -0
  39. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbml_parser_fuzz.py +0 -0
  40. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbt_integration.py +0 -0
  41. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbt_naming.py +0 -0
  42. {model2data-1.7.1 → model2data-1.7.3}/tests/test_dbt_project.py +0 -0
  43. {model2data-1.7.1 → model2data-1.7.3}/tests/test_distributions.py +0 -0
  44. {model2data-1.7.1 → model2data-1.7.3}/tests/test_faker_name_inference.py +0 -0
  45. {model2data-1.7.1 → model2data-1.7.3}/tests/test_lone_country.py +0 -0
  46. {model2data-1.7.1 → model2data-1.7.3}/tests/test_options.py +0 -0
  47. {model2data-1.7.1 → model2data-1.7.3}/tests/test_release_stress.py +0 -0
  48. {model2data-1.7.1 → model2data-1.7.3}/tests/test_row_identity.py +0 -0
  49. {model2data-1.7.1 → model2data-1.7.3}/tests/test_shaping.py +0 -0
  50. {model2data-1.7.1 → model2data-1.7.3}/tests/test_table_seeds.py +0 -0
  51. {model2data-1.7.1 → model2data-1.7.3}/tests/test_timeline.py +0 -0
@@ -1,10 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.7.1
3
+ Version: 1.7.3
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/JB-Analytica/model2data
8
+ Project-URL: Studio, https://studio.jbanalytica.com
8
9
  Project-URL: Repository, https://github.com/JB-Analytica/model2data
9
10
  Project-URL: Issues, https://github.com/JB-Analytica/model2data/issues
10
11
  Project-URL: Changelog, https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md
@@ -63,6 +64,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
63
64
  cd dbt_ecommerce && dbt build
64
65
  ```
65
66
 
67
+ > **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
68
+ > engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
69
+ > column will generate before you generate it, then export CSVs or a runnable dbt project.
70
+ > Nothing to install, free to start.
71
+
66
72
  That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
67
73
  DuckDB — from a schema file, in seconds:
68
74
 
@@ -118,6 +124,29 @@ access required.
118
124
 
119
125
  ---
120
126
 
127
+ ## model2data studio — the same engine, in the browser
128
+
129
+ [**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
130
+ web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
131
+ model2data, and it's the better fit while a schema is still being designed:
132
+
133
+ - **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
134
+ entity diagram redraws as you type. Click a column to trace what actually joins to it.
135
+ - **See what you'll get before you generate.** Every column shows an example of the value it will
136
+ produce, and columns nothing recognises are marked — so placeholder data is visible rather than
137
+ silent.
138
+ - **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
139
+ seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
140
+ you previewed.
141
+ - **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
142
+ Confluence or wiki page.
143
+
144
+ Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
145
+ The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
146
+ model gets designed and shown.
147
+
148
+ ---
149
+
121
150
  ## Installation
122
151
 
123
152
  ```bash
@@ -323,4 +352,6 @@ MIT License. See LICENSE for details.
323
352
  <br>
324
353
  Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
325
354
  Data & Analytics Engineering · Data Platform Architecture · Modern BI.
355
+ <br>
356
+ Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
326
357
  </p>
@@ -26,6 +26,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
26
26
  cd dbt_ecommerce && dbt build
27
27
  ```
28
28
 
29
+ > **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
30
+ > engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
31
+ > column will generate before you generate it, then export CSVs or a runnable dbt project.
32
+ > Nothing to install, free to start.
33
+
29
34
  ---
30
35
 
31
36
  ## Why this exists
@@ -112,6 +117,30 @@ flowchart LR
112
117
 
113
118
  ---
114
119
 
120
+ ## model2data studio — the same engine, in the browser
121
+
122
+ [**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
123
+ web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
124
+ model2data, and it's the better fit while a schema is still being designed:
125
+
126
+ - **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
127
+ entity diagram redraws as you type. Click a column to trace what actually joins to it.
128
+ - **See what you'll get before you generate.** Every column shows an example of the value it will
129
+ produce, and columns nothing recognises are marked — so placeholder data is visible rather than
130
+ silent.
131
+ - **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
132
+ seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
133
+ you previewed.
134
+ - **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
135
+ Confluence or wiki page.
136
+
137
+ Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com)
138
+ ([plans and pricing](https://www.jbanalytica.com/model2data/pricing/)).
139
+ The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
140
+ model gets designed and shown.
141
+
142
+ ---
143
+
115
144
  ## Installation
116
145
 
117
146
  ```bash
@@ -428,4 +457,6 @@ MIT License. See LICENSE for details.
428
457
  <br>
429
458
  Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
430
459
  Data & Analytics Engineering · Data Platform Architecture · Modern BI.
460
+ <br>
461
+ Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
431
462
  </p>
@@ -19,6 +19,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
19
19
  cd dbt_ecommerce && dbt build
20
20
  ```
21
21
 
22
+ > **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
23
+ > engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
24
+ > column will generate before you generate it, then export CSVs or a runnable dbt project.
25
+ > Nothing to install, free to start.
26
+
22
27
  That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
23
28
  DuckDB — from a schema file, in seconds:
24
29
 
@@ -74,6 +79,29 @@ access required.
74
79
 
75
80
  ---
76
81
 
82
+ ## model2data studio — the same engine, in the browser
83
+
84
+ [**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
85
+ web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
86
+ model2data, and it's the better fit while a schema is still being designed:
87
+
88
+ - **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
89
+ entity diagram redraws as you type. Click a column to trace what actually joins to it.
90
+ - **See what you'll get before you generate.** Every column shows an example of the value it will
91
+ produce, and columns nothing recognises are marked — so placeholder data is visible rather than
92
+ silent.
93
+ - **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
94
+ seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
95
+ you previewed.
96
+ - **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
97
+ Confluence or wiki page.
98
+
99
+ Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
100
+ The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
101
+ model gets designed and shown.
102
+
103
+ ---
104
+
77
105
  ## Installation
78
106
 
79
107
  ```bash
@@ -279,4 +307,6 @@ MIT License. See LICENSE for details.
279
307
  <br>
280
308
  Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
281
309
  Data & Analytics Engineering · Data Platform Architecture · Modern BI.
310
+ <br>
311
+ Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
282
312
  </p>
@@ -283,19 +283,23 @@ def generate_data_from_dbml(
283
283
  continue
284
284
 
285
285
  # find FK linking child → parent
286
- fk_column = next(
286
+ fk_ref = next(
287
287
  (
288
- r["source_column"]
288
+ r
289
289
  for r in fk_refs
290
290
  if r["source_table"] == table_name and r["target_table"] == parent_table
291
291
  ),
292
292
  None,
293
293
  )
294
+ if fk_ref is None:
295
+ continue
294
296
 
295
- if not fk_column or fk_column not in df.columns:
297
+ fk_column = fk_ref["source_column"]
298
+ parent_key = fk_ref["target_column"]
299
+ if fk_column not in df.columns or parent_key not in parent_df.columns:
296
300
  continue
297
301
 
298
- lookup = parent_df.groupby("id")[parent_column].first().to_dict()
302
+ lookup = parent_df.groupby(parent_key)[parent_column].first().to_dict()
299
303
 
300
304
  df[child_column] = df[fk_column].map(lookup)
301
305
 
@@ -0,0 +1,95 @@
1
+ from typing import Dict, List, Optional, Tuple
2
+
3
+ from model2data.parse.dbml import ColumnDef, TableDef
4
+
5
+
6
+ # ---------------------------------------------------------
7
+ # Public API
8
+ # ---------------------------------------------------------
9
+ def classify_refs(
10
+ tables: Dict[str, TableDef],
11
+ refs: List[Dict],
12
+ ) -> Tuple[List[Dict], List[Dict]]:
13
+ """
14
+ Classify references into:
15
+ - fk_refs: Foreign keys (target column is a key of its table)
16
+ - attribute_refs: Non-FK dependencies (mirroring parent attributes)
17
+
18
+ A Ref onto a primary key, or a column named "id", is always a foreign key.
19
+ A Ref onto a unique column is one too -- dbt projects commonly declare
20
+ their keys with `unique` + `not_null` tests rather than a primary key
21
+ constraint -- unless the child already has a primary-key FK to the same
22
+ parent. Then it stays an attribute ref and is mirrored through that FK, so
23
+ `orders.customer_email > customers.email` next to
24
+ `orders.customer_id > customers.id` keeps the email the customer's own.
25
+ """
26
+ kinds = [_key_kind(tables, ref) for ref in refs]
27
+ pk_pairs = {
28
+ (ref["source_table"], ref["target_table"])
29
+ for ref, kind in zip(refs, kinds, strict=True)
30
+ if kind == "pk"
31
+ }
32
+
33
+ fk_refs = []
34
+ attribute_refs = []
35
+ for ref, kind in zip(refs, kinds, strict=True):
36
+ pair = (ref["source_table"], ref["target_table"])
37
+ if kind == "pk" or (kind == "unique" and pair not in pk_pairs):
38
+ fk_refs.append(ref)
39
+ else:
40
+ attribute_refs.append(ref)
41
+
42
+ return fk_refs, attribute_refs
43
+
44
+
45
+ def _key_kind(tables: Dict[str, TableDef], ref: Dict) -> Optional[str]:
46
+ """Whether the column a Ref points at is a "pk", a "unique" key, or neither (None)."""
47
+ table = tables.get(ref["target_table"])
48
+ if table is None:
49
+ return None
50
+ column = next((c for c in table.columns if c.name == ref["target_column"]), None)
51
+ if column is None:
52
+ return None
53
+ if _is_primary_key(table, column):
54
+ return "pk"
55
+ if _is_unique(table, column):
56
+ return "unique"
57
+ return None
58
+
59
+
60
+ def _single_column_keys(table: TableDef, key_type: str) -> set:
61
+ return {
62
+ key["columns"][0]
63
+ for key in table.composite_keys
64
+ if key["type"] == key_type and len(key["columns"]) == 1
65
+ }
66
+
67
+
68
+ def _is_primary_key(table: TableDef, column: ColumnDef) -> bool:
69
+ return (
70
+ "pk" in column.settings
71
+ or "primary key" in column.settings
72
+ or column.name.lower() == "id"
73
+ or column.name in _single_column_keys(table, "pk")
74
+ )
75
+
76
+
77
+ def _is_unique(table: TableDef, column: ColumnDef) -> bool:
78
+ return "unique" in column.settings or column.name in _single_column_keys(table, "unique")
79
+
80
+
81
+ def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
82
+ """
83
+ Build a lookup dictionary for FK relationships.
84
+
85
+ Returns:
86
+ {
87
+ (child_table, child_column): (parent_table, parent_column)
88
+ }
89
+ """
90
+ lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
91
+ for ref in fk_refs:
92
+ key = (ref["source_table"], ref["source_column"])
93
+ value = (ref["target_table"], ref["target_column"])
94
+ lookup[key] = value
95
+ return lookup
@@ -1,10 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.7.1
3
+ Version: 1.7.3
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/JB-Analytica/model2data
8
+ Project-URL: Studio, https://studio.jbanalytica.com
8
9
  Project-URL: Repository, https://github.com/JB-Analytica/model2data
9
10
  Project-URL: Issues, https://github.com/JB-Analytica/model2data/issues
10
11
  Project-URL: Changelog, https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md
@@ -63,6 +64,11 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
63
64
  cd dbt_ecommerce && dbt build
64
65
  ```
65
66
 
67
+ > **Prefer a browser?** [model2data studio](https://studio.jbanalytica.com) is the same
68
+ > engine as a web app: write DBML, watch the entity diagram redraw as you type, see what every
69
+ > column will generate before you generate it, then export CSVs or a runnable dbt project.
70
+ > Nothing to install, free to start.
71
+
66
72
  That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
67
73
  DuckDB — from a schema file, in seconds:
68
74
 
@@ -118,6 +124,29 @@ access required.
118
124
 
119
125
  ---
120
126
 
127
+ ## model2data studio — the same engine, in the browser
128
+
129
+ [**model2data studio**](https://studio.jbanalytica.com) puts everything on this page behind a
130
+ web UI. It is built by JB Analytica on top of this library, it's the fastest way to try
131
+ model2data, and it's the better fit while a schema is still being designed:
132
+
133
+ - **Type DBML, see the diagram.** Syntax highlighting, autocomplete and live error checking; the
134
+ entity diagram redraws as you type. Click a column to trace what actually joins to it.
135
+ - **See what you'll get before you generate.** Every column shows an example of the value it will
136
+ produce, and columns nothing recognises are marked — so placeholder data is visible rather than
137
+ silent.
138
+ - **Generate and export.** Per-table row counts, then CSVs or a complete dbt project: the same
139
+ seeds, staging models, tests and DuckDB profile this CLI produces, reproducing the exact rows
140
+ you previewed.
141
+ - **Share the model.** A share link that also embeds as a chrome-free diagram in a Notion,
142
+ Confluence or wiki page.
143
+
144
+ Free to start, nothing to install: [studio.jbanalytica.com](https://studio.jbanalytica.com).
145
+ The CLI stays the right tool for scripting, CI and fixtures you commit; the studio is where a
146
+ model gets designed and shown.
147
+
148
+ ---
149
+
121
150
  ## Installation
122
151
 
123
152
  ```bash
@@ -323,4 +352,6 @@ MIT License. See LICENSE for details.
323
352
  <br>
324
353
  Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
325
354
  Data & Analytics Engineering · Data Platform Architecture · Modern BI.
355
+ <br>
356
+ Try <a href="https://studio.jbanalytica.com"><strong>model2data studio</strong></a> — model2data in the browser, nothing to install.
326
357
  </p>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.7.1"
7
+ version = "1.7.3"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -60,6 +60,7 @@ dev = [
60
60
 
61
61
  [project.urls]
62
62
  Homepage = "https://github.com/JB-Analytica/model2data"
63
+ Studio = "https://studio.jbanalytica.com"
63
64
  Repository = "https://github.com/JB-Analytica/model2data"
64
65
  Issues = "https://github.com/JB-Analytica/model2data/issues"
65
66
  Changelog = "https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md"
@@ -638,3 +638,41 @@ def test_generic_tests_nest_parameters_under_arguments(tmp_path):
638
638
  assert set(relationships) == {"arguments"}, "params must live under `arguments:`"
639
639
  assert relationships["arguments"]["to"] == "ref('stg_customers')"
640
640
  assert relationships["arguments"]["field"] == "id"
641
+
642
+
643
+ def test_relationships_test_emitted_for_ref_onto_unique_non_pk_column(tmp_path):
644
+ """A key declared with `unique` + `not null` rather than `pk` is still a
645
+ foreign-key target: the generator draws the child from the parent's
646
+ values, so the project gets the relationships test that proves it.
647
+ """
648
+ tables = {
649
+ "customers": TableDef(
650
+ name="customers",
651
+ columns=[ColumnDef("customer_id", "int", {"not null", "unique"})],
652
+ ),
653
+ "orders": TableDef(
654
+ name="orders",
655
+ columns=[
656
+ ColumnDef("order_id", "int", {"not null", "unique"}),
657
+ ColumnDef("customer_id", "int", {"not null"}),
658
+ ],
659
+ ),
660
+ }
661
+ refs = [
662
+ {
663
+ "source_table": "orders",
664
+ "source_column": "customer_id",
665
+ "target_table": "customers",
666
+ "target_column": "customer_id",
667
+ }
668
+ ]
669
+ generate_dbt_yml(tmp_path, tables, refs, source_name="shop")
670
+
671
+ stg_yaml = yaml.safe_load((tmp_path / "models" / "staging" / "stg_orders.yml").read_text())
672
+ orders_col = next(c for c in stg_yaml["models"][0]["columns"] if c["name"] == "customer_id")
673
+ relationship_test = next(
674
+ t["relationships"]
675
+ for t in orders_col["tests"]
676
+ if isinstance(t, dict) and "relationships" in t
677
+ )
678
+ assert relationship_test["arguments"]["field"] == "customer_id"
@@ -958,3 +958,168 @@ def test_row_overrides_are_deterministic_with_a_seed():
958
958
 
959
959
  assert first["users"].equals(second["users"])
960
960
  assert first["orders"].equals(second["orders"])
961
+
962
+
963
+ def _customers_and_orders_on_a_unique_key() -> tuple[dict, list[dict]]:
964
+ # The shape a dbt import produces when the key is declared with
965
+ # `unique` + `not_null` tests rather than a primary key constraint.
966
+ tables = {
967
+ "customers": TableDef(
968
+ name="customers",
969
+ columns=[
970
+ ColumnDef("customer_id", "int", {"not null", "unique"}),
971
+ ColumnDef("name", "varchar"),
972
+ ],
973
+ ),
974
+ "orders": TableDef(
975
+ name="orders",
976
+ columns=[
977
+ ColumnDef("order_id", "int", {"not null", "unique"}),
978
+ ColumnDef("customer_id", "int", {"not null"}),
979
+ ],
980
+ ),
981
+ }
982
+ refs = [
983
+ {
984
+ "source_table": "orders",
985
+ "source_column": "customer_id",
986
+ "target_table": "customers",
987
+ "target_column": "customer_id",
988
+ }
989
+ ]
990
+ return tables, refs
991
+
992
+
993
+ def test_ref_onto_a_unique_non_pk_column_is_a_foreign_key():
994
+ # At the default 100 rows the column's own integer range is wider than the
995
+ # parent's keys, so a ref treated as unrelated data fails the dbt
996
+ # relationships test on about half of all seeds.
997
+ tables, refs = _customers_and_orders_on_a_unique_key()
998
+ for seed in range(42, 62):
999
+ data = generate_data_from_dbml(tables, refs, base_rows=100, seed=seed)
1000
+ parent_keys = set(data["customers"]["customer_id"])
1001
+ assert set(data["orders"]["customer_id"]) <= parent_keys, seed
1002
+
1003
+
1004
+ def test_ref_onto_a_unique_index_column_is_a_foreign_key():
1005
+ tables, refs = _customers_and_orders_on_a_unique_key()
1006
+ tables["customers"].columns[0].settings = {"not null"}
1007
+ tables["customers"].composite_keys = [{"columns": ["customer_id"], "type": "unique"}]
1008
+ for seed in range(42, 47):
1009
+ data = generate_data_from_dbml(tables, refs, base_rows=100, seed=seed)
1010
+ parent_keys = set(data["customers"]["customer_id"])
1011
+ assert set(data["orders"]["customer_id"]) <= parent_keys, seed
1012
+
1013
+
1014
+ def test_ref_onto_a_unique_column_beside_a_pk_fk_is_still_mirrored():
1015
+ tables = {
1016
+ "customers": TableDef(
1017
+ name="customers",
1018
+ columns=[
1019
+ ColumnDef("id", "int", {"pk"}),
1020
+ ColumnDef("email", "varchar", {"unique"}),
1021
+ ],
1022
+ ),
1023
+ "orders": TableDef(
1024
+ name="orders",
1025
+ columns=[
1026
+ ColumnDef("id", "int", {"pk"}),
1027
+ ColumnDef("customer_id", "int"),
1028
+ ColumnDef("customer_email", "varchar"),
1029
+ ],
1030
+ ),
1031
+ }
1032
+ refs = [
1033
+ {
1034
+ "source_table": "orders",
1035
+ "source_column": "customer_id",
1036
+ "target_table": "customers",
1037
+ "target_column": "id",
1038
+ },
1039
+ {
1040
+ "source_table": "orders",
1041
+ "source_column": "customer_email",
1042
+ "target_table": "customers",
1043
+ "target_column": "email",
1044
+ },
1045
+ ]
1046
+ data = generate_data_from_dbml(tables, refs, base_rows=50, seed=7)
1047
+ email_of = data["customers"].set_index("id")["email"]
1048
+ orders = data["orders"]
1049
+ assert orders["customer_email"].equals(orders["customer_id"].map(email_of))
1050
+
1051
+
1052
+ def test_mirroring_goes_through_a_foreign_key_onto_a_non_id_key():
1053
+ tables, refs = _customers_and_orders_on_a_unique_key()
1054
+ tables["orders"].columns.append(ColumnDef("customer_name", "varchar"))
1055
+ refs.append(
1056
+ {
1057
+ "source_table": "orders",
1058
+ "source_column": "customer_name",
1059
+ "target_table": "customers",
1060
+ "target_column": "name",
1061
+ }
1062
+ )
1063
+ data = generate_data_from_dbml(tables, refs, base_rows=50, seed=7)
1064
+ name_of = data["customers"].set_index("customer_id")["name"]
1065
+ orders = data["orders"]
1066
+ assert orders["customer_name"].equals(orders["customer_id"].map(name_of))
1067
+
1068
+
1069
+ def test_mirroring_is_skipped_when_the_fk_column_is_not_in_the_child_table():
1070
+ # A Ref naming a child column the table doesn't declare still classifies
1071
+ # as a foreign key; the mirror that would ride on it has nothing to read.
1072
+ tables = {
1073
+ "customers": TableDef(
1074
+ name="customers",
1075
+ columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("name", "varchar")],
1076
+ ),
1077
+ "orders": TableDef(
1078
+ name="orders",
1079
+ columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("customer_name", "varchar")],
1080
+ ),
1081
+ }
1082
+ refs = [
1083
+ {
1084
+ "source_table": "orders",
1085
+ "source_column": "customer_id",
1086
+ "target_table": "customers",
1087
+ "target_column": "id",
1088
+ },
1089
+ {
1090
+ "source_table": "orders",
1091
+ "source_column": "customer_name",
1092
+ "target_table": "customers",
1093
+ "target_column": "name",
1094
+ },
1095
+ ]
1096
+ data = generate_data_from_dbml(tables, refs, base_rows=20, seed=3)
1097
+ orders = data["orders"]
1098
+ assert "customer_id" not in orders.columns
1099
+ assert not orders["customer_name"].isin(data["customers"]["name"]).all()
1100
+
1101
+
1102
+ def test_attribute_ref_without_fk_is_skipped_when_the_parent_is_generated_first():
1103
+ # Tables with no FK between them are generated in name order, so here the
1104
+ # parent ("accounts") already exists when the child's mirror pass runs and
1105
+ # it is the missing FK, not the missing parent, that skips the mirror.
1106
+ tables = {
1107
+ "accounts": TableDef(
1108
+ name="accounts",
1109
+ columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("name", "varchar")],
1110
+ ),
1111
+ "orders": TableDef(
1112
+ name="orders",
1113
+ columns=[ColumnDef("id", "int", {"pk"}), ColumnDef("account_name", "varchar")],
1114
+ ),
1115
+ }
1116
+ refs = [
1117
+ {
1118
+ "source_table": "orders",
1119
+ "source_column": "account_name",
1120
+ "target_table": "accounts",
1121
+ "target_column": "name",
1122
+ }
1123
+ ]
1124
+ data = generate_data_from_dbml(tables, refs, base_rows=20, seed=2)
1125
+ assert not data["orders"]["account_name"].isin(data["accounts"]["name"]).all()
@@ -1,52 +0,0 @@
1
- from typing import Dict, List, Tuple
2
-
3
- from model2data.parse.dbml import TableDef
4
-
5
-
6
- # ---------------------------------------------------------
7
- # Public API
8
- # ---------------------------------------------------------
9
- def classify_refs(
10
- tables: Dict[str, TableDef],
11
- refs: List[Dict],
12
- ) -> Tuple[List[Dict], List[Dict]]:
13
- """
14
- Classify references into:
15
- - fk_refs: Foreign keys (target column looks like a PK)
16
- - attribute_refs: Non-FK dependencies (mirroring parent attributes)
17
- """
18
- fk_refs = []
19
- attribute_refs = []
20
-
21
- for ref in refs:
22
- target_table = tables.get(ref["target_table"])
23
- target_col = None
24
- if target_table:
25
- target_col = next(
26
- (c for c in target_table.columns if c.name == ref["target_column"]), None
27
- )
28
-
29
- # FK if target column is a primary key or named "id"
30
- if target_col and ("pk" in target_col.settings or target_col.name.lower() == "id"):
31
- fk_refs.append(ref)
32
- else:
33
- attribute_refs.append(ref)
34
-
35
- return fk_refs, attribute_refs
36
-
37
-
38
- def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
39
- """
40
- Build a lookup dictionary for FK relationships.
41
-
42
- Returns:
43
- {
44
- (child_table, child_column): (parent_table, parent_column)
45
- }
46
- """
47
- lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
48
- for ref in fk_refs:
49
- key = (ref["source_table"], ref["source_column"])
50
- value = (ref["target_table"], ref["target_column"])
51
- lookup[key] = value
52
- return lookup
File without changes
File without changes
File without changes
File without changes