model2data 0.1.0__tar.gz → 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-0.1.0 → model2data-0.1.1}/PKG-INFO +2 -1
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/PKG-INFO +2 -1
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/SOURCES.txt +0 -5
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/requires.txt +1 -0
- {model2data-0.1.0 → model2data-0.1.1}/pyproject.toml +2 -1
- model2data-0.1.0/model2data/dbt/project.py +0 -74
- model2data-0.1.0/model2data/generate/core.py +0 -171
- model2data-0.1.0/model2data/generate/faker.py +0 -113
- model2data-0.1.0/model2data/generate/relationships.py +0 -49
- model2data-0.1.0/model2data/parse/dbml.py +0 -162
- {model2data-0.1.0 → model2data-0.1.1}/LICENSE +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/README.md +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/model2data/cli.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/model2data/utils.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/top_level.txt +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/setup.cfg +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/tests/test_cli_smoke.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbml_parser.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbt_naming.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbt_tests.py +0 -0
- {model2data-0.1.0 → model2data-0.1.1}/tests/test_generation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.1
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Requires-Python: >=3.9
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -13,6 +13,7 @@ Requires-Dist: pyyaml>=6.0.3
|
|
|
13
13
|
Requires-Dist: typer>=0.20.0
|
|
14
14
|
Provides-Extra: dev
|
|
15
15
|
Requires-Dist: pytest; extra == "dev"
|
|
16
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
16
17
|
Dynamic: license-file
|
|
17
18
|
|
|
18
19
|
# model2data
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: model2data
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.1
|
|
4
4
|
Summary: Generate analytics-ready datasets from DBML models
|
|
5
5
|
Requires-Python: >=3.9
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -13,6 +13,7 @@ Requires-Dist: pyyaml>=6.0.3
|
|
|
13
13
|
Requires-Dist: typer>=0.20.0
|
|
14
14
|
Provides-Extra: dev
|
|
15
15
|
Requires-Dist: pytest; extra == "dev"
|
|
16
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
16
17
|
Dynamic: license-file
|
|
17
18
|
|
|
18
19
|
# model2data
|
|
@@ -9,11 +9,6 @@ model2data.egg-info/dependency_links.txt
|
|
|
9
9
|
model2data.egg-info/entry_points.txt
|
|
10
10
|
model2data.egg-info/requires.txt
|
|
11
11
|
model2data.egg-info/top_level.txt
|
|
12
|
-
model2data/dbt/project.py
|
|
13
|
-
model2data/generate/core.py
|
|
14
|
-
model2data/generate/faker.py
|
|
15
|
-
model2data/generate/relationships.py
|
|
16
|
-
model2data/parse/dbml.py
|
|
17
12
|
tests/test_cli_smoke.py
|
|
18
13
|
tests/test_dbml_parser.py
|
|
19
14
|
tests/test_dbt_naming.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "model2data"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.1"
|
|
8
8
|
description = "Generate analytics-ready datasets from DBML models"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -20,6 +20,7 @@ dependencies = [
|
|
|
20
20
|
[project.optional-dependencies]
|
|
21
21
|
dev = [
|
|
22
22
|
"pytest",
|
|
23
|
+
"pytest-cov",
|
|
23
24
|
]
|
|
24
25
|
|
|
25
26
|
[project.scripts]
|
|
@@ -1,74 +0,0 @@
|
|
|
1
|
-
from pathlib import Path
|
|
2
|
-
import shutil
|
|
3
|
-
import jinja2
|
|
4
|
-
|
|
5
|
-
TEMPLATES_DIR = Path(__file__).parent / "templates"
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
def create_project_scaffold(dest: Path, project_name: str, profile_name: str) -> None:
|
|
9
|
-
dest.mkdir(parents=True, exist_ok=True)
|
|
10
|
-
|
|
11
|
-
# dbt folders
|
|
12
|
-
(dest / "models" / "staging").mkdir(parents=True, exist_ok=True)
|
|
13
|
-
(dest / "seeds" / "raw").mkdir(parents=True, exist_ok=True)
|
|
14
|
-
(dest / "analysis").mkdir(exist_ok=True)
|
|
15
|
-
(dest / "macros").mkdir(exist_ok=True)
|
|
16
|
-
(dest / "tests").mkdir(exist_ok=True)
|
|
17
|
-
(dest / "snapshots").mkdir(exist_ok=True)
|
|
18
|
-
|
|
19
|
-
# dbt_project.yml
|
|
20
|
-
_render_template(
|
|
21
|
-
template_name="dbt_project.yml.jinja",
|
|
22
|
-
output_path=dest / "dbt_project.yml",
|
|
23
|
-
context={"project_name": project_name, "profile_name": profile_name},
|
|
24
|
-
)
|
|
25
|
-
|
|
26
|
-
# Copy over any macros from templates
|
|
27
|
-
template_macros_dir = Path("model2data/dbt/templates/macros")
|
|
28
|
-
if template_macros_dir.exists():
|
|
29
|
-
for macro_file in template_macros_dir.glob("*.sql"):
|
|
30
|
-
target_file = dest / "macros" / macro_file.name
|
|
31
|
-
if not target_file.exists():
|
|
32
|
-
target_file.write_text(macro_file.read_text())
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def create_staging_models(dest: Path, project_name: str) -> None:
|
|
36
|
-
"""
|
|
37
|
-
Creates staging models in models/staging/ folder that reference raw seed tables as sources.
|
|
38
|
-
"""
|
|
39
|
-
seeds_path = dest / "seeds" / "raw"
|
|
40
|
-
models_path = dest / "models" / "staging"
|
|
41
|
-
models_path.mkdir(parents=True, exist_ok=True)
|
|
42
|
-
|
|
43
|
-
for csv_file in seeds_path.glob("*.csv"):
|
|
44
|
-
table_name = csv_file.stem # keep full seed name, e.g., raw_stories
|
|
45
|
-
model_file = models_path / f"stg_{table_name}.sql"
|
|
46
|
-
|
|
47
|
-
if not model_file.exists():
|
|
48
|
-
sql_content = f"""\
|
|
49
|
-
-- Auto-generated staging model for {table_name}
|
|
50
|
-
select *
|
|
51
|
-
from {{{{ source('raw', '{table_name}') }}}}
|
|
52
|
-
"""
|
|
53
|
-
model_file.write_text(sql_content)
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
def create_profiles_yml(dest: Path, profile_name: str) -> None:
|
|
57
|
-
profiles_file = dest / "profiles.yml"
|
|
58
|
-
if profiles_file.exists():
|
|
59
|
-
content = profiles_file.read_text()
|
|
60
|
-
if profile_name in content:
|
|
61
|
-
return
|
|
62
|
-
_render_template(
|
|
63
|
-
template_name="profiles.yml.jinja",
|
|
64
|
-
output_path=profiles_file,
|
|
65
|
-
context={"profile_name": profile_name},
|
|
66
|
-
)
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
def _render_template(template_name: str, output_path: Path, context: dict) -> None:
|
|
70
|
-
template_path = TEMPLATES_DIR / template_name
|
|
71
|
-
if not template_path.exists():
|
|
72
|
-
raise FileNotFoundError(f"Template not found: {template_path}")
|
|
73
|
-
template = jinja2.Template(template_path.read_text())
|
|
74
|
-
output_path.write_text(template.render(**context))
|
|
@@ -1,171 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import random
|
|
4
|
-
from collections import defaultdict, deque
|
|
5
|
-
from typing import Optional
|
|
6
|
-
|
|
7
|
-
import pandas as pd
|
|
8
|
-
from faker import Faker
|
|
9
|
-
|
|
10
|
-
from model2data.generate.faker import generate_column_values
|
|
11
|
-
from model2data.generate.relationships import (
|
|
12
|
-
classify_refs,
|
|
13
|
-
build_fk_lookup,
|
|
14
|
-
)
|
|
15
|
-
from model2data.parse.dbml import TableDef
|
|
16
|
-
|
|
17
|
-
fake = Faker()
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
# ---------------------------------------------------------
|
|
21
|
-
# Public API
|
|
22
|
-
# ---------------------------------------------------------
|
|
23
|
-
def generate_data_from_dbml(
|
|
24
|
-
tables: dict[str, TableDef],
|
|
25
|
-
refs: list[dict],
|
|
26
|
-
base_rows: int = 100,
|
|
27
|
-
seed: Optional[int] = None,
|
|
28
|
-
) -> dict[str, pd.DataFrame]:
|
|
29
|
-
"""
|
|
30
|
-
Generate synthetic datasets from parsed DBML definitions.
|
|
31
|
-
|
|
32
|
-
This function is deterministic if a seed is provided.
|
|
33
|
-
It performs no filesystem I/O and returns pandas DataFrames.
|
|
34
|
-
"""
|
|
35
|
-
if seed is not None:
|
|
36
|
-
random.seed(seed)
|
|
37
|
-
Faker.seed(seed)
|
|
38
|
-
|
|
39
|
-
# ---------------------------------------------------------
|
|
40
|
-
# Classify references
|
|
41
|
-
# ---------------------------------------------------------
|
|
42
|
-
fk_refs, attribute_refs = classify_refs(tables, refs)
|
|
43
|
-
fk_lookup = build_fk_lookup(fk_refs)
|
|
44
|
-
|
|
45
|
-
# ---------------------------------------------------------
|
|
46
|
-
# Generate tables in dependency order
|
|
47
|
-
# ---------------------------------------------------------
|
|
48
|
-
ordered_tables = _topological_table_order(tables, fk_refs)
|
|
49
|
-
generated: dict[str, pd.DataFrame] = {}
|
|
50
|
-
|
|
51
|
-
for table_name in ordered_tables:
|
|
52
|
-
table_def = tables[table_name]
|
|
53
|
-
row_count = _determine_row_count(table_def.name, base_rows)
|
|
54
|
-
|
|
55
|
-
data: dict[str, list] = {}
|
|
56
|
-
|
|
57
|
-
# -----------------------
|
|
58
|
-
# First pass: columns + FKs
|
|
59
|
-
# -----------------------
|
|
60
|
-
for column in table_def.columns:
|
|
61
|
-
fk_series = None
|
|
62
|
-
fk_target = fk_lookup.get((table_name, column.name))
|
|
63
|
-
|
|
64
|
-
if fk_target:
|
|
65
|
-
parent_table, parent_column = fk_target
|
|
66
|
-
parent_df = generated.get(parent_table)
|
|
67
|
-
if parent_df is not None and parent_column in parent_df.columns:
|
|
68
|
-
fk_series = parent_df[parent_column]
|
|
69
|
-
|
|
70
|
-
ensure_unique = "pk" in column.settings
|
|
71
|
-
data[column.name] = generate_column_values(
|
|
72
|
-
column=column,
|
|
73
|
-
row_count=row_count,
|
|
74
|
-
fk_series=fk_series,
|
|
75
|
-
ensure_unique=ensure_unique,
|
|
76
|
-
)
|
|
77
|
-
|
|
78
|
-
df = pd.DataFrame(data)
|
|
79
|
-
|
|
80
|
-
# -----------------------------------------------------
|
|
81
|
-
# Second pass: attribute mirroring (non-FK refs)
|
|
82
|
-
# -----------------------------------------------------
|
|
83
|
-
for ref in attribute_refs:
|
|
84
|
-
if ref["source_table"] != table_name:
|
|
85
|
-
continue
|
|
86
|
-
|
|
87
|
-
parent_table = ref["target_table"]
|
|
88
|
-
parent_column = ref["target_column"]
|
|
89
|
-
child_column = ref["source_column"]
|
|
90
|
-
|
|
91
|
-
parent_df = generated.get(parent_table)
|
|
92
|
-
if parent_df is None:
|
|
93
|
-
continue
|
|
94
|
-
|
|
95
|
-
# find FK linking child → parent
|
|
96
|
-
fk_column = next(
|
|
97
|
-
(
|
|
98
|
-
r["source_column"]
|
|
99
|
-
for r in fk_refs
|
|
100
|
-
if r["source_table"] == table_name
|
|
101
|
-
and r["target_table"] == parent_table
|
|
102
|
-
),
|
|
103
|
-
None,
|
|
104
|
-
)
|
|
105
|
-
|
|
106
|
-
if not fk_column or fk_column not in df.columns:
|
|
107
|
-
continue
|
|
108
|
-
|
|
109
|
-
lookup = (
|
|
110
|
-
parent_df.groupby("id")[parent_column]
|
|
111
|
-
.first()
|
|
112
|
-
.to_dict()
|
|
113
|
-
)
|
|
114
|
-
|
|
115
|
-
df[child_column] = df[fk_column].map(lookup)
|
|
116
|
-
|
|
117
|
-
generated[table_name] = df
|
|
118
|
-
|
|
119
|
-
return generated
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
# ---------------------------------------------------------
|
|
123
|
-
# Internal helpers
|
|
124
|
-
# ---------------------------------------------------------
|
|
125
|
-
def _determine_row_count(table_name: str, base_rows: int) -> int:
|
|
126
|
-
"""
|
|
127
|
-
Return the base number of rows for all tables.
|
|
128
|
-
"""
|
|
129
|
-
return base_rows
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
def _topological_table_order(
|
|
133
|
-
tables: dict[str, TableDef],
|
|
134
|
-
fk_refs: list[dict],
|
|
135
|
-
) -> list[str]:
|
|
136
|
-
"""
|
|
137
|
-
Order tables so parent tables are generated before children.
|
|
138
|
-
"""
|
|
139
|
-
graph: dict[str, set[str]] = defaultdict(set)
|
|
140
|
-
indegree: dict[str, int] = {name: 0 for name in tables.keys()}
|
|
141
|
-
|
|
142
|
-
for ref in fk_refs:
|
|
143
|
-
parent = ref["target_table"]
|
|
144
|
-
child = ref["source_table"]
|
|
145
|
-
|
|
146
|
-
if parent == child:
|
|
147
|
-
continue
|
|
148
|
-
if parent not in tables or child not in tables:
|
|
149
|
-
continue
|
|
150
|
-
|
|
151
|
-
if child not in graph[parent]:
|
|
152
|
-
graph[parent].add(child)
|
|
153
|
-
indegree[child] += 1
|
|
154
|
-
|
|
155
|
-
queue = deque(sorted(name for name, deg in indegree.items() if deg == 0))
|
|
156
|
-
order: list[str] = []
|
|
157
|
-
|
|
158
|
-
while queue:
|
|
159
|
-
node = queue.popleft()
|
|
160
|
-
order.append(node)
|
|
161
|
-
for neighbor in sorted(graph.get(node, [])):
|
|
162
|
-
indegree[neighbor] -= 1
|
|
163
|
-
if indegree[neighbor] == 0:
|
|
164
|
-
queue.append(neighbor)
|
|
165
|
-
|
|
166
|
-
# Safety net for disconnected tables
|
|
167
|
-
for name in tables.keys():
|
|
168
|
-
if name not in order:
|
|
169
|
-
order.append(name)
|
|
170
|
-
|
|
171
|
-
return order
|
|
@@ -1,113 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import random
|
|
4
|
-
import uuid
|
|
5
|
-
from datetime import datetime, timedelta
|
|
6
|
-
from typing import Optional
|
|
7
|
-
|
|
8
|
-
import pandas as pd
|
|
9
|
-
from faker import Faker
|
|
10
|
-
|
|
11
|
-
from model2data.parse.dbml import ColumnDef
|
|
12
|
-
|
|
13
|
-
fake = Faker()
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
# ---------------------------------------------------------
|
|
17
|
-
# Public API
|
|
18
|
-
# ---------------------------------------------------------
|
|
19
|
-
def generate_column_values(
|
|
20
|
-
column: ColumnDef,
|
|
21
|
-
row_count: int,
|
|
22
|
-
fk_series: Optional[pd.Series] = None,
|
|
23
|
-
ensure_unique: bool = False,
|
|
24
|
-
) -> list:
|
|
25
|
-
"""
|
|
26
|
-
Generate synthetic values for a single column.
|
|
27
|
-
Respects FKs, uniqueness, and optional min/max hints in column notes.
|
|
28
|
-
"""
|
|
29
|
-
if fk_series is not None and not fk_series.empty:
|
|
30
|
-
fk_values = fk_series.tolist()
|
|
31
|
-
return [random.choice(fk_values) for _ in range(row_count)]
|
|
32
|
-
|
|
33
|
-
dtype = column.data_type.lower()
|
|
34
|
-
base_type = dtype.split("(")[0].strip()
|
|
35
|
-
values: list = []
|
|
36
|
-
|
|
37
|
-
# -----------------------------------------------------
|
|
38
|
-
# UUIDs / hashes
|
|
39
|
-
# -----------------------------------------------------
|
|
40
|
-
if "uuid" in base_type or "hash" in base_type:
|
|
41
|
-
values = [str(uuid.uuid4()) for _ in range(row_count)]
|
|
42
|
-
|
|
43
|
-
# -----------------------------------------------------
|
|
44
|
-
# Integers
|
|
45
|
-
# -----------------------------------------------------
|
|
46
|
-
elif any(key in base_type for key in ["int", "integer", "bigint", "smallint"]):
|
|
47
|
-
min_val = 0
|
|
48
|
-
max_val = 100
|
|
49
|
-
if column.note:
|
|
50
|
-
if "min" in column.note:
|
|
51
|
-
min_val = column.note["min"]
|
|
52
|
-
if "max" in column.note:
|
|
53
|
-
max_val = column.note["max"]
|
|
54
|
-
values = [random.randint(min_val, max_val) for _ in range(row_count)]
|
|
55
|
-
|
|
56
|
-
# -----------------------------------------------------
|
|
57
|
-
# Floats / decimals
|
|
58
|
-
# -----------------------------------------------------
|
|
59
|
-
elif any(key in base_type for key in ["decimal", "numeric", "float", "double"]):
|
|
60
|
-
values = [round(random.uniform(0, 10_000), 2) for _ in range(row_count)]
|
|
61
|
-
|
|
62
|
-
# -----------------------------------------------------
|
|
63
|
-
# Booleans
|
|
64
|
-
# -----------------------------------------------------
|
|
65
|
-
elif "boolean" in base_type or "bool" in base_type:
|
|
66
|
-
values = [random.choice([True, False]) for _ in range(row_count)]
|
|
67
|
-
|
|
68
|
-
# -----------------------------------------------------
|
|
69
|
-
# Dates
|
|
70
|
-
# -----------------------------------------------------
|
|
71
|
-
elif "date" in base_type and "time" not in base_type:
|
|
72
|
-
values = [fake.date_between(start_date="-2y", end_date="today") for _ in range(row_count)]
|
|
73
|
-
|
|
74
|
-
elif "time" in base_type and "stamp" not in base_type:
|
|
75
|
-
values = [fake.time() for _ in range(row_count)]
|
|
76
|
-
|
|
77
|
-
elif any(key in base_type for key in ["timestamp", "datetime"]):
|
|
78
|
-
values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
|
|
79
|
-
|
|
80
|
-
# -----------------------------------------------------
|
|
81
|
-
# Fallback to Faker providers
|
|
82
|
-
# -----------------------------------------------------
|
|
83
|
-
else:
|
|
84
|
-
try:
|
|
85
|
-
values = [fake.format(base_type) for _ in range(row_count)]
|
|
86
|
-
except:
|
|
87
|
-
if column.name.lower().endswith("_id") or ensure_unique:
|
|
88
|
-
values = [str(uuid.uuid4()) for _ in range(row_count)]
|
|
89
|
-
else:
|
|
90
|
-
values = [fake.sentence(nb_words=3) for _ in range(row_count)]
|
|
91
|
-
|
|
92
|
-
# -----------------------------------------------------
|
|
93
|
-
# Nullability
|
|
94
|
-
# -----------------------------------------------------
|
|
95
|
-
if "not null" not in column.settings:
|
|
96
|
-
null_fraction = max(0, min(0.2, 1 - (row_count / (row_count + 50))))
|
|
97
|
-
sample_size = int(row_count * null_fraction)
|
|
98
|
-
if sample_size:
|
|
99
|
-
for idx in random.sample(range(row_count), k=sample_size):
|
|
100
|
-
values[idx] = None
|
|
101
|
-
|
|
102
|
-
return values
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
# ---------------------------------------------------------
|
|
106
|
-
# Internal helpers
|
|
107
|
-
# ---------------------------------------------------------
|
|
108
|
-
def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
|
|
109
|
-
start = datetime.now() + timedelta(days=start_days)
|
|
110
|
-
end = datetime.now() + timedelta(days=end_days)
|
|
111
|
-
delta = end - start
|
|
112
|
-
random_second = random.randint(0, int(delta.total_seconds()))
|
|
113
|
-
return start + timedelta(seconds=random_second)
|
|
@@ -1,49 +0,0 @@
|
|
|
1
|
-
from typing import Tuple, List, Dict
|
|
2
|
-
from model2data.parse.dbml import TableDef
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
# ---------------------------------------------------------
|
|
6
|
-
# Public API
|
|
7
|
-
# ---------------------------------------------------------
|
|
8
|
-
def classify_refs(
|
|
9
|
-
tables: Dict[str, TableDef],
|
|
10
|
-
refs: List[Dict],
|
|
11
|
-
) -> Tuple[List[Dict], List[Dict]]:
|
|
12
|
-
"""
|
|
13
|
-
Classify references into:
|
|
14
|
-
- fk_refs: Foreign keys (target column looks like a PK)
|
|
15
|
-
- attribute_refs: Non-FK dependencies (mirroring parent attributes)
|
|
16
|
-
"""
|
|
17
|
-
fk_refs = []
|
|
18
|
-
attribute_refs = []
|
|
19
|
-
|
|
20
|
-
for ref in refs:
|
|
21
|
-
target_table = tables.get(ref["target_table"])
|
|
22
|
-
target_col = None
|
|
23
|
-
if target_table:
|
|
24
|
-
target_col = next((c for c in target_table.columns if c.name == ref["target_column"]), None)
|
|
25
|
-
|
|
26
|
-
# FK if target column is a primary key or named "id"
|
|
27
|
-
if target_col and ("pk" in target_col.settings or target_col.name.lower() == "id"):
|
|
28
|
-
fk_refs.append(ref)
|
|
29
|
-
else:
|
|
30
|
-
attribute_refs.append(ref)
|
|
31
|
-
|
|
32
|
-
return fk_refs, attribute_refs
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
|
|
36
|
-
"""
|
|
37
|
-
Build a lookup dictionary for FK relationships.
|
|
38
|
-
|
|
39
|
-
Returns:
|
|
40
|
-
{
|
|
41
|
-
(child_table, child_column): (parent_table, parent_column)
|
|
42
|
-
}
|
|
43
|
-
"""
|
|
44
|
-
lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
|
|
45
|
-
for ref in fk_refs:
|
|
46
|
-
key = (ref["source_table"], ref["source_column"])
|
|
47
|
-
value = (ref["target_table"], ref["target_column"])
|
|
48
|
-
lookup[key] = value
|
|
49
|
-
return lookup
|
|
@@ -1,162 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
from pathlib import Path
|
|
3
|
-
from typing import Optional
|
|
4
|
-
from dataclasses import dataclass, field
|
|
5
|
-
import re
|
|
6
|
-
import json
|
|
7
|
-
import uuid
|
|
8
|
-
|
|
9
|
-
# -------------------------------
|
|
10
|
-
# Dataclasses
|
|
11
|
-
# -------------------------------
|
|
12
|
-
|
|
13
|
-
@dataclass
|
|
14
|
-
class ColumnDef:
|
|
15
|
-
name: str
|
|
16
|
-
data_type: str
|
|
17
|
-
settings: set[str] = field(default_factory=set)
|
|
18
|
-
note: Optional[dict] = None
|
|
19
|
-
|
|
20
|
-
@dataclass
|
|
21
|
-
class TableDef:
|
|
22
|
-
name: str
|
|
23
|
-
columns: list[ColumnDef] = field(default_factory=list)
|
|
24
|
-
|
|
25
|
-
# -------------------------------
|
|
26
|
-
# Helpers
|
|
27
|
-
# -------------------------------
|
|
28
|
-
|
|
29
|
-
def _strip_quotes(value: str) -> str:
|
|
30
|
-
return value.strip().strip('"').strip("'")
|
|
31
|
-
|
|
32
|
-
def _parse_column_settings(raw: Optional[str]) -> set[str]:
|
|
33
|
-
if not raw:
|
|
34
|
-
return set()
|
|
35
|
-
parts = [part.strip() for part in raw.split(",")]
|
|
36
|
-
return {part.strip("'").strip('"').lower() for part in parts if part}
|
|
37
|
-
|
|
38
|
-
def normalize_identifier(value: str) -> str:
|
|
39
|
-
cleaned = re.sub(r"[^0-9A-Za-z]+", "_", value).strip("_").lower()
|
|
40
|
-
if not cleaned:
|
|
41
|
-
cleaned = "table"
|
|
42
|
-
if cleaned[0].isdigit():
|
|
43
|
-
cleaned = f"t_{cleaned}"
|
|
44
|
-
return cleaned
|
|
45
|
-
|
|
46
|
-
def parse_dbml(dbml_path: Path) -> tuple[dict[str, TableDef], list[dict]]:
|
|
47
|
-
text = dbml_path.read_text(encoding="utf-8")
|
|
48
|
-
lines = text.splitlines()
|
|
49
|
-
tables: dict[str, TableDef] = {}
|
|
50
|
-
refs: list[dict] = []
|
|
51
|
-
|
|
52
|
-
current_table: Optional[TableDef] = None
|
|
53
|
-
in_indexes_block = False
|
|
54
|
-
note_block_depth = 0
|
|
55
|
-
in_ref_block = False # NEW
|
|
56
|
-
|
|
57
|
-
for raw_line in lines:
|
|
58
|
-
line = raw_line.strip()
|
|
59
|
-
if not line or line.startswith("//"):
|
|
60
|
-
continue
|
|
61
|
-
|
|
62
|
-
cleaned = line.split("//", 1)[0].strip()
|
|
63
|
-
if not cleaned:
|
|
64
|
-
continue
|
|
65
|
-
|
|
66
|
-
triple_quote_count = cleaned.count("'''")
|
|
67
|
-
if triple_quote_count:
|
|
68
|
-
note_block_depth = (note_block_depth + triple_quote_count) % 2
|
|
69
|
-
if cleaned.startswith("Note:"):
|
|
70
|
-
continue
|
|
71
|
-
if note_block_depth:
|
|
72
|
-
continue
|
|
73
|
-
|
|
74
|
-
# ----------------------
|
|
75
|
-
# TABLE PARSING
|
|
76
|
-
# ----------------------
|
|
77
|
-
if cleaned.lower().startswith("table "):
|
|
78
|
-
table_name_section = cleaned[6:].split("{", 1)[0].strip()
|
|
79
|
-
if "[" in table_name_section:
|
|
80
|
-
table_name_section = table_name_section.split("[", 1)[0].strip()
|
|
81
|
-
table_name = _strip_quotes(table_name_section)
|
|
82
|
-
current_table = TableDef(name=table_name)
|
|
83
|
-
continue
|
|
84
|
-
|
|
85
|
-
if current_table:
|
|
86
|
-
if cleaned.startswith("indexes"):
|
|
87
|
-
in_indexes_block = True
|
|
88
|
-
continue
|
|
89
|
-
if in_indexes_block:
|
|
90
|
-
if cleaned.endswith("}"):
|
|
91
|
-
in_indexes_block = False
|
|
92
|
-
continue
|
|
93
|
-
if cleaned.startswith("}"):
|
|
94
|
-
tables[current_table.name] = current_table
|
|
95
|
-
current_table = None
|
|
96
|
-
continue
|
|
97
|
-
if cleaned.startswith("Note:"):
|
|
98
|
-
continue
|
|
99
|
-
|
|
100
|
-
col_match = re.match(
|
|
101
|
-
r'(".*?"|`.*?`|[\w]+)\s+([^\[]+?)(?:\s+\[(.+)\])?$',
|
|
102
|
-
cleaned,
|
|
103
|
-
)
|
|
104
|
-
if not col_match:
|
|
105
|
-
continue
|
|
106
|
-
|
|
107
|
-
col_name = _strip_quotes(col_match.group(1))
|
|
108
|
-
col_type = col_match.group(2).strip()
|
|
109
|
-
settings = _parse_column_settings(col_match.group(3))
|
|
110
|
-
|
|
111
|
-
current_table.columns.append(
|
|
112
|
-
ColumnDef(
|
|
113
|
-
name=col_name,
|
|
114
|
-
data_type=col_type,
|
|
115
|
-
settings=settings,
|
|
116
|
-
note=None,
|
|
117
|
-
)
|
|
118
|
-
)
|
|
119
|
-
continue
|
|
120
|
-
|
|
121
|
-
# ----------------------
|
|
122
|
-
# REF BLOCK START
|
|
123
|
-
# ----------------------
|
|
124
|
-
if cleaned.startswith("Ref"):
|
|
125
|
-
in_ref_block = True
|
|
126
|
-
continue
|
|
127
|
-
|
|
128
|
-
if in_ref_block:
|
|
129
|
-
if cleaned.startswith("}"):
|
|
130
|
-
in_ref_block = False
|
|
131
|
-
continue
|
|
132
|
-
|
|
133
|
-
# Match: "table"."column" > "table"."column"
|
|
134
|
-
ref_match = re.match(
|
|
135
|
-
r'(".*?"|`.*?`|[\w]+)\.(".*?"|`.*?`|[\w]+)\s*([<>])\s*'
|
|
136
|
-
r'(".*?"|`.*?`|[\w]+)\.(".*?"|`.*?`|[\w]+)',
|
|
137
|
-
cleaned,
|
|
138
|
-
)
|
|
139
|
-
if not ref_match:
|
|
140
|
-
continue
|
|
141
|
-
|
|
142
|
-
left_table, left_column, operator, right_table, right_column = ref_match.groups()
|
|
143
|
-
|
|
144
|
-
# Ignore <> and other non-FK relations
|
|
145
|
-
if operator not in (">", "<"):
|
|
146
|
-
continue
|
|
147
|
-
|
|
148
|
-
if operator == "<":
|
|
149
|
-
left_table, right_table = right_table, left_table
|
|
150
|
-
left_column, right_column = right_column, left_column
|
|
151
|
-
|
|
152
|
-
refs.append(
|
|
153
|
-
{
|
|
154
|
-
"source_table": _strip_quotes(left_table),
|
|
155
|
-
"source_column": _strip_quotes(left_column),
|
|
156
|
-
"target_table": _strip_quotes(right_table),
|
|
157
|
-
"target_column": _strip_quotes(right_column),
|
|
158
|
-
}
|
|
159
|
-
)
|
|
160
|
-
continue
|
|
161
|
-
|
|
162
|
-
return tables, refs
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|