model2data 0.1.0__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {model2data-0.1.0 → model2data-0.1.1}/PKG-INFO +2 -1
  2. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/PKG-INFO +2 -1
  3. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/SOURCES.txt +0 -5
  4. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/requires.txt +1 -0
  5. {model2data-0.1.0 → model2data-0.1.1}/pyproject.toml +2 -1
  6. model2data-0.1.0/model2data/dbt/project.py +0 -74
  7. model2data-0.1.0/model2data/generate/core.py +0 -171
  8. model2data-0.1.0/model2data/generate/faker.py +0 -113
  9. model2data-0.1.0/model2data/generate/relationships.py +0 -49
  10. model2data-0.1.0/model2data/parse/dbml.py +0 -162
  11. {model2data-0.1.0 → model2data-0.1.1}/LICENSE +0 -0
  12. {model2data-0.1.0 → model2data-0.1.1}/README.md +0 -0
  13. {model2data-0.1.0 → model2data-0.1.1}/model2data/cli.py +0 -0
  14. {model2data-0.1.0 → model2data-0.1.1}/model2data/utils.py +0 -0
  15. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/dependency_links.txt +0 -0
  16. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/entry_points.txt +0 -0
  17. {model2data-0.1.0 → model2data-0.1.1}/model2data.egg-info/top_level.txt +0 -0
  18. {model2data-0.1.0 → model2data-0.1.1}/setup.cfg +0 -0
  19. {model2data-0.1.0 → model2data-0.1.1}/tests/test_cli_smoke.py +0 -0
  20. {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbml_parser.py +0 -0
  21. {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbt_naming.py +0 -0
  22. {model2data-0.1.0 → model2data-0.1.1}/tests/test_dbt_tests.py +0 -0
  23. {model2data-0.1.0 → model2data-0.1.1}/tests/test_generation.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 0.1.0
3
+ Version: 0.1.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Requires-Python: >=3.9
6
6
  Description-Content-Type: text/markdown
@@ -13,6 +13,7 @@ Requires-Dist: pyyaml>=6.0.3
13
13
  Requires-Dist: typer>=0.20.0
14
14
  Provides-Extra: dev
15
15
  Requires-Dist: pytest; extra == "dev"
16
+ Requires-Dist: pytest-cov; extra == "dev"
16
17
  Dynamic: license-file
17
18
 
18
19
  # model2data
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 0.1.0
3
+ Version: 0.1.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Requires-Python: >=3.9
6
6
  Description-Content-Type: text/markdown
@@ -13,6 +13,7 @@ Requires-Dist: pyyaml>=6.0.3
13
13
  Requires-Dist: typer>=0.20.0
14
14
  Provides-Extra: dev
15
15
  Requires-Dist: pytest; extra == "dev"
16
+ Requires-Dist: pytest-cov; extra == "dev"
16
17
  Dynamic: license-file
17
18
 
18
19
  # model2data
@@ -9,11 +9,6 @@ model2data.egg-info/dependency_links.txt
9
9
  model2data.egg-info/entry_points.txt
10
10
  model2data.egg-info/requires.txt
11
11
  model2data.egg-info/top_level.txt
12
- model2data/dbt/project.py
13
- model2data/generate/core.py
14
- model2data/generate/faker.py
15
- model2data/generate/relationships.py
16
- model2data/parse/dbml.py
17
12
  tests/test_cli_smoke.py
18
13
  tests/test_dbml_parser.py
19
14
  tests/test_dbt_naming.py
@@ -7,3 +7,4 @@ typer>=0.20.0
7
7
 
8
8
  [dev]
9
9
  pytest
10
+ pytest-cov
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "0.1.0"
7
+ version = "0.1.1"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -20,6 +20,7 @@ dependencies = [
20
20
  [project.optional-dependencies]
21
21
  dev = [
22
22
  "pytest",
23
+ "pytest-cov",
23
24
  ]
24
25
 
25
26
  [project.scripts]
@@ -1,74 +0,0 @@
1
- from pathlib import Path
2
- import shutil
3
- import jinja2
4
-
5
- TEMPLATES_DIR = Path(__file__).parent / "templates"
6
-
7
-
8
- def create_project_scaffold(dest: Path, project_name: str, profile_name: str) -> None:
9
- dest.mkdir(parents=True, exist_ok=True)
10
-
11
- # dbt folders
12
- (dest / "models" / "staging").mkdir(parents=True, exist_ok=True)
13
- (dest / "seeds" / "raw").mkdir(parents=True, exist_ok=True)
14
- (dest / "analysis").mkdir(exist_ok=True)
15
- (dest / "macros").mkdir(exist_ok=True)
16
- (dest / "tests").mkdir(exist_ok=True)
17
- (dest / "snapshots").mkdir(exist_ok=True)
18
-
19
- # dbt_project.yml
20
- _render_template(
21
- template_name="dbt_project.yml.jinja",
22
- output_path=dest / "dbt_project.yml",
23
- context={"project_name": project_name, "profile_name": profile_name},
24
- )
25
-
26
- # Copy over any macros from templates
27
- template_macros_dir = Path("model2data/dbt/templates/macros")
28
- if template_macros_dir.exists():
29
- for macro_file in template_macros_dir.glob("*.sql"):
30
- target_file = dest / "macros" / macro_file.name
31
- if not target_file.exists():
32
- target_file.write_text(macro_file.read_text())
33
-
34
-
35
- def create_staging_models(dest: Path, project_name: str) -> None:
36
- """
37
- Creates staging models in models/staging/ folder that reference raw seed tables as sources.
38
- """
39
- seeds_path = dest / "seeds" / "raw"
40
- models_path = dest / "models" / "staging"
41
- models_path.mkdir(parents=True, exist_ok=True)
42
-
43
- for csv_file in seeds_path.glob("*.csv"):
44
- table_name = csv_file.stem # keep full seed name, e.g., raw_stories
45
- model_file = models_path / f"stg_{table_name}.sql"
46
-
47
- if not model_file.exists():
48
- sql_content = f"""\
49
- -- Auto-generated staging model for {table_name}
50
- select *
51
- from {{{{ source('raw', '{table_name}') }}}}
52
- """
53
- model_file.write_text(sql_content)
54
-
55
-
56
- def create_profiles_yml(dest: Path, profile_name: str) -> None:
57
- profiles_file = dest / "profiles.yml"
58
- if profiles_file.exists():
59
- content = profiles_file.read_text()
60
- if profile_name in content:
61
- return
62
- _render_template(
63
- template_name="profiles.yml.jinja",
64
- output_path=profiles_file,
65
- context={"profile_name": profile_name},
66
- )
67
-
68
-
69
- def _render_template(template_name: str, output_path: Path, context: dict) -> None:
70
- template_path = TEMPLATES_DIR / template_name
71
- if not template_path.exists():
72
- raise FileNotFoundError(f"Template not found: {template_path}")
73
- template = jinja2.Template(template_path.read_text())
74
- output_path.write_text(template.render(**context))
@@ -1,171 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import random
4
- from collections import defaultdict, deque
5
- from typing import Optional
6
-
7
- import pandas as pd
8
- from faker import Faker
9
-
10
- from model2data.generate.faker import generate_column_values
11
- from model2data.generate.relationships import (
12
- classify_refs,
13
- build_fk_lookup,
14
- )
15
- from model2data.parse.dbml import TableDef
16
-
17
- fake = Faker()
18
-
19
-
20
- # ---------------------------------------------------------
21
- # Public API
22
- # ---------------------------------------------------------
23
- def generate_data_from_dbml(
24
- tables: dict[str, TableDef],
25
- refs: list[dict],
26
- base_rows: int = 100,
27
- seed: Optional[int] = None,
28
- ) -> dict[str, pd.DataFrame]:
29
- """
30
- Generate synthetic datasets from parsed DBML definitions.
31
-
32
- This function is deterministic if a seed is provided.
33
- It performs no filesystem I/O and returns pandas DataFrames.
34
- """
35
- if seed is not None:
36
- random.seed(seed)
37
- Faker.seed(seed)
38
-
39
- # ---------------------------------------------------------
40
- # Classify references
41
- # ---------------------------------------------------------
42
- fk_refs, attribute_refs = classify_refs(tables, refs)
43
- fk_lookup = build_fk_lookup(fk_refs)
44
-
45
- # ---------------------------------------------------------
46
- # Generate tables in dependency order
47
- # ---------------------------------------------------------
48
- ordered_tables = _topological_table_order(tables, fk_refs)
49
- generated: dict[str, pd.DataFrame] = {}
50
-
51
- for table_name in ordered_tables:
52
- table_def = tables[table_name]
53
- row_count = _determine_row_count(table_def.name, base_rows)
54
-
55
- data: dict[str, list] = {}
56
-
57
- # -----------------------
58
- # First pass: columns + FKs
59
- # -----------------------
60
- for column in table_def.columns:
61
- fk_series = None
62
- fk_target = fk_lookup.get((table_name, column.name))
63
-
64
- if fk_target:
65
- parent_table, parent_column = fk_target
66
- parent_df = generated.get(parent_table)
67
- if parent_df is not None and parent_column in parent_df.columns:
68
- fk_series = parent_df[parent_column]
69
-
70
- ensure_unique = "pk" in column.settings
71
- data[column.name] = generate_column_values(
72
- column=column,
73
- row_count=row_count,
74
- fk_series=fk_series,
75
- ensure_unique=ensure_unique,
76
- )
77
-
78
- df = pd.DataFrame(data)
79
-
80
- # -----------------------------------------------------
81
- # Second pass: attribute mirroring (non-FK refs)
82
- # -----------------------------------------------------
83
- for ref in attribute_refs:
84
- if ref["source_table"] != table_name:
85
- continue
86
-
87
- parent_table = ref["target_table"]
88
- parent_column = ref["target_column"]
89
- child_column = ref["source_column"]
90
-
91
- parent_df = generated.get(parent_table)
92
- if parent_df is None:
93
- continue
94
-
95
- # find FK linking child → parent
96
- fk_column = next(
97
- (
98
- r["source_column"]
99
- for r in fk_refs
100
- if r["source_table"] == table_name
101
- and r["target_table"] == parent_table
102
- ),
103
- None,
104
- )
105
-
106
- if not fk_column or fk_column not in df.columns:
107
- continue
108
-
109
- lookup = (
110
- parent_df.groupby("id")[parent_column]
111
- .first()
112
- .to_dict()
113
- )
114
-
115
- df[child_column] = df[fk_column].map(lookup)
116
-
117
- generated[table_name] = df
118
-
119
- return generated
120
-
121
-
122
- # ---------------------------------------------------------
123
- # Internal helpers
124
- # ---------------------------------------------------------
125
- def _determine_row_count(table_name: str, base_rows: int) -> int:
126
- """
127
- Return the base number of rows for all tables.
128
- """
129
- return base_rows
130
-
131
-
132
- def _topological_table_order(
133
- tables: dict[str, TableDef],
134
- fk_refs: list[dict],
135
- ) -> list[str]:
136
- """
137
- Order tables so parent tables are generated before children.
138
- """
139
- graph: dict[str, set[str]] = defaultdict(set)
140
- indegree: dict[str, int] = {name: 0 for name in tables.keys()}
141
-
142
- for ref in fk_refs:
143
- parent = ref["target_table"]
144
- child = ref["source_table"]
145
-
146
- if parent == child:
147
- continue
148
- if parent not in tables or child not in tables:
149
- continue
150
-
151
- if child not in graph[parent]:
152
- graph[parent].add(child)
153
- indegree[child] += 1
154
-
155
- queue = deque(sorted(name for name, deg in indegree.items() if deg == 0))
156
- order: list[str] = []
157
-
158
- while queue:
159
- node = queue.popleft()
160
- order.append(node)
161
- for neighbor in sorted(graph.get(node, [])):
162
- indegree[neighbor] -= 1
163
- if indegree[neighbor] == 0:
164
- queue.append(neighbor)
165
-
166
- # Safety net for disconnected tables
167
- for name in tables.keys():
168
- if name not in order:
169
- order.append(name)
170
-
171
- return order
@@ -1,113 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import random
4
- import uuid
5
- from datetime import datetime, timedelta
6
- from typing import Optional
7
-
8
- import pandas as pd
9
- from faker import Faker
10
-
11
- from model2data.parse.dbml import ColumnDef
12
-
13
- fake = Faker()
14
-
15
-
16
- # ---------------------------------------------------------
17
- # Public API
18
- # ---------------------------------------------------------
19
- def generate_column_values(
20
- column: ColumnDef,
21
- row_count: int,
22
- fk_series: Optional[pd.Series] = None,
23
- ensure_unique: bool = False,
24
- ) -> list:
25
- """
26
- Generate synthetic values for a single column.
27
- Respects FKs, uniqueness, and optional min/max hints in column notes.
28
- """
29
- if fk_series is not None and not fk_series.empty:
30
- fk_values = fk_series.tolist()
31
- return [random.choice(fk_values) for _ in range(row_count)]
32
-
33
- dtype = column.data_type.lower()
34
- base_type = dtype.split("(")[0].strip()
35
- values: list = []
36
-
37
- # -----------------------------------------------------
38
- # UUIDs / hashes
39
- # -----------------------------------------------------
40
- if "uuid" in base_type or "hash" in base_type:
41
- values = [str(uuid.uuid4()) for _ in range(row_count)]
42
-
43
- # -----------------------------------------------------
44
- # Integers
45
- # -----------------------------------------------------
46
- elif any(key in base_type for key in ["int", "integer", "bigint", "smallint"]):
47
- min_val = 0
48
- max_val = 100
49
- if column.note:
50
- if "min" in column.note:
51
- min_val = column.note["min"]
52
- if "max" in column.note:
53
- max_val = column.note["max"]
54
- values = [random.randint(min_val, max_val) for _ in range(row_count)]
55
-
56
- # -----------------------------------------------------
57
- # Floats / decimals
58
- # -----------------------------------------------------
59
- elif any(key in base_type for key in ["decimal", "numeric", "float", "double"]):
60
- values = [round(random.uniform(0, 10_000), 2) for _ in range(row_count)]
61
-
62
- # -----------------------------------------------------
63
- # Booleans
64
- # -----------------------------------------------------
65
- elif "boolean" in base_type or "bool" in base_type:
66
- values = [random.choice([True, False]) for _ in range(row_count)]
67
-
68
- # -----------------------------------------------------
69
- # Dates
70
- # -----------------------------------------------------
71
- elif "date" in base_type and "time" not in base_type:
72
- values = [fake.date_between(start_date="-2y", end_date="today") for _ in range(row_count)]
73
-
74
- elif "time" in base_type and "stamp" not in base_type:
75
- values = [fake.time() for _ in range(row_count)]
76
-
77
- elif any(key in base_type for key in ["timestamp", "datetime"]):
78
- values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
79
-
80
- # -----------------------------------------------------
81
- # Fallback to Faker providers
82
- # -----------------------------------------------------
83
- else:
84
- try:
85
- values = [fake.format(base_type) for _ in range(row_count)]
86
- except:
87
- if column.name.lower().endswith("_id") or ensure_unique:
88
- values = [str(uuid.uuid4()) for _ in range(row_count)]
89
- else:
90
- values = [fake.sentence(nb_words=3) for _ in range(row_count)]
91
-
92
- # -----------------------------------------------------
93
- # Nullability
94
- # -----------------------------------------------------
95
- if "not null" not in column.settings:
96
- null_fraction = max(0, min(0.2, 1 - (row_count / (row_count + 50))))
97
- sample_size = int(row_count * null_fraction)
98
- if sample_size:
99
- for idx in random.sample(range(row_count), k=sample_size):
100
- values[idx] = None
101
-
102
- return values
103
-
104
-
105
- # ---------------------------------------------------------
106
- # Internal helpers
107
- # ---------------------------------------------------------
108
- def _random_datetime(start_days: int = -365, end_days: int = 0) -> datetime:
109
- start = datetime.now() + timedelta(days=start_days)
110
- end = datetime.now() + timedelta(days=end_days)
111
- delta = end - start
112
- random_second = random.randint(0, int(delta.total_seconds()))
113
- return start + timedelta(seconds=random_second)
@@ -1,49 +0,0 @@
1
- from typing import Tuple, List, Dict
2
- from model2data.parse.dbml import TableDef
3
-
4
-
5
- # ---------------------------------------------------------
6
- # Public API
7
- # ---------------------------------------------------------
8
- def classify_refs(
9
- tables: Dict[str, TableDef],
10
- refs: List[Dict],
11
- ) -> Tuple[List[Dict], List[Dict]]:
12
- """
13
- Classify references into:
14
- - fk_refs: Foreign keys (target column looks like a PK)
15
- - attribute_refs: Non-FK dependencies (mirroring parent attributes)
16
- """
17
- fk_refs = []
18
- attribute_refs = []
19
-
20
- for ref in refs:
21
- target_table = tables.get(ref["target_table"])
22
- target_col = None
23
- if target_table:
24
- target_col = next((c for c in target_table.columns if c.name == ref["target_column"]), None)
25
-
26
- # FK if target column is a primary key or named "id"
27
- if target_col and ("pk" in target_col.settings or target_col.name.lower() == "id"):
28
- fk_refs.append(ref)
29
- else:
30
- attribute_refs.append(ref)
31
-
32
- return fk_refs, attribute_refs
33
-
34
-
35
- def build_fk_lookup(fk_refs: List[Dict]) -> Dict[Tuple[str, str], Tuple[str, str]]:
36
- """
37
- Build a lookup dictionary for FK relationships.
38
-
39
- Returns:
40
- {
41
- (child_table, child_column): (parent_table, parent_column)
42
- }
43
- """
44
- lookup: Dict[Tuple[str, str], Tuple[str, str]] = {}
45
- for ref in fk_refs:
46
- key = (ref["source_table"], ref["source_column"])
47
- value = (ref["target_table"], ref["target_column"])
48
- lookup[key] = value
49
- return lookup
@@ -1,162 +0,0 @@
1
- from __future__ import annotations
2
- from pathlib import Path
3
- from typing import Optional
4
- from dataclasses import dataclass, field
5
- import re
6
- import json
7
- import uuid
8
-
9
- # -------------------------------
10
- # Dataclasses
11
- # -------------------------------
12
-
13
- @dataclass
14
- class ColumnDef:
15
- name: str
16
- data_type: str
17
- settings: set[str] = field(default_factory=set)
18
- note: Optional[dict] = None
19
-
20
- @dataclass
21
- class TableDef:
22
- name: str
23
- columns: list[ColumnDef] = field(default_factory=list)
24
-
25
- # -------------------------------
26
- # Helpers
27
- # -------------------------------
28
-
29
- def _strip_quotes(value: str) -> str:
30
- return value.strip().strip('"').strip("'")
31
-
32
- def _parse_column_settings(raw: Optional[str]) -> set[str]:
33
- if not raw:
34
- return set()
35
- parts = [part.strip() for part in raw.split(",")]
36
- return {part.strip("'").strip('"').lower() for part in parts if part}
37
-
38
- def normalize_identifier(value: str) -> str:
39
- cleaned = re.sub(r"[^0-9A-Za-z]+", "_", value).strip("_").lower()
40
- if not cleaned:
41
- cleaned = "table"
42
- if cleaned[0].isdigit():
43
- cleaned = f"t_{cleaned}"
44
- return cleaned
45
-
46
- def parse_dbml(dbml_path: Path) -> tuple[dict[str, TableDef], list[dict]]:
47
- text = dbml_path.read_text(encoding="utf-8")
48
- lines = text.splitlines()
49
- tables: dict[str, TableDef] = {}
50
- refs: list[dict] = []
51
-
52
- current_table: Optional[TableDef] = None
53
- in_indexes_block = False
54
- note_block_depth = 0
55
- in_ref_block = False # NEW
56
-
57
- for raw_line in lines:
58
- line = raw_line.strip()
59
- if not line or line.startswith("//"):
60
- continue
61
-
62
- cleaned = line.split("//", 1)[0].strip()
63
- if not cleaned:
64
- continue
65
-
66
- triple_quote_count = cleaned.count("'''")
67
- if triple_quote_count:
68
- note_block_depth = (note_block_depth + triple_quote_count) % 2
69
- if cleaned.startswith("Note:"):
70
- continue
71
- if note_block_depth:
72
- continue
73
-
74
- # ----------------------
75
- # TABLE PARSING
76
- # ----------------------
77
- if cleaned.lower().startswith("table "):
78
- table_name_section = cleaned[6:].split("{", 1)[0].strip()
79
- if "[" in table_name_section:
80
- table_name_section = table_name_section.split("[", 1)[0].strip()
81
- table_name = _strip_quotes(table_name_section)
82
- current_table = TableDef(name=table_name)
83
- continue
84
-
85
- if current_table:
86
- if cleaned.startswith("indexes"):
87
- in_indexes_block = True
88
- continue
89
- if in_indexes_block:
90
- if cleaned.endswith("}"):
91
- in_indexes_block = False
92
- continue
93
- if cleaned.startswith("}"):
94
- tables[current_table.name] = current_table
95
- current_table = None
96
- continue
97
- if cleaned.startswith("Note:"):
98
- continue
99
-
100
- col_match = re.match(
101
- r'(".*?"|`.*?`|[\w]+)\s+([^\[]+?)(?:\s+\[(.+)\])?$',
102
- cleaned,
103
- )
104
- if not col_match:
105
- continue
106
-
107
- col_name = _strip_quotes(col_match.group(1))
108
- col_type = col_match.group(2).strip()
109
- settings = _parse_column_settings(col_match.group(3))
110
-
111
- current_table.columns.append(
112
- ColumnDef(
113
- name=col_name,
114
- data_type=col_type,
115
- settings=settings,
116
- note=None,
117
- )
118
- )
119
- continue
120
-
121
- # ----------------------
122
- # REF BLOCK START
123
- # ----------------------
124
- if cleaned.startswith("Ref"):
125
- in_ref_block = True
126
- continue
127
-
128
- if in_ref_block:
129
- if cleaned.startswith("}"):
130
- in_ref_block = False
131
- continue
132
-
133
- # Match: "table"."column" > "table"."column"
134
- ref_match = re.match(
135
- r'(".*?"|`.*?`|[\w]+)\.(".*?"|`.*?`|[\w]+)\s*([<>])\s*'
136
- r'(".*?"|`.*?`|[\w]+)\.(".*?"|`.*?`|[\w]+)',
137
- cleaned,
138
- )
139
- if not ref_match:
140
- continue
141
-
142
- left_table, left_column, operator, right_table, right_column = ref_match.groups()
143
-
144
- # Ignore <> and other non-FK relations
145
- if operator not in (">", "<"):
146
- continue
147
-
148
- if operator == "<":
149
- left_table, right_table = right_table, left_table
150
- left_column, right_column = right_column, left_column
151
-
152
- refs.append(
153
- {
154
- "source_table": _strip_quotes(left_table),
155
- "source_column": _strip_quotes(left_column),
156
- "target_table": _strip_quotes(right_table),
157
- "target_column": _strip_quotes(right_column),
158
- }
159
- )
160
- continue
161
-
162
- return tables, refs
File without changes
File without changes
File without changes
File without changes