model2data 1.1.0__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {model2data-1.1.0/model2data.egg-info → model2data-1.3.0}/PKG-INFO +1 -1
- {model2data-1.1.0 → model2data-1.3.0}/README.md +5 -3
- model2data-1.3.0/model2data/__init__.py +12 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/cli.py +11 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/core.py +18 -2
- model2data-1.3.0/model2data/generate/faker.py +696 -0
- {model2data-1.1.0 → model2data-1.3.0/model2data.egg-info}/PKG-INFO +1 -1
- {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/SOURCES.txt +2 -1
- {model2data-1.1.0 → model2data-1.3.0}/pyproject.toml +1 -1
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_faker_name_inference.py +67 -0
- model2data-1.3.0/tests/test_row_identity.py +372 -0
- model2data-1.1.0/model2data/__init__.py +0 -3
- model2data-1.1.0/model2data/generate/faker.py +0 -365
- {model2data-1.1.0 → model2data-1.3.0}/LICENSE +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/README_PYPI.md +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/project.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/tests.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/relationships.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/parse/__init__.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/parse/dbml.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data/utils.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/dependency_links.txt +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/entry_points.txt +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/requires.txt +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/top_level.txt +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/setup.cfg +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_cli.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_coverage_gaps.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbml_parser.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbml_parser_fuzz.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_integration.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_naming.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_project.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_tests.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_generation.py +0 -0
- {model2data-1.1.0 → model2data-1.3.0}/tests/test_release_stress.py +0 -0
|
@@ -39,7 +39,8 @@ access required.
|
|
|
39
39
|
- **Privacy-safe.** Nothing but a schema definition goes in; nothing but synthetic data comes out.
|
|
40
40
|
- **Realistic, not random.** Column names are matched against ~35 common patterns — `email`,
|
|
41
41
|
`first_name`, `city`, `phone`, `company`, ... — so a column called `email` gets real-looking
|
|
42
|
-
emails, not `Lorem ipsum` text.
|
|
42
|
+
emails, not `Lorem ipsum` text. Type a column with any Faker provider (`billing_country state`,
|
|
43
|
+
`sku ean13`) to pick its generator outright when the name is wrong for the data.
|
|
43
44
|
- **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
|
|
44
45
|
dependency order.
|
|
45
46
|
- **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
|
|
@@ -97,8 +98,9 @@ flowchart LR
|
|
|
97
98
|
|
|
98
99
|
1. **Parse.** Reads tables, columns, types, and `Ref` relationships from a DBML file.
|
|
99
100
|
2. **Generate.** Produces synthetic values per column — typed generation for known SQL types
|
|
100
|
-
(int, date, timestamp, ...),
|
|
101
|
-
`city`, ...), foreign keys
|
|
101
|
+
(int, date, timestamp, ...), then a Faker provider named as the type (`sku ean13`), then
|
|
102
|
+
name-aware inference for everything else (`email`, `phone`, `city`, ...), foreign keys
|
|
103
|
+
resolved against already-generated parent rows.
|
|
102
104
|
3. **Scaffold.** Writes a complete dbt project around that data: CSV seeds, staging models that
|
|
103
105
|
`ref()` those seeds, `not_null`/`unique`/`relationships` tests, `accepted_values` tests for
|
|
104
106
|
DBML `Enum`-typed columns, singular SQL tests for composite primary/unique keys, table and
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Model2Data: Generate analytics-ready datasets from DBML models."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
# Read the installed distribution's version rather than repeating it here.
|
|
7
|
+
# The literal that used to live in this file said 0.1.1 for every release up
|
|
8
|
+
# to and including 1.2.0: nothing reads `__version__`, so nothing caught it
|
|
9
|
+
# drifting. Deriving it means it cannot drift again.
|
|
10
|
+
__version__ = version("model2data")
|
|
11
|
+
except PackageNotFoundError: # pragma: no cover - running from a source tree
|
|
12
|
+
__version__ = "0.0.0.dev0"
|
|
@@ -18,6 +18,7 @@ from model2data.generate.core import (
|
|
|
18
18
|
get_unresolved_composite_keys,
|
|
19
19
|
)
|
|
20
20
|
from model2data.generate.faker import (
|
|
21
|
+
DEFAULT_LOCALE,
|
|
21
22
|
get_duplicate_unique_columns,
|
|
22
23
|
get_unmapped_columns,
|
|
23
24
|
reset_stats,
|
|
@@ -125,6 +126,15 @@ def main(
|
|
|
125
126
|
"Using the same seed will always produce identical datasets."
|
|
126
127
|
),
|
|
127
128
|
),
|
|
129
|
+
locale: Optional[str] = typer.Option(
|
|
130
|
+
None,
|
|
131
|
+
"--locale",
|
|
132
|
+
help=(
|
|
133
|
+
f"Faker locale for generated people and addresses "
|
|
134
|
+
f"(default: {DEFAULT_LOCALE}).\n"
|
|
135
|
+
"Examples: en_GB, nl_BE, fr_FR, de_DE."
|
|
136
|
+
),
|
|
137
|
+
),
|
|
128
138
|
name: Optional[str] = typer.Option(
|
|
129
139
|
None,
|
|
130
140
|
"--name",
|
|
@@ -214,6 +224,7 @@ def main(
|
|
|
214
224
|
base_rows=rows,
|
|
215
225
|
seed=seed,
|
|
216
226
|
row_overrides=row_overrides,
|
|
227
|
+
locale=locale,
|
|
217
228
|
)
|
|
218
229
|
|
|
219
230
|
# -------------------------
|
|
@@ -10,7 +10,10 @@ from faker import Faker
|
|
|
10
10
|
|
|
11
11
|
from model2data.generate.faker import (
|
|
12
12
|
generate_column_values,
|
|
13
|
+
release_row_pools,
|
|
13
14
|
reset_duplicate_unique_columns,
|
|
15
|
+
reset_row_pools,
|
|
16
|
+
set_locale,
|
|
14
17
|
)
|
|
15
18
|
from model2data.generate.relationships import (
|
|
16
19
|
build_fk_lookup,
|
|
@@ -18,8 +21,6 @@ from model2data.generate.relationships import (
|
|
|
18
21
|
)
|
|
19
22
|
from model2data.parse.dbml import TableDef
|
|
20
23
|
|
|
21
|
-
fake = Faker()
|
|
22
|
-
|
|
23
24
|
# Tables the most recent generate_data_from_dbml() call found stuck in an
|
|
24
25
|
# unresolved FK cycle (never reached indegree 0 during the topological
|
|
25
26
|
# sort). Exposed out-of-band, mirroring generate.faker's
|
|
@@ -66,6 +67,7 @@ def generate_data_from_dbml(
|
|
|
66
67
|
base_rows: int = 100,
|
|
67
68
|
seed: Optional[int] = None,
|
|
68
69
|
row_overrides: Optional[Mapping[str, int]] = None,
|
|
70
|
+
locale: Optional[str] = None,
|
|
69
71
|
) -> dict[str, pd.DataFrame]:
|
|
70
72
|
"""
|
|
71
73
|
Generate synthetic datasets from parsed DBML definitions.
|
|
@@ -78,9 +80,19 @@ def generate_data_from_dbml(
|
|
|
78
80
|
present in `row_overrides` fall back to `base_rows`; unknown names are
|
|
79
81
|
ignored.
|
|
80
82
|
|
|
83
|
+
`locale` picks the Faker locale every generated person and address is drawn
|
|
84
|
+
from -- `"nl_BE"`, `"fr_FR"`, `"en_GB"` -- defaulting to `DEFAULT_LOCALE`.
|
|
85
|
+
It is a per-run setting rather than a per-column one on purpose: a table
|
|
86
|
+
holding one Belgian and one American address is the incoherence the row
|
|
87
|
+
pools exist to remove.
|
|
88
|
+
|
|
81
89
|
This function is deterministic if a seed is provided.
|
|
82
90
|
It performs no filesystem I/O and returns pandas DataFrames.
|
|
83
91
|
"""
|
|
92
|
+
# Locale first, then the seed: switching locale builds a new Faker, and the
|
|
93
|
+
# seed has to be the last word on the generator that actually runs.
|
|
94
|
+
set_locale(locale)
|
|
95
|
+
|
|
84
96
|
if seed is not None:
|
|
85
97
|
random.seed(seed)
|
|
86
98
|
Faker.seed(seed)
|
|
@@ -88,6 +100,7 @@ def generate_data_from_dbml(
|
|
|
88
100
|
reset_cycle_state()
|
|
89
101
|
reset_dedup_state()
|
|
90
102
|
reset_duplicate_unique_columns()
|
|
103
|
+
reset_row_pools()
|
|
91
104
|
|
|
92
105
|
# ---------------------------------------------------------
|
|
93
106
|
# Classify references
|
|
@@ -193,6 +206,9 @@ def generate_data_from_dbml(
|
|
|
193
206
|
|
|
194
207
|
df = _coerce_integer_dtypes(df, table_def)
|
|
195
208
|
generated[table_name] = df
|
|
209
|
+
# This table is finished: nothing will read its people or addresses
|
|
210
|
+
# again, and on a million-row table they are worth tens of megabytes.
|
|
211
|
+
release_row_pools(table_name)
|
|
196
212
|
|
|
197
213
|
return generated
|
|
198
214
|
|