model2data 1.1.0__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {model2data-1.1.0/model2data.egg-info → model2data-1.3.0}/PKG-INFO +1 -1
  2. {model2data-1.1.0 → model2data-1.3.0}/README.md +5 -3
  3. model2data-1.3.0/model2data/__init__.py +12 -0
  4. {model2data-1.1.0 → model2data-1.3.0}/model2data/cli.py +11 -0
  5. {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/core.py +18 -2
  6. model2data-1.3.0/model2data/generate/faker.py +696 -0
  7. {model2data-1.1.0 → model2data-1.3.0/model2data.egg-info}/PKG-INFO +1 -1
  8. {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/SOURCES.txt +2 -1
  9. {model2data-1.1.0 → model2data-1.3.0}/pyproject.toml +1 -1
  10. {model2data-1.1.0 → model2data-1.3.0}/tests/test_faker_name_inference.py +67 -0
  11. model2data-1.3.0/tests/test_row_identity.py +372 -0
  12. model2data-1.1.0/model2data/__init__.py +0 -3
  13. model2data-1.1.0/model2data/generate/faker.py +0 -365
  14. {model2data-1.1.0 → model2data-1.3.0}/LICENSE +0 -0
  15. {model2data-1.1.0 → model2data-1.3.0}/README_PYPI.md +0 -0
  16. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/__init__.py +0 -0
  17. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/project.py +0 -0
  18. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  19. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  20. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  21. {model2data-1.1.0 → model2data-1.3.0}/model2data/dbt/tests.py +0 -0
  22. {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/__init__.py +0 -0
  23. {model2data-1.1.0 → model2data-1.3.0}/model2data/generate/relationships.py +0 -0
  24. {model2data-1.1.0 → model2data-1.3.0}/model2data/parse/__init__.py +0 -0
  25. {model2data-1.1.0 → model2data-1.3.0}/model2data/parse/dbml.py +0 -0
  26. {model2data-1.1.0 → model2data-1.3.0}/model2data/utils.py +0 -0
  27. {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/dependency_links.txt +0 -0
  28. {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/entry_points.txt +0 -0
  29. {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/requires.txt +0 -0
  30. {model2data-1.1.0 → model2data-1.3.0}/model2data.egg-info/top_level.txt +0 -0
  31. {model2data-1.1.0 → model2data-1.3.0}/setup.cfg +0 -0
  32. {model2data-1.1.0 → model2data-1.3.0}/tests/test_cli.py +0 -0
  33. {model2data-1.1.0 → model2data-1.3.0}/tests/test_coverage_gaps.py +0 -0
  34. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbml_parser.py +0 -0
  35. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbml_parser_fuzz.py +0 -0
  36. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_integration.py +0 -0
  37. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_naming.py +0 -0
  38. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_project.py +0 -0
  39. {model2data-1.1.0 → model2data-1.3.0}/tests/test_dbt_tests.py +0 -0
  40. {model2data-1.1.0 → model2data-1.3.0}/tests/test_generation.py +0 -0
  41. {model2data-1.1.0 → model2data-1.3.0}/tests/test_release_stress.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.1.0
3
+ Version: 1.3.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -39,7 +39,8 @@ access required.
39
39
  - **Privacy-safe.** Nothing but a schema definition goes in; nothing but synthetic data comes out.
40
40
  - **Realistic, not random.** Column names are matched against ~35 common patterns — `email`,
41
41
  `first_name`, `city`, `phone`, `company`, ... — so a column called `email` gets real-looking
42
- emails, not `Lorem ipsum` text.
42
+ emails, not `Lorem ipsum` text. Type a column with any Faker provider (`billing_country state`,
43
+ `sku ean13`) to pick its generator outright when the name is wrong for the data.
43
44
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
44
45
  dependency order.
45
46
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
@@ -97,8 +98,9 @@ flowchart LR
97
98
 
98
99
  1. **Parse.** Reads tables, columns, types, and `Ref` relationships from a DBML file.
99
100
  2. **Generate.** Produces synthetic values per column — typed generation for known SQL types
100
- (int, date, timestamp, ...), name-aware inference for everything else (`email`, `phone`,
101
- `city`, ...), foreign keys resolved against already-generated parent rows.
101
+ (int, date, timestamp, ...), then a Faker provider named as the type (`sku ean13`), then
102
+ name-aware inference for everything else (`email`, `phone`, `city`, ...), foreign keys
103
+ resolved against already-generated parent rows.
102
104
  3. **Scaffold.** Writes a complete dbt project around that data: CSV seeds, staging models that
103
105
  `ref()` those seeds, `not_null`/`unique`/`relationships` tests, `accepted_values` tests for
104
106
  DBML `Enum`-typed columns, singular SQL tests for composite primary/unique keys, table and
@@ -0,0 +1,12 @@
1
+ """Model2Data: Generate analytics-ready datasets from DBML models."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ # Read the installed distribution's version rather than repeating it here.
7
+ # The literal that used to live in this file said 0.1.1 for every release up
8
+ # to and including 1.2.0: nothing reads `__version__`, so nothing caught it
9
+ # drifting. Deriving it means it cannot drift again.
10
+ __version__ = version("model2data")
11
+ except PackageNotFoundError: # pragma: no cover - running from a source tree
12
+ __version__ = "0.0.0.dev0"
@@ -18,6 +18,7 @@ from model2data.generate.core import (
18
18
  get_unresolved_composite_keys,
19
19
  )
20
20
  from model2data.generate.faker import (
21
+ DEFAULT_LOCALE,
21
22
  get_duplicate_unique_columns,
22
23
  get_unmapped_columns,
23
24
  reset_stats,
@@ -125,6 +126,15 @@ def main(
125
126
  "Using the same seed will always produce identical datasets."
126
127
  ),
127
128
  ),
129
+ locale: Optional[str] = typer.Option(
130
+ None,
131
+ "--locale",
132
+ help=(
133
+ f"Faker locale for generated people and addresses "
134
+ f"(default: {DEFAULT_LOCALE}).\n"
135
+ "Examples: en_GB, nl_BE, fr_FR, de_DE."
136
+ ),
137
+ ),
128
138
  name: Optional[str] = typer.Option(
129
139
  None,
130
140
  "--name",
@@ -214,6 +224,7 @@ def main(
214
224
  base_rows=rows,
215
225
  seed=seed,
216
226
  row_overrides=row_overrides,
227
+ locale=locale,
217
228
  )
218
229
 
219
230
  # -------------------------
@@ -10,7 +10,10 @@ from faker import Faker
10
10
 
11
11
  from model2data.generate.faker import (
12
12
  generate_column_values,
13
+ release_row_pools,
13
14
  reset_duplicate_unique_columns,
15
+ reset_row_pools,
16
+ set_locale,
14
17
  )
15
18
  from model2data.generate.relationships import (
16
19
  build_fk_lookup,
@@ -18,8 +21,6 @@ from model2data.generate.relationships import (
18
21
  )
19
22
  from model2data.parse.dbml import TableDef
20
23
 
21
- fake = Faker()
22
-
23
24
  # Tables the most recent generate_data_from_dbml() call found stuck in an
24
25
  # unresolved FK cycle (never reached indegree 0 during the topological
25
26
  # sort). Exposed out-of-band, mirroring generate.faker's
@@ -66,6 +67,7 @@ def generate_data_from_dbml(
66
67
  base_rows: int = 100,
67
68
  seed: Optional[int] = None,
68
69
  row_overrides: Optional[Mapping[str, int]] = None,
70
+ locale: Optional[str] = None,
69
71
  ) -> dict[str, pd.DataFrame]:
70
72
  """
71
73
  Generate synthetic datasets from parsed DBML definitions.
@@ -78,9 +80,19 @@ def generate_data_from_dbml(
78
80
  present in `row_overrides` fall back to `base_rows`; unknown names are
79
81
  ignored.
80
82
 
83
+ `locale` picks the Faker locale every generated person and address is drawn
84
+ from -- `"nl_BE"`, `"fr_FR"`, `"en_GB"` -- defaulting to `DEFAULT_LOCALE`.
85
+ It is a per-run setting rather than a per-column one on purpose: a table
86
+ holding one Belgian and one American address is the incoherence the row
87
+ pools exist to remove.
88
+
81
89
  This function is deterministic if a seed is provided.
82
90
  It performs no filesystem I/O and returns pandas DataFrames.
83
91
  """
92
+ # Locale first, then the seed: switching locale builds a new Faker, and the
93
+ # seed has to be the last word on the generator that actually runs.
94
+ set_locale(locale)
95
+
84
96
  if seed is not None:
85
97
  random.seed(seed)
86
98
  Faker.seed(seed)
@@ -88,6 +100,7 @@ def generate_data_from_dbml(
88
100
  reset_cycle_state()
89
101
  reset_dedup_state()
90
102
  reset_duplicate_unique_columns()
103
+ reset_row_pools()
91
104
 
92
105
  # ---------------------------------------------------------
93
106
  # Classify references
@@ -193,6 +206,9 @@ def generate_data_from_dbml(
193
206
 
194
207
  df = _coerce_integer_dtypes(df, table_def)
195
208
  generated[table_name] = df
209
+ # This table is finished: nothing will read its people or addresses
210
+ # again, and on a million-row table they are worth tens of megabytes.
211
+ release_row_pools(table_name)
196
212
 
197
213
  return generated
198
214