model2data 1.1.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {model2data-1.1.0/model2data.egg-info → model2data-1.2.0}/PKG-INFO +1 -1
  2. {model2data-1.1.0 → model2data-1.2.0}/README.md +5 -3
  3. {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/faker.py +34 -7
  4. {model2data-1.1.0 → model2data-1.2.0/model2data.egg-info}/PKG-INFO +1 -1
  5. {model2data-1.1.0 → model2data-1.2.0}/pyproject.toml +1 -1
  6. {model2data-1.1.0 → model2data-1.2.0}/tests/test_faker_name_inference.py +67 -0
  7. {model2data-1.1.0 → model2data-1.2.0}/LICENSE +0 -0
  8. {model2data-1.1.0 → model2data-1.2.0}/README_PYPI.md +0 -0
  9. {model2data-1.1.0 → model2data-1.2.0}/model2data/__init__.py +0 -0
  10. {model2data-1.1.0 → model2data-1.2.0}/model2data/cli.py +0 -0
  11. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/__init__.py +0 -0
  12. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/project.py +0 -0
  13. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  14. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  15. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  16. {model2data-1.1.0 → model2data-1.2.0}/model2data/dbt/tests.py +0 -0
  17. {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/__init__.py +0 -0
  18. {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/core.py +0 -0
  19. {model2data-1.1.0 → model2data-1.2.0}/model2data/generate/relationships.py +0 -0
  20. {model2data-1.1.0 → model2data-1.2.0}/model2data/parse/__init__.py +0 -0
  21. {model2data-1.1.0 → model2data-1.2.0}/model2data/parse/dbml.py +0 -0
  22. {model2data-1.1.0 → model2data-1.2.0}/model2data/utils.py +0 -0
  23. {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/SOURCES.txt +0 -0
  24. {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/dependency_links.txt +0 -0
  25. {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/entry_points.txt +0 -0
  26. {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/requires.txt +0 -0
  27. {model2data-1.1.0 → model2data-1.2.0}/model2data.egg-info/top_level.txt +0 -0
  28. {model2data-1.1.0 → model2data-1.2.0}/setup.cfg +0 -0
  29. {model2data-1.1.0 → model2data-1.2.0}/tests/test_cli.py +0 -0
  30. {model2data-1.1.0 → model2data-1.2.0}/tests/test_coverage_gaps.py +0 -0
  31. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbml_parser.py +0 -0
  32. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbml_parser_fuzz.py +0 -0
  33. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_integration.py +0 -0
  34. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_naming.py +0 -0
  35. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_project.py +0 -0
  36. {model2data-1.1.0 → model2data-1.2.0}/tests/test_dbt_tests.py +0 -0
  37. {model2data-1.1.0 → model2data-1.2.0}/tests/test_generation.py +0 -0
  38. {model2data-1.1.0 → model2data-1.2.0}/tests/test_release_stress.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -39,7 +39,8 @@ access required.
39
39
  - **Privacy-safe.** Nothing but a schema definition goes in; nothing but synthetic data comes out.
40
40
  - **Realistic, not random.** Column names are matched against ~35 common patterns — `email`,
41
41
  `first_name`, `city`, `phone`, `company`, ... — so a column called `email` gets real-looking
42
- emails, not `Lorem ipsum` text.
42
+ emails, not `Lorem ipsum` text. Type a column with any Faker provider (`billing_country state`,
43
+ `sku ean13`) to pick its generator outright when the name is wrong for the data.
43
44
  - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
44
45
  dependency order.
45
46
  - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
@@ -97,8 +98,9 @@ flowchart LR
97
98
 
98
99
  1. **Parse.** Reads tables, columns, types, and `Ref` relationships from a DBML file.
99
100
  2. **Generate.** Produces synthetic values per column — typed generation for known SQL types
100
- (int, date, timestamp, ...), name-aware inference for everything else (`email`, `phone`,
101
- `city`, ...), foreign keys resolved against already-generated parent rows.
101
+ (int, date, timestamp, ...), then a Faker provider named as the type (`sku ean13`), then
102
+ name-aware inference for everything else (`email`, `phone`, `city`, ...), foreign keys
103
+ resolved against already-generated parent rows.
102
104
  3. **Scaffold.** Writes a complete dbt project around that data: CSV seeds, staging models that
103
105
  `ref()` those seeds, `not_null`/`unique`/`relationships` tests, `accepted_values` tests for
104
106
  DBML `Enum`-typed columns, singular SQL tests for composite primary/unique keys, table and
@@ -150,6 +150,33 @@ def _infer_by_name(column_name: str) -> Optional[Callable[[], object]]:
150
150
  return None
151
151
 
152
152
 
153
+ # Type names that are ordinary SQL types first and Faker providers only by
154
+ # coincidence. Someone typing `email text` means the SQL type, so these must
155
+ # not count as a deliberate choice of provider -- otherwise the commonest
156
+ # declaration in any schema would stop generating emails.
157
+ _SQL_TYPES_SHADOWING_A_PROVIDER = frozenset({"text", "json", "jsonb", "xml", "binary", "year"})
158
+
159
+
160
+ def _infer_by_type(base_type: str) -> Optional[Callable[[], object]]:
161
+ """The provider a column's declared type names, if it names one deliberately.
162
+
163
+ `sku ean13` and `home_state state` are the user saying which generator they
164
+ want, in the only place DBML gives them to say it. That has to outrank the
165
+ guess made from the column's name: a name pattern is inferred, a type is
166
+ declared, and `first_name email` silently generating first names -- the
167
+ type having no effect whatsoever -- is the single most confusing thing this
168
+ module did.
169
+ """
170
+ if base_type in _SQL_TYPES_SHADOWING_A_PROVIDER:
171
+ return None
172
+ try:
173
+ fake.format(base_type)
174
+ except (AttributeError, TypeError):
175
+ # Not a provider, or one that needs arguments: nothing was declared.
176
+ return None
177
+ return lambda: fake.format(base_type)
178
+
179
+
153
180
  # ---------------------------------------------------------
154
181
  # Public API
155
182
  # ---------------------------------------------------------
@@ -278,16 +305,16 @@ def generate_column_values(
278
305
  values = [_random_datetime().isoformat(sep=" ") for _ in range(row_count)]
279
306
 
280
307
  # -----------------------------------------------------
281
- # Untyped / generic string columns: infer intent from the
282
- # column name first (email, city, phone...), then fall back
283
- # to a literal Faker provider name, then to a generic value.
308
+ # Untyped / generic string columns: honour a type that names
309
+ # a Faker provider (`sku ean13`), then infer intent from the
310
+ # column name (email, city, phone...), then a generic value.
284
311
  # -----------------------------------------------------
285
312
  else:
286
- name_generator = _infer_by_name(column.name)
287
- if name_generator is not None:
288
- values = [name_generator() for _ in range(row_count)]
313
+ generator = _infer_by_type(base_type) or _infer_by_name(column.name)
314
+ if generator is not None:
315
+ values = [generator() for _ in range(row_count)]
289
316
  values = (
290
- _deduplicate(values, name_generator, column_name=unique_label)
317
+ _deduplicate(values, generator, column_name=unique_label)
291
318
  if ensure_unique
292
319
  else values
293
320
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.1.0"
7
+ version = "1.2.0"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -82,6 +82,73 @@ class TestNameInference:
82
82
  assert result == ["x", "x", "x"]
83
83
 
84
84
 
85
+ class TestTypeBeatsName:
86
+ """A declared Faker provider type outranks the guess made from the name.
87
+
88
+ DBML has one place to say which generator a column should use -- its type --
89
+ and until this held, a recognised column name silently overruled it:
90
+ `home_state state` generated countries and `first_name email` generated
91
+ first names, with the declared type having no effect at all.
92
+ """
93
+
94
+ def test_a_declared_provider_overrules_the_name_pattern(self):
95
+ values = generate_column_values(
96
+ ColumnDef(name="first_name", data_type="email", settings={"not null"}),
97
+ row_count=20,
98
+ )
99
+ assert all(EMAIL_RE.match(v) for v in values)
100
+
101
+ def test_a_column_can_be_typed_against_its_own_name(self):
102
+ # The case the feature exists for: a state column that isn't called one.
103
+ # Without this the `country` in its name won and it generated countries.
104
+ from faker.providers.address.en_US import Provider as UsAddress
105
+
106
+ values = generate_column_values(
107
+ ColumnDef(name="billing_country", data_type="state", settings={"not null"}),
108
+ row_count=40,
109
+ )
110
+ assert set(values) <= set(UsAddress.states)
111
+
112
+ def test_a_plain_sql_type_does_not_count_as_a_declaration(self):
113
+ # `text` and `json` are Faker providers by coincidence; a column typed
114
+ # that way means the SQL type, and its name must still be inferred.
115
+ for sql_type in ("text", "json", "varchar", "char(64)"):
116
+ values = generate_column_values(
117
+ ColumnDef(name="email", data_type=sql_type, settings={"not null"}),
118
+ row_count=10,
119
+ )
120
+ assert all(EMAIL_RE.match(v) for v in values), sql_type
121
+
122
+ def test_an_unknown_type_still_falls_back_to_the_name(self):
123
+ reset_stats()
124
+ values = generate_column_values(
125
+ ColumnDef(name="city", data_type="weird_custom_type", settings={"not null"}),
126
+ row_count=10,
127
+ )
128
+ assert all(isinstance(value, str) and value for value in values)
129
+ assert get_unmapped_columns() == []
130
+
131
+ def test_a_declared_provider_is_still_deduplicated_when_unique(self):
132
+ values = generate_column_values(
133
+ ColumnDef(name="first_name", data_type="email", settings={"not null", "unique"}),
134
+ row_count=50,
135
+ ensure_unique=True,
136
+ )
137
+ assert len(values) == len(set(values)) == 50
138
+
139
+ def test_structured_types_are_untouched_by_this(self):
140
+ # `date` and `boolean` are providers too, but they are decided long
141
+ # before either inference runs, and must keep their real Python types.
142
+ dates = generate_column_values(
143
+ ColumnDef(name="signup_date", data_type="date", settings={"not null"}), row_count=5
144
+ )
145
+ assert all(hasattr(value, "year") for value in dates)
146
+ flags = generate_column_values(
147
+ ColumnDef(name="is_active", data_type="boolean", settings={"not null"}), row_count=5
148
+ )
149
+ assert set(flags) <= {True, False}
150
+
151
+
85
152
  class TestEnumGeneration:
86
153
  def test_enum_values_only_ever_come_from_the_enum_set(self):
87
154
  allowed = {"active", "inactive", "pending"}
File without changes
File without changes
File without changes
File without changes
File without changes