model2data 1.7.0__tar.gz → 1.7.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {model2data-1.7.0/model2data.egg-info → model2data-1.7.1}/PKG-INFO +1 -1
  2. {model2data-1.7.0 → model2data-1.7.1}/README.md +4 -1
  3. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/core.py +33 -0
  4. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/faker.py +60 -2
  5. {model2data-1.7.0 → model2data-1.7.1/model2data.egg-info}/PKG-INFO +1 -1
  6. {model2data-1.7.0 → model2data-1.7.1}/model2data.egg-info/SOURCES.txt +1 -0
  7. {model2data-1.7.0 → model2data-1.7.1}/pyproject.toml +1 -1
  8. model2data-1.7.1/tests/test_lone_country.py +132 -0
  9. {model2data-1.7.0 → model2data-1.7.1}/LICENSE +0 -0
  10. {model2data-1.7.0 → model2data-1.7.1}/README_PYPI.md +0 -0
  11. {model2data-1.7.0 → model2data-1.7.1}/model2data/__init__.py +0 -0
  12. {model2data-1.7.0 → model2data-1.7.1}/model2data/cli.py +0 -0
  13. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/__init__.py +0 -0
  14. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/project.py +0 -0
  15. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  16. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  17. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  18. {model2data-1.7.0 → model2data-1.7.1}/model2data/dbt/tests.py +0 -0
  19. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/__init__.py +0 -0
  20. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/hints.py +0 -0
  21. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/options.py +0 -0
  22. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/relationships.py +0 -0
  23. {model2data-1.7.0 → model2data-1.7.1}/model2data/generate/timeline.py +0 -0
  24. {model2data-1.7.0 → model2data-1.7.1}/model2data/parse/__init__.py +0 -0
  25. {model2data-1.7.0 → model2data-1.7.1}/model2data/parse/dbml.py +0 -0
  26. {model2data-1.7.0 → model2data-1.7.1}/model2data/utils.py +0 -0
  27. {model2data-1.7.0 → model2data-1.7.1}/model2data.egg-info/dependency_links.txt +0 -0
  28. {model2data-1.7.0 → model2data-1.7.1}/model2data.egg-info/entry_points.txt +0 -0
  29. {model2data-1.7.0 → model2data-1.7.1}/model2data.egg-info/requires.txt +0 -0
  30. {model2data-1.7.0 → model2data-1.7.1}/model2data.egg-info/top_level.txt +0 -0
  31. {model2data-1.7.0 → model2data-1.7.1}/setup.cfg +0 -0
  32. {model2data-1.7.0 → model2data-1.7.1}/tests/test_as_of_anchor.py +0 -0
  33. {model2data-1.7.0 → model2data-1.7.1}/tests/test_cli.py +0 -0
  34. {model2data-1.7.0 → model2data-1.7.1}/tests/test_column_time_hints.py +0 -0
  35. {model2data-1.7.0 → model2data-1.7.1}/tests/test_coverage_gaps.py +0 -0
  36. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbml_parser.py +0 -0
  37. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbml_parser_fuzz.py +0 -0
  38. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbt_integration.py +0 -0
  39. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbt_naming.py +0 -0
  40. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbt_project.py +0 -0
  41. {model2data-1.7.0 → model2data-1.7.1}/tests/test_dbt_tests.py +0 -0
  42. {model2data-1.7.0 → model2data-1.7.1}/tests/test_distributions.py +0 -0
  43. {model2data-1.7.0 → model2data-1.7.1}/tests/test_faker_name_inference.py +0 -0
  44. {model2data-1.7.0 → model2data-1.7.1}/tests/test_generation.py +0 -0
  45. {model2data-1.7.0 → model2data-1.7.1}/tests/test_options.py +0 -0
  46. {model2data-1.7.0 → model2data-1.7.1}/tests/test_release_stress.py +0 -0
  47. {model2data-1.7.0 → model2data-1.7.1}/tests/test_row_identity.py +0 -0
  48. {model2data-1.7.0 → model2data-1.7.1}/tests/test_shaping.py +0 -0
  49. {model2data-1.7.0 → model2data-1.7.1}/tests/test_table_seeds.py +0 -0
  50. {model2data-1.7.0 → model2data-1.7.1}/tests/test_timeline.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.7.0
3
+ Version: 1.7.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -159,7 +159,10 @@ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --table-seed orde
159
159
  ```
160
160
 
161
161
  `--locale` picks the country every generated person and address comes from (`en_US` by default);
162
- it's a per-run setting, so a table can't end up holding one Belgian and one American address:
162
+ it's a per-run setting, so a table can't end up holding one Belgian and one American address. A
163
+ `country` column that sits beside a `city`/`street`/`state`/`postcode` column always agrees with
164
+ that place; a `country` column with none of those beside it isn't describing anyone's address, so
165
+ it reads as an international mix instead, with the locale's own country the most common:
163
166
 
164
167
  ```bash
165
168
  model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --locale nl_BE
@@ -15,6 +15,7 @@ from model2data.generate.faker import (
15
15
  release_row_pools,
16
16
  reset_duplicate_unique_columns,
17
17
  reset_row_pools,
18
+ resolve_address_pool_field,
18
19
  set_locale,
19
20
  )
20
21
  from model2data.generate.hints import validate_hints
@@ -201,6 +202,12 @@ def generate_data_from_dbml(
201
202
  for column_name in key.get("columns") or []
202
203
  }
203
204
 
205
+ # A table's own shape, worked out before a single value is drawn: does
206
+ # this table have a `country` column with no `city`/`street`/`state`/
207
+ # `postcode` beside it to keep coherent with. See
208
+ # _lone_country_columns.
209
+ lone_country_columns = _lone_country_columns(table_def)
210
+
204
211
  # -----------------------
205
212
  # First pass: columns + FKs
206
213
  # -----------------------
@@ -230,6 +237,7 @@ def generate_data_from_dbml(
230
237
  as_of=as_of,
231
238
  time_profile=profile,
232
239
  skew=skew,
240
+ lone_country=column.name in lone_country_columns,
233
241
  )
234
242
 
235
243
  df = pd.DataFrame(data)
@@ -303,6 +311,31 @@ def generate_data_from_dbml(
303
311
  # ---------------------------------------------------------
304
312
  # Internal helpers
305
313
  # ---------------------------------------------------------
314
+ def _lone_country_columns(table_def: TableDef) -> set[str]:
315
+ """Names of this table's *lone* country columns.
316
+
317
+ A `country` column reads as the locale's own country on every row when it
318
+ sits beside a `city`/`street`/`state`/`postcode` column -- together they
319
+ describe one place, and the country has to agree with the rest of it.
320
+ Alone, repeating that same country on every row reads as a single-country
321
+ customer base rather than an international one, so
322
+ `generate_column_values` draws it from a home-heavy mix instead (see
323
+ `faker._HOME_COUNTRY_SHARE`). `country` columns don't count as company
324
+ for each other -- only a *different* address-pool field does.
325
+ """
326
+ address_fields = {
327
+ column.name: field
328
+ for column in table_def.columns
329
+ for field in [resolve_address_pool_field(column)]
330
+ if field is not None
331
+ }
332
+ country_columns = {name for name, field in address_fields.items() if field == "country"}
333
+ if not country_columns:
334
+ return set()
335
+ has_place_column = any(field != "country" for field in address_fields.values())
336
+ return set() if has_place_column else country_columns
337
+
338
+
306
339
  def _coerce_integer_dtypes(df: pd.DataFrame, table_def: TableDef) -> pd.DataFrame:
307
340
  """
308
341
  Cast int/bigint/smallint-typed columns to pandas' nullable "Int64" dtype.
@@ -505,6 +505,42 @@ def _infer_by_type(base_type: str) -> Optional[_Provider]:
505
505
  return lambda: fake.format(base_type)
506
506
 
507
507
 
508
+ def resolve_address_pool_field(column: ColumnDef) -> Optional[str]:
509
+ """The address-pool field (`street`, `full`, `city`, `state`, `postcode`,
510
+ `country`) this column would draw from, if any -- else None.
511
+
512
+ Same declared-type-then-name precedence `generate_column_values`'s own
513
+ untyped-column branch uses (`_infer_by_type` before `_infer_by_name`),
514
+ narrowed to the address pool. Exposed for `generate.core` to tell a
515
+ *lone* country column -- the only address-pool-shaped column in its
516
+ table -- from one that sits beside a `city`/`street`/`state`/`postcode`
517
+ column, before a single value of the table has been generated.
518
+
519
+ Restricted to columns that would actually reach that branch: an enum
520
+ column, or one whose type is a structured (int/date/uuid/...) type, never
521
+ gets there in `generate_column_values` -- and `_infer_by_type` probes an
522
+ unrecognized type by actually calling it (`fake.format(base_type)`), so
523
+ running it over a `date` or `int` column here would consume real draws
524
+ from the shared RNG and shift every value generated after it, breaking
525
+ reproducibility for reasons invisible to whoever hits it.
526
+ """
527
+ if column.enum_values or not is_free_text_type(column.data_type):
528
+ return None
529
+ base_type = column.data_type.lower().split("(")[0].strip()
530
+ generator = _infer_by_type(base_type) or _infer_by_name(column.name)
531
+ if isinstance(generator, _FromRow) and generator.pool == "address":
532
+ return generator.field
533
+ return None
534
+
535
+
536
+ # A lone country column (see resolve_address_pool_field's caller) mixes the
537
+ # locale's own country in with the rest of the world rather than repeating it
538
+ # on every row. 0.6 is a default, not a claim about any real market -- a
539
+ # business selling internationally still has a home market, and most of its
540
+ # rows are plausibly it, but "most" is not "all".
541
+ _HOME_COUNTRY_SHARE = 0.6
542
+
543
+
508
544
  def _column_time_profile(
509
545
  column: ColumnDef, time_profile: Optional[TimeProfile]
510
546
  ) -> Optional[TimeProfile]:
@@ -580,6 +616,7 @@ def generate_column_values(
580
616
  as_of: AsOf = None,
581
617
  time_profile: Optional[TimeProfile] = None,
582
618
  skew: float = 0.0,
619
+ lone_country: bool = False,
583
620
  ) -> list:
584
621
  """
585
622
  Generate synthetic values for a single column.
@@ -591,6 +628,16 @@ def generate_column_values(
591
628
  every path that draws a value -- the main pass, the self-referencing FK
592
629
  repair, the composite-key retry -- draws it the same way.
593
630
 
631
+ `lone_country` tells the address-pool branch this column is the *only*
632
+ address-shaped column in its table (see
633
+ `generate.core._lone_country_columns`). A `country` column that sits
634
+ beside a `city`/`street`/`state`/`postcode` column still reads that
635
+ place's own country, byte-identical to earlier releases; a lone one
636
+ instead draws a home-heavy mix of the locale's country and the wider
637
+ world, since "Belgium" on every row of a customers table with no other
638
+ address column reads as a single-country customer base rather than an
639
+ international one.
640
+
594
641
  `as_of` is the date every generated date and timestamp is placed relative
595
642
  to, defaulting to today. Pass it to make a seeded run reproduce on any
596
643
  later day rather than only on the day it first ran.
@@ -629,6 +676,7 @@ def generate_column_values(
629
676
  as_of=as_of,
630
677
  time_profile=time_profile,
631
678
  skew=skew,
679
+ lone_country=lone_country,
632
680
  )
633
681
  values = random.choices(pool, k=row_count)
634
682
  if not force_not_null and "not null" not in column.settings and "pk" not in column.settings:
@@ -827,8 +875,18 @@ def generate_column_values(
827
875
  else:
828
876
  generator = _infer_by_type(base_type) or _infer_by_name(column.name)
829
877
  if isinstance(generator, _FromRow):
830
- rows = _row_pool(generator.pool, table_name, row_count)
831
- values = [getattr(rows[index], generator.field) for index in range(row_count)]
878
+ if lone_country and generator.pool == "address" and generator.field == "country":
879
+ # No sibling city/street/state/postcode column to keep this
880
+ # one coherent with, so it isn't "this row's place" at all --
881
+ # draw a home-heavy mix instead of repeating the locale's own
882
+ # country on every row.
883
+ values = [
884
+ _country_name if random.random() < _HOME_COUNTRY_SHARE else fake.country()
885
+ for _ in range(row_count)
886
+ ]
887
+ else:
888
+ rows = _row_pool(generator.pool, table_name, row_count)
889
+ values = [getattr(rows[index], generator.field) for index in range(row_count)]
832
890
  if ensure_unique:
833
891
  values = _deduplicate_identity(values)
834
892
  elif generator is not None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.7.0
3
+ Version: 1.7.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -39,6 +39,7 @@ tests/test_dbt_tests.py
39
39
  tests/test_distributions.py
40
40
  tests/test_faker_name_inference.py
41
41
  tests/test_generation.py
42
+ tests/test_lone_country.py
42
43
  tests/test_options.py
43
44
  tests/test_release_stress.py
44
45
  tests/test_row_identity.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.7.0"
7
+ version = "1.7.1"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -0,0 +1,132 @@
1
+ """A lone `country` column reads as an international mix, not one repeated
2
+ country.
3
+
4
+ Since 1.3.0 every row draws one address from a per-table pool, and that
5
+ pool's country is always the locale's own -- right when a sibling
6
+ city/street/state/postcode column needs it to agree, wrong when `country` is
7
+ the only address-shaped column in the table. These tests pin: the mix (a
8
+ majority-but-not-all home-country share, several distinct countries), that a
9
+ sibling place column switches it back off, that a declared `country` type
10
+ behaves like the name, and that `distinct`/`null_rate`/determinism/locale
11
+ keep working the same way they do everywhere else.
12
+ """
13
+
14
+ from model2data.generate.core import generate_data_from_dbml
15
+ from model2data.parse.dbml import ColumnDef, TableDef
16
+
17
+
18
+ def _customers_table(*extra_columns: ColumnDef) -> TableDef:
19
+ return TableDef(
20
+ name="customers",
21
+ columns=[
22
+ ColumnDef(name="id", data_type="int", settings={"pk"}),
23
+ ColumnDef(name="country", data_type="varchar", settings={"not null"}),
24
+ *extra_columns,
25
+ ],
26
+ )
27
+
28
+
29
+ class TestLoneCountryColumn:
30
+ def test_lone_country_column_reads_as_an_international_mix(self):
31
+ df = generate_data_from_dbml(
32
+ tables={"customers": _customers_table()},
33
+ refs=[],
34
+ base_rows=500,
35
+ seed=1,
36
+ locale="nl_BE",
37
+ )["customers"]
38
+
39
+ counts = df["country"].value_counts(normalize=True)
40
+ assert len(counts) >= 5
41
+
42
+ home_share = counts.get("Belgium", 0.0)
43
+ assert 0.5 <= home_share <= 0.7
44
+ assert counts.idxmax() == "Belgium"
45
+
46
+ def test_a_sibling_place_column_keeps_the_country_single(self):
47
+ df = generate_data_from_dbml(
48
+ tables={
49
+ "customers": _customers_table(
50
+ ColumnDef(name="city", data_type="varchar", settings={"not null"})
51
+ )
52
+ },
53
+ refs=[],
54
+ base_rows=200,
55
+ seed=2,
56
+ locale="nl_BE",
57
+ )["customers"]
58
+
59
+ assert set(df["country"]) == {"Belgium"}
60
+
61
+ def test_declared_country_type_behaves_like_the_name(self):
62
+ df = generate_data_from_dbml(
63
+ tables={
64
+ "customers": TableDef(
65
+ name="customers",
66
+ columns=[
67
+ ColumnDef(name="id", data_type="int", settings={"pk"}),
68
+ # Named generically; the *declared type* is what
69
+ # says "country".
70
+ ColumnDef(name="hq", data_type="country", settings={"not null"}),
71
+ ],
72
+ )
73
+ },
74
+ refs=[],
75
+ base_rows=500,
76
+ seed=3,
77
+ locale="nl_BE",
78
+ )["customers"]
79
+
80
+ counts = df["hq"].value_counts(normalize=True)
81
+ assert len(counts) >= 5
82
+ assert 0.5 <= counts.get("Belgium", 0.0) <= 0.7
83
+
84
+ def test_distinct_hint_still_bounds_the_pool(self):
85
+ df = generate_data_from_dbml(
86
+ tables={
87
+ "customers": TableDef(
88
+ name="customers",
89
+ columns=[
90
+ ColumnDef(name="id", data_type="int", settings={"pk"}),
91
+ ColumnDef(
92
+ name="country",
93
+ data_type="varchar",
94
+ settings={"not null"},
95
+ note={"distinct": 3},
96
+ ),
97
+ ],
98
+ )
99
+ },
100
+ refs=[],
101
+ base_rows=300,
102
+ seed=4,
103
+ locale="nl_BE",
104
+ )["customers"]
105
+
106
+ assert df["country"].nunique() <= 3
107
+
108
+ def test_same_seed_reproduces_the_same_mix(self):
109
+ def run():
110
+ return generate_data_from_dbml(
111
+ tables={"customers": _customers_table()},
112
+ refs=[],
113
+ base_rows=200,
114
+ seed=5,
115
+ locale="nl_BE",
116
+ )["customers"]
117
+
118
+ first_run, second_run = run(), run()
119
+ assert list(first_run["country"]) == list(second_run["country"])
120
+
121
+ def test_default_locale_also_produces_a_mix(self):
122
+ df = generate_data_from_dbml(
123
+ tables={"customers": _customers_table()},
124
+ refs=[],
125
+ base_rows=500,
126
+ seed=6,
127
+ )["customers"]
128
+
129
+ counts = df["country"].value_counts(normalize=True)
130
+ assert len(counts) >= 5
131
+ assert 0.5 <= counts.get("United States", 0.0) <= 0.7
132
+ assert counts.idxmax() == "United States"
File without changes
File without changes
File without changes
File without changes
File without changes