model2data 1.3.0__tar.gz → 1.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {model2data-1.3.0/model2data.egg-info → model2data-1.3.1}/PKG-INFO +1 -1
  2. {model2data-1.3.0 → model2data-1.3.1}/model2data/generate/faker.py +21 -2
  3. {model2data-1.3.0 → model2data-1.3.1/model2data.egg-info}/PKG-INFO +1 -1
  4. {model2data-1.3.0 → model2data-1.3.1}/pyproject.toml +1 -1
  5. {model2data-1.3.0 → model2data-1.3.1}/tests/test_row_identity.py +42 -0
  6. {model2data-1.3.0 → model2data-1.3.1}/LICENSE +0 -0
  7. {model2data-1.3.0 → model2data-1.3.1}/README.md +0 -0
  8. {model2data-1.3.0 → model2data-1.3.1}/README_PYPI.md +0 -0
  9. {model2data-1.3.0 → model2data-1.3.1}/model2data/__init__.py +0 -0
  10. {model2data-1.3.0 → model2data-1.3.1}/model2data/cli.py +0 -0
  11. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/__init__.py +0 -0
  12. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/project.py +0 -0
  13. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/templates/dbt_project.yml.jinja +0 -0
  14. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/templates/macros/generate_schema_name.sql +0 -0
  15. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/templates/profiles.yml.jinja +0 -0
  16. {model2data-1.3.0 → model2data-1.3.1}/model2data/dbt/tests.py +0 -0
  17. {model2data-1.3.0 → model2data-1.3.1}/model2data/generate/__init__.py +0 -0
  18. {model2data-1.3.0 → model2data-1.3.1}/model2data/generate/core.py +0 -0
  19. {model2data-1.3.0 → model2data-1.3.1}/model2data/generate/relationships.py +0 -0
  20. {model2data-1.3.0 → model2data-1.3.1}/model2data/parse/__init__.py +0 -0
  21. {model2data-1.3.0 → model2data-1.3.1}/model2data/parse/dbml.py +0 -0
  22. {model2data-1.3.0 → model2data-1.3.1}/model2data/utils.py +0 -0
  23. {model2data-1.3.0 → model2data-1.3.1}/model2data.egg-info/SOURCES.txt +0 -0
  24. {model2data-1.3.0 → model2data-1.3.1}/model2data.egg-info/dependency_links.txt +0 -0
  25. {model2data-1.3.0 → model2data-1.3.1}/model2data.egg-info/entry_points.txt +0 -0
  26. {model2data-1.3.0 → model2data-1.3.1}/model2data.egg-info/requires.txt +0 -0
  27. {model2data-1.3.0 → model2data-1.3.1}/model2data.egg-info/top_level.txt +0 -0
  28. {model2data-1.3.0 → model2data-1.3.1}/setup.cfg +0 -0
  29. {model2data-1.3.0 → model2data-1.3.1}/tests/test_cli.py +0 -0
  30. {model2data-1.3.0 → model2data-1.3.1}/tests/test_coverage_gaps.py +0 -0
  31. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbml_parser.py +0 -0
  32. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbml_parser_fuzz.py +0 -0
  33. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbt_integration.py +0 -0
  34. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbt_naming.py +0 -0
  35. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbt_project.py +0 -0
  36. {model2data-1.3.0 → model2data-1.3.1}/tests/test_dbt_tests.py +0 -0
  37. {model2data-1.3.0 → model2data-1.3.1}/tests/test_faker_name_inference.py +0 -0
  38. {model2data-1.3.0 → model2data-1.3.1}/tests/test_generation.py +0 -0
  39. {model2data-1.3.0 → model2data-1.3.1}/tests/test_release_stress.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.3.0
3
+ Version: 1.3.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -2,6 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  import random
4
4
  import re
5
+ import unicodedata
5
6
  import uuid
6
7
  from dataclasses import dataclass
7
8
  from datetime import datetime, timedelta
@@ -147,9 +148,27 @@ _person_state: dict[str, list[_Person]] = {}
147
148
  _address_state: dict[str, list[_Address]] = {}
148
149
 
149
150
 
151
+ # Letters that NFKD does not take apart, because they are their own letters
152
+ # rather than a base plus an accent. Without these, "ø" and "ß" would simply
153
+ # vanish along with the accents.
154
+ _UNDECOMPOSED_LETTERS = str.maketrans(
155
+ {"ø": "o", "æ": "ae", "œ": "oe", "ß": "ss", "ł": "l", "đ": "d", "ð": "d", "þ": "th", "ı": "i"}
156
+ )
157
+
158
+
150
159
  def _slug(value: str) -> str:
151
- """Reduce a name to something that can sit inside an email or a username."""
152
- return re.sub(r"[^a-z0-9]+", "", value.lower()) or "user"
160
+ """Reduce a name to something that can sit inside an email or a username.
161
+
162
+ Accents are folded, not dropped. Stripping them outright turned `Aimée` into
163
+ `aime` and `Müller` into `mller` -- not that person's name, and conspicuously
164
+ broken in exactly the European locales the locale option exists to serve.
165
+ NFKD splits most accented letters into a base letter plus a combining mark,
166
+ which encoding to ASCII then discards; the letters that do not decompose are
167
+ mapped first.
168
+ """
169
+ folded = value.lower().translate(_UNDECOMPOSED_LETTERS)
170
+ ascii_only = unicodedata.normalize("NFKD", folded).encode("ascii", "ignore").decode("ascii")
171
+ return re.sub(r"[^a-z0-9]+", "", ascii_only) or "user"
153
172
 
154
173
 
155
174
  # Resolved once per locale, not once per row. Both of these are constant for a
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: model2data
3
- Version: 1.3.0
3
+ Version: 1.3.1
4
4
  Summary: Generate analytics-ready datasets from DBML models
5
5
  Author: JB Analytica
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "model2data"
7
- version = "1.3.0"
7
+ version = "1.3.1"
8
8
  description = "Generate analytics-ready datasets from DBML models"
9
9
  readme = "README_PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -370,3 +370,45 @@ class TestThinLocales:
370
370
  # picked once beats a different random country on every row.
371
371
  assert len(countries) == 1
372
372
  assert countries != {""}
373
+
374
+
375
+ class TestNameFolding:
376
+ def test_accents_are_folded_not_dropped(self):
377
+ # Dropping them turned Aimée into "aime" and Müller into "mller" -- not
378
+ # that person's name, and most visible in exactly the European locales
379
+ # the locale option exists to serve.
380
+ assert faker_module._slug("Aimée") == "aimee"
381
+ assert faker_module._slug("Müller") == "muller"
382
+ assert faker_module._slug("Björn") == "bjorn"
383
+
384
+ def test_letters_that_do_not_decompose_are_mapped(self):
385
+ # NFKD leaves these intact because they are their own letters, not a
386
+ # base plus an accent, so they would vanish with the combining marks.
387
+ assert faker_module._slug("Søren") == "soren"
388
+ assert faker_module._slug("Weiß") == "weiss"
389
+ assert faker_module._slug("Łukasz") == "lukasz"
390
+ assert faker_module._slug("Æther") == "aether"
391
+
392
+ def test_punctuation_and_spacing_still_go(self):
393
+ assert faker_module._slug("O'Brien") == "obrien"
394
+ assert faker_module._slug("Van Der Berg") == "vanderberg"
395
+
396
+ def test_a_name_with_no_latin_letters_falls_back(self):
397
+ # Documented limitation rather than a target: deriving an address from a
398
+ # CJK name needs romanization, and Faker exposes no romanized first/last
399
+ # pair to build one from. Such a locale gets "user", disambiguated by the
400
+ # unique suffixing, which is meaningless but at least stable and unique.
401
+ assert faker_module._slug("日本") == "user"
402
+
403
+ def test_generated_emails_are_ascii_in_an_accented_locale(self):
404
+ df = generate_data_from_dbml(
405
+ tables={"customers": _person_table()},
406
+ refs=[],
407
+ base_rows=40,
408
+ seed=31,
409
+ locale="fr_FR",
410
+ )["customers"]
411
+ for _, row in df.iterrows():
412
+ _assert_row_is_one_person(row)
413
+ # An address has to be usable as an address, whatever the name says.
414
+ row["email"].encode("ascii")
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes