model2data 1.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. model2data-1.5.0/LICENSE +21 -0
  2. model2data-1.5.0/PKG-INFO +326 -0
  3. model2data-1.5.0/README.md +390 -0
  4. model2data-1.5.0/README_PYPI.md +282 -0
  5. model2data-1.5.0/model2data/__init__.py +12 -0
  6. model2data-1.5.0/model2data/cli.py +456 -0
  7. model2data-1.5.0/model2data/dbt/__init__.py +1 -0
  8. model2data-1.5.0/model2data/dbt/project.py +85 -0
  9. model2data-1.5.0/model2data/dbt/templates/dbt_project.yml.jinja +33 -0
  10. model2data-1.5.0/model2data/dbt/templates/macros/generate_schema_name.sql +9 -0
  11. model2data-1.5.0/model2data/dbt/templates/profiles.yml.jinja +18 -0
  12. model2data-1.5.0/model2data/dbt/tests.py +305 -0
  13. model2data-1.5.0/model2data/generate/__init__.py +1 -0
  14. model2data-1.5.0/model2data/generate/core.py +576 -0
  15. model2data-1.5.0/model2data/generate/faker.py +910 -0
  16. model2data-1.5.0/model2data/generate/hints.py +201 -0
  17. model2data-1.5.0/model2data/generate/options.py +80 -0
  18. model2data-1.5.0/model2data/generate/relationships.py +52 -0
  19. model2data-1.5.0/model2data/generate/timeline.py +454 -0
  20. model2data-1.5.0/model2data/parse/__init__.py +1 -0
  21. model2data-1.5.0/model2data/parse/dbml.py +698 -0
  22. model2data-1.5.0/model2data/utils.py +10 -0
  23. model2data-1.5.0/model2data.egg-info/PKG-INFO +326 -0
  24. model2data-1.5.0/model2data.egg-info/SOURCES.txt +45 -0
  25. model2data-1.5.0/model2data.egg-info/dependency_links.txt +1 -0
  26. model2data-1.5.0/model2data.egg-info/entry_points.txt +2 -0
  27. model2data-1.5.0/model2data.egg-info/requires.txt +18 -0
  28. model2data-1.5.0/model2data.egg-info/top_level.txt +1 -0
  29. model2data-1.5.0/pyproject.toml +111 -0
  30. model2data-1.5.0/setup.cfg +4 -0
  31. model2data-1.5.0/tests/test_as_of_anchor.py +139 -0
  32. model2data-1.5.0/tests/test_cli.py +951 -0
  33. model2data-1.5.0/tests/test_coverage_gaps.py +233 -0
  34. model2data-1.5.0/tests/test_dbml_parser.py +2076 -0
  35. model2data-1.5.0/tests/test_dbml_parser_fuzz.py +281 -0
  36. model2data-1.5.0/tests/test_dbt_integration.py +145 -0
  37. model2data-1.5.0/tests/test_dbt_naming.py +56 -0
  38. model2data-1.5.0/tests/test_dbt_project.py +352 -0
  39. model2data-1.5.0/tests/test_dbt_tests.py +640 -0
  40. model2data-1.5.0/tests/test_faker_name_inference.py +258 -0
  41. model2data-1.5.0/tests/test_generation.py +960 -0
  42. model2data-1.5.0/tests/test_options.py +52 -0
  43. model2data-1.5.0/tests/test_release_stress.py +351 -0
  44. model2data-1.5.0/tests/test_row_identity.py +414 -0
  45. model2data-1.5.0/tests/test_shaping.py +493 -0
  46. model2data-1.5.0/tests/test_table_seeds.py +225 -0
  47. model2data-1.5.0/tests/test_timeline.py +421 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 JB Analytica
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,326 @@
1
+ Metadata-Version: 2.4
2
+ Name: model2data
3
+ Version: 1.5.0
4
+ Summary: Generate analytics-ready datasets from DBML models
5
+ Author: JB Analytica
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/JB-Analytica/model2data
8
+ Project-URL: Repository, https://github.com/JB-Analytica/model2data
9
+ Project-URL: Issues, https://github.com/JB-Analytica/model2data/issues
10
+ Project-URL: Changelog, https://github.com/JB-Analytica/model2data/blob/main/CHANGELOG.md
11
+ Keywords: dbt,dbml,synthetic-data,test-data,analytics-engineering,data-engineering,duckdb,seed-data
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Programming Language :: Python :: 3.14
21
+ Classifier: Topic :: Database
22
+ Classifier: Topic :: Software Development :: Testing
23
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
24
+ Requires-Python: >=3.10
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Requires-Dist: dbt-core>=1.11
28
+ Requires-Dist: dbt-duckdb>=1.11
29
+ Requires-Dist: faker>=37.12.0
30
+ Requires-Dist: pandas>=2.3.3
31
+ Requires-Dist: pyyaml>=6.0.3
32
+ Requires-Dist: typer>=0.20.0
33
+ Provides-Extra: postgres
34
+ Requires-Dist: dbt-postgres>=1.11; extra == "postgres"
35
+ Provides-Extra: dev
36
+ Requires-Dist: pytest; extra == "dev"
37
+ Requires-Dist: pytest-cov; extra == "dev"
38
+ Requires-Dist: pre-commit; extra == "dev"
39
+ Requires-Dist: ruff>=0.1.0; extra == "dev"
40
+ Requires-Dist: ty>=0.0.4; extra == "dev"
41
+ Requires-Dist: types-pyyaml; extra == "dev"
42
+ Requires-Dist: poethepoet>=0.38.0; extra == "dev"
43
+ Dynamic: license-file
44
+
45
+ # model2data
46
+
47
+ [![PyPI](https://img.shields.io/pypi/v/model2data)](https://pypi.org/project/model2data/)
48
+ [![CI](https://github.com/JB-Analytica/model2data/actions/workflows/ci.yml/badge.svg)](https://github.com/JB-Analytica/model2data/actions/workflows/ci.yml)
49
+ [![codecov](https://codecov.io/gh/JB-Analytica/model2data/branch/main/graph/badge.svg)](https://codecov.io/gh/JB-Analytica/model2data)
50
+ [![License](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/JB-Analytica/model2data/blob/main/LICENSE)
51
+
52
+ **Turn a data model into a running analytics stack in one command.**
53
+
54
+ Give `model2data` a [DBML](https://dbml.dbdiagram.io/docs/) schema — hand-written or exported
55
+ from an existing database — and it generates realistic, relationship-preserving synthetic data
56
+ *and* a complete, runnable dbt project around it: seeds, staging models, tests, and a
57
+ DuckDB or Postgres profile. No sample data to hunt down, no dbt boilerplate to hand-write, no
58
+ production data to risk exposing.
59
+
60
+ ```bash
61
+ pip install model2data
62
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
63
+ cd dbt_ecommerce && dbt build
64
+ ```
65
+
66
+ That's a working analytics stack — real (synthetic) data, tested dbt models, queryable in
67
+ DuckDB — from a schema file, in seconds:
68
+
69
+ ![model2data generating a project and running it with dbt](https://raw.githubusercontent.com/JB-Analytica/model2data/main/assets/demo.gif)
70
+
71
+ ---
72
+
73
+ ## Why this exists
74
+
75
+ Analytics engineers hit the same wall constantly: you need realistic data to build or test a
76
+ pipeline, but production data is off-limits (privacy, access, scale), and hand-rolling mock CSVs
77
+ is tedious and doesn't scale past two tables. `model2data` closes that gap — from a schema
78
+ definition to a seeded, tested dbt project you can actually run, with no database or production
79
+ access required.
80
+
81
+ - **Privacy-safe.** Nothing but a schema definition goes in; nothing but synthetic data comes out.
82
+ - **Realistic, not random.** Column names are matched against ~35 common patterns — `email`,
83
+ `first_name`, `city`, `phone`, `company`, ... — so a column called `email` gets real-looking
84
+ emails, not `Lorem ipsum` text.
85
+ - **Relationship-preserving.** Foreign keys resolve to real parent rows; tables are generated in
86
+ dependency order.
87
+ - **Deterministic.** Pass `--seed` and the same schema always produces the same data — safe to
88
+ commit fixtures, safe to diff across CI runs. Add `--as-of` to pin the date the data is anchored
89
+ on, and the run reproduces on any later day rather than only on the day it first ran.
90
+ - **Re-rollable one table at a time.** `--table-seed orders=7` regenerates a single table and
91
+ leaves every other table byte-identical, so you can keep the four tables that look right.
92
+ - **A real dbt project, not just CSVs.** Seeds, staging models that `ref()` them, schema tests,
93
+ and a ready-to-use profile — the thing you'd otherwise spend an afternoon scaffolding by hand.
94
+ A single `dbt build` loads, transforms, and tests the whole thing.
95
+
96
+ ## Who is model2data for?
97
+
98
+ - **Analytics engineers** — generate realistic datasets and a working dbt project without
99
+ waiting on production access.
100
+ - **Data engineers** — produce deterministic test data from an existing schema for pipeline and
101
+ migration testing.
102
+ - **Software & data teams** — prototype integrations and analytics workflows without exposing
103
+ production data.
104
+ - **Consultants & architects** — spin up realistic environments for demos, workshops, and
105
+ architecture validation in minutes, not hours.
106
+
107
+ ## How it works
108
+
109
+ 1. **Parse.** Reads tables, columns, types, and `Ref` relationships from a DBML file.
110
+ 2. **Generate.** Produces synthetic values per column — typed generation for known SQL types
111
+ (int, date, timestamp, ...), name-aware inference for everything else (`email`, `phone`,
112
+ `city`, ...), foreign keys resolved against already-generated parent rows.
113
+ 3. **Scaffold.** Writes a complete dbt project around that data: CSV seeds, staging models that
114
+ `ref()` those seeds, `not_null`/`unique`/`relationships` tests, `accepted_values` tests for
115
+ DBML `Enum`-typed columns, singular SQL tests for composite primary/unique keys, table and
116
+ column `description:` fields pulled from DBML notes, and a profile for DuckDB (zero-config,
117
+ file-based) or Postgres.
118
+
119
+ ---
120
+
121
+ ## Installation
122
+
123
+ ```bash
124
+ pip install model2data
125
+ ```
126
+
127
+ ---
128
+
129
+ ## Quick start
130
+
131
+ We bundle several example schemas in `examples/` — this walkthrough uses the e-commerce one
132
+ (`examples/ecommerce.dbml`: customers, products, orders, order items, and reviews).
133
+
134
+ Generate a project with synthetic data:
135
+
136
+ ```bash
137
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42
138
+ ```
139
+
140
+ This creates a `dbt_ecommerce/` folder with your data and dbt setup.
141
+
142
+ `--seed` reproduces a run's numbers, but dates and timestamps are generated relative to the
143
+ current date, so the same seed drifts once the day turns over. `--as-of` pins the date they're
144
+ anchored on, and the whole dataset reproduces on any later day — which is what makes a generated
145
+ fixture safe to commit. If one table comes out wrong and the rest looks right, `--table-seed`
146
+ re-rolls just that table, leaving every other table's seed CSV byte-identical. `--locale` picks
147
+ the country every generated person and address comes from:
148
+
149
+ ```bash
150
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 \
151
+ --as-of 2026-01-31 --table-seed orders=7 --locale nl_BE
152
+ ```
153
+
154
+ Run dbt to load, transform, and test the data:
155
+
156
+ ```bash
157
+ cd dbt_ecommerce
158
+ dbt build
159
+ ```
160
+
161
+ Staging models `ref()` their seeds, so a single `dbt build` loads the seeds, builds the models,
162
+ and runs every generated test in one dependency-ordered pass — no separate `dbt seed`/`dbt run`
163
+ needed, even on a brand-new database. (The individual `dbt deps`, `dbt seed`, and `dbt run`
164
+ commands still work if you'd rather drive the steps yourself; the generated project declares no
165
+ packages, so `dbt deps` is a no-op.)
166
+
167
+ Your analytics-ready dataset is now in DuckDB!
168
+
169
+ To target Postgres instead, install the extra and pass `--adapter postgres`:
170
+
171
+ ```bash
172
+ pip install "model2data[postgres]"
173
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --adapter postgres
174
+ ```
175
+
176
+ Connection details are read from environment variables (`MODEL2DATA_PG_HOST`, `MODEL2DATA_PG_PORT`, `MODEL2DATA_PG_USER`, `MODEL2DATA_PG_PASSWORD`, `MODEL2DATA_PG_DATABASE`), defaulting to `localhost:5432` with a `postgres`/`postgres` user for local development.
177
+
178
+ After generation, the CLI prints a short summary — tables and rows generated, relationships found in the DBML, and any columns that fell back to generic placeholder text because neither their type nor name could be matched.
179
+
180
+ Pass `--unit-tests` to also generate deterministic dbt unit test fixtures (`models/staging/ut_stg_<table>.yml`) from the actually-generated seed rows:
181
+
182
+ ```bash
183
+ model2data --file examples/ecommerce.dbml --rows 200 --seed 42 --unit-tests
184
+ ```
185
+
186
+ This targets dbt-core's native unit testing feature, which works out of the box with the base
187
+ install — see [dbt-core versions](#dbt-core-versions) below.
188
+
189
+ ---
190
+
191
+ ## Generated dbt project structure
192
+
193
+ The generated dbt project includes:
194
+
195
+ ```
196
+ dbt_{project_name}/
197
+ ├── seeds/
198
+ │ └── raw/
199
+ │ ├── __seed_config.yml # seed descriptions + column-type overrides
200
+ │ ├── table1.csv
201
+ │ └── table2.csv
202
+ ├── models/
203
+ │ └── staging/
204
+ │ ├── stg_table1.sql
205
+ │ ├── stg_table1.yml
206
+ │ ├── ut_stg_table1.yml # only with --unit-tests
207
+ │ └── ...
208
+ ├── data-tests/
209
+ │ └── unique_combination_stg_table1_col_a_col_b.sql # only for composite pk/unique keys
210
+ ├── macros/
211
+ │ └── generate_schema_name.sql
212
+ ├── dbt_project.yml
213
+ ├── profiles.yml # DuckDB or Postgres config, depending on --adapter
214
+ └── {project_name}_profile.duckdb # DuckDB adapter only
215
+ ```
216
+
217
+ - **Seeds**: CSV files with generated synthetic data, plus `__seed_config.yml` — each seed's
218
+ `description:` (from the table's DBML `Note`) and the column-type overrides that keep
219
+ all-digit text columns (barcodes, zero-padded postcodes, ...) from being loaded as integers.
220
+ - **Staging Models**: Basic dbt models that `ref()` their seed. Using `ref()` rather than
221
+ declaring the seeds as dbt `sources` is what gives each model a real DAG edge to the seed
222
+ behind it, so one `dbt build` orders seeds before models on a fresh database.
223
+ - **Tests**: A YAML per staging model with column tests (`not_null`, `unique`, `relationships`,
224
+ and `accepted_values` for DBML `Enum`-typed columns). Column `Note` text from the DBML becomes
225
+ `description:` fields.
226
+ - **Composite key tests**: Composite primary/unique keys declared in an `indexes { }` block get
227
+ a singular SQL test under `data-tests/`, dbt's configured `test-paths`.
228
+ - **Profiles**: Pre-configured for DuckDB (file-based) or Postgres (via env vars), with schema handling.
229
+ - **Unit tests** (opt-in via `--unit-tests`): `models/staging/ut_stg_<table>.yml` fixtures built
230
+ from real generated rows, co-located with each staging model so dbt (which only parses unit
231
+ tests from `model-paths`) picks them up.
232
+
233
+ ---
234
+
235
+ ## Using model2data with an LLM
236
+
237
+ If you want to go from a plain-English description of a data model straight to a running,
238
+ demo-ready dbt project, [LLMS.md](https://github.com/JB-Analytica/model2data/blob/main/LLMS.md) is written for an LLM/agent to read: it covers the
239
+ full DBML feature set model2data understands (enums, notes, defaults, composite keys, both
240
+ relationship syntaxes, self-references) and the exact command sequence to run. Point an
241
+ LLM-backed coding assistant at it and describe your data model — it can author the DBML and run
242
+ model2data for you.
243
+
244
+ ---
245
+
246
+ ## dbt-core versions
247
+
248
+ model2data requires **dbt-core >= 1.11**, tracking [dbt's own version support
249
+ policy](https://docs.getdbt.com/docs/dbt-versions): dbt Labs supports each minor release for one
250
+ year, and 1.11 is the oldest that still is. Generated projects build cleanly — no deprecation
251
+ warnings — on every supported dbt-core version, and CI proves it on each push by running a real
252
+ `dbt build` against both the stated floor and the newest release.
253
+
254
+ If you're pinned to an older dbt-core, use model2data 0.5.x, which supported down to 1.8.5.
255
+
256
+ ---
257
+
258
+ ## Design decisions / non-goals
259
+
260
+ - **DuckDB Default**: Chosen for its zero-config, file-based nature, making it easy to get started without database setup. Postgres is supported via `--adapter postgres`; other adapters can be configured manually.
261
+ - **dbt Integration**: Leverages dbt's transformation capabilities for a familiar workflow in analytics engineering.
262
+ - **Synthetic Data**: Uses deterministic generation for reproducibility; not intended for production use or as a replacement for real data.
263
+ - **Non-goals**: This is not a data migration tool, ETL pipeline, or real-time data generator. It focuses on static, synthetic datasets for testing and prototyping.
264
+
265
+ ---
266
+
267
+ ## Limitations
268
+
269
+ - Synthetic data generation is heuristic-based (typed generation, name-aware inference, enum/default awareness) and may not perfectly mimic real-world distributions or edge cases.
270
+ - DuckDB and Postgres are supported today; other databases require manual profile adjustments.
271
+ - No support for incremental models or advanced dbt features in generated projects.
272
+ - Composite foreign keys (across a bridge/join table) are generated as independent single-column FKs — each column's values are individually valid, but the *combination* isn't guaranteed to match a real parent composite key unless that key is separately enforced via `indexes { }`.
273
+ - Any DBML the parser can't fully make sense of (a malformed line, a ref pointing at an unknown table, an unrecognized column definition) is reported as a warning in the CLI's summary rather than silently dropped — check that summary after generating from a schema you didn't author yourself.
274
+
275
+ ---
276
+
277
+ ## Project status
278
+
279
+ As of `1.0.0`, model2data is considered **feature-complete for its intended use case**: turning a
280
+ DBML schema into realistic synthetic data and a runnable dbt project, reliably. There's no active
281
+ roadmap of new capabilities planned — the focus from here is maintenance: bug fixes, keeping pace
282
+ with new dbt-core releases, and reviewing community contributions.
283
+
284
+ Ideas that came up during development but were deliberately left out of scope, in case anyone
285
+ wants to pick them up as a contribution:
286
+
287
+ - Additional database adapters (e.g. Snowflake, BigQuery).
288
+ - A rule-based semantic layer scaffold (`semantic_models.yml`/basic metrics) derived from the
289
+ parsed schema shape.
290
+ - Example mart-layer models on top of staging (the generated `dbt_project.yml` carries a
291
+ ready-to-uncomment `marts` schema/materialization config for this).
292
+
293
+ See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) if you'd like to work on any of these.
294
+
295
+ ---
296
+
297
+ ## Contributing
298
+
299
+ We welcome contributions!
300
+
301
+ - Open issues for bugs or feature requests.
302
+ - Submit PRs to add new DBML examples, custom data generators, or improvements.
303
+ - Ensure all new features include tests if possible.
304
+
305
+ See [CONTRIBUTING.md](https://github.com/JB-Analytica/model2data/blob/main/CONTRIBUTING.md) for detailed guidelines, and [DEVELOPMENT.md](https://github.com/JB-Analytica/model2data/blob/main/DEVELOPMENT.md) for the local dev setup and release process.
306
+
307
+ ## Code of Conduct
308
+
309
+ Please read our [Code of Conduct](https://github.com/JB-Analytica/model2data/blob/main/CODE_OF_CONDUCT.md) to understand our community standards.
310
+
311
+ ---
312
+
313
+ ## License
314
+
315
+ MIT License. See LICENSE for details.
316
+
317
+ ---
318
+
319
+ <p align="center">
320
+ <a href="https://www.jbanalytica.com">
321
+ <img src="https://raw.githubusercontent.com/JB-Analytica/model2data/main/assets/jba-icon-dark-bg.svg" alt="JB Analytica" height="40">
322
+ </a>
323
+ <br>
324
+ Built and maintained by <a href="https://www.jbanalytica.com"><strong>JB Analytica</strong></a> —
325
+ Data & Analytics Engineering · Data Platform Architecture · Modern BI.
326
+ </p>