dbt-cst 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. dbt_cst-0.1.0/.gitignore +6 -0
  2. dbt_cst-0.1.0/.python-version +1 -0
  3. dbt_cst-0.1.0/Cargo.lock +236 -0
  4. dbt_cst-0.1.0/Cargo.toml +20 -0
  5. dbt_cst-0.1.0/LICENSE +21 -0
  6. dbt_cst-0.1.0/PKG-INFO +165 -0
  7. dbt_cst-0.1.0/README.md +153 -0
  8. dbt_cst-0.1.0/examples/dump.rs +22 -0
  9. dbt_cst-0.1.0/pyproject.toml +33 -0
  10. dbt_cst-0.1.0/python/dbt_cst/__init__.py +27 -0
  11. dbt_cst-0.1.0/python/dbt_cst/_native.pyi +84 -0
  12. dbt_cst-0.1.0/python/dbt_cst/py.typed +0 -0
  13. dbt_cst-0.1.0/rustfmt.toml +1 -0
  14. dbt_cst-0.1.0/scripts/publish.sh +44 -0
  15. dbt_cst-0.1.0/src/cst.rs +273 -0
  16. dbt_cst-0.1.0/src/edit.rs +353 -0
  17. dbt_cst-0.1.0/src/jinja.rs +463 -0
  18. dbt_cst-0.1.0/src/lexer.rs +344 -0
  19. dbt_cst-0.1.0/src/lib.rs +32 -0
  20. dbt_cst-0.1.0/src/parser.rs +513 -0
  21. dbt_cst-0.1.0/src/python.rs +596 -0
  22. dbt_cst-0.1.0/src/query.rs +237 -0
  23. dbt_cst-0.1.0/tests/corpus.rs +179 -0
  24. dbt_cst-0.1.0/tests/corpus_repos.txt +17 -0
  25. dbt_cst-0.1.0/tests/cst.rs +670 -0
  26. dbt_cst-0.1.0/tests/fixtures/bare_keyword.sql +1 -0
  27. dbt_cst-0.1.0/tests/fixtures/comment_only.sql +4 -0
  28. dbt_cst-0.1.0/tests/fixtures/crlf_no_trailing_newline.sql +4 -0
  29. dbt_cst-0.1.0/tests/fixtures/edge_cases.sql +52 -0
  30. dbt_cst-0.1.0/tests/fixtures/empty.sql +0 -0
  31. dbt_cst-0.1.0/tests/fixtures/jinja_delimiters.sql +4 -0
  32. dbt_cst-0.1.0/tests/fixtures/multiple_statements.sql +4 -0
  33. dbt_cst-0.1.0/tests/fixtures/unbalanced_brackets.sql +2 -0
  34. dbt_cst-0.1.0/tests/fixtures/unicode.sql +5 -0
  35. dbt_cst-0.1.0/tests/fixtures/unpaired_jinja_blocks.sql +6 -0
  36. dbt_cst-0.1.0/tests/fixtures/unterminated_block_comment.sql +2 -0
  37. dbt_cst-0.1.0/tests/fixtures/unterminated_dollar_quote.sql +2 -0
  38. dbt_cst-0.1.0/tests/fixtures/unterminated_quoted_identifier.sql +2 -0
  39. dbt_cst-0.1.0/tests/fixtures/unterminated_string.sql +2 -0
  40. dbt_cst-0.1.0/tests/python/test_albert_models.py +64 -0
  41. dbt_cst-0.1.0/tests/python/test_api.py +189 -0
  42. dbt_cst-0.1.0/tests/python/test_corpus.py +119 -0
  43. dbt_cst-0.1.0/tests/python/test_unicode.py +81 -0
  44. dbt_cst-0.1.0/tests/support/corpus.rs +101 -0
  45. dbt_cst-0.1.0/tests/support/mod.rs +1 -0
  46. dbt_cst-0.1.0/uv.lock +218 -0
@@ -0,0 +1,6 @@
1
+ /target
2
+ .venv/
3
+ __pycache__/
4
+ *.so
5
+ .pytest_cache/
6
+ /dist
@@ -0,0 +1 @@
1
+ 3.11
@@ -0,0 +1,236 @@
1
+ # This file is automatically @generated by Cargo.
2
+ # It is not intended for manual editing.
3
+ version = 4
4
+
5
+ [[package]]
6
+ name = "bitflags"
7
+ version = "2.13.2"
8
+ source = "registry+https://github.com/rust-lang/crates.io-index"
9
+ checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06"
10
+
11
+ [[package]]
12
+ name = "cfg-if"
13
+ version = "1.0.5"
14
+ source = "registry+https://github.com/rust-lang/crates.io-index"
15
+ checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
16
+
17
+ [[package]]
18
+ name = "dbt-cst"
19
+ version = "0.1.0"
20
+ dependencies = [
21
+ "nom",
22
+ "parking_lot",
23
+ "pyo3",
24
+ ]
25
+
26
+ [[package]]
27
+ name = "heck"
28
+ version = "0.5.0"
29
+ source = "registry+https://github.com/rust-lang/crates.io-index"
30
+ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
31
+
32
+ [[package]]
33
+ name = "inventory"
34
+ version = "0.3.25"
35
+ source = "registry+https://github.com/rust-lang/crates.io-index"
36
+ checksum = "6928282826c822ad91bf1c9a1cb90a30ba1c26770749929b4656cd6be829cd7c"
37
+ dependencies = [
38
+ "rustversion",
39
+ ]
40
+
41
+ [[package]]
42
+ name = "libc"
43
+ version = "0.2.190"
44
+ source = "registry+https://github.com/rust-lang/crates.io-index"
45
+ checksum = "ce5d3ddc6d3fa000eb1536d85e147bfe31aacaba692ed6a876f95cb7c855be78"
46
+
47
+ [[package]]
48
+ name = "lock_api"
49
+ version = "0.4.14"
50
+ source = "registry+https://github.com/rust-lang/crates.io-index"
51
+ checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965"
52
+ dependencies = [
53
+ "scopeguard",
54
+ ]
55
+
56
+ [[package]]
57
+ name = "memchr"
58
+ version = "2.8.3"
59
+ source = "registry+https://github.com/rust-lang/crates.io-index"
60
+ checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
61
+
62
+ [[package]]
63
+ name = "nom"
64
+ version = "8.0.0"
65
+ source = "registry+https://github.com/rust-lang/crates.io-index"
66
+ checksum = "df9761775871bdef83bee530e60050f7e54b1105350d6884eb0fb4f46c2f9405"
67
+ dependencies = [
68
+ "memchr",
69
+ ]
70
+
71
+ [[package]]
72
+ name = "once_cell"
73
+ version = "1.21.4"
74
+ source = "registry+https://github.com/rust-lang/crates.io-index"
75
+ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
76
+
77
+ [[package]]
78
+ name = "parking_lot"
79
+ version = "0.12.5"
80
+ source = "registry+https://github.com/rust-lang/crates.io-index"
81
+ checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a"
82
+ dependencies = [
83
+ "lock_api",
84
+ "parking_lot_core",
85
+ ]
86
+
87
+ [[package]]
88
+ name = "parking_lot_core"
89
+ version = "0.9.12"
90
+ source = "registry+https://github.com/rust-lang/crates.io-index"
91
+ checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1"
92
+ dependencies = [
93
+ "cfg-if",
94
+ "libc",
95
+ "redox_syscall",
96
+ "smallvec",
97
+ "windows-link",
98
+ ]
99
+
100
+ [[package]]
101
+ name = "portable-atomic"
102
+ version = "1.15.0"
103
+ source = "registry+https://github.com/rust-lang/crates.io-index"
104
+ checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
105
+
106
+ [[package]]
107
+ name = "proc-macro2"
108
+ version = "1.0.107"
109
+ source = "registry+https://github.com/rust-lang/crates.io-index"
110
+ checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
111
+ dependencies = [
112
+ "unicode-ident",
113
+ ]
114
+
115
+ [[package]]
116
+ name = "pyo3"
117
+ version = "0.29.3"
118
+ source = "registry+https://github.com/rust-lang/crates.io-index"
119
+ checksum = "700d18fa267b73b9b521fd7e13580e2f446916f176cacee1ab63fcc8191f1655"
120
+ dependencies = [
121
+ "inventory",
122
+ "libc",
123
+ "once_cell",
124
+ "portable-atomic",
125
+ "pyo3-build-config",
126
+ "pyo3-ffi",
127
+ "pyo3-macros",
128
+ ]
129
+
130
+ [[package]]
131
+ name = "pyo3-build-config"
132
+ version = "0.29.3"
133
+ source = "registry+https://github.com/rust-lang/crates.io-index"
134
+ checksum = "7b3fc0c4d08f6bb10e71fe39dfb9e2f59c6eb6854e22ec8092f50c69a4499adb"
135
+ dependencies = [
136
+ "target-lexicon",
137
+ ]
138
+
139
+ [[package]]
140
+ name = "pyo3-ffi"
141
+ version = "0.29.3"
142
+ source = "registry+https://github.com/rust-lang/crates.io-index"
143
+ checksum = "dfc0b8e19df29aad7086cf977bb0c2a2f143e30567eb113e9cf72b62ca698330"
144
+ dependencies = [
145
+ "libc",
146
+ "pyo3-build-config",
147
+ ]
148
+
149
+ [[package]]
150
+ name = "pyo3-macros"
151
+ version = "0.29.3"
152
+ source = "registry+https://github.com/rust-lang/crates.io-index"
153
+ checksum = "6100e8a4b5eba53afaa5ed078364851a0b2499c553a44c31026b929049b49dc6"
154
+ dependencies = [
155
+ "proc-macro2",
156
+ "pyo3-macros-backend",
157
+ "quote",
158
+ "syn",
159
+ ]
160
+
161
+ [[package]]
162
+ name = "pyo3-macros-backend"
163
+ version = "0.29.3"
164
+ source = "registry+https://github.com/rust-lang/crates.io-index"
165
+ checksum = "6143877a16e82b5a727b7127ff4cd86858a24a28d745d72f43e6f227c7b1bdb3"
166
+ dependencies = [
167
+ "heck",
168
+ "proc-macro2",
169
+ "quote",
170
+ "syn",
171
+ ]
172
+
173
+ [[package]]
174
+ name = "quote"
175
+ version = "1.0.47"
176
+ source = "registry+https://github.com/rust-lang/crates.io-index"
177
+ checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
178
+ dependencies = [
179
+ "proc-macro2",
180
+ ]
181
+
182
+ [[package]]
183
+ name = "redox_syscall"
184
+ version = "0.5.18"
185
+ source = "registry+https://github.com/rust-lang/crates.io-index"
186
+ checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d"
187
+ dependencies = [
188
+ "bitflags",
189
+ ]
190
+
191
+ [[package]]
192
+ name = "rustversion"
193
+ version = "1.0.23"
194
+ source = "registry+https://github.com/rust-lang/crates.io-index"
195
+ checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f"
196
+
197
+ [[package]]
198
+ name = "scopeguard"
199
+ version = "1.2.0"
200
+ source = "registry+https://github.com/rust-lang/crates.io-index"
201
+ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
202
+
203
+ [[package]]
204
+ name = "smallvec"
205
+ version = "1.16.2"
206
+ source = "registry+https://github.com/rust-lang/crates.io-index"
207
+ checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e"
208
+
209
+ [[package]]
210
+ name = "syn"
211
+ version = "2.0.119"
212
+ source = "registry+https://github.com/rust-lang/crates.io-index"
213
+ checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
214
+ dependencies = [
215
+ "proc-macro2",
216
+ "quote",
217
+ "unicode-ident",
218
+ ]
219
+
220
+ [[package]]
221
+ name = "target-lexicon"
222
+ version = "0.13.5"
223
+ source = "registry+https://github.com/rust-lang/crates.io-index"
224
+ checksum = "adb6935a6f5c20170eeceb1a3835a49e12e19d792f6dd344ccc76a985ca5a6ca"
225
+
226
+ [[package]]
227
+ name = "unicode-ident"
228
+ version = "1.0.26"
229
+ source = "registry+https://github.com/rust-lang/crates.io-index"
230
+ checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
231
+
232
+ [[package]]
233
+ name = "windows-link"
234
+ version = "0.2.1"
235
+ source = "registry+https://github.com/rust-lang/crates.io-index"
236
+ checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5"
@@ -0,0 +1,20 @@
1
+ [package]
2
+ name = "dbt-cst"
3
+ version = "0.1.0"
4
+ edition = "2024"
5
+ description = "Lossless concrete syntax tree parser for dbt models (Jinja templated SQL) with Python bindings"
6
+ readme = "README.md"
7
+
8
+ [lib]
9
+ name = "dbt_cst"
10
+ crate-type = ["cdylib", "rlib"]
11
+
12
+ [features]
13
+ default = []
14
+ # Enabled by maturin when building the Python extension module.
15
+ python = ["dep:pyo3"]
16
+
17
+ [dependencies]
18
+ nom = "8.0.0"
19
+ parking_lot = "0.12.5"
20
+ pyo3 = { version = "0.29.3", features = ["abi3-py39", "extension-module", "multiple-pymethods"], optional = true }
dbt_cst-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ # MIT License
2
+
3
+ Copyright (c) 2026 Albert
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
dbt_cst-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,165 @@
1
+ Metadata-Version: 2.4
2
+ Name: dbt-cst
3
+ Version: 0.1.0
4
+ Classifier: Programming Language :: Rust
5
+ Classifier: Programming Language :: Python :: Implementation :: CPython
6
+ License-File: LICENSE
7
+ Summary: Lossless dbt model parser: read, modify and rewrite Jinja templated SQL with its formatting intact
8
+ Requires-Python: >=3.9
9
+ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
10
+ Project-URL: Repository, https://github.com/livinlefevreloca/dbt-cst
11
+
12
+ # dbt-cst
13
+
14
+ A lossless dbt model parser written in Rust with Python bindings. It parses a model (Jinja
15
+ templated SQL) into a concrete syntax tree (CST) that keeps every comment, blank line,
16
+ indentation and Jinja tag, so a file can be read, modified and written back out with its
17
+ formatting intact:
18
+
19
+ ```python
20
+ from dbt_cst import parse
21
+
22
+ doc = parse(source)
23
+ assert str(doc) == source # byte for byte, for every model in albert-dbt-analytics-transforms
24
+ ```
25
+
26
+ The tree groups the SQL into queries, clauses, select columns and CTEs, and the Jinja into
27
+ blocks, without interpreting either: a column is the text between two commas, a WHERE clause
28
+ is the text between its keyword and the next clause keyword. That is enough to add and remove
29
+ columns in a model someone has customized by hand, which rewriting the file from a template
30
+ cannot do. The Python package is an abi3 wheel built against Python 3.9, so one wheel works on
31
+ Python 3.9 and 3.11.
32
+
33
+ ## Python usage
34
+
35
+ ```python
36
+ from dbt_cst import parse, DbtSyntaxError
37
+
38
+ doc = parse(path.read_text())
39
+
40
+ doc.config["materialized"] # "view"
41
+ doc.config["materialized"] = "table" # quoted like the call's other strings
42
+ doc.config["tags"] = ["pii"] # added in the call's layout: one per line, trailing comma
43
+ doc.config.set_raw("schema", 'var("schema")')
44
+ del doc.config["alias"]
45
+
46
+ # Example: Modifying multiple config values
47
+ doc.config["materialized"] = "table"
48
+ doc.config["schema"] = "analytics"
49
+ doc.config["tags"] = ["pii", "finance"]
50
+ doc.config["enabled"] = False
51
+
52
+ # Example: Using raw expressions for dynamic values
53
+ doc.config.set_raw("alias", 'var("model_alias")')
54
+ doc.config.set_raw("schema", 'var("schema")')
55
+
56
+ # Example: Removing config values
57
+ if "description" in doc.config:
58
+ del doc.config["description"]
59
+
60
+ doc.sources # [("trading", "apex_apexentry")]
61
+ doc.refs # [("users",), ("package", "model")]
62
+
63
+ query = doc.query # the model's own SELECT, after any CTEs
64
+ select = query.select
65
+ select.names # ["account_id", "activity_id", "percent"]
66
+ "account_id" in select # True, compared ignoring case
67
+
68
+ select.add("user_id") # appended in the select's comma style
69
+ select.add("amount / 100 AS dollars", index=select.index("account_id") + 1)
70
+ select.remove("activity_id") # with its comma and the comments above it
71
+ select.find("percent").text = '"percent" AS pct'
72
+
73
+ renamed = query.cte("renamed").query.select # CTE bodies are queries too
74
+ for query in doc.queries(): # every query, subqueries included
75
+ ...
76
+
77
+ path.write_text(str(doc))
78
+ ```
79
+
80
+ ### API
81
+
82
+ `parse(source) -> Document` raises `DbtSyntaxError` (a `ValueError`) with `line`, `column` and
83
+ `offset` attributes. Only a Jinja tag that is never closed is an error. SQL that does not fit
84
+ the expected shape still parses: it ends up as unstructured tokens and still round trips.
85
+
86
+ | Object | Members |
87
+ | ---------- | ----------------------------------------------------------------------------------------------- |
88
+ | `Document` | `config`, `query`, `queries()`, `sources`, `refs`, `str()` |
89
+ | `Config` | `c[key]`, `get(key, default)`, `c[key] = value`, `del c[key]`, `raw(key)`, `set_raw(key, src)`, `keys()`, `in`, `len`, `iter` |
90
+ | `Query` | `select` (the first), `selects` (one per UNION branch), `ctes`, `cte(name)`, `clauses` |
91
+ | `Cte` | `name`, `query` |
92
+ | `Clause` | `name` (`FROM`, `GROUP BY`, `UNION ALL`, ...), `str()` |
93
+ | `Select` | `columns`, `names`, `find(name)`, `index(name)`, `add(text, index=None)`, `remove(name_or_column)`, `in`, `len`, `iter` |
94
+ | `Column` | `text` (settable), `expr`, `alias`, `name`, `is_templated`, `select`, `detach()` |
95
+
96
+ `Config` values are Jinja literals: strings, numbers, booleans, `None`, lists and dicts.
97
+ Reading an argument whose value is an expression such as `var("x")` raises `ValueError`; use
98
+ `raw()` for its source text.
99
+
100
+ `Column.name` is the alias, or the last part of a plain column reference (`t.id` gives `id`),
101
+ unquoted. It is `None` for an unaliased expression or a bare Jinja tag such as
102
+ `{{ dbt_utils.star(...) }}`.
103
+
104
+ Handles are live: a `Column` from `select.columns` stays attached to the document, and two
105
+ handles to the same node compare equal.
106
+
107
+ ## How it parses
108
+
109
+ 1. **Lexing** (`src/lexer.rs`) splits the file into tokens covering every byte. Jinja tags end
110
+ where Jinja says they do, skipping string literals and nested brackets, so
111
+ `{{ {'a': 1} }}` and `{{ "}}" }}` are one tag each. A tag inside a SQL string or comment is
112
+ absorbed into that token. `{% raw %}` sections are one token.
113
+ 2. **Grouping** (`src/parser.rs`) pairs `(`/`)`, `[`/`]` and Jinja block tags with one stack.
114
+ Any `{% name %}` whose `{% endname %}` appears in the file is opened tentatively, so tags
115
+ from Jinja, dbt (`materialization`, `test`, `snapshot`) and packages all work without a
116
+ list of names. A tag that never closes, like a one line `{% set %}`, or one whose end would
117
+ cross a parenthesis, is put back as a plain token. `elif` and `else` split blocks into
118
+ branches.
119
+ 3. **Structure**: a top level `SELECT` or `WITH` starts a query, which is split into clauses at
120
+ top level clause keywords. SELECT items are split at top level commas, and WITH clauses into
121
+ CTEs. "Top level" means outside any bracket or Jinja block, so commas in function calls and
122
+ keywords in subqueries never split anything. A Jinja block that supplies its own commas,
123
+ like `{% for c in cols %}{{ c }},{% endfor %}`, separates the columns around it.
124
+
125
+ The lexer is written with nom, like looker-cst's parser. The one hand written scan is the end
126
+ of a `{{ }}` or `{% %}` tag, which tracks bracket depth and skips string literals the way
127
+ Jinja's own lexer does.
128
+
129
+ ## Rust usage
130
+
131
+ ```rust
132
+ let doc = dbt_cst::parse(&source)?;
133
+ let query = doc.query().unwrap();
134
+ let select = dbt_cst::query::clauses_named(&query.read(), "SELECT").remove(0);
135
+ dbt_cst::edit::insert_item(&mut select.write(), usize::MAX, "user_id")?;
136
+ assert!(doc.to_string().contains("user_id"));
137
+ ```
138
+
139
+ `cargo run --example dump -- path/to/model.sql` prints the tree.
140
+
141
+ ## Development
142
+
143
+ ```sh
144
+ cargo test # Rust tests, cloning the corpus into target/corpus on first run
145
+ uv run pytest # builds the extension with maturin, then runs the Python tests
146
+ DBT_EXTRA_REPOS=../albert-dbt-analytics-transforms cargo test --release --test corpus
147
+ ```
148
+
149
+ The corpus is the 14 public dbt projects listed in `tests/corpus_repos.txt` (dbt-utils,
150
+ dbt-expectations, automate-dv, the Fivetran packages and others, about 1,700 `.sql` files).
151
+ Both test suites shallow clone them into `target/corpus` on first use, or into `DBT_CORPUS`
152
+ when set. A failed clone skips those files unless `REQUIRE_DBT_CORPUS` is set, as in CI.
153
+ `DBT_EXTRA_REPOS` adds more checkouts, separated by `:`.
154
+
155
+ - `tests/corpus.rs` round trips every file, adds and removes a column in every SELECT,
156
+ removes each column in turn, and adds and removes a config argument.
157
+ - `tests/python/test_corpus.py` does the same read, modify and write cycle through the Python
158
+ API, one test per file.
159
+ - `tests/fixtures/` holds awkward inputs (CRLF, unicode, unterminated quotes, unbalanced
160
+ brackets, unpaired Jinja blocks) that both suites round trip, and `edge_cases.sql`, a
161
+ customized model whose structure both suites check.
162
+ - `tests/python/test_albert_models.py` checks every model in a sibling
163
+ `albert-dbt-analytics-transforms` checkout (or `ALBERT_DBT_REPO`) against the field list
164
+ DBTGenerator reads today.
165
+
@@ -0,0 +1,153 @@
1
+ # dbt-cst
2
+
3
+ A lossless dbt model parser written in Rust with Python bindings. It parses a model (Jinja
4
+ templated SQL) into a concrete syntax tree (CST) that keeps every comment, blank line,
5
+ indentation and Jinja tag, so a file can be read, modified and written back out with its
6
+ formatting intact:
7
+
8
+ ```python
9
+ from dbt_cst import parse
10
+
11
+ doc = parse(source)
12
+ assert str(doc) == source # byte for byte, for every model in albert-dbt-analytics-transforms
13
+ ```
14
+
15
+ The tree groups the SQL into queries, clauses, select columns and CTEs, and the Jinja into
16
+ blocks, without interpreting either: a column is the text between two commas, a WHERE clause
17
+ is the text between its keyword and the next clause keyword. That is enough to add and remove
18
+ columns in a model someone has customized by hand, which rewriting the file from a template
19
+ cannot do. The Python package is an abi3 wheel built against Python 3.9, so one wheel works on
20
+ Python 3.9 and 3.11.
21
+
22
+ ## Python usage
23
+
24
+ ```python
25
+ from dbt_cst import parse, DbtSyntaxError
26
+
27
+ doc = parse(path.read_text())
28
+
29
+ doc.config["materialized"] # "view"
30
+ doc.config["materialized"] = "table" # quoted like the call's other strings
31
+ doc.config["tags"] = ["pii"] # added in the call's layout: one per line, trailing comma
32
+ doc.config.set_raw("schema", 'var("schema")')
33
+ del doc.config["alias"]
34
+
35
+ # Example: Modifying multiple config values
36
+ doc.config["materialized"] = "table"
37
+ doc.config["schema"] = "analytics"
38
+ doc.config["tags"] = ["pii", "finance"]
39
+ doc.config["enabled"] = False
40
+
41
+ # Example: Using raw expressions for dynamic values
42
+ doc.config.set_raw("alias", 'var("model_alias")')
43
+ doc.config.set_raw("schema", 'var("schema")')
44
+
45
+ # Example: Removing config values
46
+ if "description" in doc.config:
47
+ del doc.config["description"]
48
+
49
+ doc.sources # [("trading", "apex_apexentry")]
50
+ doc.refs # [("users",), ("package", "model")]
51
+
52
+ query = doc.query # the model's own SELECT, after any CTEs
53
+ select = query.select
54
+ select.names # ["account_id", "activity_id", "percent"]
55
+ "account_id" in select # True, compared ignoring case
56
+
57
+ select.add("user_id") # appended in the select's comma style
58
+ select.add("amount / 100 AS dollars", index=select.index("account_id") + 1)
59
+ select.remove("activity_id") # with its comma and the comments above it
60
+ select.find("percent").text = '"percent" AS pct'
61
+
62
+ renamed = query.cte("renamed").query.select # CTE bodies are queries too
63
+ for query in doc.queries(): # every query, subqueries included
64
+ ...
65
+
66
+ path.write_text(str(doc))
67
+ ```
68
+
69
+ ### API
70
+
71
+ `parse(source) -> Document` raises `DbtSyntaxError` (a `ValueError`) with `line`, `column` and
72
+ `offset` attributes. Only a Jinja tag that is never closed is an error. SQL that does not fit
73
+ the expected shape still parses: it ends up as unstructured tokens and still round trips.
74
+
75
+ | Object | Members |
76
+ | ---------- | ----------------------------------------------------------------------------------------------- |
77
+ | `Document` | `config`, `query`, `queries()`, `sources`, `refs`, `str()` |
78
+ | `Config` | `c[key]`, `get(key, default)`, `c[key] = value`, `del c[key]`, `raw(key)`, `set_raw(key, src)`, `keys()`, `in`, `len`, `iter` |
79
+ | `Query` | `select` (the first), `selects` (one per UNION branch), `ctes`, `cte(name)`, `clauses` |
80
+ | `Cte` | `name`, `query` |
81
+ | `Clause` | `name` (`FROM`, `GROUP BY`, `UNION ALL`, ...), `str()` |
82
+ | `Select` | `columns`, `names`, `find(name)`, `index(name)`, `add(text, index=None)`, `remove(name_or_column)`, `in`, `len`, `iter` |
83
+ | `Column` | `text` (settable), `expr`, `alias`, `name`, `is_templated`, `select`, `detach()` |
84
+
85
+ `Config` values are Jinja literals: strings, numbers, booleans, `None`, lists and dicts.
86
+ Reading an argument whose value is an expression such as `var("x")` raises `ValueError`; use
87
+ `raw()` for its source text.
88
+
89
+ `Column.name` is the alias, or the last part of a plain column reference (`t.id` gives `id`),
90
+ unquoted. It is `None` for an unaliased expression or a bare Jinja tag such as
91
+ `{{ dbt_utils.star(...) }}`.
92
+
93
+ Handles are live: a `Column` from `select.columns` stays attached to the document, and two
94
+ handles to the same node compare equal.
95
+
96
+ ## How it parses
97
+
98
+ 1. **Lexing** (`src/lexer.rs`) splits the file into tokens covering every byte. Jinja tags end
99
+ where Jinja says they do, skipping string literals and nested brackets, so
100
+ `{{ {'a': 1} }}` and `{{ "}}" }}` are one tag each. A tag inside a SQL string or comment is
101
+ absorbed into that token. `{% raw %}` sections are one token.
102
+ 2. **Grouping** (`src/parser.rs`) pairs `(`/`)`, `[`/`]` and Jinja block tags with one stack.
103
+ Any `{% name %}` whose `{% endname %}` appears in the file is opened tentatively, so tags
104
+ from Jinja, dbt (`materialization`, `test`, `snapshot`) and packages all work without a
105
+ list of names. A tag that never closes, like a one line `{% set %}`, or one whose end would
106
+ cross a parenthesis, is put back as a plain token. `elif` and `else` split blocks into
107
+ branches.
108
+ 3. **Structure**: a top level `SELECT` or `WITH` starts a query, which is split into clauses at
109
+ top level clause keywords. SELECT items are split at top level commas, and WITH clauses into
110
+ CTEs. "Top level" means outside any bracket or Jinja block, so commas in function calls and
111
+ keywords in subqueries never split anything. A Jinja block that supplies its own commas,
112
+ like `{% for c in cols %}{{ c }},{% endfor %}`, separates the columns around it.
113
+
114
+ The lexer is written with nom, like looker-cst's parser. The one hand written scan is the end
115
+ of a `{{ }}` or `{% %}` tag, which tracks bracket depth and skips string literals the way
116
+ Jinja's own lexer does.
117
+
118
+ ## Rust usage
119
+
120
+ ```rust
121
+ let doc = dbt_cst::parse(&source)?;
122
+ let query = doc.query().unwrap();
123
+ let select = dbt_cst::query::clauses_named(&query.read(), "SELECT").remove(0);
124
+ dbt_cst::edit::insert_item(&mut select.write(), usize::MAX, "user_id")?;
125
+ assert!(doc.to_string().contains("user_id"));
126
+ ```
127
+
128
+ `cargo run --example dump -- path/to/model.sql` prints the tree.
129
+
130
+ ## Development
131
+
132
+ ```sh
133
+ cargo test # Rust tests, cloning the corpus into target/corpus on first run
134
+ uv run pytest # builds the extension with maturin, then runs the Python tests
135
+ DBT_EXTRA_REPOS=../albert-dbt-analytics-transforms cargo test --release --test corpus
136
+ ```
137
+
138
+ The corpus is the 14 public dbt projects listed in `tests/corpus_repos.txt` (dbt-utils,
139
+ dbt-expectations, automate-dv, the Fivetran packages and others, about 1,700 `.sql` files).
140
+ Both test suites shallow clone them into `target/corpus` on first use, or into `DBT_CORPUS`
141
+ when set. A failed clone skips those files unless `REQUIRE_DBT_CORPUS` is set, as in CI.
142
+ `DBT_EXTRA_REPOS` adds more checkouts, separated by `:`.
143
+
144
+ - `tests/corpus.rs` round trips every file, adds and removes a column in every SELECT,
145
+ removes each column in turn, and adds and removes a config argument.
146
+ - `tests/python/test_corpus.py` does the same read, modify and write cycle through the Python
147
+ API, one test per file.
148
+ - `tests/fixtures/` holds awkward inputs (CRLF, unicode, unterminated quotes, unbalanced
149
+ brackets, unpaired Jinja blocks) that both suites round trip, and `edge_cases.sql`, a
150
+ customized model whose structure both suites check.
151
+ - `tests/python/test_albert_models.py` checks every model in a sibling
152
+ `albert-dbt-analytics-transforms` checkout (or `ALBERT_DBT_REPO`) against the field list
153
+ DBTGenerator reads today.
@@ -0,0 +1,22 @@
1
+ //! Prints the tree of a dbt file, one node or token per line.
2
+ //!
3
+ //! cargo run --example dump -- path/to/model.sql
4
+
5
+ use dbt_cst::{Child, Node, parse};
6
+
7
+ fn main() {
8
+ let path = std::env::args().nth(1).expect("usage: dump <file.sql>");
9
+ let source = std::fs::read_to_string(&path).expect("readable file");
10
+ let doc = parse(&source).unwrap_or_else(|e| panic!("{path}: {e}"));
11
+ print_node(&doc.root.read(), 0);
12
+ }
13
+
14
+ fn print_node(node: &Node, depth: usize) {
15
+ println!("{}{:?}", " ".repeat(depth), node.kind);
16
+ for child in &node.children {
17
+ match child {
18
+ Child::Token(token) => println!("{}{:?} {:?}", " ".repeat(depth + 1), token.kind, token.text),
19
+ Child::Node(node) => print_node(&node.read(), depth + 1),
20
+ }
21
+ }
22
+ }
@@ -0,0 +1,33 @@
1
+ [build-system]
2
+ requires = ["maturin>=1.9,<2.0"]
3
+ build-backend = "maturin"
4
+
5
+ [project]
6
+ name = "dbt-cst"
7
+ version = "0.1.0"
8
+ description = "Lossless dbt model parser: read, modify and rewrite Jinja templated SQL with its formatting intact"
9
+ requires-python = ">=3.9"
10
+ readme = "README.md"
11
+ license = { file = "LICENSE" }
12
+ classifiers = [
13
+ "Programming Language :: Rust",
14
+ "Programming Language :: Python :: Implementation :: CPython",
15
+ ]
16
+
17
+ [project.urls]
18
+ Repository = "https://github.com/livinlefevreloca/dbt-cst"
19
+
20
+ [dependency-groups]
21
+ dev = ["pytest>=8"]
22
+
23
+ [tool.maturin]
24
+ python-source = "python"
25
+ module-name = "dbt_cst._native"
26
+ features = ["python"]
27
+
28
+ [tool.pytest.ini_options]
29
+ testpaths = ["tests/python"]
30
+
31
+ [tool.uv]
32
+ # Rebuild the extension when the Rust sources change.
33
+ cache-keys = [{ file = "pyproject.toml" }, { file = "Cargo.toml" }, { file = "src/**/*.rs" }]