dbt-cst 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dbt_cst-0.1.0/.gitignore +6 -0
- dbt_cst-0.1.0/.python-version +1 -0
- dbt_cst-0.1.0/Cargo.lock +236 -0
- dbt_cst-0.1.0/Cargo.toml +20 -0
- dbt_cst-0.1.0/LICENSE +21 -0
- dbt_cst-0.1.0/PKG-INFO +165 -0
- dbt_cst-0.1.0/README.md +153 -0
- dbt_cst-0.1.0/examples/dump.rs +22 -0
- dbt_cst-0.1.0/pyproject.toml +33 -0
- dbt_cst-0.1.0/python/dbt_cst/__init__.py +27 -0
- dbt_cst-0.1.0/python/dbt_cst/_native.pyi +84 -0
- dbt_cst-0.1.0/python/dbt_cst/py.typed +0 -0
- dbt_cst-0.1.0/rustfmt.toml +1 -0
- dbt_cst-0.1.0/scripts/publish.sh +44 -0
- dbt_cst-0.1.0/src/cst.rs +273 -0
- dbt_cst-0.1.0/src/edit.rs +353 -0
- dbt_cst-0.1.0/src/jinja.rs +463 -0
- dbt_cst-0.1.0/src/lexer.rs +344 -0
- dbt_cst-0.1.0/src/lib.rs +32 -0
- dbt_cst-0.1.0/src/parser.rs +513 -0
- dbt_cst-0.1.0/src/python.rs +596 -0
- dbt_cst-0.1.0/src/query.rs +237 -0
- dbt_cst-0.1.0/tests/corpus.rs +179 -0
- dbt_cst-0.1.0/tests/corpus_repos.txt +17 -0
- dbt_cst-0.1.0/tests/cst.rs +670 -0
- dbt_cst-0.1.0/tests/fixtures/bare_keyword.sql +1 -0
- dbt_cst-0.1.0/tests/fixtures/comment_only.sql +4 -0
- dbt_cst-0.1.0/tests/fixtures/crlf_no_trailing_newline.sql +4 -0
- dbt_cst-0.1.0/tests/fixtures/edge_cases.sql +52 -0
- dbt_cst-0.1.0/tests/fixtures/empty.sql +0 -0
- dbt_cst-0.1.0/tests/fixtures/jinja_delimiters.sql +4 -0
- dbt_cst-0.1.0/tests/fixtures/multiple_statements.sql +4 -0
- dbt_cst-0.1.0/tests/fixtures/unbalanced_brackets.sql +2 -0
- dbt_cst-0.1.0/tests/fixtures/unicode.sql +5 -0
- dbt_cst-0.1.0/tests/fixtures/unpaired_jinja_blocks.sql +6 -0
- dbt_cst-0.1.0/tests/fixtures/unterminated_block_comment.sql +2 -0
- dbt_cst-0.1.0/tests/fixtures/unterminated_dollar_quote.sql +2 -0
- dbt_cst-0.1.0/tests/fixtures/unterminated_quoted_identifier.sql +2 -0
- dbt_cst-0.1.0/tests/fixtures/unterminated_string.sql +2 -0
- dbt_cst-0.1.0/tests/python/test_albert_models.py +64 -0
- dbt_cst-0.1.0/tests/python/test_api.py +189 -0
- dbt_cst-0.1.0/tests/python/test_corpus.py +119 -0
- dbt_cst-0.1.0/tests/python/test_unicode.py +81 -0
- dbt_cst-0.1.0/tests/support/corpus.rs +101 -0
- dbt_cst-0.1.0/tests/support/mod.rs +1 -0
- dbt_cst-0.1.0/uv.lock +218 -0
dbt_cst-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.11
|
dbt_cst-0.1.0/Cargo.lock
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
# This file is automatically @generated by Cargo.
|
|
2
|
+
# It is not intended for manual editing.
|
|
3
|
+
version = 4
|
|
4
|
+
|
|
5
|
+
[[package]]
|
|
6
|
+
name = "bitflags"
|
|
7
|
+
version = "2.13.2"
|
|
8
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
9
|
+
checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06"
|
|
10
|
+
|
|
11
|
+
[[package]]
|
|
12
|
+
name = "cfg-if"
|
|
13
|
+
version = "1.0.5"
|
|
14
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
15
|
+
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
|
16
|
+
|
|
17
|
+
[[package]]
|
|
18
|
+
name = "dbt-cst"
|
|
19
|
+
version = "0.1.0"
|
|
20
|
+
dependencies = [
|
|
21
|
+
"nom",
|
|
22
|
+
"parking_lot",
|
|
23
|
+
"pyo3",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[[package]]
|
|
27
|
+
name = "heck"
|
|
28
|
+
version = "0.5.0"
|
|
29
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
30
|
+
checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
|
31
|
+
|
|
32
|
+
[[package]]
|
|
33
|
+
name = "inventory"
|
|
34
|
+
version = "0.3.25"
|
|
35
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
36
|
+
checksum = "6928282826c822ad91bf1c9a1cb90a30ba1c26770749929b4656cd6be829cd7c"
|
|
37
|
+
dependencies = [
|
|
38
|
+
"rustversion",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[[package]]
|
|
42
|
+
name = "libc"
|
|
43
|
+
version = "0.2.190"
|
|
44
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
45
|
+
checksum = "ce5d3ddc6d3fa000eb1536d85e147bfe31aacaba692ed6a876f95cb7c855be78"
|
|
46
|
+
|
|
47
|
+
[[package]]
|
|
48
|
+
name = "lock_api"
|
|
49
|
+
version = "0.4.14"
|
|
50
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
51
|
+
checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965"
|
|
52
|
+
dependencies = [
|
|
53
|
+
"scopeguard",
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
[[package]]
|
|
57
|
+
name = "memchr"
|
|
58
|
+
version = "2.8.3"
|
|
59
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
60
|
+
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
|
61
|
+
|
|
62
|
+
[[package]]
|
|
63
|
+
name = "nom"
|
|
64
|
+
version = "8.0.0"
|
|
65
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
66
|
+
checksum = "df9761775871bdef83bee530e60050f7e54b1105350d6884eb0fb4f46c2f9405"
|
|
67
|
+
dependencies = [
|
|
68
|
+
"memchr",
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
[[package]]
|
|
72
|
+
name = "once_cell"
|
|
73
|
+
version = "1.21.4"
|
|
74
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
75
|
+
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
76
|
+
|
|
77
|
+
[[package]]
|
|
78
|
+
name = "parking_lot"
|
|
79
|
+
version = "0.12.5"
|
|
80
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
81
|
+
checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a"
|
|
82
|
+
dependencies = [
|
|
83
|
+
"lock_api",
|
|
84
|
+
"parking_lot_core",
|
|
85
|
+
]
|
|
86
|
+
|
|
87
|
+
[[package]]
|
|
88
|
+
name = "parking_lot_core"
|
|
89
|
+
version = "0.9.12"
|
|
90
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
91
|
+
checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1"
|
|
92
|
+
dependencies = [
|
|
93
|
+
"cfg-if",
|
|
94
|
+
"libc",
|
|
95
|
+
"redox_syscall",
|
|
96
|
+
"smallvec",
|
|
97
|
+
"windows-link",
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
[[package]]
|
|
101
|
+
name = "portable-atomic"
|
|
102
|
+
version = "1.15.0"
|
|
103
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
104
|
+
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
|
105
|
+
|
|
106
|
+
[[package]]
|
|
107
|
+
name = "proc-macro2"
|
|
108
|
+
version = "1.0.107"
|
|
109
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
110
|
+
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
|
111
|
+
dependencies = [
|
|
112
|
+
"unicode-ident",
|
|
113
|
+
]
|
|
114
|
+
|
|
115
|
+
[[package]]
|
|
116
|
+
name = "pyo3"
|
|
117
|
+
version = "0.29.3"
|
|
118
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
119
|
+
checksum = "700d18fa267b73b9b521fd7e13580e2f446916f176cacee1ab63fcc8191f1655"
|
|
120
|
+
dependencies = [
|
|
121
|
+
"inventory",
|
|
122
|
+
"libc",
|
|
123
|
+
"once_cell",
|
|
124
|
+
"portable-atomic",
|
|
125
|
+
"pyo3-build-config",
|
|
126
|
+
"pyo3-ffi",
|
|
127
|
+
"pyo3-macros",
|
|
128
|
+
]
|
|
129
|
+
|
|
130
|
+
[[package]]
|
|
131
|
+
name = "pyo3-build-config"
|
|
132
|
+
version = "0.29.3"
|
|
133
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
134
|
+
checksum = "7b3fc0c4d08f6bb10e71fe39dfb9e2f59c6eb6854e22ec8092f50c69a4499adb"
|
|
135
|
+
dependencies = [
|
|
136
|
+
"target-lexicon",
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
[[package]]
|
|
140
|
+
name = "pyo3-ffi"
|
|
141
|
+
version = "0.29.3"
|
|
142
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
143
|
+
checksum = "dfc0b8e19df29aad7086cf977bb0c2a2f143e30567eb113e9cf72b62ca698330"
|
|
144
|
+
dependencies = [
|
|
145
|
+
"libc",
|
|
146
|
+
"pyo3-build-config",
|
|
147
|
+
]
|
|
148
|
+
|
|
149
|
+
[[package]]
|
|
150
|
+
name = "pyo3-macros"
|
|
151
|
+
version = "0.29.3"
|
|
152
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
153
|
+
checksum = "6100e8a4b5eba53afaa5ed078364851a0b2499c553a44c31026b929049b49dc6"
|
|
154
|
+
dependencies = [
|
|
155
|
+
"proc-macro2",
|
|
156
|
+
"pyo3-macros-backend",
|
|
157
|
+
"quote",
|
|
158
|
+
"syn",
|
|
159
|
+
]
|
|
160
|
+
|
|
161
|
+
[[package]]
|
|
162
|
+
name = "pyo3-macros-backend"
|
|
163
|
+
version = "0.29.3"
|
|
164
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
165
|
+
checksum = "6143877a16e82b5a727b7127ff4cd86858a24a28d745d72f43e6f227c7b1bdb3"
|
|
166
|
+
dependencies = [
|
|
167
|
+
"heck",
|
|
168
|
+
"proc-macro2",
|
|
169
|
+
"quote",
|
|
170
|
+
"syn",
|
|
171
|
+
]
|
|
172
|
+
|
|
173
|
+
[[package]]
|
|
174
|
+
name = "quote"
|
|
175
|
+
version = "1.0.47"
|
|
176
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
177
|
+
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
|
178
|
+
dependencies = [
|
|
179
|
+
"proc-macro2",
|
|
180
|
+
]
|
|
181
|
+
|
|
182
|
+
[[package]]
|
|
183
|
+
name = "redox_syscall"
|
|
184
|
+
version = "0.5.18"
|
|
185
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
186
|
+
checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d"
|
|
187
|
+
dependencies = [
|
|
188
|
+
"bitflags",
|
|
189
|
+
]
|
|
190
|
+
|
|
191
|
+
[[package]]
|
|
192
|
+
name = "rustversion"
|
|
193
|
+
version = "1.0.23"
|
|
194
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
195
|
+
checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f"
|
|
196
|
+
|
|
197
|
+
[[package]]
|
|
198
|
+
name = "scopeguard"
|
|
199
|
+
version = "1.2.0"
|
|
200
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
201
|
+
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
|
|
202
|
+
|
|
203
|
+
[[package]]
|
|
204
|
+
name = "smallvec"
|
|
205
|
+
version = "1.16.2"
|
|
206
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
207
|
+
checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e"
|
|
208
|
+
|
|
209
|
+
[[package]]
|
|
210
|
+
name = "syn"
|
|
211
|
+
version = "2.0.119"
|
|
212
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
213
|
+
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
|
214
|
+
dependencies = [
|
|
215
|
+
"proc-macro2",
|
|
216
|
+
"quote",
|
|
217
|
+
"unicode-ident",
|
|
218
|
+
]
|
|
219
|
+
|
|
220
|
+
[[package]]
|
|
221
|
+
name = "target-lexicon"
|
|
222
|
+
version = "0.13.5"
|
|
223
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
224
|
+
checksum = "adb6935a6f5c20170eeceb1a3835a49e12e19d792f6dd344ccc76a985ca5a6ca"
|
|
225
|
+
|
|
226
|
+
[[package]]
|
|
227
|
+
name = "unicode-ident"
|
|
228
|
+
version = "1.0.26"
|
|
229
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
230
|
+
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
|
231
|
+
|
|
232
|
+
[[package]]
|
|
233
|
+
name = "windows-link"
|
|
234
|
+
version = "0.2.1"
|
|
235
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
236
|
+
checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5"
|
dbt_cst-0.1.0/Cargo.toml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
[package]
|
|
2
|
+
name = "dbt-cst"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
edition = "2024"
|
|
5
|
+
description = "Lossless concrete syntax tree parser for dbt models (Jinja templated SQL) with Python bindings"
|
|
6
|
+
readme = "README.md"
|
|
7
|
+
|
|
8
|
+
[lib]
|
|
9
|
+
name = "dbt_cst"
|
|
10
|
+
crate-type = ["cdylib", "rlib"]
|
|
11
|
+
|
|
12
|
+
[features]
|
|
13
|
+
default = []
|
|
14
|
+
# Enabled by maturin when building the Python extension module.
|
|
15
|
+
python = ["dep:pyo3"]
|
|
16
|
+
|
|
17
|
+
[dependencies]
|
|
18
|
+
nom = "8.0.0"
|
|
19
|
+
parking_lot = "0.12.5"
|
|
20
|
+
pyo3 = { version = "0.29.3", features = ["abi3-py39", "extension-module", "multiple-pymethods"], optional = true }
|
dbt_cst-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Albert
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
dbt_cst-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dbt-cst
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Classifier: Programming Language :: Rust
|
|
5
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Summary: Lossless dbt model parser: read, modify and rewrite Jinja templated SQL with its formatting intact
|
|
8
|
+
Requires-Python: >=3.9
|
|
9
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
10
|
+
Project-URL: Repository, https://github.com/livinlefevreloca/dbt-cst
|
|
11
|
+
|
|
12
|
+
# dbt-cst
|
|
13
|
+
|
|
14
|
+
A lossless dbt model parser written in Rust with Python bindings. It parses a model (Jinja
|
|
15
|
+
templated SQL) into a concrete syntax tree (CST) that keeps every comment, blank line,
|
|
16
|
+
indentation and Jinja tag, so a file can be read, modified and written back out with its
|
|
17
|
+
formatting intact:
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
from dbt_cst import parse
|
|
21
|
+
|
|
22
|
+
doc = parse(source)
|
|
23
|
+
assert str(doc) == source # byte for byte, for every model in albert-dbt-analytics-transforms
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
The tree groups the SQL into queries, clauses, select columns and CTEs, and the Jinja into
|
|
27
|
+
blocks, without interpreting either: a column is the text between two commas, a WHERE clause
|
|
28
|
+
is the text between its keyword and the next clause keyword. That is enough to add and remove
|
|
29
|
+
columns in a model someone has customized by hand, which rewriting the file from a template
|
|
30
|
+
cannot do. The Python package is an abi3 wheel built against Python 3.9, so one wheel works on
|
|
31
|
+
Python 3.9 and 3.11.
|
|
32
|
+
|
|
33
|
+
## Python usage
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from dbt_cst import parse, DbtSyntaxError
|
|
37
|
+
|
|
38
|
+
doc = parse(path.read_text())
|
|
39
|
+
|
|
40
|
+
doc.config["materialized"] # "view"
|
|
41
|
+
doc.config["materialized"] = "table" # quoted like the call's other strings
|
|
42
|
+
doc.config["tags"] = ["pii"] # added in the call's layout: one per line, trailing comma
|
|
43
|
+
doc.config.set_raw("schema", 'var("schema")')
|
|
44
|
+
del doc.config["alias"]
|
|
45
|
+
|
|
46
|
+
# Example: Modifying multiple config values
|
|
47
|
+
doc.config["materialized"] = "table"
|
|
48
|
+
doc.config["schema"] = "analytics"
|
|
49
|
+
doc.config["tags"] = ["pii", "finance"]
|
|
50
|
+
doc.config["enabled"] = False
|
|
51
|
+
|
|
52
|
+
# Example: Using raw expressions for dynamic values
|
|
53
|
+
doc.config.set_raw("alias", 'var("model_alias")')
|
|
54
|
+
doc.config.set_raw("schema", 'var("schema")')
|
|
55
|
+
|
|
56
|
+
# Example: Removing config values
|
|
57
|
+
if "description" in doc.config:
|
|
58
|
+
del doc.config["description"]
|
|
59
|
+
|
|
60
|
+
doc.sources # [("trading", "apex_apexentry")]
|
|
61
|
+
doc.refs # [("users",), ("package", "model")]
|
|
62
|
+
|
|
63
|
+
query = doc.query # the model's own SELECT, after any CTEs
|
|
64
|
+
select = query.select
|
|
65
|
+
select.names # ["account_id", "activity_id", "percent"]
|
|
66
|
+
"account_id" in select # True, compared ignoring case
|
|
67
|
+
|
|
68
|
+
select.add("user_id") # appended in the select's comma style
|
|
69
|
+
select.add("amount / 100 AS dollars", index=select.index("account_id") + 1)
|
|
70
|
+
select.remove("activity_id") # with its comma and the comments above it
|
|
71
|
+
select.find("percent").text = '"percent" AS pct'
|
|
72
|
+
|
|
73
|
+
renamed = query.cte("renamed").query.select # CTE bodies are queries too
|
|
74
|
+
for query in doc.queries(): # every query, subqueries included
|
|
75
|
+
...
|
|
76
|
+
|
|
77
|
+
path.write_text(str(doc))
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### API
|
|
81
|
+
|
|
82
|
+
`parse(source) -> Document` raises `DbtSyntaxError` (a `ValueError`) with `line`, `column` and
|
|
83
|
+
`offset` attributes. Only a Jinja tag that is never closed is an error. SQL that does not fit
|
|
84
|
+
the expected shape still parses: it ends up as unstructured tokens and still round trips.
|
|
85
|
+
|
|
86
|
+
| Object | Members |
|
|
87
|
+
| ---------- | ----------------------------------------------------------------------------------------------- |
|
|
88
|
+
| `Document` | `config`, `query`, `queries()`, `sources`, `refs`, `str()` |
|
|
89
|
+
| `Config` | `c[key]`, `get(key, default)`, `c[key] = value`, `del c[key]`, `raw(key)`, `set_raw(key, src)`, `keys()`, `in`, `len`, `iter` |
|
|
90
|
+
| `Query` | `select` (the first), `selects` (one per UNION branch), `ctes`, `cte(name)`, `clauses` |
|
|
91
|
+
| `Cte` | `name`, `query` |
|
|
92
|
+
| `Clause` | `name` (`FROM`, `GROUP BY`, `UNION ALL`, ...), `str()` |
|
|
93
|
+
| `Select` | `columns`, `names`, `find(name)`, `index(name)`, `add(text, index=None)`, `remove(name_or_column)`, `in`, `len`, `iter` |
|
|
94
|
+
| `Column` | `text` (settable), `expr`, `alias`, `name`, `is_templated`, `select`, `detach()` |
|
|
95
|
+
|
|
96
|
+
`Config` values are Jinja literals: strings, numbers, booleans, `None`, lists and dicts.
|
|
97
|
+
Reading an argument whose value is an expression such as `var("x")` raises `ValueError`; use
|
|
98
|
+
`raw()` for its source text.
|
|
99
|
+
|
|
100
|
+
`Column.name` is the alias, or the last part of a plain column reference (`t.id` gives `id`),
|
|
101
|
+
unquoted. It is `None` for an unaliased expression or a bare Jinja tag such as
|
|
102
|
+
`{{ dbt_utils.star(...) }}`.
|
|
103
|
+
|
|
104
|
+
Handles are live: a `Column` from `select.columns` stays attached to the document, and two
|
|
105
|
+
handles to the same node compare equal.
|
|
106
|
+
|
|
107
|
+
## How it parses
|
|
108
|
+
|
|
109
|
+
1. **Lexing** (`src/lexer.rs`) splits the file into tokens covering every byte. Jinja tags end
|
|
110
|
+
where Jinja says they do, skipping string literals and nested brackets, so
|
|
111
|
+
`{{ {'a': 1} }}` and `{{ "}}" }}` are one tag each. A tag inside a SQL string or comment is
|
|
112
|
+
absorbed into that token. `{% raw %}` sections are one token.
|
|
113
|
+
2. **Grouping** (`src/parser.rs`) pairs `(`/`)`, `[`/`]` and Jinja block tags with one stack.
|
|
114
|
+
Any `{% name %}` whose `{% endname %}` appears in the file is opened tentatively, so tags
|
|
115
|
+
from Jinja, dbt (`materialization`, `test`, `snapshot`) and packages all work without a
|
|
116
|
+
list of names. A tag that never closes, like a one line `{% set %}`, or one whose end would
|
|
117
|
+
cross a parenthesis, is put back as a plain token. `elif` and `else` split blocks into
|
|
118
|
+
branches.
|
|
119
|
+
3. **Structure**: a top level `SELECT` or `WITH` starts a query, which is split into clauses at
|
|
120
|
+
top level clause keywords. SELECT items are split at top level commas, and WITH clauses into
|
|
121
|
+
CTEs. "Top level" means outside any bracket or Jinja block, so commas in function calls and
|
|
122
|
+
keywords in subqueries never split anything. A Jinja block that supplies its own commas,
|
|
123
|
+
like `{% for c in cols %}{{ c }},{% endfor %}`, separates the columns around it.
|
|
124
|
+
|
|
125
|
+
The lexer is written with nom, like looker-cst's parser. The one hand written scan is the end
|
|
126
|
+
of a `{{ }}` or `{% %}` tag, which tracks bracket depth and skips string literals the way
|
|
127
|
+
Jinja's own lexer does.
|
|
128
|
+
|
|
129
|
+
## Rust usage
|
|
130
|
+
|
|
131
|
+
```rust
|
|
132
|
+
let doc = dbt_cst::parse(&source)?;
|
|
133
|
+
let query = doc.query().unwrap();
|
|
134
|
+
let select = dbt_cst::query::clauses_named(&query.read(), "SELECT").remove(0);
|
|
135
|
+
dbt_cst::edit::insert_item(&mut select.write(), usize::MAX, "user_id")?;
|
|
136
|
+
assert!(doc.to_string().contains("user_id"));
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
`cargo run --example dump -- path/to/model.sql` prints the tree.
|
|
140
|
+
|
|
141
|
+
## Development
|
|
142
|
+
|
|
143
|
+
```sh
|
|
144
|
+
cargo test # Rust tests, cloning the corpus into target/corpus on first run
|
|
145
|
+
uv run pytest # builds the extension with maturin, then runs the Python tests
|
|
146
|
+
DBT_EXTRA_REPOS=../albert-dbt-analytics-transforms cargo test --release --test corpus
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
The corpus is the 14 public dbt projects listed in `tests/corpus_repos.txt` (dbt-utils,
|
|
150
|
+
dbt-expectations, automate-dv, the Fivetran packages and others, about 1,700 `.sql` files).
|
|
151
|
+
Both test suites shallow clone them into `target/corpus` on first use, or into `DBT_CORPUS`
|
|
152
|
+
when set. A failed clone skips those files unless `REQUIRE_DBT_CORPUS` is set, as in CI.
|
|
153
|
+
`DBT_EXTRA_REPOS` adds more checkouts, separated by `:`.
|
|
154
|
+
|
|
155
|
+
- `tests/corpus.rs` round trips every file, adds and removes a column in every SELECT,
|
|
156
|
+
removes each column in turn, and adds and removes a config argument.
|
|
157
|
+
- `tests/python/test_corpus.py` does the same read, modify and write cycle through the Python
|
|
158
|
+
API, one test per file.
|
|
159
|
+
- `tests/fixtures/` holds awkward inputs (CRLF, unicode, unterminated quotes, unbalanced
|
|
160
|
+
brackets, unpaired Jinja blocks) that both suites round trip, and `edge_cases.sql`, a
|
|
161
|
+
customized model whose structure both suites check.
|
|
162
|
+
- `tests/python/test_albert_models.py` checks every model in a sibling
|
|
163
|
+
`albert-dbt-analytics-transforms` checkout (or `ALBERT_DBT_REPO`) against the field list
|
|
164
|
+
DBTGenerator reads today.
|
|
165
|
+
|
dbt_cst-0.1.0/README.md
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# dbt-cst
|
|
2
|
+
|
|
3
|
+
A lossless dbt model parser written in Rust with Python bindings. It parses a model (Jinja
|
|
4
|
+
templated SQL) into a concrete syntax tree (CST) that keeps every comment, blank line,
|
|
5
|
+
indentation and Jinja tag, so a file can be read, modified and written back out with its
|
|
6
|
+
formatting intact:
|
|
7
|
+
|
|
8
|
+
```python
|
|
9
|
+
from dbt_cst import parse
|
|
10
|
+
|
|
11
|
+
doc = parse(source)
|
|
12
|
+
assert str(doc) == source # byte for byte, for every model in albert-dbt-analytics-transforms
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
The tree groups the SQL into queries, clauses, select columns and CTEs, and the Jinja into
|
|
16
|
+
blocks, without interpreting either: a column is the text between two commas, a WHERE clause
|
|
17
|
+
is the text between its keyword and the next clause keyword. That is enough to add and remove
|
|
18
|
+
columns in a model someone has customized by hand, which rewriting the file from a template
|
|
19
|
+
cannot do. The Python package is an abi3 wheel built against Python 3.9, so one wheel works on
|
|
20
|
+
Python 3.9 and 3.11.
|
|
21
|
+
|
|
22
|
+
## Python usage
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
from dbt_cst import parse, DbtSyntaxError
|
|
26
|
+
|
|
27
|
+
doc = parse(path.read_text())
|
|
28
|
+
|
|
29
|
+
doc.config["materialized"] # "view"
|
|
30
|
+
doc.config["materialized"] = "table" # quoted like the call's other strings
|
|
31
|
+
doc.config["tags"] = ["pii"] # added in the call's layout: one per line, trailing comma
|
|
32
|
+
doc.config.set_raw("schema", 'var("schema")')
|
|
33
|
+
del doc.config["alias"]
|
|
34
|
+
|
|
35
|
+
# Example: Modifying multiple config values
|
|
36
|
+
doc.config["materialized"] = "table"
|
|
37
|
+
doc.config["schema"] = "analytics"
|
|
38
|
+
doc.config["tags"] = ["pii", "finance"]
|
|
39
|
+
doc.config["enabled"] = False
|
|
40
|
+
|
|
41
|
+
# Example: Using raw expressions for dynamic values
|
|
42
|
+
doc.config.set_raw("alias", 'var("model_alias")')
|
|
43
|
+
doc.config.set_raw("schema", 'var("schema")')
|
|
44
|
+
|
|
45
|
+
# Example: Removing config values
|
|
46
|
+
if "description" in doc.config:
|
|
47
|
+
del doc.config["description"]
|
|
48
|
+
|
|
49
|
+
doc.sources # [("trading", "apex_apexentry")]
|
|
50
|
+
doc.refs # [("users",), ("package", "model")]
|
|
51
|
+
|
|
52
|
+
query = doc.query # the model's own SELECT, after any CTEs
|
|
53
|
+
select = query.select
|
|
54
|
+
select.names # ["account_id", "activity_id", "percent"]
|
|
55
|
+
"account_id" in select # True, compared ignoring case
|
|
56
|
+
|
|
57
|
+
select.add("user_id") # appended in the select's comma style
|
|
58
|
+
select.add("amount / 100 AS dollars", index=select.index("account_id") + 1)
|
|
59
|
+
select.remove("activity_id") # with its comma and the comments above it
|
|
60
|
+
select.find("percent").text = '"percent" AS pct'
|
|
61
|
+
|
|
62
|
+
renamed = query.cte("renamed").query.select # CTE bodies are queries too
|
|
63
|
+
for query in doc.queries(): # every query, subqueries included
|
|
64
|
+
...
|
|
65
|
+
|
|
66
|
+
path.write_text(str(doc))
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### API
|
|
70
|
+
|
|
71
|
+
`parse(source) -> Document` raises `DbtSyntaxError` (a `ValueError`) with `line`, `column` and
|
|
72
|
+
`offset` attributes. Only a Jinja tag that is never closed is an error. SQL that does not fit
|
|
73
|
+
the expected shape still parses: it ends up as unstructured tokens and still round trips.
|
|
74
|
+
|
|
75
|
+
| Object | Members |
|
|
76
|
+
| ---------- | ----------------------------------------------------------------------------------------------- |
|
|
77
|
+
| `Document` | `config`, `query`, `queries()`, `sources`, `refs`, `str()` |
|
|
78
|
+
| `Config` | `c[key]`, `get(key, default)`, `c[key] = value`, `del c[key]`, `raw(key)`, `set_raw(key, src)`, `keys()`, `in`, `len`, `iter` |
|
|
79
|
+
| `Query` | `select` (the first), `selects` (one per UNION branch), `ctes`, `cte(name)`, `clauses` |
|
|
80
|
+
| `Cte` | `name`, `query` |
|
|
81
|
+
| `Clause` | `name` (`FROM`, `GROUP BY`, `UNION ALL`, ...), `str()` |
|
|
82
|
+
| `Select` | `columns`, `names`, `find(name)`, `index(name)`, `add(text, index=None)`, `remove(name_or_column)`, `in`, `len`, `iter` |
|
|
83
|
+
| `Column` | `text` (settable), `expr`, `alias`, `name`, `is_templated`, `select`, `detach()` |
|
|
84
|
+
|
|
85
|
+
`Config` values are Jinja literals: strings, numbers, booleans, `None`, lists and dicts.
|
|
86
|
+
Reading an argument whose value is an expression such as `var("x")` raises `ValueError`; use
|
|
87
|
+
`raw()` for its source text.
|
|
88
|
+
|
|
89
|
+
`Column.name` is the alias, or the last part of a plain column reference (`t.id` gives `id`),
|
|
90
|
+
unquoted. It is `None` for an unaliased expression or a bare Jinja tag such as
|
|
91
|
+
`{{ dbt_utils.star(...) }}`.
|
|
92
|
+
|
|
93
|
+
Handles are live: a `Column` from `select.columns` stays attached to the document, and two
|
|
94
|
+
handles to the same node compare equal.
|
|
95
|
+
|
|
96
|
+
## How it parses
|
|
97
|
+
|
|
98
|
+
1. **Lexing** (`src/lexer.rs`) splits the file into tokens covering every byte. Jinja tags end
|
|
99
|
+
where Jinja says they do, skipping string literals and nested brackets, so
|
|
100
|
+
`{{ {'a': 1} }}` and `{{ "}}" }}` are one tag each. A tag inside a SQL string or comment is
|
|
101
|
+
absorbed into that token. `{% raw %}` sections are one token.
|
|
102
|
+
2. **Grouping** (`src/parser.rs`) pairs `(`/`)`, `[`/`]` and Jinja block tags with one stack.
|
|
103
|
+
Any `{% name %}` whose `{% endname %}` appears in the file is opened tentatively, so tags
|
|
104
|
+
from Jinja, dbt (`materialization`, `test`, `snapshot`) and packages all work without a
|
|
105
|
+
list of names. A tag that never closes, like a one line `{% set %}`, or one whose end would
|
|
106
|
+
cross a parenthesis, is put back as a plain token. `elif` and `else` split blocks into
|
|
107
|
+
branches.
|
|
108
|
+
3. **Structure**: a top level `SELECT` or `WITH` starts a query, which is split into clauses at
|
|
109
|
+
top level clause keywords. SELECT items are split at top level commas, and WITH clauses into
|
|
110
|
+
CTEs. "Top level" means outside any bracket or Jinja block, so commas in function calls and
|
|
111
|
+
keywords in subqueries never split anything. A Jinja block that supplies its own commas,
|
|
112
|
+
like `{% for c in cols %}{{ c }},{% endfor %}`, separates the columns around it.
|
|
113
|
+
|
|
114
|
+
The lexer is written with nom, like looker-cst's parser. The one hand written scan is the end
|
|
115
|
+
of a `{{ }}` or `{% %}` tag, which tracks bracket depth and skips string literals the way
|
|
116
|
+
Jinja's own lexer does.
|
|
117
|
+
|
|
118
|
+
## Rust usage
|
|
119
|
+
|
|
120
|
+
```rust
|
|
121
|
+
let doc = dbt_cst::parse(&source)?;
|
|
122
|
+
let query = doc.query().unwrap();
|
|
123
|
+
let select = dbt_cst::query::clauses_named(&query.read(), "SELECT").remove(0);
|
|
124
|
+
dbt_cst::edit::insert_item(&mut select.write(), usize::MAX, "user_id")?;
|
|
125
|
+
assert!(doc.to_string().contains("user_id"));
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
`cargo run --example dump -- path/to/model.sql` prints the tree.
|
|
129
|
+
|
|
130
|
+
## Development
|
|
131
|
+
|
|
132
|
+
```sh
|
|
133
|
+
cargo test # Rust tests, cloning the corpus into target/corpus on first run
|
|
134
|
+
uv run pytest # builds the extension with maturin, then runs the Python tests
|
|
135
|
+
DBT_EXTRA_REPOS=../albert-dbt-analytics-transforms cargo test --release --test corpus
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
The corpus is the 14 public dbt projects listed in `tests/corpus_repos.txt` (dbt-utils,
|
|
139
|
+
dbt-expectations, automate-dv, the Fivetran packages and others, about 1,700 `.sql` files).
|
|
140
|
+
Both test suites shallow clone them into `target/corpus` on first use, or into `DBT_CORPUS`
|
|
141
|
+
when set. A failed clone skips those files unless `REQUIRE_DBT_CORPUS` is set, as in CI.
|
|
142
|
+
`DBT_EXTRA_REPOS` adds more checkouts, separated by `:`.
|
|
143
|
+
|
|
144
|
+
- `tests/corpus.rs` round trips every file, adds and removes a column in every SELECT,
|
|
145
|
+
removes each column in turn, and adds and removes a config argument.
|
|
146
|
+
- `tests/python/test_corpus.py` does the same read, modify and write cycle through the Python
|
|
147
|
+
API, one test per file.
|
|
148
|
+
- `tests/fixtures/` holds awkward inputs (CRLF, unicode, unterminated quotes, unbalanced
|
|
149
|
+
brackets, unpaired Jinja blocks) that both suites round trip, and `edge_cases.sql`, a
|
|
150
|
+
customized model whose structure both suites check.
|
|
151
|
+
- `tests/python/test_albert_models.py` checks every model in a sibling
|
|
152
|
+
`albert-dbt-analytics-transforms` checkout (or `ALBERT_DBT_REPO`) against the field list
|
|
153
|
+
DBTGenerator reads today.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
//! Prints the tree of a dbt file, one node or token per line.
|
|
2
|
+
//!
|
|
3
|
+
//! cargo run --example dump -- path/to/model.sql
|
|
4
|
+
|
|
5
|
+
use dbt_cst::{Child, Node, parse};
|
|
6
|
+
|
|
7
|
+
fn main() {
|
|
8
|
+
let path = std::env::args().nth(1).expect("usage: dump <file.sql>");
|
|
9
|
+
let source = std::fs::read_to_string(&path).expect("readable file");
|
|
10
|
+
let doc = parse(&source).unwrap_or_else(|e| panic!("{path}: {e}"));
|
|
11
|
+
print_node(&doc.root.read(), 0);
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
fn print_node(node: &Node, depth: usize) {
|
|
15
|
+
println!("{}{:?}", " ".repeat(depth), node.kind);
|
|
16
|
+
for child in &node.children {
|
|
17
|
+
match child {
|
|
18
|
+
Child::Token(token) => println!("{}{:?} {:?}", " ".repeat(depth + 1), token.kind, token.text),
|
|
19
|
+
Child::Node(node) => print_node(&node.read(), depth + 1),
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["maturin>=1.9,<2.0"]
|
|
3
|
+
build-backend = "maturin"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dbt-cst"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Lossless dbt model parser: read, modify and rewrite Jinja templated SQL with its formatting intact"
|
|
9
|
+
requires-python = ">=3.9"
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
license = { file = "LICENSE" }
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Programming Language :: Rust",
|
|
14
|
+
"Programming Language :: Python :: Implementation :: CPython",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.urls]
|
|
18
|
+
Repository = "https://github.com/livinlefevreloca/dbt-cst"
|
|
19
|
+
|
|
20
|
+
[dependency-groups]
|
|
21
|
+
dev = ["pytest>=8"]
|
|
22
|
+
|
|
23
|
+
[tool.maturin]
|
|
24
|
+
python-source = "python"
|
|
25
|
+
module-name = "dbt_cst._native"
|
|
26
|
+
features = ["python"]
|
|
27
|
+
|
|
28
|
+
[tool.pytest.ini_options]
|
|
29
|
+
testpaths = ["tests/python"]
|
|
30
|
+
|
|
31
|
+
[tool.uv]
|
|
32
|
+
# Rebuild the extension when the Rust sources change.
|
|
33
|
+
cache-keys = [{ file = "pyproject.toml" }, { file = "Cargo.toml" }, { file = "src/**/*.rs" }]
|