ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/engine/jinja.py
ADDED
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
"""Convert Jinja-templated dbt SQL into parseable SQL.
|
|
2
|
+
|
|
3
|
+
This module handles the transformation of dbt's Jinja-templated SQL into
|
|
4
|
+
standard SQL that can be parsed by SQLGlot for column lineage extraction.
|
|
5
|
+
|
|
6
|
+
Key transformations:
|
|
7
|
+
- {{ ref('model') }} → model
|
|
8
|
+
- {{ source('schema', 'table') }} → schema__table
|
|
9
|
+
- {{ config(...) }} → removed
|
|
10
|
+
- {% if/for/... %} → removed
|
|
11
|
+
- Custom macros → NULL placeholder
|
|
12
|
+
|
|
13
|
+
Security:
|
|
14
|
+
- ReDoS protection via size limits on regex processing
|
|
15
|
+
- Bounded regex patterns to prevent catastrophic backtracking
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import logging
|
|
19
|
+
import re
|
|
20
|
+
from typing import Final
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
# Maximum SQL size for regex processing (500KB) - prevents ReDoS on huge files
|
|
25
|
+
MAX_SQL_SIZE_FOR_REGEX: Final[int] = 500 * 1024
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def convert_jinja_to_sql(raw_sql: str) -> str:
|
|
29
|
+
"""Convert Jinja-templated SQL to parseable SQL.
|
|
30
|
+
|
|
31
|
+
This is the full conversion that handles:
|
|
32
|
+
|
|
33
|
+
Core dbt functions:
|
|
34
|
+
- {{ ref('model') }} → model
|
|
35
|
+
- {{ source('src', 'table') }} → src__table
|
|
36
|
+
- {{ config(...) }} → removed
|
|
37
|
+
- {{ var('x') }} → 'x' (variable reference)
|
|
38
|
+
- {{ this }} → this_model
|
|
39
|
+
|
|
40
|
+
dbt-utils macros:
|
|
41
|
+
- {{ group_by(n) }} → GROUP BY n
|
|
42
|
+
- {{ dbt_utils.star(...) }} → *
|
|
43
|
+
- {{ dbt_utils.surrogate_key(...) }} → md5(...)
|
|
44
|
+
|
|
45
|
+
SQL normalization:
|
|
46
|
+
- Escaped quotes: ''value'' → 'value'
|
|
47
|
+
- Dollar-quoted strings: $$text$$ → 'text'
|
|
48
|
+
- Prepared statement placeholders: $1, ?, :name → 'param'
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
raw_sql: SQL with Jinja templates
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
Clean SQL parseable by SQLGlot
|
|
55
|
+
|
|
56
|
+
Security:
|
|
57
|
+
Returns raw SQL unchanged if size exceeds MAX_SQL_SIZE_FOR_REGEX
|
|
58
|
+
to prevent ReDoS attacks.
|
|
59
|
+
"""
|
|
60
|
+
if len(raw_sql) > MAX_SQL_SIZE_FOR_REGEX:
|
|
61
|
+
logger.warning(
|
|
62
|
+
f"SQL too large for Jinja cleaning ({len(raw_sql)} bytes > {MAX_SQL_SIZE_FOR_REGEX}). "
|
|
63
|
+
"Returning raw SQL to avoid ReDoS risk."
|
|
64
|
+
)
|
|
65
|
+
return raw_sql
|
|
66
|
+
|
|
67
|
+
cleaned = raw_sql
|
|
68
|
+
|
|
69
|
+
# ==========================================================================
|
|
70
|
+
# SQL NORMALIZATION FOR CROSS-DIALECT PARSING
|
|
71
|
+
# These patterns handle syntax variations across Snowflake, BigQuery,
|
|
72
|
+
# Databricks, Trino, Redshift, Postgres that sqlglot may struggle with.
|
|
73
|
+
# ==========================================================================
|
|
74
|
+
|
|
75
|
+
# 1. Escaped single quotes: ''value'' → 'value'
|
|
76
|
+
# Common in Snowflake/Trino compiled SQL for Jinja variables
|
|
77
|
+
cleaned = re.sub(r"''([^']+)''", r"'\1'", cleaned)
|
|
78
|
+
|
|
79
|
+
# 2. Dollar-quoted strings (PostgreSQL/Redshift): $$text$$ → 'text'
|
|
80
|
+
cleaned = re.sub(r"\$\$([^$]*)\$\$", r"'\1'", cleaned)
|
|
81
|
+
cleaned = re.sub(r"\$(\w+)\$([^$]*)\$\1\$", r"'\2'", cleaned) # Tagged: $tag$text$tag$
|
|
82
|
+
|
|
83
|
+
# 3. Unicode escape sequences: U&'text' → 'text'
|
|
84
|
+
cleaned = re.sub(r"U&'([^']*)'", r"'\1'", cleaned, flags=re.IGNORECASE)
|
|
85
|
+
|
|
86
|
+
# 4. National character strings: N'text' → 'text'
|
|
87
|
+
cleaned = re.sub(r"(?<!\w)N'([^']*)'", r"'\1'", cleaned)
|
|
88
|
+
|
|
89
|
+
# 5. Raw string literals (BigQuery): r'pattern' → 'pattern'
|
|
90
|
+
cleaned = re.sub(r"(?<!\w)[rR]'([^']*)'", r"'\1'", cleaned)
|
|
91
|
+
cleaned = re.sub(r'(?<!\w)[rR]"([^"]*)"', r"'\1'", cleaned)
|
|
92
|
+
|
|
93
|
+
# 6. Byte literals (BigQuery): b'bytes' → 'bytes'
|
|
94
|
+
cleaned = re.sub(r"(?<!\w)[bB]'([^']*)'", r"'\1'", cleaned)
|
|
95
|
+
cleaned = re.sub(r'(?<!\w)[bB]"([^"]*)"', r"'\1'", cleaned)
|
|
96
|
+
|
|
97
|
+
# 7. Prepared statement placeholders → literal placeholder
|
|
98
|
+
cleaned = re.sub(r"\$(\d+)", r"'param\1'", cleaned) # $1, $2 (Postgres)
|
|
99
|
+
cleaned = re.sub(r"(?<!\?)\?(?!\?)", "'param'", cleaned) # ? (JDBC)
|
|
100
|
+
cleaned = re.sub(
|
|
101
|
+
r"(?<!:):(\w+)(?=\s|,|\)|$|AND|OR|=|>|<|\+|-|\*|/)",
|
|
102
|
+
r"'\1'",
|
|
103
|
+
cleaned,
|
|
104
|
+
flags=re.IGNORECASE,
|
|
105
|
+
) # :name (Oracle)
|
|
106
|
+
|
|
107
|
+
# 8. Interval literals with quoted units
|
|
108
|
+
cleaned = re.sub(r"INTERVAL\s+'(\d+)'\s+(\w+)", r"INTERVAL \1 \2", cleaned, flags=re.IGNORECASE)
|
|
109
|
+
|
|
110
|
+
# ==========================================================================
|
|
111
|
+
# CORE DBT FUNCTIONS
|
|
112
|
+
# ==========================================================================
|
|
113
|
+
|
|
114
|
+
# ref('model') → model
|
|
115
|
+
cleaned = re.sub(r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"\1", cleaned)
|
|
116
|
+
|
|
117
|
+
# source('schema', 'table') → schema__table
|
|
118
|
+
cleaned = re.sub(
|
|
119
|
+
r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
|
|
120
|
+
r"\1__\2",
|
|
121
|
+
cleaned,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# config() blocks (can be multiline, nested)
|
|
125
|
+
cleaned = re.sub(r"\{\{\s*config\s*\(.*?\)\s*\}\}", "", cleaned, flags=re.DOTALL)
|
|
126
|
+
cleaned = re.sub(r"\{\{\s*config\([^}]*\}\}\s*\}\}", "", cleaned)
|
|
127
|
+
|
|
128
|
+
# ==========================================================================
|
|
129
|
+
# DBT-UTILS MACROS
|
|
130
|
+
# ==========================================================================
|
|
131
|
+
|
|
132
|
+
cleaned = re.sub(r"\{\{\s*group_by\s*\(\s*(\d+)\s*\)\s*\}\}", r"GROUP BY \1", cleaned)
|
|
133
|
+
cleaned = re.sub(r"\{\{\s*order_by\s*\(\s*(\d+)\s*\)\s*\}\}", r"ORDER BY \1", cleaned)
|
|
134
|
+
cleaned = re.sub(r"\{\{\s*limit\s*\(\s*(\d+)\s*\)\s*\}\}", r"LIMIT \1", cleaned)
|
|
135
|
+
|
|
136
|
+
# dbt_utils.star() → *
|
|
137
|
+
cleaned = re.sub(r"\{\{\s*dbt_utils\.star\s*\([^}]*\)\s*\}\}", "*", cleaned)
|
|
138
|
+
|
|
139
|
+
# dbt_utils.surrogate_key() → md5('key')
|
|
140
|
+
cleaned = re.sub(r"\{\{\s*dbt_utils\.surrogate_key\s*\([^)]*\)\s*\}\}", "md5('key')", cleaned)
|
|
141
|
+
cleaned = re.sub(
|
|
142
|
+
r"\{\{\s*dbt_utils\.generate_surrogate_key\s*\([^)]*\)\s*\}\}",
|
|
143
|
+
"md5('key')",
|
|
144
|
+
cleaned,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
# dbt_utils.pivot/unpivot → comment placeholder
|
|
148
|
+
cleaned = re.sub(r"\{\{\s*dbt_utils\.pivot\s*\([^)]*\)\s*\}\}", "/* pivot */", cleaned)
|
|
149
|
+
cleaned = re.sub(r"\{\{\s*dbt_utils\.unpivot\s*\([^)]*\)\s*\}\}", "/* unpivot */", cleaned)
|
|
150
|
+
|
|
151
|
+
# dbt_utils.generate_series() → subquery
|
|
152
|
+
cleaned = re.sub(
|
|
153
|
+
r"\{\{\s*dbt_utils\.generate_series\s*\([^)]*\)\s*\}\}",
|
|
154
|
+
"(SELECT 1 AS n)",
|
|
155
|
+
cleaned,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
# dbt_utils.get_column_values() → tuple
|
|
159
|
+
cleaned = re.sub(
|
|
160
|
+
r"\{\{\s*dbt_utils\.get_column_values\s*\([^)]*\)\s*\}\}",
|
|
161
|
+
"('value1', 'value2')",
|
|
162
|
+
cleaned,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
# dbt_utils.safe_cast() → TRY_CAST
|
|
166
|
+
cleaned = re.sub(
|
|
167
|
+
r"\{\{\s*dbt_utils\.safe_cast\s*\(\s*['\"]?([^'\"]+)['\"]?\s*,\s*['\"]?([^'\"]+)['\"]?\s*\)\s*\}\}",
|
|
168
|
+
r"TRY_CAST(\1 AS \2)",
|
|
169
|
+
cleaned,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
# dbt_utils.date_trunc() → DATE_TRUNC
|
|
173
|
+
cleaned = re.sub(
|
|
174
|
+
r"\{\{\s*dbt_utils\.date_trunc\s*\(\s*['\"](\w+)['\"]\s*,\s*([^)]+)\s*\)\s*\}\}",
|
|
175
|
+
r"DATE_TRUNC('\1', \2)",
|
|
176
|
+
cleaned,
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
# Generic dbt_utils.* → NULL
|
|
180
|
+
cleaned = re.sub(r"\{\{\s*dbt_utils\.\w+\s*\([^)]*\)\s*\}\}", "NULL", cleaned)
|
|
181
|
+
|
|
182
|
+
# ==========================================================================
|
|
183
|
+
# OTHER DBT MACROS
|
|
184
|
+
# ==========================================================================
|
|
185
|
+
|
|
186
|
+
# dbt_date macros
|
|
187
|
+
cleaned = re.sub(
|
|
188
|
+
r"\{\{\s*dbt_date\.date_spine\s*\([^)]*\)\s*\}\}",
|
|
189
|
+
"(SELECT CURRENT_DATE AS date_day)",
|
|
190
|
+
cleaned,
|
|
191
|
+
)
|
|
192
|
+
cleaned = re.sub(r"\{\{\s*dbt_date\.\w+\s*\([^)]*\)\s*\}\}", "CURRENT_DATE", cleaned)
|
|
193
|
+
|
|
194
|
+
# fivetran_utils macros
|
|
195
|
+
cleaned = re.sub(
|
|
196
|
+
r"\{\{\s*fivetran_utils\.\w+\s*\([^)]*\)\s*\}\}", "/* fivetran_macro */", cleaned
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
# dbt_expectations macros
|
|
200
|
+
cleaned = re.sub(r"\{\{\s*dbt_expectations\.\w+\s*\([^)]*\)\s*\}\}", "TRUE", cleaned)
|
|
201
|
+
|
|
202
|
+
# ==========================================================================
|
|
203
|
+
# VARIABLE REFERENCES
|
|
204
|
+
# ==========================================================================
|
|
205
|
+
|
|
206
|
+
cleaned = re.sub(r"\{\{\s*var\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"'\1'", cleaned)
|
|
207
|
+
cleaned = re.sub(r"\{\{\s*env_var\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"'\1'", cleaned)
|
|
208
|
+
cleaned = re.sub(r"\{\{\s*this\s*\}\}", "this_model", cleaned)
|
|
209
|
+
cleaned = re.sub(r"\{\{\s*target\.\w+\s*\}\}", "'target_value'", cleaned)
|
|
210
|
+
cleaned = re.sub(r"\{\{\s*run_started_at\s*\}\}", "CURRENT_TIMESTAMP", cleaned)
|
|
211
|
+
cleaned = re.sub(r"\{\{\s*invocation_id\s*\}\}", "'invocation_id'", cleaned)
|
|
212
|
+
|
|
213
|
+
# ==========================================================================
|
|
214
|
+
# CONTROL FLOW AND CLEANUP
|
|
215
|
+
# ==========================================================================
|
|
216
|
+
|
|
217
|
+
# {% ... %} control blocks carry no lineage
|
|
218
|
+
cleaned = re.sub(r"\{%[\s\S]*?%\}", "", cleaned)
|
|
219
|
+
|
|
220
|
+
# Any remaining {{ ... }} → NULL
|
|
221
|
+
cleaned = re.sub(r"\{\{[^}]*\}\}", "NULL", cleaned)
|
|
222
|
+
|
|
223
|
+
# Clean up leftover Jinja fragments
|
|
224
|
+
cleaned = re.sub(r'["\')]*\s*\}\}', "", cleaned)
|
|
225
|
+
cleaned = re.sub(r'\{\{\s*["\(\']*', "", cleaned)
|
|
226
|
+
cleaned = re.sub(r'^\s*["\']+["\')]*\s*$', "", cleaned, flags=re.MULTILINE)
|
|
227
|
+
|
|
228
|
+
# a stripped config block leaves its indentation behind
|
|
229
|
+
cleaned = cleaned.lstrip()
|
|
230
|
+
|
|
231
|
+
return cleaned
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def convert_jinja_to_sql_minimal(raw_sql: str) -> str:
|
|
235
|
+
"""Minimal Jinja conversion - only guaranteed-safe transformations.
|
|
236
|
+
|
|
237
|
+
Use this as a fallback when full conversion breaks valid SQL.
|
|
238
|
+
Only handles patterns that cannot produce invalid SQL.
|
|
239
|
+
|
|
240
|
+
Safe transformations:
|
|
241
|
+
- {{ ref('model') }} → model
|
|
242
|
+
- {{ source('src', 'table') }} → src__table
|
|
243
|
+
- {{ config(...) }} → removed
|
|
244
|
+
- {% ... %} → removed
|
|
245
|
+
- Other {{ ... }} → NULL
|
|
246
|
+
|
|
247
|
+
Does NOT apply:
|
|
248
|
+
- SQL string literal normalization
|
|
249
|
+
- Prepared statement placeholder handling
|
|
250
|
+
- Dialect-specific pattern fixes
|
|
251
|
+
|
|
252
|
+
Args:
|
|
253
|
+
raw_sql: SQL with Jinja templates
|
|
254
|
+
|
|
255
|
+
Returns:
|
|
256
|
+
Minimally cleaned SQL
|
|
257
|
+
"""
|
|
258
|
+
if len(raw_sql) > MAX_SQL_SIZE_FOR_REGEX:
|
|
259
|
+
return raw_sql
|
|
260
|
+
|
|
261
|
+
cleaned = raw_sql
|
|
262
|
+
|
|
263
|
+
# Only the safest Jinja transformations
|
|
264
|
+
cleaned = re.sub(r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"\1", cleaned)
|
|
265
|
+
cleaned = re.sub(
|
|
266
|
+
r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
|
|
267
|
+
r"\1__\2",
|
|
268
|
+
cleaned,
|
|
269
|
+
)
|
|
270
|
+
cleaned = re.sub(r"\{\{\s*config\s*\(.*?\)\s*\}\}", "", cleaned, flags=re.DOTALL)
|
|
271
|
+
|
|
272
|
+
# Control flow blocks
|
|
273
|
+
cleaned = re.sub(r"\{%[\s\S]*?%\}", "", cleaned)
|
|
274
|
+
|
|
275
|
+
# Remaining Jinja macros → NULL
|
|
276
|
+
cleaned = re.sub(r"\{\{[^}]*\}\}", "NULL", cleaned)
|
|
277
|
+
|
|
278
|
+
# Minimal cleanup
|
|
279
|
+
cleaned = re.sub(r'["\')]*\s*\}\}', "", cleaned)
|
|
280
|
+
cleaned = re.sub(r'\{\{\s*["\(\']*', "", cleaned)
|
|
281
|
+
|
|
282
|
+
return cleaned
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
"""JSON-access source extraction for select items.
|
|
2
|
+
|
|
3
|
+
Handles column references reached through JSON path expressions
|
|
4
|
+
(JSON_EXTRACT, ->, data:field) and bracket syntax (data['key']), preserving
|
|
5
|
+
the JSON path on the emitted source entry.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from sqlglot import exp
|
|
11
|
+
|
|
12
|
+
from ripple.engine.column_ref import (
|
|
13
|
+
resolve_column_ref,
|
|
14
|
+
under_subquery_where,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
# built once: per-column construction showed up in profiles
|
|
18
|
+
# Covers all dialects:
|
|
19
|
+
# - Snowflake: data:field::string (parses to JSONExtract)
|
|
20
|
+
# - BigQuery: JSON_EXTRACT, JSON_EXTRACT_SCALAR, JSON_VALUE
|
|
21
|
+
# - Trino: json_extract, json_extract_scalar
|
|
22
|
+
# - Redshift: json_extract_path_text
|
|
23
|
+
# - Postgres: ->, ->> (JSONExtract), #>, #>> (JSONBExtract)
|
|
24
|
+
# - Databricks: get_json_object
|
|
25
|
+
JSON_EXPR_TYPES: tuple[type, ...] = (
|
|
26
|
+
exp.JSONExtract, # Most dialects: JSON_EXTRACT, ->, data:field
|
|
27
|
+
exp.JSONExtractScalar, # JSON_EXTRACT_SCALAR, ->>, JSON_VALUE
|
|
28
|
+
)
|
|
29
|
+
# the JSONB variants exist only in newer sqlglot
|
|
30
|
+
if hasattr(exp, "JSONBExtract"):
|
|
31
|
+
JSON_EXPR_TYPES = JSON_EXPR_TYPES + (exp.JSONBExtract,)
|
|
32
|
+
if hasattr(exp, "JSONBExtractScalar"):
|
|
33
|
+
JSON_EXPR_TYPES = JSON_EXPR_TYPES + (exp.JSONBExtractScalar,)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _extract_json_path(json_path_expr: "exp.Expression | None") -> str:
|
|
37
|
+
"""Extract JSON path string from various JSON path expression types.
|
|
38
|
+
|
|
39
|
+
Handles:
|
|
40
|
+
- JSONPath expressions: $.field.subfield -> "field.subfield"
|
|
41
|
+
- String literals: '$.path' -> "path"
|
|
42
|
+
- Trino array arguments: 'key1', 'key2' -> "key1.key2"
|
|
43
|
+
- Redshift json_extract_path_text: col, 'key1', 'key2' -> "key1.key2"
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
Extracted path string (without leading $.) or empty string if not extractable.
|
|
47
|
+
"""
|
|
48
|
+
if json_path_expr is None:
|
|
49
|
+
return ""
|
|
50
|
+
|
|
51
|
+
if hasattr(json_path_expr, "expressions"):
|
|
52
|
+
path_parts = []
|
|
53
|
+
for path_part in json_path_expr.expressions:
|
|
54
|
+
# Skip JSONPathRoot ($)
|
|
55
|
+
if type(path_part).__name__ == "JSONPathRoot":
|
|
56
|
+
continue
|
|
57
|
+
if hasattr(path_part, "this"):
|
|
58
|
+
part_name = str(path_part.this)
|
|
59
|
+
path_parts.append(part_name)
|
|
60
|
+
elif hasattr(path_part, "name"):
|
|
61
|
+
path_parts.append(path_part.name)
|
|
62
|
+
if path_parts:
|
|
63
|
+
return ".".join(path_parts)
|
|
64
|
+
|
|
65
|
+
if hasattr(json_path_expr, "this"):
|
|
66
|
+
path_str = str(json_path_expr.this)
|
|
67
|
+
# Strip quotes and leading $.
|
|
68
|
+
return path_str.strip("'\"").lstrip("$.")
|
|
69
|
+
|
|
70
|
+
# Fallback: try string representation
|
|
71
|
+
path_str = str(json_path_expr)
|
|
72
|
+
return path_str.strip("'\"").lstrip("$.")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def collect_json_sources(
|
|
76
|
+
source_expr: "exp.Expression",
|
|
77
|
+
known_relations: set[str],
|
|
78
|
+
alias_map: dict[str, str],
|
|
79
|
+
array_expansion_sources: dict[str, dict[str, str]],
|
|
80
|
+
default_table: str,
|
|
81
|
+
qualified_via_schema: bool,
|
|
82
|
+
sources: list[dict[str, Any]],
|
|
83
|
+
seen: set,
|
|
84
|
+
) -> set[int]:
|
|
85
|
+
"""Append JSON-path and bracket-access sources for one select item.
|
|
86
|
+
|
|
87
|
+
Mutates sources/seen in place and returns the ids of Column nodes
|
|
88
|
+
already handled, so the plain column sweep can skip them.
|
|
89
|
+
"""
|
|
90
|
+
json_columns_handled = set()
|
|
91
|
+
for json_expr in source_expr.find_all(*JSON_EXPR_TYPES):
|
|
92
|
+
base_col = json_expr.this
|
|
93
|
+
json_path_expr = json_expr.expression
|
|
94
|
+
|
|
95
|
+
# JSON_EXTRACT(JSON_EXTRACT(...)): the inner call owns the column
|
|
96
|
+
while isinstance(base_col, JSON_EXPR_TYPES):
|
|
97
|
+
base_col = base_col.this
|
|
98
|
+
|
|
99
|
+
if isinstance(base_col, exp.Column):
|
|
100
|
+
if under_subquery_where(base_col, source_expr):
|
|
101
|
+
json_columns_handled.add(id(base_col))
|
|
102
|
+
continue
|
|
103
|
+
alias, column, _struct_path, _certain = resolve_column_ref(base_col, known_relations)
|
|
104
|
+
|
|
105
|
+
json_path = _extract_json_path(json_path_expr)
|
|
106
|
+
|
|
107
|
+
# Mark this column as handled via JSON path
|
|
108
|
+
json_columns_handled.add(id(base_col))
|
|
109
|
+
|
|
110
|
+
source_entry: dict[str, Any] = {
|
|
111
|
+
"column": column,
|
|
112
|
+
"inferred": False,
|
|
113
|
+
"json_path": json_path if json_path else None,
|
|
114
|
+
"is_json_access": True,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
if not alias and column in array_expansion_sources:
|
|
118
|
+
# the expansion alias used bare: elem->>'k' reads the value
|
|
119
|
+
# jsonb_array_elements produced, so the true source is the
|
|
120
|
+
# expanded array column (FAC findings_text, holdout round 4)
|
|
121
|
+
expansion_info = array_expansion_sources[column]
|
|
122
|
+
table = expansion_info.get("source_table", "")
|
|
123
|
+
confidence = 0.5
|
|
124
|
+
trust_level = "moderate"
|
|
125
|
+
source_entry["from_array_expansion"] = True
|
|
126
|
+
source_entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
127
|
+
src_col = expansion_info.get("source_column", column)
|
|
128
|
+
source_entry["source_column_in_array"] = src_col
|
|
129
|
+
if src_col:
|
|
130
|
+
source_entry["expanded_as"] = column
|
|
131
|
+
source_entry["column"] = src_col
|
|
132
|
+
elif alias:
|
|
133
|
+
if alias in array_expansion_sources:
|
|
134
|
+
expansion_info = array_expansion_sources[alias]
|
|
135
|
+
table = expansion_info.get("source_table", alias)
|
|
136
|
+
confidence = 0.5
|
|
137
|
+
trust_level = "moderate"
|
|
138
|
+
source_entry["from_array_expansion"] = True
|
|
139
|
+
source_entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
140
|
+
source_entry["source_column_in_array"] = expansion_info.get(
|
|
141
|
+
"source_column", column
|
|
142
|
+
)
|
|
143
|
+
else:
|
|
144
|
+
table = alias_map.get(alias, alias)
|
|
145
|
+
confidence = 1.0
|
|
146
|
+
trust_level = "verified" if qualified_via_schema else "high_confidence"
|
|
147
|
+
elif default_table:
|
|
148
|
+
table = default_table
|
|
149
|
+
confidence = 0.8
|
|
150
|
+
trust_level = "high_confidence"
|
|
151
|
+
source_entry["inferred"] = True
|
|
152
|
+
else:
|
|
153
|
+
table = ""
|
|
154
|
+
confidence = 0.5
|
|
155
|
+
trust_level = "moderate"
|
|
156
|
+
|
|
157
|
+
source_entry["table"] = table
|
|
158
|
+
source_entry["confidence"] = confidence
|
|
159
|
+
source_entry["trust_level"] = trust_level
|
|
160
|
+
|
|
161
|
+
key = (table, column, json_path)
|
|
162
|
+
if key not in seen:
|
|
163
|
+
seen.add(key)
|
|
164
|
+
sources.append(source_entry)
|
|
165
|
+
|
|
166
|
+
# This pattern is used in Trino, Redshift (SUPER), and some other dialects
|
|
167
|
+
for bracket_expr in source_expr.find_all(exp.Bracket):
|
|
168
|
+
base = bracket_expr.this
|
|
169
|
+
# Skip if already handled via JSON expression
|
|
170
|
+
if any(id(base) in json_columns_handled for base in [bracket_expr.this]):
|
|
171
|
+
continue
|
|
172
|
+
|
|
173
|
+
# Walk up to find the base column
|
|
174
|
+
json_path_parts = []
|
|
175
|
+
while isinstance(base, exp.Bracket):
|
|
176
|
+
if base.expressions:
|
|
177
|
+
key = base.expressions[0]
|
|
178
|
+
if hasattr(key, "this"):
|
|
179
|
+
json_path_parts.insert(0, str(key.this))
|
|
180
|
+
elif hasattr(key, "name"):
|
|
181
|
+
json_path_parts.insert(0, key.name)
|
|
182
|
+
else:
|
|
183
|
+
json_path_parts.insert(0, str(key))
|
|
184
|
+
base = base.this
|
|
185
|
+
|
|
186
|
+
if isinstance(base, exp.Column):
|
|
187
|
+
json_columns_handled.add(id(base))
|
|
188
|
+
if under_subquery_where(base, source_expr):
|
|
189
|
+
continue
|
|
190
|
+
alias, column, _struct_path, _certain = resolve_column_ref(base, known_relations)
|
|
191
|
+
json_path = ".".join(json_path_parts) if json_path_parts else None
|
|
192
|
+
|
|
193
|
+
source_entry: dict[str, Any] = {
|
|
194
|
+
"column": column,
|
|
195
|
+
"inferred": False,
|
|
196
|
+
"json_path": json_path,
|
|
197
|
+
"is_json_access": True,
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
if not alias and column in array_expansion_sources:
|
|
201
|
+
# bare expansion alias (see the JSONExtract branch above)
|
|
202
|
+
expansion_info = array_expansion_sources[column]
|
|
203
|
+
table = expansion_info.get("source_table", "")
|
|
204
|
+
confidence = 0.5
|
|
205
|
+
trust_level = "moderate"
|
|
206
|
+
source_entry["from_array_expansion"] = True
|
|
207
|
+
src_col = expansion_info.get("source_column", column)
|
|
208
|
+
if src_col:
|
|
209
|
+
source_entry["expanded_as"] = column
|
|
210
|
+
source_entry["column"] = src_col
|
|
211
|
+
elif alias:
|
|
212
|
+
if alias in array_expansion_sources:
|
|
213
|
+
expansion_info = array_expansion_sources[alias]
|
|
214
|
+
table = expansion_info.get("source_table", alias)
|
|
215
|
+
confidence = 0.5
|
|
216
|
+
trust_level = "moderate"
|
|
217
|
+
source_entry["from_array_expansion"] = True
|
|
218
|
+
else:
|
|
219
|
+
table = alias_map.get(alias, alias)
|
|
220
|
+
confidence = 1.0
|
|
221
|
+
trust_level = "verified" if qualified_via_schema else "high_confidence"
|
|
222
|
+
elif default_table:
|
|
223
|
+
table = default_table
|
|
224
|
+
confidence = 0.8
|
|
225
|
+
trust_level = "high_confidence"
|
|
226
|
+
source_entry["inferred"] = True
|
|
227
|
+
else:
|
|
228
|
+
table = ""
|
|
229
|
+
confidence = 0.5
|
|
230
|
+
trust_level = "moderate"
|
|
231
|
+
|
|
232
|
+
source_entry["table"] = table
|
|
233
|
+
source_entry["confidence"] = confidence
|
|
234
|
+
source_entry["trust_level"] = trust_level
|
|
235
|
+
|
|
236
|
+
key = (table, column, json_path)
|
|
237
|
+
if key not in seen:
|
|
238
|
+
seen.add(key)
|
|
239
|
+
sources.append(source_entry)
|
|
240
|
+
|
|
241
|
+
return json_columns_handled
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Primary-source detection from dbt macro patterns.
|
|
2
|
+
|
|
3
|
+
Regex-only: finds the source() or ref() call that names a model's primary
|
|
4
|
+
data source, before any SQL parsing happens.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def extract_primary_source_from_macro(raw_sql: str) -> dict | None:
|
|
13
|
+
"""Extract primary source from staging macro patterns.
|
|
14
|
+
|
|
15
|
+
Handles:
|
|
16
|
+
1. ANY custom macro wrapping source() or ref() - not just known names
|
|
17
|
+
2. Direct source()/ref() calls as first statement
|
|
18
|
+
3. Various whitespace and formatting patterns
|
|
19
|
+
4. Multi-line macro calls
|
|
20
|
+
|
|
21
|
+
Strategy:
|
|
22
|
+
- First, strip config blocks to find the "main body"
|
|
23
|
+
- Look for source() or ref() as the FIRST reference in the main body
|
|
24
|
+
- This is almost always the primary data source (not WHERE clause refs)
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
{
|
|
28
|
+
"type": "source"|"ref",
|
|
29
|
+
"source": "source_name", # for source type
|
|
30
|
+
"table": "table_name", # for source type
|
|
31
|
+
"ref": "model_name", # for ref type
|
|
32
|
+
"detection_method": "macro_wrapper"|"direct_call"|"first_ref"
|
|
33
|
+
}
|
|
34
|
+
or None if no pattern found
|
|
35
|
+
"""
|
|
36
|
+
import re
|
|
37
|
+
|
|
38
|
+
if not raw_sql or not raw_sql.strip():
|
|
39
|
+
return None
|
|
40
|
+
|
|
41
|
+
# Step 1: Strip config block(s) to get to the main SQL body
|
|
42
|
+
config_pattern = re.compile(
|
|
43
|
+
r"\{\{\s*config\s*\([^)]*(?:\([^)]*\)[^)]*)*\)\s*\}\}", re.IGNORECASE | re.DOTALL
|
|
44
|
+
)
|
|
45
|
+
main_body = config_pattern.sub("", raw_sql).strip()
|
|
46
|
+
|
|
47
|
+
# Step 2: Try to find ANY macro that wraps source() or ref()
|
|
48
|
+
# Generic macro wrapping source()
|
|
49
|
+
generic_macro_source = re.compile(
|
|
50
|
+
r"\{\{\s*(\w+)\s*\(\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)",
|
|
51
|
+
re.IGNORECASE | re.DOTALL,
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
match = generic_macro_source.search(main_body)
|
|
55
|
+
if match:
|
|
56
|
+
macro_name = match.group(1).lower()
|
|
57
|
+
skip_macros = {"config", "if", "elif", "else", "endif", "for", "endfor", "set", "do"}
|
|
58
|
+
if macro_name not in skip_macros:
|
|
59
|
+
return {
|
|
60
|
+
"type": "source",
|
|
61
|
+
"source": match.group(2),
|
|
62
|
+
"table": match.group(3),
|
|
63
|
+
"detection_method": "macro_wrapper",
|
|
64
|
+
"macro_name": macro_name,
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
# Generic macro wrapping ref()
|
|
68
|
+
generic_macro_ref = re.compile(
|
|
69
|
+
r"\{\{\s*(\w+)\s*\(\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)", re.IGNORECASE | re.DOTALL
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
match = generic_macro_ref.search(main_body)
|
|
73
|
+
if match:
|
|
74
|
+
macro_name = match.group(1).lower()
|
|
75
|
+
skip_macros = {"config", "if", "elif", "else", "endif", "for", "endfor", "set", "do"}
|
|
76
|
+
if macro_name not in skip_macros:
|
|
77
|
+
return {
|
|
78
|
+
"type": "ref",
|
|
79
|
+
"ref": match.group(2),
|
|
80
|
+
"detection_method": "macro_wrapper",
|
|
81
|
+
"macro_name": macro_name,
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
# Step 3: Look for direct source() or ref() as first Jinja call
|
|
85
|
+
first_source = re.search(
|
|
86
|
+
r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
|
|
87
|
+
main_body,
|
|
88
|
+
re.IGNORECASE,
|
|
89
|
+
)
|
|
90
|
+
first_ref = re.search(
|
|
91
|
+
r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", main_body, re.IGNORECASE
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
where_pos = main_body.upper().find("WHERE")
|
|
95
|
+
if where_pos == -1:
|
|
96
|
+
where_pos = len(main_body)
|
|
97
|
+
|
|
98
|
+
candidates = []
|
|
99
|
+
if first_source and first_source.start() < where_pos:
|
|
100
|
+
candidates.append(
|
|
101
|
+
(
|
|
102
|
+
first_source.start(),
|
|
103
|
+
{
|
|
104
|
+
"type": "source",
|
|
105
|
+
"source": first_source.group(1),
|
|
106
|
+
"table": first_source.group(2),
|
|
107
|
+
"detection_method": "direct_call",
|
|
108
|
+
},
|
|
109
|
+
)
|
|
110
|
+
)
|
|
111
|
+
if first_ref and first_ref.start() < where_pos:
|
|
112
|
+
candidates.append(
|
|
113
|
+
(
|
|
114
|
+
first_ref.start(),
|
|
115
|
+
{
|
|
116
|
+
"type": "ref",
|
|
117
|
+
"ref": first_ref.group(1),
|
|
118
|
+
"detection_method": "direct_call",
|
|
119
|
+
},
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
if candidates:
|
|
124
|
+
candidates.sort(key=lambda x: x[0])
|
|
125
|
+
return candidates[0][1]
|
|
126
|
+
|
|
127
|
+
return None
|