ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,466 @@
|
|
|
1
|
+
"""Column dependency extraction - WHERE clauses and JOIN keys.
|
|
2
|
+
|
|
3
|
+
This module extracts C_ref columns (referenced but not contributing to output)
|
|
4
|
+
following the LineageX paper's terminology:
|
|
5
|
+
|
|
6
|
+
- C_con: Columns that contribute to output values (handled by extraction.py)
|
|
7
|
+
- C_ref: Columns referenced in WHERE, HAVING, JOIN ON (handled here)
|
|
8
|
+
|
|
9
|
+
Knowing which columns gate rows completes the picture for impact analysis.
|
|
10
|
+
|
|
11
|
+
Example:
|
|
12
|
+
SELECT a.name, a.value
|
|
13
|
+
FROM orders a
|
|
14
|
+
JOIN customers b ON a.customer_id = b.id
|
|
15
|
+
WHERE b.status = 'active' AND a.created_at > '2024-01-01'
|
|
16
|
+
|
|
17
|
+
C_con: name, value (output columns)
|
|
18
|
+
C_ref: customer_id, id, status, created_at (filter/join columns)
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import logging
|
|
22
|
+
from typing import TYPE_CHECKING, Literal
|
|
23
|
+
|
|
24
|
+
from ripple.engine.column_ref import table_and_column
|
|
25
|
+
from ripple.engine.types import (
|
|
26
|
+
ColumnDependencies,
|
|
27
|
+
FilterDependency,
|
|
28
|
+
JoinKeyPair,
|
|
29
|
+
WarehouseColumns,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
if TYPE_CHECKING:
|
|
33
|
+
from sqlglot import exp
|
|
34
|
+
|
|
35
|
+
logger = logging.getLogger(__name__)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def extract_column_dependencies(
|
|
39
|
+
sql: str,
|
|
40
|
+
dialect: str = "snowflake",
|
|
41
|
+
warehouse_columns: "WarehouseColumns | None" = None,
|
|
42
|
+
) -> ColumnDependencies:
|
|
43
|
+
"""Extract columns used in WHERE clauses and JOIN conditions.
|
|
44
|
+
|
|
45
|
+
This complements extract_column_lineage_fast() by tracking C_ref columns
|
|
46
|
+
that affect which rows are returned, not the output values.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
sql: SQL query to analyze
|
|
50
|
+
dialect: SQL dialect for parsing
|
|
51
|
+
warehouse_columns: optional schemas; when given, an unqualified key
|
|
52
|
+
resolves to its owning table instead of the value's table
|
|
53
|
+
(cycle-10 review, F2)
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
ColumnDependencies with filter_columns and join_keys
|
|
57
|
+
"""
|
|
58
|
+
import sqlglot
|
|
59
|
+
|
|
60
|
+
result = ColumnDependencies()
|
|
61
|
+
|
|
62
|
+
try:
|
|
63
|
+
# this reparse must see the same pre-parse pipeline the lineage
|
|
64
|
+
# parse got, or a {placeholder} estate zeroes dependency extraction
|
|
65
|
+
from ripple.engine.preprocess import prepare_sql_for_parse
|
|
66
|
+
|
|
67
|
+
sql, _ = prepare_sql_for_parse(sql, dialect)
|
|
68
|
+
parsed = None
|
|
69
|
+
if warehouse_columns:
|
|
70
|
+
from ripple.engine.schema_qualification import qualify_sql_with_schema
|
|
71
|
+
|
|
72
|
+
qualified_ast, was_qualified = qualify_sql_with_schema(sql, dialect, warehouse_columns)
|
|
73
|
+
if was_qualified and qualified_ast is not None:
|
|
74
|
+
parsed = qualified_ast
|
|
75
|
+
if parsed is None:
|
|
76
|
+
parsed = sqlglot.parse_one(sql, read=dialect)
|
|
77
|
+
if (dialect or "").lower() == "tsql":
|
|
78
|
+
# this reparse never saw the XML-method rewrite the lineage AST
|
|
79
|
+
# got, so a .exist() receiver in a predicate stayed invisible
|
|
80
|
+
# (cycle-9 review, F6)
|
|
81
|
+
from ripple.engine.tsql_xml import rewrite_xml_method_calls
|
|
82
|
+
|
|
83
|
+
parsed = rewrite_xml_method_calls(parsed)
|
|
84
|
+
|
|
85
|
+
result.filter_columns.extend(_extract_where_columns(parsed))
|
|
86
|
+
|
|
87
|
+
result.filter_columns.extend(_extract_having_columns(parsed))
|
|
88
|
+
|
|
89
|
+
result.filter_columns.extend(_extract_qualify_columns(parsed))
|
|
90
|
+
|
|
91
|
+
result.join_keys.extend(_extract_join_keys(parsed))
|
|
92
|
+
|
|
93
|
+
result.window_keys.extend(_extract_window_keys(parsed, dialect))
|
|
94
|
+
|
|
95
|
+
except Exception as e:
|
|
96
|
+
logger.warning(f"Dependency extraction failed: {type(e).__name__}: {e}")
|
|
97
|
+
# Sanitize error message - don't expose internal details to users
|
|
98
|
+
result.warnings.append(f"Dependency extraction failed: {type(e).__name__}")
|
|
99
|
+
|
|
100
|
+
return result
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _extract_where_columns(
|
|
104
|
+
parsed: "exp.Expression",
|
|
105
|
+
) -> list[FilterDependency]:
|
|
106
|
+
"""Extract columns from WHERE clauses."""
|
|
107
|
+
from sqlglot import exp
|
|
108
|
+
|
|
109
|
+
results = []
|
|
110
|
+
|
|
111
|
+
for where in parsed.find_all(exp.Where):
|
|
112
|
+
for col in where.find_all(exp.Column):
|
|
113
|
+
table, column = table_and_column(col)
|
|
114
|
+
results.append(
|
|
115
|
+
FilterDependency(
|
|
116
|
+
column=column,
|
|
117
|
+
table=table,
|
|
118
|
+
predicate_type="where",
|
|
119
|
+
operator=_get_parent_operator(col),
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
return results
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _keyed_relation_names(parsed: "exp.Expression") -> set[str]:
|
|
127
|
+
"""Lowercased names a keyed column's qualifier could bind: table names,
|
|
128
|
+
dotted qualified names, and query aliases."""
|
|
129
|
+
from sqlglot import exp
|
|
130
|
+
|
|
131
|
+
from ripple.engine.column_ref import qualified_table_name
|
|
132
|
+
|
|
133
|
+
known: set[str] = set()
|
|
134
|
+
for table in parsed.find_all(exp.Table):
|
|
135
|
+
if not table.name:
|
|
136
|
+
continue
|
|
137
|
+
known.add(table.name.lower())
|
|
138
|
+
qualified = qualified_table_name(table).lower()
|
|
139
|
+
if qualified:
|
|
140
|
+
known.add(qualified)
|
|
141
|
+
if table.alias:
|
|
142
|
+
known.add(table.alias.lower())
|
|
143
|
+
for cte in parsed.find_all(exp.CTE):
|
|
144
|
+
if cte.alias_or_name:
|
|
145
|
+
known.add(cte.alias_or_name.lower())
|
|
146
|
+
return known
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _resolve_keyed_column(col: "exp.Column", known: set[str]) -> tuple[str | None, str]:
|
|
150
|
+
"""(table, column) for a window/selector key, qualification-aware.
|
|
151
|
+
|
|
152
|
+
table_and_column's struct guess read db.sch.t.s as table db, column
|
|
153
|
+
sch, a confident edge to a relation the query never reads (cycle-10
|
|
154
|
+
review, F1). A dotted prefix matching a relation the statement
|
|
155
|
+
names is a qualification, not a struct path.
|
|
156
|
+
"""
|
|
157
|
+
from ripple.engine.column_ref import column_parts
|
|
158
|
+
|
|
159
|
+
parts = column_parts(col)
|
|
160
|
+
if len(parts) <= 2 or parts[0].lower() in known:
|
|
161
|
+
return table_and_column(col)
|
|
162
|
+
prefix = ".".join(parts[:-1])
|
|
163
|
+
if prefix.lower() in known or parts[-2].lower() in known:
|
|
164
|
+
return prefix, parts[-1]
|
|
165
|
+
return table_and_column(col)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _lateral_alias_columns(
|
|
169
|
+
col: "exp.Column", known: set[str]
|
|
170
|
+
) -> "list[tuple[str | None, str]] | None":
|
|
171
|
+
"""The real columns behind a keyed name that is a same-select alias.
|
|
172
|
+
|
|
173
|
+
rank_col AS selector, then max_by(v, selector): the key is the aliased
|
|
174
|
+
expression's own columns, never a column named selector (cycle-10
|
|
175
|
+
review, F3). A key sitting inside the expression its own alias
|
|
176
|
+
names is the real column, so it stays textual. None when the name
|
|
177
|
+
matches no alias.
|
|
178
|
+
"""
|
|
179
|
+
from sqlglot import exp
|
|
180
|
+
|
|
181
|
+
from ripple.engine.column_ref import owning_select
|
|
182
|
+
|
|
183
|
+
owner = owning_select(col)
|
|
184
|
+
if owner is None:
|
|
185
|
+
return None
|
|
186
|
+
for item in owner.selects:
|
|
187
|
+
if not isinstance(item, exp.Alias) or (item.alias or "").lower() != col.name.lower():
|
|
188
|
+
continue
|
|
189
|
+
if any(inner is col for inner in item.this.find_all(exp.Column)):
|
|
190
|
+
return None
|
|
191
|
+
return [_resolve_keyed_column(inner, known) for inner in item.this.find_all(exp.Column)]
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _extract_window_keys(
|
|
196
|
+
parsed: "exp.Expression",
|
|
197
|
+
dialect: str = "",
|
|
198
|
+
) -> list[FilterDependency]:
|
|
199
|
+
"""Extract the PARTITION BY and ORDER BY columns of every window.
|
|
200
|
+
|
|
201
|
+
They frame the window rather than feed its value, so they are C_ref like a
|
|
202
|
+
WHERE column. Kept because renaming a partition key really does break the
|
|
203
|
+
window; dropping them would trade one wrong answer for a blind spot.
|
|
204
|
+
The selector argument of arg_max/arg_min plays the same role as a
|
|
205
|
+
window's ORDER BY, so it rides along here.
|
|
206
|
+
"""
|
|
207
|
+
from sqlglot import exp
|
|
208
|
+
|
|
209
|
+
from ripple.engine.column_ref import column_parts
|
|
210
|
+
from ripple.engine.types import LATERAL_ALIAS_DIALECTS
|
|
211
|
+
|
|
212
|
+
keyed: list[exp.Expression] = []
|
|
213
|
+
for window in parsed.find_all(exp.Window):
|
|
214
|
+
keyed.extend(window.args.get("partition_by") or [])
|
|
215
|
+
order = window.args.get("order")
|
|
216
|
+
if order is not None:
|
|
217
|
+
keyed.append(order)
|
|
218
|
+
for agg in parsed.find_all(exp.ArgMax, exp.ArgMin):
|
|
219
|
+
keyed.extend(
|
|
220
|
+
arg for arg in (agg.args.get("expression"), agg.args.get("count")) if arg is not None
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
known = _keyed_relation_names(parsed)
|
|
224
|
+
allow_lateral = (dialect or "").lower() in LATERAL_ALIAS_DIALECTS
|
|
225
|
+
|
|
226
|
+
results = []
|
|
227
|
+
seen: set[tuple[str | None, str]] = set()
|
|
228
|
+
for part in keyed:
|
|
229
|
+
for col in part.find_all(exp.Column):
|
|
230
|
+
resolved = None
|
|
231
|
+
if allow_lateral and len(column_parts(col)) == 1:
|
|
232
|
+
resolved = _lateral_alias_columns(col, known)
|
|
233
|
+
if resolved is None:
|
|
234
|
+
resolved = [_resolve_keyed_column(col, known)]
|
|
235
|
+
for table, column in resolved:
|
|
236
|
+
if not column or (table, column) in seen:
|
|
237
|
+
continue
|
|
238
|
+
seen.add((table, column))
|
|
239
|
+
results.append(
|
|
240
|
+
FilterDependency(
|
|
241
|
+
column=column,
|
|
242
|
+
table=table,
|
|
243
|
+
predicate_type="window",
|
|
244
|
+
)
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
return results
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _extract_having_columns(
|
|
251
|
+
parsed: "exp.Expression",
|
|
252
|
+
) -> list[FilterDependency]:
|
|
253
|
+
"""Extract columns from HAVING clauses."""
|
|
254
|
+
from sqlglot import exp
|
|
255
|
+
|
|
256
|
+
results = []
|
|
257
|
+
|
|
258
|
+
for having in parsed.find_all(exp.Having):
|
|
259
|
+
for col in having.find_all(exp.Column):
|
|
260
|
+
table, column = table_and_column(col)
|
|
261
|
+
results.append(
|
|
262
|
+
FilterDependency(
|
|
263
|
+
column=column,
|
|
264
|
+
table=table,
|
|
265
|
+
predicate_type="having",
|
|
266
|
+
operator=_get_parent_operator(col),
|
|
267
|
+
)
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
return results
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _extract_qualify_columns(
|
|
274
|
+
parsed: "exp.Expression",
|
|
275
|
+
) -> list[FilterDependency]:
|
|
276
|
+
"""Extract columns from QUALIFY clauses (Snowflake/Databricks)."""
|
|
277
|
+
from sqlglot import exp
|
|
278
|
+
|
|
279
|
+
results = []
|
|
280
|
+
|
|
281
|
+
for qualify in parsed.find_all(exp.Qualify):
|
|
282
|
+
for col in qualify.find_all(exp.Column):
|
|
283
|
+
table, column = table_and_column(col)
|
|
284
|
+
results.append(
|
|
285
|
+
FilterDependency(
|
|
286
|
+
column=column,
|
|
287
|
+
table=table,
|
|
288
|
+
predicate_type="qualify",
|
|
289
|
+
operator=_get_parent_operator(col),
|
|
290
|
+
)
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
return results
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _extract_join_keys(
|
|
297
|
+
parsed: "exp.Expression",
|
|
298
|
+
) -> list[JoinKeyPair]:
|
|
299
|
+
"""Extract column pairs from JOIN ON conditions."""
|
|
300
|
+
from sqlglot import exp
|
|
301
|
+
|
|
302
|
+
results = []
|
|
303
|
+
|
|
304
|
+
for join in parsed.find_all(exp.Join):
|
|
305
|
+
join_type = _get_join_type(join)
|
|
306
|
+
on_clause = join.args.get("on")
|
|
307
|
+
|
|
308
|
+
if on_clause:
|
|
309
|
+
# Find equality conditions (a.col = b.col)
|
|
310
|
+
for eq in on_clause.find_all(exp.EQ):
|
|
311
|
+
left = eq.left
|
|
312
|
+
right = eq.right
|
|
313
|
+
|
|
314
|
+
if isinstance(left, exp.Column) and isinstance(right, exp.Column):
|
|
315
|
+
left_table, left_column = table_and_column(left)
|
|
316
|
+
right_table, right_column = table_and_column(right)
|
|
317
|
+
results.append(
|
|
318
|
+
JoinKeyPair(
|
|
319
|
+
left_column=left_column,
|
|
320
|
+
left_table=left_table or "",
|
|
321
|
+
right_column=right_column,
|
|
322
|
+
right_table=right_table or "",
|
|
323
|
+
join_type=join_type,
|
|
324
|
+
)
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
return results
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _get_parent_operator(col: "exp.Column") -> str | None:
|
|
331
|
+
"""Get the operator containing this column (=, IN, BETWEEN, etc.)."""
|
|
332
|
+
from sqlglot import exp
|
|
333
|
+
|
|
334
|
+
parent = col.parent
|
|
335
|
+
while parent:
|
|
336
|
+
if isinstance(parent, exp.EQ):
|
|
337
|
+
return "="
|
|
338
|
+
elif isinstance(parent, exp.NEQ):
|
|
339
|
+
return "!="
|
|
340
|
+
elif isinstance(parent, exp.In):
|
|
341
|
+
return "IN"
|
|
342
|
+
elif isinstance(parent, exp.Between):
|
|
343
|
+
return "BETWEEN"
|
|
344
|
+
elif isinstance(parent, exp.GT):
|
|
345
|
+
return ">"
|
|
346
|
+
elif isinstance(parent, exp.GTE):
|
|
347
|
+
return ">="
|
|
348
|
+
elif isinstance(parent, exp.LT):
|
|
349
|
+
return "<"
|
|
350
|
+
elif isinstance(parent, exp.LTE):
|
|
351
|
+
return "<="
|
|
352
|
+
elif isinstance(parent, exp.Like):
|
|
353
|
+
return "LIKE"
|
|
354
|
+
elif isinstance(parent, exp.ILike):
|
|
355
|
+
return "ILIKE"
|
|
356
|
+
elif isinstance(parent, exp.Is):
|
|
357
|
+
return "IS"
|
|
358
|
+
elif isinstance(parent, exp.RegexpLike):
|
|
359
|
+
return "REGEXP"
|
|
360
|
+
parent = parent.parent
|
|
361
|
+
return None
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _get_join_type(join: "exp.Join") -> Literal["inner", "left", "right", "full", "cross"]:
|
|
365
|
+
"""Determine JOIN type from expression.
|
|
366
|
+
|
|
367
|
+
Handles: INNER, LEFT, RIGHT, FULL, CROSS, NATURAL, SEMI, ANTI joins.
|
|
368
|
+
NATURAL/SEMI/ANTI are mapped to "inner" as they don't have a distinct type in our model.
|
|
369
|
+
"""
|
|
370
|
+
from typing import Literal as Lit
|
|
371
|
+
from typing import cast
|
|
372
|
+
|
|
373
|
+
if join.side:
|
|
374
|
+
side = join.side.lower()
|
|
375
|
+
if side in ("left", "right", "full"):
|
|
376
|
+
return cast(Lit["left", "right", "full"], side)
|
|
377
|
+
if join.kind:
|
|
378
|
+
kind = join.kind.lower()
|
|
379
|
+
if kind == "cross":
|
|
380
|
+
return "cross"
|
|
381
|
+
# NATURAL, SEMI, ANTI joins default to "inner" behavior
|
|
382
|
+
return "inner"
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def extract_all_referenced_columns(
|
|
386
|
+
sql: str,
|
|
387
|
+
dialect: str = "snowflake",
|
|
388
|
+
) -> list[dict[str, str | None]]:
|
|
389
|
+
"""Extract ALL column references from SQL (SELECT, WHERE, JOIN, GROUP BY, etc.).
|
|
390
|
+
|
|
391
|
+
This is a convenience function that returns every column reference found,
|
|
392
|
+
tagged with its context (select, where, join, group_by, order_by, etc.).
|
|
393
|
+
|
|
394
|
+
Useful for impact analysis: "which columns does this query touch?"
|
|
395
|
+
|
|
396
|
+
Args:
|
|
397
|
+
sql: SQL query to analyze
|
|
398
|
+
dialect: SQL dialect
|
|
399
|
+
|
|
400
|
+
Returns:
|
|
401
|
+
List of column references with context
|
|
402
|
+
"""
|
|
403
|
+
import sqlglot
|
|
404
|
+
from sqlglot import exp
|
|
405
|
+
|
|
406
|
+
results = []
|
|
407
|
+
|
|
408
|
+
try:
|
|
409
|
+
from ripple.engine.preprocess import prepare_sql_for_parse
|
|
410
|
+
|
|
411
|
+
sql, _ = prepare_sql_for_parse(sql, dialect)
|
|
412
|
+
parsed = sqlglot.parse_one(sql, read=dialect)
|
|
413
|
+
|
|
414
|
+
# Track which columns we've seen to avoid duplicates
|
|
415
|
+
seen: set[tuple[str, str, str]] = set()
|
|
416
|
+
|
|
417
|
+
def add_column(col: exp.Column, context: str) -> None:
|
|
418
|
+
table, column = table_and_column(col)
|
|
419
|
+
key = (table or "", column, context)
|
|
420
|
+
if key not in seen:
|
|
421
|
+
seen.add(key)
|
|
422
|
+
results.append(
|
|
423
|
+
{
|
|
424
|
+
"column": column,
|
|
425
|
+
"table": table,
|
|
426
|
+
"context": context,
|
|
427
|
+
}
|
|
428
|
+
)
|
|
429
|
+
|
|
430
|
+
# SELECT columns
|
|
431
|
+
for select in parsed.find_all(exp.Select):
|
|
432
|
+
for expr in select.selects:
|
|
433
|
+
for col in expr.find_all(exp.Column):
|
|
434
|
+
add_column(col, "select")
|
|
435
|
+
|
|
436
|
+
# WHERE columns
|
|
437
|
+
for where in parsed.find_all(exp.Where):
|
|
438
|
+
for col in where.find_all(exp.Column):
|
|
439
|
+
add_column(col, "where")
|
|
440
|
+
|
|
441
|
+
# HAVING columns
|
|
442
|
+
for having in parsed.find_all(exp.Having):
|
|
443
|
+
for col in having.find_all(exp.Column):
|
|
444
|
+
add_column(col, "having")
|
|
445
|
+
|
|
446
|
+
# JOIN ON columns
|
|
447
|
+
for join in parsed.find_all(exp.Join):
|
|
448
|
+
on_clause = join.args.get("on")
|
|
449
|
+
if on_clause:
|
|
450
|
+
for col in on_clause.find_all(exp.Column):
|
|
451
|
+
add_column(col, "join")
|
|
452
|
+
|
|
453
|
+
# GROUP BY columns
|
|
454
|
+
for group in parsed.find_all(exp.Group):
|
|
455
|
+
for col in group.find_all(exp.Column):
|
|
456
|
+
add_column(col, "group_by")
|
|
457
|
+
|
|
458
|
+
# ORDER BY columns
|
|
459
|
+
for order in parsed.find_all(exp.Order):
|
|
460
|
+
for col in order.find_all(exp.Column):
|
|
461
|
+
add_column(col, "order_by")
|
|
462
|
+
|
|
463
|
+
except Exception as e:
|
|
464
|
+
logger.warning(f"Failed to extract all column references: {e}")
|
|
465
|
+
|
|
466
|
+
return results
|
ripple/engine/dialect.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Map dbt adapter types to SQLGlot dialect names.
|
|
2
|
+
|
|
3
|
+
dbt manifests include metadata.adapter_type which tells us what database
|
|
4
|
+
the project targets. This module maps those adapter names to SQLGlot's
|
|
5
|
+
dialect identifiers for correct SQL parsing.
|
|
6
|
+
|
|
7
|
+
Example:
|
|
8
|
+
manifest["metadata"]["adapter_type"] = "snowflake"
|
|
9
|
+
→ get_sqlglot_dialect(manifest) returns "snowflake"
|
|
10
|
+
|
|
11
|
+
manifest["metadata"]["adapter_type"] = "athena"
|
|
12
|
+
→ get_sqlglot_dialect(manifest) returns "trino" (Athena uses Presto/Trino SQL)
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
from typing import Any, Final
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
# Mapping from dbt adapter_type → SQLGlot dialect name
|
|
21
|
+
#
|
|
22
|
+
# Why some mappings aren't 1:1:
|
|
23
|
+
# - athena → trino: AWS Athena uses Presto/Trino SQL syntax
|
|
24
|
+
# - synapse → tsql: Azure Synapse uses T-SQL (SQL Server syntax)
|
|
25
|
+
# - sqlserver → tsql: SQL Server uses T-SQL
|
|
26
|
+
#
|
|
27
|
+
ADAPTER_TO_SQLGLOT_DIALECT: Final[dict[str, str]] = {
|
|
28
|
+
# Cloud data warehouses
|
|
29
|
+
"snowflake": "snowflake",
|
|
30
|
+
"bigquery": "bigquery",
|
|
31
|
+
"databricks": "databricks",
|
|
32
|
+
"redshift": "redshift",
|
|
33
|
+
# Open source / on-prem
|
|
34
|
+
"postgres": "postgres",
|
|
35
|
+
"trino": "trino",
|
|
36
|
+
"presto": "trino", # Presto uses Trino SQL syntax
|
|
37
|
+
"starburst": "trino", # Starburst is enterprise Trino
|
|
38
|
+
"spark": "spark",
|
|
39
|
+
"duckdb": "duckdb",
|
|
40
|
+
"clickhouse": "clickhouse",
|
|
41
|
+
# AWS
|
|
42
|
+
"athena": "trino", # Athena uses Presto/Trino SQL syntax
|
|
43
|
+
"glue": "spark", # AWS Glue uses Spark SQL
|
|
44
|
+
# Microsoft
|
|
45
|
+
"synapse": "tsql", # Azure Synapse uses T-SQL
|
|
46
|
+
"sqlserver": "tsql", # SQL Server uses T-SQL
|
|
47
|
+
"fabric": "tsql", # Microsoft Fabric uses T-SQL
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
# Default dialect when adapter is unknown or not specified
|
|
51
|
+
DEFAULT_DIALECT: Final[str] = "snowflake"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def get_sqlglot_dialect(manifest: dict[str, Any]) -> str:
|
|
55
|
+
"""Get SQLGlot dialect from dbt manifest metadata.
|
|
56
|
+
|
|
57
|
+
Extracts the adapter_type from manifest.metadata and maps it
|
|
58
|
+
to the corresponding SQLGlot dialect for SQL parsing.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
manifest: Parsed dbt manifest.json dictionary
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
SQLGlot dialect string (e.g., "snowflake", "bigquery", "trino")
|
|
65
|
+
Defaults to "snowflake" if adapter is unknown.
|
|
66
|
+
|
|
67
|
+
Example:
|
|
68
|
+
>>> manifest = {"metadata": {"adapter_type": "databricks"}}
|
|
69
|
+
>>> get_sqlglot_dialect(manifest)
|
|
70
|
+
'databricks'
|
|
71
|
+
"""
|
|
72
|
+
adapter_type = manifest.get("metadata", {}).get("adapter_type", "")
|
|
73
|
+
dialect = ADAPTER_TO_SQLGLOT_DIALECT.get(adapter_type.lower(), DEFAULT_DIALECT)
|
|
74
|
+
|
|
75
|
+
if adapter_type and adapter_type.lower() not in ADAPTER_TO_SQLGLOT_DIALECT:
|
|
76
|
+
logger.warning(
|
|
77
|
+
f"Unknown dbt adapter '{adapter_type}', defaulting to '{DEFAULT_DIALECT}' dialect"
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
logger.debug(f"dbt adapter '{adapter_type}' → SQLGlot dialect '{dialect}'")
|
|
81
|
+
return dialect
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def get_dialect_from_adapter_type(adapter_type: str | None) -> str:
|
|
85
|
+
"""Get SQLGlot dialect from a dbt adapter type string.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
adapter_type: dbt adapter type (e.g., "snowflake", "bigquery", "databricks")
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
SQLGlot dialect string. Defaults to "snowflake" if unknown.
|
|
92
|
+
|
|
93
|
+
Example:
|
|
94
|
+
>>> get_dialect_from_adapter_type("databricks")
|
|
95
|
+
'databricks'
|
|
96
|
+
>>> get_dialect_from_adapter_type("athena")
|
|
97
|
+
'trino'
|
|
98
|
+
"""
|
|
99
|
+
if not adapter_type:
|
|
100
|
+
return DEFAULT_DIALECT
|
|
101
|
+
|
|
102
|
+
dialect = ADAPTER_TO_SQLGLOT_DIALECT.get(adapter_type.lower(), DEFAULT_DIALECT)
|
|
103
|
+
|
|
104
|
+
if adapter_type.lower() not in ADAPTER_TO_SQLGLOT_DIALECT:
|
|
105
|
+
logger.warning(
|
|
106
|
+
f"Unknown dbt adapter '{adapter_type}', defaulting to '{DEFAULT_DIALECT}' dialect"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
return dialect
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def get_dialect_from_config(config: dict[str, Any] | None) -> str:
|
|
113
|
+
"""Get the dialect from a config dict that may carry adapter_type.
|
|
114
|
+
|
|
115
|
+
Args:
|
|
116
|
+
config: dict that may contain "adapter_type".
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
SQLGlot dialect string, or DEFAULT_DIALECT if absent.
|
|
120
|
+
"""
|
|
121
|
+
if not config:
|
|
122
|
+
return DEFAULT_DIALECT
|
|
123
|
+
return get_dialect_from_adapter_type(config.get("adapter_type"))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def get_supported_adapters() -> list[str]:
|
|
127
|
+
"""Get list of supported dbt adapter types.
|
|
128
|
+
|
|
129
|
+
Returns:
|
|
130
|
+
List of adapter type strings that have SQLGlot dialect mappings.
|
|
131
|
+
"""
|
|
132
|
+
return list(ADAPTER_TO_SQLGLOT_DIALECT.keys())
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Engine entry point.
|
|
2
|
+
|
|
3
|
+
Every caller goes through this module for lineage extraction, so a different
|
|
4
|
+
engine could slot in behind it later without touching the call sites. Today it
|
|
5
|
+
calls the Python engine directly.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from ripple.engine.extraction import extract_lineage_complete
|
|
11
|
+
|
|
12
|
+
__all__ = ["extract_lineage_complete"]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Column lineage extraction - public entry points (facade).
|
|
2
|
+
|
|
3
|
+
The implementation is split by responsibility:
|
|
4
|
+
- macro_source.py: primary-source detection from dbt macro patterns
|
|
5
|
+
- scope.py: FROM/JOIN/LATERAL/UNNEST alias and expansion registration
|
|
6
|
+
- json_sources.py: JSON path and bracket access source extraction
|
|
7
|
+
- select_sources.py: per-select-item source classification
|
|
8
|
+
- statement.py: statement-level orchestration (extract_column_lineage_fast)
|
|
9
|
+
- pipeline.py: CTE mappings, dialect fallback, unified C_con + C_ref API
|
|
10
|
+
|
|
11
|
+
This module only re-exports the public functions so existing imports
|
|
12
|
+
(`from ripple.engine.extraction import ...`) keep working.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from ripple.engine.macro_source import extract_primary_source_from_macro
|
|
16
|
+
from ripple.engine.pipeline import (
|
|
17
|
+
extract_column_lineage_with_ctes,
|
|
18
|
+
extract_lineage_complete,
|
|
19
|
+
)
|
|
20
|
+
from ripple.engine.statement import extract_column_lineage_fast
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"extract_column_lineage_fast",
|
|
24
|
+
"extract_column_lineage_with_ctes",
|
|
25
|
+
"extract_lineage_complete",
|
|
26
|
+
"extract_primary_source_from_macro",
|
|
27
|
+
]
|