ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/ci_signature.py
ADDED
|
@@ -0,0 +1,374 @@
|
|
|
1
|
+
"""What a changed model's SQL means, as a comparable signature.
|
|
2
|
+
|
|
3
|
+
`ripple ci` diffs the base and head versions of every changed model. The
|
|
4
|
+
signature per output column is (upstream columns, top-level expression, the
|
|
5
|
+
inner expressions its lineage passes through), plus one entry for the
|
|
6
|
+
statement's row-deciding closure and one for the inner expressions no output
|
|
7
|
+
column reaches. Pure functions over SQL text; nothing here touches git or the
|
|
8
|
+
graph.
|
|
9
|
+
|
|
10
|
+
The row-deciding closure is the one rule behind the predicates entry: every
|
|
11
|
+
clause that decides which rows survive or how they group (WHERE, HAVING,
|
|
12
|
+
QUALIFY, JOIN, GROUP BY, DISTINCT, a row cap with the ORDER BY it ranks by)
|
|
13
|
+
plus the whole chain of expressions behind every column those clauses read,
|
|
14
|
+
through CTEs, derived tables and set operations, down to the physical tables.
|
|
15
|
+
A change anywhere in it widens to the whole model. When sqlglot cannot resolve
|
|
16
|
+
the statement the raw clause text is the floor: the failure mode is a spare
|
|
17
|
+
rebuild, never a missed one.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import sqlglot
|
|
23
|
+
from sqlglot import exp
|
|
24
|
+
from sqlglot.lineage import lineage
|
|
25
|
+
from sqlglot.optimizer.qualify import qualify
|
|
26
|
+
from sqlglot.optimizer.scope import Scope, ScopeType, build_scope
|
|
27
|
+
|
|
28
|
+
from ripple.engine.safe_gen import safe_sql
|
|
29
|
+
|
|
30
|
+
PREDICATES_KEY = "(predicates)"
|
|
31
|
+
CTE_EXPRESSIONS_KEY = "(cte expressions)"
|
|
32
|
+
|
|
33
|
+
_KEY_HOLDERS = (exp.Group, exp.Tuple, exp.GroupingSets, exp.Cube, exp.Rollup, exp.Distinct)
|
|
34
|
+
_CORRELATING = (ScopeType.SUBQUERY, ScopeType.UDTF, ScopeType.UNION)
|
|
35
|
+
# the spellings of a row cap: TOP and FETCH FIRST land in `limit` on most
|
|
36
|
+
# dialects, but not all versions of every one
|
|
37
|
+
_CAPS = ("limit", "offset", "fetch", "top")
|
|
38
|
+
# clauses that reshape the row set without filtering a column
|
|
39
|
+
_ROW_SHAPERS = tuple(
|
|
40
|
+
node
|
|
41
|
+
for name in ("TableSample", "Connect", "Pivot", "Unpivot", "MatchRecognize")
|
|
42
|
+
if (node := getattr(exp, name, None)) is not None
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _lineage_signature(sql: str, dialect: str) -> dict[str, tuple]:
|
|
47
|
+
"""column -> (upstream columns, defining expression, inner chain), plus
|
|
48
|
+
the row-deciding closure entry and the leftover inner expressions entry.
|
|
49
|
+
The expression matters: amount/100.0 to amount changes every downstream
|
|
50
|
+
number. The closure matters more: WHERE status='pending' to 'paid'
|
|
51
|
+
changes every downstream row while touching no column definition."""
|
|
52
|
+
from ripple.engine import extract_lineage_complete
|
|
53
|
+
from ripple.engine.jinja import convert_jinja_to_sql
|
|
54
|
+
from ripple.engine.preprocess import prepare_sql_for_parse
|
|
55
|
+
|
|
56
|
+
result = extract_lineage_complete(sql, dialect=dialect, clean_jinja_func=convert_jinja_to_sql)
|
|
57
|
+
# signature reparses share the engine's pre-parse pipeline, or a braced
|
|
58
|
+
# model diffs as an empty signature (cycle-10 review, F8 class)
|
|
59
|
+
cleaned, _ = prepare_sql_for_parse(sql, dialect, convert_jinja_to_sql)
|
|
60
|
+
expressions = _column_expressions(cleaned, dialect)
|
|
61
|
+
traced = _trace_statement(cleaned, dialect)
|
|
62
|
+
chains, leftover, closure = traced or ({}, _cte_expression_signature(cleaned, dialect), [])
|
|
63
|
+
signature = {
|
|
64
|
+
column: (
|
|
65
|
+
frozenset(f"{s.get('table', '?')}.{s.get('column', '?')}" for s in sources),
|
|
66
|
+
expressions.get(column.lower(), ""),
|
|
67
|
+
chains.get(column.lower(), ""),
|
|
68
|
+
)
|
|
69
|
+
for column, sources in result.contributing.items()
|
|
70
|
+
}
|
|
71
|
+
signature[PREDICATES_KEY] = (frozenset(), _predicate_signature(cleaned, dialect, closure))
|
|
72
|
+
signature[CTE_EXPRESSIONS_KEY] = (frozenset(), leftover)
|
|
73
|
+
return signature
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _trace_statement(cleaned: str, dialect: str) -> tuple[dict[str, str], str, list[str]] | None:
|
|
77
|
+
"""Per output column, the inner expressions its lineage passes through;
|
|
78
|
+
the leftover signature of the inner expressions no output column passes
|
|
79
|
+
through; the row-deciding closure texts. Per column because one hash over
|
|
80
|
+
every CTE item widened any one-column edit in `with source, renamed
|
|
81
|
+
select *` to the whole model (jaffle-shop stg_orders). None when sqlglot
|
|
82
|
+
cannot resolve the statement; the caller falls back to the single hash."""
|
|
83
|
+
try:
|
|
84
|
+
parsed = sqlglot.parse_one(cleaned, dialect=dialect)
|
|
85
|
+
qualified = qualify(parsed, dialect=dialect, validate_qualify_columns=False)
|
|
86
|
+
root = build_scope(qualified)
|
|
87
|
+
if root is None:
|
|
88
|
+
return None
|
|
89
|
+
closure = _Closure(root, dialect)
|
|
90
|
+
closure.run(qualified)
|
|
91
|
+
traced = (
|
|
92
|
+
_lineage_chains(qualified, root, dialect) if isinstance(qualified, exp.Select) else None
|
|
93
|
+
)
|
|
94
|
+
except Exception:
|
|
95
|
+
return None
|
|
96
|
+
if traced is None:
|
|
97
|
+
return {}, _cte_expression_signature(cleaned, dialect), closure.texts
|
|
98
|
+
chains, reached, items = traced
|
|
99
|
+
# items on the closure stay in the leftover so the edit also reports as
|
|
100
|
+
# reworked; two tests pin that label for a CTE flag read by a WHERE
|
|
101
|
+
leftover = sorted(
|
|
102
|
+
item.sql(dialect=dialect).lower() for key, item in items.items() if key not in reached
|
|
103
|
+
)
|
|
104
|
+
return chains, "|".join(leftover), closure.texts
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _lineage_chains(
|
|
108
|
+
qualified: exp.Select, root: Scope, dialect: str
|
|
109
|
+
) -> tuple[dict[str, str], set[int], dict[int, exp.Expression]] | None:
|
|
110
|
+
"""Per output column, the inner items its lineage passes through; the ids
|
|
111
|
+
of every expression any output column reaches; the inner items by id.
|
|
112
|
+
None for an output sqlglot cannot name."""
|
|
113
|
+
items = {
|
|
114
|
+
id(item): item
|
|
115
|
+
for select in qualified.find_all(exp.Select)
|
|
116
|
+
if select is not qualified
|
|
117
|
+
for item in select.expressions
|
|
118
|
+
}
|
|
119
|
+
chains: dict[str, str] = {}
|
|
120
|
+
reached: set[int] = set()
|
|
121
|
+
for item in qualified.expressions:
|
|
122
|
+
name = item.alias_or_name
|
|
123
|
+
if not name or name == "*":
|
|
124
|
+
return None
|
|
125
|
+
stack = list(lineage(name, qualified, scope=root, dialect=dialect).downstream)
|
|
126
|
+
parts = []
|
|
127
|
+
while stack:
|
|
128
|
+
node = stack.pop()
|
|
129
|
+
key = id(node.expression)
|
|
130
|
+
reached.add(key)
|
|
131
|
+
if key in items:
|
|
132
|
+
parts.append(node.expression.sql(dialect=dialect).lower())
|
|
133
|
+
stack.extend(node.downstream)
|
|
134
|
+
chains[name.lower()] = "|".join(sorted(parts))
|
|
135
|
+
return chains, reached, items
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
class _Closure:
|
|
139
|
+
"""One walk over the row-deciding clauses of a qualified statement:
|
|
140
|
+
`texts` is the clause texts plus every expression behind the columns they
|
|
141
|
+
read, `absorbed` the ids of everything already in `texts`."""
|
|
142
|
+
|
|
143
|
+
def __init__(self, root: Scope, dialect: str) -> None:
|
|
144
|
+
self.root = root
|
|
145
|
+
self.dialect = dialect
|
|
146
|
+
self.owners = {id(scope.expression): scope for scope in root.traverse()}
|
|
147
|
+
self.texts: list[str] = []
|
|
148
|
+
self.absorbed: set[int] = set()
|
|
149
|
+
self.asked: set[tuple[int, str, str]] = set()
|
|
150
|
+
|
|
151
|
+
def run(self, qualified: exp.Expression) -> None:
|
|
152
|
+
for clause in _row_deciding_clauses(qualified):
|
|
153
|
+
self.absorb(clause)
|
|
154
|
+
for item in _ordinal_items(clause):
|
|
155
|
+
self.absorb(item)
|
|
156
|
+
|
|
157
|
+
def scope_of(self, node: exp.Expression) -> Scope:
|
|
158
|
+
return self.owners.get(id(node.find_ancestor(exp.Select)), self.root)
|
|
159
|
+
|
|
160
|
+
def ask(self, scope: Scope, table: str, name: str) -> None:
|
|
161
|
+
"""Absorb what defines `table.name` seen from `scope`; a star passes
|
|
162
|
+
the name through to its own sources."""
|
|
163
|
+
key = (id(scope), table.lower(), name.lower())
|
|
164
|
+
if key in self.asked:
|
|
165
|
+
return
|
|
166
|
+
self.asked.add(key)
|
|
167
|
+
for item, item_scope in _defining_items(scope, table.lower(), name.lower()):
|
|
168
|
+
if item.is_star:
|
|
169
|
+
self.ask(item_scope, item.table if isinstance(item, exp.Column) else "", name)
|
|
170
|
+
else:
|
|
171
|
+
self.absorb(item)
|
|
172
|
+
|
|
173
|
+
def absorb(self, item: exp.Expression) -> None:
|
|
174
|
+
if id(item) in self.absorbed:
|
|
175
|
+
return
|
|
176
|
+
self.absorbed.add(id(item))
|
|
177
|
+
self.texts.append(item.sql(dialect=self.dialect).lower())
|
|
178
|
+
for column in item.find_all(exp.Column):
|
|
179
|
+
self.ask(self.scope_of(column), column.table, column.name)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _row_deciding_clauses(statement: exp.Expression):
|
|
183
|
+
"""Every clause in the statement that decides which rows survive or how
|
|
184
|
+
they group."""
|
|
185
|
+
yield from statement.find_all(exp.Where, exp.Having, exp.Qualify, exp.Group)
|
|
186
|
+
yield from statement.find_all(*_ROW_SHAPERS)
|
|
187
|
+
for join in statement.find_all(exp.Join):
|
|
188
|
+
yield from _join_clauses(join)
|
|
189
|
+
for query in statement.find_all(exp.Select, exp.SetOperation):
|
|
190
|
+
yield from _row_set_clauses(query)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _join_clauses(join: exp.Join):
|
|
194
|
+
"""A join decides rows by its kind, its source and its condition. What an
|
|
195
|
+
inline subquery source projects decides none of the three."""
|
|
196
|
+
kinds = " ".join(join.text(part) for part in ("method", "side", "kind"))
|
|
197
|
+
pivots = " ".join(_rendered(pivot) for pivot in join.args.get("pivots") or [])
|
|
198
|
+
shape = f"join {kinds} {_source_shape(join.this)} {pivots}"
|
|
199
|
+
yield exp.Literal.string(" ".join(shape.lower().split()))
|
|
200
|
+
for part in ("on", "match_condition"):
|
|
201
|
+
if (condition := join.args.get(part)) is not None:
|
|
202
|
+
yield condition
|
|
203
|
+
yield from join.args.get("using") or []
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _source_shape(source: exp.Expression | None) -> str:
|
|
207
|
+
"""Where a join source draws its rows from: a table by name, a derived
|
|
208
|
+
table by its own FROM, WHERE and GROUP BY. Anything else renders whole."""
|
|
209
|
+
if source is None:
|
|
210
|
+
return ""
|
|
211
|
+
inner = source.this if isinstance(source, exp.Subquery) else None
|
|
212
|
+
if not isinstance(inner, exp.Select):
|
|
213
|
+
return _rendered(source)
|
|
214
|
+
# "from_" is the sqlglot 30 arg name, "from" the older one
|
|
215
|
+
args = (inner.args.get(name) for name in ("from", "from_", "where", "group"))
|
|
216
|
+
return " ".join(_rendered(arg) for arg in args if arg is not None)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _row_set_clauses(query: exp.Expression):
|
|
220
|
+
"""A row cap or a DISTINCT ON makes the ORDER BY decide which rows
|
|
221
|
+
survive, not just the order they come in. A plain DISTINCT node carries
|
|
222
|
+
no reference to the select list, so every select item decides which rows
|
|
223
|
+
are duplicates."""
|
|
224
|
+
caps = [cap for name in _CAPS if (cap := query.args.get(name)) is not None]
|
|
225
|
+
yield from caps
|
|
226
|
+
if isinstance(query, exp.SetOperation):
|
|
227
|
+
# union keeps one row per duplicate, union all keeps both, and except
|
|
228
|
+
# is not intersect: the operator itself decides the rows
|
|
229
|
+
mode = "all" if query.args.get("distinct") is False else "distinct"
|
|
230
|
+
yield exp.Literal.string(f"{query.key} {mode}".lower())
|
|
231
|
+
return
|
|
232
|
+
distinct = query.args.get("distinct") if isinstance(query, exp.Select) else None
|
|
233
|
+
picks_a_row = bool(caps) or (distinct is not None and distinct.args.get("on") is not None)
|
|
234
|
+
if picks_a_row and (order := query.args.get("order")) is not None:
|
|
235
|
+
yield order
|
|
236
|
+
if distinct is None:
|
|
237
|
+
return
|
|
238
|
+
yield distinct
|
|
239
|
+
if distinct.args.get("on") is None:
|
|
240
|
+
# duplicates are compared by value; the output's name is not part of it
|
|
241
|
+
yield from (item.unalias() for item in query.expressions)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _rendered(node: exp.Expression) -> str:
|
|
245
|
+
return safe_sql(node, None) or ""
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _ordinal_items(clause: exp.Expression) -> list[exp.Expression]:
|
|
249
|
+
"""Select items behind `group by 1` style keys, for the dialects where
|
|
250
|
+
qualify leaves the ordinal in place."""
|
|
251
|
+
select = clause.find_ancestor(exp.Select)
|
|
252
|
+
if select is None or not isinstance(clause, (exp.Group, exp.Distinct)):
|
|
253
|
+
return []
|
|
254
|
+
found = []
|
|
255
|
+
for key in clause.find_all(exp.Literal):
|
|
256
|
+
if key.is_int and isinstance(key.parent, _KEY_HOLDERS):
|
|
257
|
+
index = int(key.this) - 1
|
|
258
|
+
if 0 <= index < len(select.expressions):
|
|
259
|
+
found.append(select.expressions[index])
|
|
260
|
+
return found
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _defining_items(scope: Scope, table: str, name: str) -> list[tuple[exp.Expression, Scope]]:
|
|
264
|
+
"""Items defining `table.name` seen from `scope`: the matching source's
|
|
265
|
+
item, every same-named item when unqualified, the select's own alias
|
|
266
|
+
(HAVING, QUALIFY, GROUP BY). Subqueries walk outward for correlated
|
|
267
|
+
references. Physical tables end the chain."""
|
|
268
|
+
while scope is not None:
|
|
269
|
+
found = [
|
|
270
|
+
pair
|
|
271
|
+
for alias, source in scope.sources.items()
|
|
272
|
+
if isinstance(source, Scope) and (not table or alias.lower() == table)
|
|
273
|
+
for pair in _named_items(source, name)
|
|
274
|
+
]
|
|
275
|
+
if not table and isinstance(scope.expression, exp.Select):
|
|
276
|
+
found += [
|
|
277
|
+
(item, scope)
|
|
278
|
+
for item in scope.expression.expressions
|
|
279
|
+
if item.alias_or_name.lower() == name
|
|
280
|
+
]
|
|
281
|
+
aliases = {alias.lower() for alias in scope.sources}
|
|
282
|
+
if found or table in aliases or scope.scope_type not in _CORRELATING:
|
|
283
|
+
return found
|
|
284
|
+
scope = scope.parent
|
|
285
|
+
return []
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _named_items(source: Scope, name: str) -> list[tuple[exp.Expression, Scope]]:
|
|
289
|
+
"""Items putting `name` on a source's output, stars included; both
|
|
290
|
+
branches at that position for a set operation; a table function whole."""
|
|
291
|
+
expression = source.expression
|
|
292
|
+
if isinstance(expression, exp.SetOperation):
|
|
293
|
+
names = _output_names(source)
|
|
294
|
+
return _items_at(source, names.index(name)) if name in names else []
|
|
295
|
+
if not isinstance(expression, exp.Select):
|
|
296
|
+
return [(expression, source)]
|
|
297
|
+
return [
|
|
298
|
+
(item, source)
|
|
299
|
+
for item in expression.expressions
|
|
300
|
+
if item.is_star or item.alias_or_name.lower() == name
|
|
301
|
+
]
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _output_names(scope: Scope) -> list[str]:
|
|
305
|
+
if isinstance(scope.expression, exp.SetOperation):
|
|
306
|
+
return _output_names(scope.union_scopes[0]) if scope.union_scopes else []
|
|
307
|
+
if isinstance(scope.expression, exp.Select):
|
|
308
|
+
return [item.alias_or_name.lower() for item in scope.expression.expressions]
|
|
309
|
+
return []
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _items_at(scope: Scope, position: int) -> list[tuple[exp.Expression, Scope]]:
|
|
313
|
+
if isinstance(scope.expression, exp.SetOperation):
|
|
314
|
+
return [pair for branch in scope.union_scopes for pair in _items_at(branch, position)]
|
|
315
|
+
items = scope.expression.expressions if isinstance(scope.expression, exp.Select) else []
|
|
316
|
+
return [(items[position], scope)] if position < len(items) else []
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _cte_expression_signature(sql: str, dialect: str) -> str:
|
|
320
|
+
"""Normalized SELECT expressions from every CTE and subquery: the
|
|
321
|
+
whole-model fallback when per-column tracing fails. A formula change
|
|
322
|
+
inside a CTE changes every downstream number while leaving the top-level
|
|
323
|
+
expressions untouched, so it needs its own signature."""
|
|
324
|
+
try:
|
|
325
|
+
parsed = sqlglot.parse_one(sql, dialect=dialect)
|
|
326
|
+
except Exception:
|
|
327
|
+
return ""
|
|
328
|
+
# the top-level select has its own per-column signature
|
|
329
|
+
top = parsed if isinstance(parsed, exp.Select) else parsed.find(exp.Select)
|
|
330
|
+
parts = [
|
|
331
|
+
rendered.lower()
|
|
332
|
+
for select in parsed.find_all(exp.Select)
|
|
333
|
+
if select is not top
|
|
334
|
+
for item in select.expressions
|
|
335
|
+
if (rendered := safe_sql(item, dialect)) is not None
|
|
336
|
+
]
|
|
337
|
+
return "|".join(sorted(parts))
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _predicate_signature(sql: str, dialect: str, closure: list[str]) -> str:
|
|
341
|
+
"""The row-deciding closure plus the raw text of every row-deciding
|
|
342
|
+
clause. The raw text is the floor that survives a failed qualification."""
|
|
343
|
+
parts = list(closure)
|
|
344
|
+
try:
|
|
345
|
+
parsed = sqlglot.parse_one(sql, dialect=dialect)
|
|
346
|
+
except Exception:
|
|
347
|
+
return "|".join(sorted(parts))
|
|
348
|
+
for node in _row_deciding_clauses(parsed):
|
|
349
|
+
rendered = safe_sql(node, dialect)
|
|
350
|
+
if rendered is not None:
|
|
351
|
+
parts.append(rendered.lower())
|
|
352
|
+
return "|".join(sorted(parts))
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _column_expressions(sql: str, dialect: str) -> dict[str, str]:
|
|
356
|
+
"""Normalized top-level SELECT expression per output column."""
|
|
357
|
+
try:
|
|
358
|
+
parsed = sqlglot.parse_one(sql, dialect=dialect)
|
|
359
|
+
except Exception:
|
|
360
|
+
return {}
|
|
361
|
+
select = parsed
|
|
362
|
+
while select is not None and not isinstance(select, exp.Select):
|
|
363
|
+
select = select.this if hasattr(select, "this") else None
|
|
364
|
+
if not isinstance(select, exp.Expression):
|
|
365
|
+
return {}
|
|
366
|
+
if select is None:
|
|
367
|
+
return {}
|
|
368
|
+
out = {}
|
|
369
|
+
for item in select.expressions:
|
|
370
|
+
name = item.alias_or_name
|
|
371
|
+
rendered = safe_sql(item, dialect)
|
|
372
|
+
if name and name != "*" and rendered is not None:
|
|
373
|
+
out[name.lower()] = rendered.lower()
|
|
374
|
+
return out
|