ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,653 @@
|
|
|
1
|
+
"""T-SQL table-valued function lineage.
|
|
2
|
+
|
|
3
|
+
CREATE FUNCTION ... RETURNS @out TABLE (...) mints no model from the
|
|
4
|
+
generic script path: T-SQL bodies separate statements by newline, not
|
|
5
|
+
";", so sqlglot reads the whole file as one Command. This module finds
|
|
6
|
+
the shape in the raw text, chops the body into statements on keyword
|
|
7
|
+
boundaries, and synthesizes the @out fills into one analyzable
|
|
8
|
+
SELECT/UNION: positional INSERT-SELECT mapping, if-guarded fills
|
|
9
|
+
unioned, local table-variable hops inlined to the first external
|
|
10
|
+
boundary. A table-variable read with no visible fill (a TVP parameter,
|
|
11
|
+
or a variable filled by dynamic SQL) stays cited as written with its @
|
|
12
|
+
prefix; dynamic SQL stays invisible (ceds_generate, holdout round 12).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
import sqlglot
|
|
20
|
+
from sqlglot import exp
|
|
21
|
+
from sqlglot.errors import ErrorLevel
|
|
22
|
+
|
|
23
|
+
from ripple.engine.safe_gen import safe_sql
|
|
24
|
+
|
|
25
|
+
_RETURNS_TABLE = re.compile(r"\s*RETURNS\s+(?P<var>@[\w$]+)\s+(?:as\s+)?TABLE\s*\(", re.I)
|
|
26
|
+
# AS is optional in practice: ceds' fnSplit writes RETURNS @List TABLE (...) BEGIN
|
|
27
|
+
_AS_BEGIN = re.compile(r"\s*(?:WITH\s+SCHEMABINDING\s+)?(?:AS\s*)?BEGIN\b", re.I)
|
|
28
|
+
_WORD = re.compile(r"[@#]?[A-Za-z_][\w$]*")
|
|
29
|
+
_IDENT_ATOM = re.compile(r"\[([^\]]+)\]|\"([^\"]+)\"|([A-Za-z_][\w$]*)")
|
|
30
|
+
|
|
31
|
+
_CONSTRAINT_STARTERS = {
|
|
32
|
+
"unique",
|
|
33
|
+
"primary",
|
|
34
|
+
"constraint",
|
|
35
|
+
"check",
|
|
36
|
+
"foreign",
|
|
37
|
+
"index",
|
|
38
|
+
"key",
|
|
39
|
+
"clustered",
|
|
40
|
+
"nonclustered",
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
_STATEMENT_STARTERS = {
|
|
44
|
+
"declare",
|
|
45
|
+
"insert",
|
|
46
|
+
"update",
|
|
47
|
+
"delete",
|
|
48
|
+
"set",
|
|
49
|
+
"if",
|
|
50
|
+
"while",
|
|
51
|
+
"return",
|
|
52
|
+
"merge",
|
|
53
|
+
"print",
|
|
54
|
+
"break",
|
|
55
|
+
"continue",
|
|
56
|
+
"begin",
|
|
57
|
+
"end",
|
|
58
|
+
"else",
|
|
59
|
+
"go",
|
|
60
|
+
"exec",
|
|
61
|
+
"execute",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
_SELECT_CONTINUATIONS = {"union", "all", "except", "intersect", "as", "then", "else", "(", "="}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _skip_atom(s: str, i: int) -> int | None:
|
|
68
|
+
"""End index of the string/comment/bracket atom starting at i, else None."""
|
|
69
|
+
c = s[i]
|
|
70
|
+
if c == "'":
|
|
71
|
+
j = i + 1
|
|
72
|
+
while j < len(s):
|
|
73
|
+
if s[j] == "'":
|
|
74
|
+
if s[j + 1 : j + 2] == "'":
|
|
75
|
+
j += 2
|
|
76
|
+
continue
|
|
77
|
+
return j + 1
|
|
78
|
+
j += 1
|
|
79
|
+
return len(s)
|
|
80
|
+
if c == "[":
|
|
81
|
+
j = i + 1
|
|
82
|
+
while j < len(s):
|
|
83
|
+
if s[j] == "]":
|
|
84
|
+
if s[j + 1 : j + 2] == "]":
|
|
85
|
+
j += 2
|
|
86
|
+
continue
|
|
87
|
+
return j + 1
|
|
88
|
+
j += 1
|
|
89
|
+
return len(s)
|
|
90
|
+
if c == '"':
|
|
91
|
+
j = s.find('"', i + 1)
|
|
92
|
+
return len(s) if j == -1 else j + 1
|
|
93
|
+
if c == "-" and s[i + 1 : i + 2] == "-":
|
|
94
|
+
j = s.find("\n", i)
|
|
95
|
+
return len(s) if j == -1 else j + 1
|
|
96
|
+
if c == "/" and s[i + 1 : i + 2] == "*":
|
|
97
|
+
j = s.find("*/", i)
|
|
98
|
+
return len(s) if j == -1 else j + 2
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _skip_ws_comments(s: str, i: int) -> int:
|
|
103
|
+
"""First index at or after i that is neither whitespace nor a comment."""
|
|
104
|
+
n = len(s)
|
|
105
|
+
while i < n:
|
|
106
|
+
c = s[i]
|
|
107
|
+
if c.isspace():
|
|
108
|
+
i += 1
|
|
109
|
+
continue
|
|
110
|
+
if (c == "-" and s[i + 1 : i + 2] == "-") or (c == "/" and s[i + 1 : i + 2] == "*"):
|
|
111
|
+
i = _skip_atom(s, i)
|
|
112
|
+
continue
|
|
113
|
+
return i
|
|
114
|
+
return i
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _match_word(s: str, i: int, word: str) -> int | None:
|
|
118
|
+
m = _WORD.match(s, i)
|
|
119
|
+
if m is not None and m.group(0).lower() == word:
|
|
120
|
+
return m.end()
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _find_tvf_head(sql: str, pos: int) -> tuple[int, int, str] | None:
|
|
125
|
+
"""The next CREATE [OR ALTER] FUNCTION <name> outside strings and
|
|
126
|
+
comments: (head start, index past the name, written name).
|
|
127
|
+
|
|
128
|
+
A raw regex matched commented-out function text, minted a phantom
|
|
129
|
+
model, and excised the comment; and it refused the legal comment
|
|
130
|
+
between CREATE and FUNCTION (cycle-13 review, F2/F9). This scan skips
|
|
131
|
+
string/comment/bracket atoms and allows comments between tokens."""
|
|
132
|
+
i = pos
|
|
133
|
+
n = len(sql)
|
|
134
|
+
while i < n:
|
|
135
|
+
j = _skip_atom(sql, i)
|
|
136
|
+
if j is not None:
|
|
137
|
+
i = j
|
|
138
|
+
continue
|
|
139
|
+
m = _WORD.match(sql, i)
|
|
140
|
+
if m is None:
|
|
141
|
+
i += 1
|
|
142
|
+
continue
|
|
143
|
+
head_start = m.start()
|
|
144
|
+
i = m.end()
|
|
145
|
+
if m.group(0).lower() != "create":
|
|
146
|
+
continue
|
|
147
|
+
k = _skip_ws_comments(sql, i)
|
|
148
|
+
end = _match_word(sql, k, "or")
|
|
149
|
+
if end is not None:
|
|
150
|
+
end = _match_word(sql, _skip_ws_comments(sql, end), "alter")
|
|
151
|
+
if end is None:
|
|
152
|
+
continue
|
|
153
|
+
k = _skip_ws_comments(sql, end)
|
|
154
|
+
end = _match_word(sql, k, "function")
|
|
155
|
+
if end is None:
|
|
156
|
+
continue
|
|
157
|
+
k = _skip_ws_comments(sql, end)
|
|
158
|
+
parts: list[str] = []
|
|
159
|
+
name_end = k
|
|
160
|
+
while True:
|
|
161
|
+
atom = _IDENT_ATOM.match(sql, k)
|
|
162
|
+
if atom is None:
|
|
163
|
+
parts = []
|
|
164
|
+
break
|
|
165
|
+
parts.append(next(g for g in atom.groups() if g is not None))
|
|
166
|
+
name_end = atom.end()
|
|
167
|
+
after = _skip_ws_comments(sql, name_end)
|
|
168
|
+
if after < n and sql[after] == ".":
|
|
169
|
+
k = _skip_ws_comments(sql, after + 1)
|
|
170
|
+
continue
|
|
171
|
+
break
|
|
172
|
+
if not parts:
|
|
173
|
+
continue
|
|
174
|
+
return head_start, name_end, ".".join(parts)
|
|
175
|
+
return None
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _balanced_paren(s: str, open_idx: int) -> int | None:
|
|
179
|
+
"""Index just past the ')' matching s[open_idx] == '(', else None."""
|
|
180
|
+
depth = 0
|
|
181
|
+
i = open_idx
|
|
182
|
+
while i < len(s):
|
|
183
|
+
j = _skip_atom(s, i)
|
|
184
|
+
if j is not None:
|
|
185
|
+
i = j
|
|
186
|
+
continue
|
|
187
|
+
c = s[i]
|
|
188
|
+
if c == "(":
|
|
189
|
+
depth += 1
|
|
190
|
+
elif c == ")":
|
|
191
|
+
depth -= 1
|
|
192
|
+
if depth == 0:
|
|
193
|
+
return i + 1
|
|
194
|
+
i += 1
|
|
195
|
+
return None
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _body_end(s: str, start: int) -> tuple[int, int]:
|
|
199
|
+
"""(body end, region end) for a body whose opening BEGIN is consumed.
|
|
200
|
+
|
|
201
|
+
Counts BEGIN and CASE as openers so an inner block's or CASE
|
|
202
|
+
expression's END never closes the function early. Unbalanced text
|
|
203
|
+
ends at the string's end."""
|
|
204
|
+
depth = 1
|
|
205
|
+
i = start
|
|
206
|
+
while i < len(s):
|
|
207
|
+
j = _skip_atom(s, i)
|
|
208
|
+
if j is not None:
|
|
209
|
+
i = j
|
|
210
|
+
continue
|
|
211
|
+
m = _WORD.match(s, i)
|
|
212
|
+
if m is None:
|
|
213
|
+
i += 1
|
|
214
|
+
continue
|
|
215
|
+
word = m.group(0).lower()
|
|
216
|
+
if word in ("begin", "case"):
|
|
217
|
+
depth += 1
|
|
218
|
+
elif word == "end":
|
|
219
|
+
depth -= 1
|
|
220
|
+
if depth == 0:
|
|
221
|
+
return m.start(), m.end()
|
|
222
|
+
i = m.end()
|
|
223
|
+
return len(s), len(s)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _column_names(cols_text: str) -> list[str]:
|
|
227
|
+
"""Declared column names, in order, from a table-type parenthesis body.
|
|
228
|
+
|
|
229
|
+
Constraint entries (unique clustered (...), primary key) are skipped;
|
|
230
|
+
a constraint glued to the last column without a comma still yields
|
|
231
|
+
that column's name (the ceds spelling)."""
|
|
232
|
+
segments: list[str] = []
|
|
233
|
+
depth = 0
|
|
234
|
+
start = 0
|
|
235
|
+
i = 0
|
|
236
|
+
while i < len(cols_text):
|
|
237
|
+
j = _skip_atom(cols_text, i)
|
|
238
|
+
if j is not None:
|
|
239
|
+
i = j
|
|
240
|
+
continue
|
|
241
|
+
c = cols_text[i]
|
|
242
|
+
if c == "(":
|
|
243
|
+
depth += 1
|
|
244
|
+
elif c == ")":
|
|
245
|
+
depth -= 1
|
|
246
|
+
elif c == "," and depth == 0:
|
|
247
|
+
segments.append(cols_text[start:i])
|
|
248
|
+
start = i + 1
|
|
249
|
+
i += 1
|
|
250
|
+
segments.append(cols_text[start:])
|
|
251
|
+
names: list[str] = []
|
|
252
|
+
for seg in segments:
|
|
253
|
+
m = _IDENT_ATOM.search(seg)
|
|
254
|
+
if m is None:
|
|
255
|
+
continue
|
|
256
|
+
name = next(g for g in m.groups() if g is not None)
|
|
257
|
+
if name.lower() in _CONSTRAINT_STARTERS:
|
|
258
|
+
continue
|
|
259
|
+
names.append(name)
|
|
260
|
+
return names
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _chop_body(body: str) -> list[str]:
|
|
264
|
+
"""Split a semicolonless T-SQL body into parseable statement chunks.
|
|
265
|
+
|
|
266
|
+
A statement-starter keyword at paren depth 0 (outside any CASE
|
|
267
|
+
expression) opens a new chunk. SELECT is a starter unless it
|
|
268
|
+
continues an INSERT/WITH that has no select yet, or follows a set
|
|
269
|
+
operator. A chunk that still will not parse is skipped by the
|
|
270
|
+
caller, never guessed at."""
|
|
271
|
+
chunks: list[str] = []
|
|
272
|
+
start = 0
|
|
273
|
+
i = 0
|
|
274
|
+
n = len(body)
|
|
275
|
+
paren = 0
|
|
276
|
+
case_depth = 0
|
|
277
|
+
chunk_head: str | None = None
|
|
278
|
+
chunk_has_select = False
|
|
279
|
+
chunk_has_set = False
|
|
280
|
+
last_word = ""
|
|
281
|
+
|
|
282
|
+
def cut(at: int) -> None:
|
|
283
|
+
nonlocal start
|
|
284
|
+
if body[start:at].strip():
|
|
285
|
+
chunks.append(body[start:at])
|
|
286
|
+
start = at
|
|
287
|
+
|
|
288
|
+
while i < n:
|
|
289
|
+
j = _skip_atom(body, i)
|
|
290
|
+
if j is not None:
|
|
291
|
+
i = j
|
|
292
|
+
continue
|
|
293
|
+
c = body[i]
|
|
294
|
+
if c == "(":
|
|
295
|
+
paren += 1
|
|
296
|
+
last_word = "("
|
|
297
|
+
i += 1
|
|
298
|
+
continue
|
|
299
|
+
if c == ")":
|
|
300
|
+
paren = max(0, paren - 1)
|
|
301
|
+
last_word = ")"
|
|
302
|
+
i += 1
|
|
303
|
+
continue
|
|
304
|
+
if c == ";":
|
|
305
|
+
cut(i + 1)
|
|
306
|
+
chunk_head = None
|
|
307
|
+
chunk_has_select = False
|
|
308
|
+
chunk_has_set = False
|
|
309
|
+
last_word = ""
|
|
310
|
+
i += 1
|
|
311
|
+
continue
|
|
312
|
+
m = _WORD.match(body, i)
|
|
313
|
+
if m is None:
|
|
314
|
+
i += 1
|
|
315
|
+
continue
|
|
316
|
+
word = m.group(0).lower()
|
|
317
|
+
if paren == 0:
|
|
318
|
+
if word == "case":
|
|
319
|
+
case_depth += 1
|
|
320
|
+
elif word == "end" and case_depth > 0:
|
|
321
|
+
case_depth -= 1
|
|
322
|
+
elif case_depth == 0:
|
|
323
|
+
split = word in _STATEMENT_STARTERS
|
|
324
|
+
if word == "select":
|
|
325
|
+
continues = (
|
|
326
|
+
chunk_head in ("insert", "with") and not chunk_has_select
|
|
327
|
+
) or last_word in _SELECT_CONTINUATIONS
|
|
328
|
+
split = not continues
|
|
329
|
+
elif word == "insert" and chunk_head == "with" and not chunk_has_select:
|
|
330
|
+
# the INSERT consuming an open WITH chunk continues it;
|
|
331
|
+
# cutting there discarded the CTE (cycle-13 review, F10)
|
|
332
|
+
split = False
|
|
333
|
+
elif word == "set" and chunk_head == "update" and not chunk_has_set:
|
|
334
|
+
# an UPDATE's own SET clause; cutting there severed the
|
|
335
|
+
# update from its assignments (cycle-13 review, F4)
|
|
336
|
+
split = False
|
|
337
|
+
if split:
|
|
338
|
+
cut(i)
|
|
339
|
+
chunk_head = word
|
|
340
|
+
chunk_has_select = False
|
|
341
|
+
chunk_has_set = False
|
|
342
|
+
elif chunk_head is None:
|
|
343
|
+
chunk_head = word
|
|
344
|
+
if word == "select":
|
|
345
|
+
chunk_has_select = True
|
|
346
|
+
elif word == "set" and chunk_head == "update":
|
|
347
|
+
chunk_has_set = True
|
|
348
|
+
last_word = word
|
|
349
|
+
i = m.end()
|
|
350
|
+
cut(n)
|
|
351
|
+
return chunks
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _parse_chunks(chunks: list[str]) -> list[exp.Expression]:
|
|
355
|
+
from ripple.engine.sql_script import _flatten_blocks
|
|
356
|
+
|
|
357
|
+
stmts: list[exp.Expression] = []
|
|
358
|
+
for chunk in chunks:
|
|
359
|
+
text = chunk.strip()
|
|
360
|
+
if not text:
|
|
361
|
+
continue
|
|
362
|
+
try:
|
|
363
|
+
parsed = sqlglot.parse_one(text, dialect="tsql", error_level=ErrorLevel.IGNORE)
|
|
364
|
+
except Exception:
|
|
365
|
+
continue
|
|
366
|
+
if parsed is not None:
|
|
367
|
+
stmts.append(parsed)
|
|
368
|
+
return _flatten_blocks(stmts, "tsql")
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _tablevar_name(node) -> str | None:
|
|
372
|
+
""" "@out" for a Table (or Schema-wrapped Table) over a Parameter."""
|
|
373
|
+
if isinstance(node, exp.Schema):
|
|
374
|
+
node = node.this
|
|
375
|
+
if isinstance(node, exp.Table) and isinstance(node.this, exp.Parameter):
|
|
376
|
+
name = node.this.name
|
|
377
|
+
return f"@{name}".lower() if name else None
|
|
378
|
+
return None
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _is_star_item(item: exp.Expression) -> bool:
|
|
382
|
+
return isinstance(item, exp.Star) or (
|
|
383
|
+
isinstance(item, exp.Column) and isinstance(item.this, exp.Star)
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
class _TvfBody:
|
|
388
|
+
def __init__(self, stmts: list[exp.Expression], out_var: str, out_columns: list[str]):
|
|
389
|
+
self.declared: dict[str, list[str]] = {out_var: out_columns}
|
|
390
|
+
# fills carry their statement position: a read of a table variable
|
|
391
|
+
# inlines only fills from EARLIER statements, because a later fill
|
|
392
|
+
# cannot retroactively reach an earlier read (cycle-13 review, F5)
|
|
393
|
+
self.fills: dict[str, list[tuple[int, list[str] | None, exp.Expression]]] = {}
|
|
394
|
+
self.poisoned: dict[str, set[str]] = {}
|
|
395
|
+
self.opaque: set[str] = set()
|
|
396
|
+
for pos, stmt in enumerate(stmts):
|
|
397
|
+
if isinstance(stmt, exp.Declare):
|
|
398
|
+
for item in stmt.expressions:
|
|
399
|
+
kind = item.args.get("kind")
|
|
400
|
+
if not isinstance(kind, exp.Schema):
|
|
401
|
+
continue
|
|
402
|
+
cols = [c.name for c in kind.expressions if isinstance(c, exp.ColumnDef)]
|
|
403
|
+
for param in item.this if isinstance(item.this, list) else [item.this]:
|
|
404
|
+
if isinstance(param, exp.Parameter) and param.name:
|
|
405
|
+
self.declared[f"@{param.name}".lower()] = cols
|
|
406
|
+
elif isinstance(stmt, exp.Insert):
|
|
407
|
+
var = _tablevar_name(stmt.this)
|
|
408
|
+
if var is None:
|
|
409
|
+
continue
|
|
410
|
+
idents = None
|
|
411
|
+
if isinstance(stmt.this, exp.Schema):
|
|
412
|
+
names = [c.name for c in stmt.this.expressions if isinstance(c, exp.Identifier)]
|
|
413
|
+
if len(names) == len(stmt.this.expressions):
|
|
414
|
+
idents = names
|
|
415
|
+
inner = stmt.expression
|
|
416
|
+
while isinstance(inner, (exp.Subquery, exp.Paren)):
|
|
417
|
+
inner = inner.this
|
|
418
|
+
if isinstance(inner, (exp.Select, exp.Union)):
|
|
419
|
+
with_clause = stmt.args.get("with_") or stmt.args.get("with")
|
|
420
|
+
if (
|
|
421
|
+
with_clause is not None
|
|
422
|
+
and isinstance(inner, exp.Select)
|
|
423
|
+
and not (inner.args.get("with_") or inner.args.get("with"))
|
|
424
|
+
):
|
|
425
|
+
# the chopper keeps WITH ... INSERT together; the CTE
|
|
426
|
+
# must ride on the fill or its reads dangle (F10)
|
|
427
|
+
inner = inner.copy()
|
|
428
|
+
inner.set("with_", with_clause.copy())
|
|
429
|
+
self.fills.setdefault(var, []).append((pos, idents, inner))
|
|
430
|
+
elif isinstance(stmt, exp.Update):
|
|
431
|
+
self._add_update(pos, stmt)
|
|
432
|
+
|
|
433
|
+
def _add_update(self, pos: int, stmt: exp.Update) -> None:
|
|
434
|
+
"""UPDATE @var SET col = expr is one more fill for col, its sources
|
|
435
|
+
unioning with earlier fills (cycle-13 review, F4). A SET shape that
|
|
436
|
+
cannot be synthesized degrades the touched columns to no value
|
|
437
|
+
edges instead of leaving a stale insert source as the answer."""
|
|
438
|
+
var = _tablevar_name(stmt.this)
|
|
439
|
+
if var is None:
|
|
440
|
+
return
|
|
441
|
+
items: list[exp.Alias] = []
|
|
442
|
+
cols: list[str] = []
|
|
443
|
+
unknown_touch = False
|
|
444
|
+
for eq in stmt.expressions:
|
|
445
|
+
if isinstance(eq, exp.EQ) and isinstance(eq.this, exp.Parameter):
|
|
446
|
+
continue # a scalar-variable assignment, not a column fill
|
|
447
|
+
if (
|
|
448
|
+
isinstance(eq, exp.EQ)
|
|
449
|
+
and isinstance(eq.this, exp.Column)
|
|
450
|
+
and not eq.this.table
|
|
451
|
+
and eq.expression is not None
|
|
452
|
+
):
|
|
453
|
+
cols.append(eq.this.name)
|
|
454
|
+
items.append(
|
|
455
|
+
exp.Alias(this=eq.expression.copy(), alias=exp.to_identifier(eq.this.name))
|
|
456
|
+
)
|
|
457
|
+
else:
|
|
458
|
+
unknown_touch = True
|
|
459
|
+
if unknown_touch:
|
|
460
|
+
self._poison(var, None)
|
|
461
|
+
return
|
|
462
|
+
if not items:
|
|
463
|
+
return
|
|
464
|
+
select = exp.Select(expressions=items)
|
|
465
|
+
upd_from = stmt.args.get("from_") or stmt.args.get("from")
|
|
466
|
+
if upd_from is not None:
|
|
467
|
+
select.set("from_", upd_from.copy())
|
|
468
|
+
self.fills.setdefault(var, []).append((pos, cols, select))
|
|
469
|
+
|
|
470
|
+
def _poison(self, var: str, cols: list[str] | None) -> None:
|
|
471
|
+
if cols:
|
|
472
|
+
self.poisoned.setdefault(var, set()).update(c.lower() for c in cols)
|
|
473
|
+
return
|
|
474
|
+
declared = self.declared.get(var)
|
|
475
|
+
if declared:
|
|
476
|
+
self.poisoned.setdefault(var, set()).update(c.lower() for c in declared)
|
|
477
|
+
else:
|
|
478
|
+
self.opaque.add(var)
|
|
479
|
+
|
|
480
|
+
def synthesize(
|
|
481
|
+
self, var: str, seen: frozenset[str], before: int | None = None
|
|
482
|
+
) -> exp.Expression | None:
|
|
483
|
+
if var in self.opaque:
|
|
484
|
+
return None
|
|
485
|
+
entries = [e for e in self.fills.get(var, []) if before is None or e[0] < before]
|
|
486
|
+
if not entries:
|
|
487
|
+
return None
|
|
488
|
+
branches: list[exp.Expression] = []
|
|
489
|
+
for pos, idents, select in entries:
|
|
490
|
+
node = self._map_fill(select, idents or self.declared.get(var, []))
|
|
491
|
+
if node is None:
|
|
492
|
+
continue
|
|
493
|
+
self._inline_tablevars(node, seen | {var}, pos)
|
|
494
|
+
branches.append(node)
|
|
495
|
+
if not branches:
|
|
496
|
+
return None
|
|
497
|
+
dead = self.poisoned.get(var)
|
|
498
|
+
if dead:
|
|
499
|
+
for branch in branches:
|
|
500
|
+
_null_out_columns(branch, dead)
|
|
501
|
+
combined = branches[0]
|
|
502
|
+
for branch in branches[1:]:
|
|
503
|
+
combined = exp.union(combined, branch, distinct=False, copy=False)
|
|
504
|
+
return combined
|
|
505
|
+
|
|
506
|
+
def _map_fill(self, select: exp.Expression, idents: list[str]) -> exp.Expression | None:
|
|
507
|
+
"""The fill re-aliased to the insert's (or declared) column list,
|
|
508
|
+
pairwise. Refuses instead of keeping the projection as written: a
|
|
509
|
+
kept fill left source names as the function's output schema and
|
|
510
|
+
minted confident edges to nonexistent return columns (cycle-13
|
|
511
|
+
review, F3). Union branches map individually. A lone star expands
|
|
512
|
+
positionally only through a locally declared table variable whose
|
|
513
|
+
columns are known; any other star refuses."""
|
|
514
|
+
if isinstance(select, exp.Union):
|
|
515
|
+
left = self._map_fill(select.this, idents)
|
|
516
|
+
right = self._map_fill(select.expression, idents)
|
|
517
|
+
if left is None or right is None:
|
|
518
|
+
return None
|
|
519
|
+
out = select.copy()
|
|
520
|
+
out.set("this", left)
|
|
521
|
+
out.set("expression", right)
|
|
522
|
+
return out
|
|
523
|
+
if not isinstance(select, exp.Select) or not idents:
|
|
524
|
+
return None
|
|
525
|
+
work = select.copy()
|
|
526
|
+
if any(_is_star_item(s) for s in work.selects) and not self._expand_star(work):
|
|
527
|
+
return None
|
|
528
|
+
if len(work.selects) != len(idents):
|
|
529
|
+
return None
|
|
530
|
+
realiased = []
|
|
531
|
+
for item, ident in zip(work.selects, idents, strict=True):
|
|
532
|
+
base = item.this if isinstance(item, exp.Alias) else item
|
|
533
|
+
realiased.append(exp.Alias(this=base, alias=exp.to_identifier(ident)))
|
|
534
|
+
work.set("expressions", realiased)
|
|
535
|
+
return work
|
|
536
|
+
|
|
537
|
+
def _expand_star(self, select: exp.Select) -> bool:
|
|
538
|
+
if len(select.selects) != 1:
|
|
539
|
+
return False
|
|
540
|
+
from_clause = select.args.get("from_") or select.args.get("from")
|
|
541
|
+
if from_clause is None or select.args.get("joins"):
|
|
542
|
+
return False
|
|
543
|
+
tables = list(from_clause.find_all(exp.Table))
|
|
544
|
+
if len(tables) != 1:
|
|
545
|
+
return False
|
|
546
|
+
var = _tablevar_name(tables[0])
|
|
547
|
+
cols = self.declared.get(var) if var else None
|
|
548
|
+
if not cols:
|
|
549
|
+
return False
|
|
550
|
+
select.set("expressions", [exp.column(c) for c in cols])
|
|
551
|
+
return True
|
|
552
|
+
|
|
553
|
+
def _inline_tablevars(self, tree: exp.Expression, seen: frozenset[str], pos: int) -> None:
|
|
554
|
+
"""Replace reads of locally filled table variables with their own
|
|
555
|
+
synthesized derivation from fills before this statement; a variable
|
|
556
|
+
with no reachable fill (a TVP parameter, dynamic SQL, a fill that
|
|
557
|
+
only happens later) stays cited as written."""
|
|
558
|
+
for table in list(tree.find_all(exp.Table)):
|
|
559
|
+
var = _tablevar_name(table)
|
|
560
|
+
if var is None or var in seen or var not in self.fills:
|
|
561
|
+
continue
|
|
562
|
+
sub = self.synthesize(var, seen, before=pos)
|
|
563
|
+
if sub is None:
|
|
564
|
+
continue
|
|
565
|
+
alias = table.alias or var.lstrip("@")
|
|
566
|
+
table.replace(
|
|
567
|
+
exp.Subquery(this=sub, alias=exp.TableAlias(this=exp.to_identifier(alias)))
|
|
568
|
+
)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def _null_out_columns(node: exp.Expression, dead: set[str]) -> None:
|
|
572
|
+
if isinstance(node, exp.Union):
|
|
573
|
+
_null_out_columns(node.this, dead)
|
|
574
|
+
_null_out_columns(node.expression, dead)
|
|
575
|
+
return
|
|
576
|
+
if not isinstance(node, exp.Select):
|
|
577
|
+
return
|
|
578
|
+
for item in list(node.expressions):
|
|
579
|
+
if isinstance(item, exp.Alias) and item.alias and item.alias.lower() in dead:
|
|
580
|
+
item.replace(exp.Alias(this=exp.Null(), alias=exp.to_identifier(item.alias)))
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def expand_table_functions(sql: str, warnings: list[str] | None = None):
|
|
584
|
+
"""(remaining sql, models): one ScriptModel per RETURNS @table function,
|
|
585
|
+
named exactly as the file writes it, its region removed from the text."""
|
|
586
|
+
from ripple.engine.sql_script import ScriptModel, _referenced_tables
|
|
587
|
+
|
|
588
|
+
models: list[ScriptModel] = []
|
|
589
|
+
regions: list[tuple[int, int]] = []
|
|
590
|
+
pos = 0
|
|
591
|
+
while True:
|
|
592
|
+
head = _find_tvf_head(sql, pos)
|
|
593
|
+
if head is None:
|
|
594
|
+
break
|
|
595
|
+
head_start, name_end, written_name = head
|
|
596
|
+
pos = name_end
|
|
597
|
+
i = _skip_ws_comments(sql, name_end)
|
|
598
|
+
if i < len(sql) and sql[i] == "(":
|
|
599
|
+
closed = _balanced_paren(sql, i)
|
|
600
|
+
if closed is None:
|
|
601
|
+
continue
|
|
602
|
+
i = closed
|
|
603
|
+
returns = _RETURNS_TABLE.match(sql, _skip_ws_comments(sql, i))
|
|
604
|
+
if returns is None:
|
|
605
|
+
continue # a scalar function keeps the generic path
|
|
606
|
+
cols_close = _balanced_paren(sql, returns.end() - 1)
|
|
607
|
+
if cols_close is None:
|
|
608
|
+
continue
|
|
609
|
+
as_begin = _AS_BEGIN.match(sql, _skip_ws_comments(sql, cols_close))
|
|
610
|
+
if as_begin is None:
|
|
611
|
+
continue
|
|
612
|
+
body_end, region_end = _body_end(sql, as_begin.end())
|
|
613
|
+
|
|
614
|
+
out_var = returns.group("var").lower()
|
|
615
|
+
out_columns = _column_names(sql[returns.end() : cols_close - 1])
|
|
616
|
+
body = _TvfBody(
|
|
617
|
+
_parse_chunks(_chop_body(sql[as_begin.end() : body_end])), out_var, out_columns
|
|
618
|
+
)
|
|
619
|
+
synthesized = body.synthesize(out_var, frozenset())
|
|
620
|
+
derived_sql = safe_sql(synthesized, "tsql") if synthesized is not None else None
|
|
621
|
+
if derived_sql is None:
|
|
622
|
+
# return columns stay addressable even when no fill is traceable
|
|
623
|
+
if synthesized is not None and warnings is not None:
|
|
624
|
+
warnings.append(
|
|
625
|
+
f"FUNCTION {written_name}: its synthesized body could not be "
|
|
626
|
+
"regenerated; return columns are schema-only"
|
|
627
|
+
)
|
|
628
|
+
models.append(
|
|
629
|
+
ScriptModel(
|
|
630
|
+
name=written_name, columns=out_columns, is_schema=True, is_function=True
|
|
631
|
+
)
|
|
632
|
+
)
|
|
633
|
+
else:
|
|
634
|
+
models.append(
|
|
635
|
+
ScriptModel(
|
|
636
|
+
name=written_name,
|
|
637
|
+
sql=derived_sql,
|
|
638
|
+
parents=_referenced_tables(synthesized),
|
|
639
|
+
columns=out_columns,
|
|
640
|
+
is_function=True,
|
|
641
|
+
)
|
|
642
|
+
)
|
|
643
|
+
regions.append((head_start, region_end))
|
|
644
|
+
pos = region_end
|
|
645
|
+
if regions:
|
|
646
|
+
parts = []
|
|
647
|
+
prev = 0
|
|
648
|
+
for start, end in regions:
|
|
649
|
+
parts.append(sql[prev:start])
|
|
650
|
+
prev = end
|
|
651
|
+
parts.append(sql[prev:])
|
|
652
|
+
sql = "".join(parts)
|
|
653
|
+
return sql, models
|