ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,729 @@
|
|
|
1
|
+
"""Multi-statement SQL scripts and dumps.
|
|
2
|
+
|
|
3
|
+
Real non-dbt SQL is rarely one statement per file. A Chinook dump or a Postgres
|
|
4
|
+
sample DB is a single file with hundreds to thousands of statements: DROP, CREATE
|
|
5
|
+
TABLE, INSERT, CREATE VIEW, mixed with psql `\\` meta-commands that aren't SQL at
|
|
6
|
+
all. This module splits such a file into statements resiliently (never raising),
|
|
7
|
+
sniffs the dialect from content when there's no dbt adapter to ask, and turns the
|
|
8
|
+
derivation-bearing statements into models with schema.
|
|
9
|
+
|
|
10
|
+
Nothing here fabricates lineage. A statement that will not parse is skipped, and a
|
|
11
|
+
CREATE TABLE with no SELECT contributes only its declared columns (the schema), not
|
|
12
|
+
a made-up edge.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
import sqlglot
|
|
20
|
+
from sqlglot import exp
|
|
21
|
+
from sqlglot.errors import ErrorLevel
|
|
22
|
+
|
|
23
|
+
from ripple.engine.safe_gen import safe_sql
|
|
24
|
+
|
|
25
|
+
# psql / mysql-cli meta-commands: whole lines starting with a backslash (\pset, \c,
|
|
26
|
+
# \d, \.), which are client directives, not SQL. Stripped before parsing.
|
|
27
|
+
_META_LINE = re.compile(r"^[ \t]*\\[a-zA-Z.].*$", re.M)
|
|
28
|
+
|
|
29
|
+
# T-SQL's GO is the same class of client directive: a batch separator sqlcmd
|
|
30
|
+
# understands but the server never sees. Left in place it lumps the next
|
|
31
|
+
# batch into one unparseable Command ("GO\nCREATE VIEW ...", fhir_server,
|
|
32
|
+
# holdout round 11). Rewritten to ";" so batch boundaries survive. The full
|
|
33
|
+
# separator grammar: case-insensitive GO, optional count, optional trailing
|
|
34
|
+
# line or block comment (cycle-12 review, F12). Applied per line by the
|
|
35
|
+
# quote/comment-state scan below, never as a blind substitution, so a GO
|
|
36
|
+
# inside a bracketed identifier or string stays part of the name (F1).
|
|
37
|
+
_GO_LINE = re.compile(
|
|
38
|
+
r"^[ \t]*GO(?:[ \t]+\d+)?[ \t]*(?:--[^\r\n]*|/\*.*?\*/[ \t]*)?\r?$",
|
|
39
|
+
re.I,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _rewrite_go_separators(sql: str) -> str:
|
|
44
|
+
"""Replace batch-separator GO lines with ";", tracking quote and comment
|
|
45
|
+
state so quoted identifiers and comments are never rewritten."""
|
|
46
|
+
out: list[str] = []
|
|
47
|
+
state = "" # "" | "'" | '"' | "`" | "[" | "/*"
|
|
48
|
+
for line in sql.split("\n"):
|
|
49
|
+
if not state and _GO_LINE.match(line):
|
|
50
|
+
out.append(";")
|
|
51
|
+
continue
|
|
52
|
+
i, n = 0, len(line)
|
|
53
|
+
while i < n:
|
|
54
|
+
c = line[i]
|
|
55
|
+
if state == "/*":
|
|
56
|
+
if c == "*" and line[i + 1 : i + 2] == "/":
|
|
57
|
+
state = ""
|
|
58
|
+
i += 2
|
|
59
|
+
continue
|
|
60
|
+
i += 1
|
|
61
|
+
continue
|
|
62
|
+
if state == "[":
|
|
63
|
+
if c == "]":
|
|
64
|
+
if line[i + 1 : i + 2] == "]": # ]] escapes ] inside the name
|
|
65
|
+
i += 2
|
|
66
|
+
continue
|
|
67
|
+
state = ""
|
|
68
|
+
i += 1
|
|
69
|
+
continue
|
|
70
|
+
if state:
|
|
71
|
+
if c == state:
|
|
72
|
+
# a doubled quote reads as close-then-reopen; same state
|
|
73
|
+
state = ""
|
|
74
|
+
i += 1
|
|
75
|
+
continue
|
|
76
|
+
if c == "-" and line[i + 1 : i + 2] == "-":
|
|
77
|
+
break # the rest of the line is a comment
|
|
78
|
+
if c == "/" and line[i + 1 : i + 2] == "*":
|
|
79
|
+
state = "/*"
|
|
80
|
+
i += 2
|
|
81
|
+
continue
|
|
82
|
+
if c in "'\"`[":
|
|
83
|
+
state = c
|
|
84
|
+
i += 1
|
|
85
|
+
out.append(line)
|
|
86
|
+
return "\n".join(out)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# Markers that only one dialect writes, weighted by how exclusive they are.
|
|
90
|
+
# A backtick scores for both BigQuery and MySQL because both quote with it;
|
|
91
|
+
# reading it as MySQL alone is what made filecoin-data-portal, a BigQuery
|
|
92
|
+
# project, parse as the wrong dialect.
|
|
93
|
+
_DIALECT_MARKERS: tuple[tuple[str, str, int, int], ...] = (
|
|
94
|
+
# (dialect, pattern, weight, re flags)
|
|
95
|
+
("mysql", r"\bENGINE\s*=\s*\w+|\bAUTO_INCREMENT\b|\bUNSIGNED\b", 4, re.I),
|
|
96
|
+
("mysql", r"\bTINYINT\b|\bLONGTEXT\b|\bDATETIME\(\d\)", 2, re.I),
|
|
97
|
+
("bigquery", r"\bINFORMATION_SCHEMA\b", 3, 0),
|
|
98
|
+
("bigquery", r"\bSTRUCT\s*<|\bARRAY\s*<", 3, re.I),
|
|
99
|
+
("bigquery", r"\bSAFE[._]\w+", 3, re.I),
|
|
100
|
+
("bigquery", r"`[\w-]+\.[\w-]+\.[\w-]+`", 3, 0),
|
|
101
|
+
# types and functions BigQuery alone spells this way
|
|
102
|
+
("bigquery", r"\bFLOAT64\b|\bBIGNUMERIC\b|\bINT64\b", 3, re.I),
|
|
103
|
+
("bigquery", r"\bTIMESTAMP_SECONDS\s*\(|\bCOUNTIF\s*\(|\bSAFE_DIVIDE\s*\(", 3, re.I),
|
|
104
|
+
("bigquery", r"\bGENERATE_(DATE|TIMESTAMP)_ARRAY\s*\(|\bUNNEST\s*\(", 2, re.I),
|
|
105
|
+
# the GO alternative mirrors _GO_LINE's full separator grammar,
|
|
106
|
+
# case-insensitive via a scoped flag (cycle-12 review, F13)
|
|
107
|
+
(
|
|
108
|
+
"tsql",
|
|
109
|
+
r"\bIDENTITY\s*\(|\bNVARCHAR\b|\bGETDATE\s*\("
|
|
110
|
+
r"|(?i:^[ \t]*GO(?:[ \t]+\d+)?[ \t]*(?:--[^\r\n]*|/\*.*?\*/[ \t]*)?\r?$)",
|
|
111
|
+
4,
|
|
112
|
+
re.M,
|
|
113
|
+
),
|
|
114
|
+
("snowflake", r"\bVARIANT\b|\bLATERAL\s+FLATTEN\b|\bQUALIFY\b", 3, re.I),
|
|
115
|
+
# a notebook exported from Databricks announces itself on line one;
|
|
116
|
+
# dbdemos-notebooks was read as tsql without this
|
|
117
|
+
("databricks", r"^-- Databricks notebook source\b", 6, re.M),
|
|
118
|
+
(
|
|
119
|
+
"databricks",
|
|
120
|
+
r"\bUSING\s+DELTA\b|\bTBLPROPERTIES\b|\bLIVE\s+TABLE\b|\bSTREAMING\s+TABLE\b",
|
|
121
|
+
3,
|
|
122
|
+
re.I,
|
|
123
|
+
),
|
|
124
|
+
# stl_/svl_/svv_/stv_ system tables and DISTKEY/SORTKEY exist only on
|
|
125
|
+
# Redshift; dbt-labs/redshift has no profile and guessed snowflake
|
|
126
|
+
("redshift", r"\b(stl|svl|svv|stv)_\w+", 3, re.I),
|
|
127
|
+
("redshift", r"\bDISTKEY\b|\bSORTKEY\b|\bDISTSTYLE\b", 4, re.I),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
# One backtick proves quoting style, not vendor: split evenly so a stronger
|
|
131
|
+
# marker decides.
|
|
132
|
+
_AMBIGUOUS_BACKTICK = ("bigquery", "mysql")
|
|
133
|
+
|
|
134
|
+
_MIN_DIALECT_SCORE = 3
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def sniff_dialect(sql: str) -> str | None:
|
|
138
|
+
"""Best-effort dialect from the SQL text itself, for repos with no dbt adapter
|
|
139
|
+
to declare one. Returns None when there's no strong signal (caller keeps its
|
|
140
|
+
default and says it assumed). Deliberately conservative: a wrong guess here is
|
|
141
|
+
what makes a real repo misparse.
|
|
142
|
+
|
|
143
|
+
Weighs every marker in the whole sample rather than returning on the first
|
|
144
|
+
hit. The caller assembles a slice of every file on purpose, so truncating
|
|
145
|
+
here threw that away: filecoin-data-portal's BigQuery markers sat past the
|
|
146
|
+
old 20,000 character cut and it was read as postgres.
|
|
147
|
+
"""
|
|
148
|
+
if re.search(r"\\(pset|connect|echo|copy|d[itsv]?\b|c\b)", sql):
|
|
149
|
+
return "postgres" # psql directives cannot appear in another dialect
|
|
150
|
+
|
|
151
|
+
scores: dict[str, int] = {}
|
|
152
|
+
for dialect, pattern, weight, flags in _DIALECT_MARKERS:
|
|
153
|
+
if re.search(pattern, sql, flags):
|
|
154
|
+
scores[dialect] = scores.get(dialect, 0) + weight
|
|
155
|
+
if "`" in sql:
|
|
156
|
+
for dialect in _AMBIGUOUS_BACKTICK:
|
|
157
|
+
scores[dialect] = scores.get(dialect, 0) + 1
|
|
158
|
+
|
|
159
|
+
if not scores:
|
|
160
|
+
return None
|
|
161
|
+
best = max(scores, key=lambda d: (scores[d], d))
|
|
162
|
+
if scores[best] < _MIN_DIALECT_SCORE:
|
|
163
|
+
return None
|
|
164
|
+
return best
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _naive_split(sql: str):
|
|
168
|
+
"""Split on semicolons that are not inside a string or comment. A fallback for
|
|
169
|
+
when sqlglot's own tokenizer refuses the whole blob."""
|
|
170
|
+
out, buf = [], []
|
|
171
|
+
i, n = 0, len(sql)
|
|
172
|
+
quote = None
|
|
173
|
+
while i < n:
|
|
174
|
+
c = sql[i]
|
|
175
|
+
if quote:
|
|
176
|
+
buf.append(c)
|
|
177
|
+
if c == quote:
|
|
178
|
+
quote = None
|
|
179
|
+
i += 1
|
|
180
|
+
continue
|
|
181
|
+
if c in ("'", '"', "`"):
|
|
182
|
+
quote = c
|
|
183
|
+
buf.append(c)
|
|
184
|
+
elif c == "-" and i + 1 < n and sql[i + 1] == "-":
|
|
185
|
+
j = sql.find("\n", i)
|
|
186
|
+
j = n if j == -1 else j
|
|
187
|
+
buf.append(sql[i:j])
|
|
188
|
+
i = j
|
|
189
|
+
continue
|
|
190
|
+
elif c == "/" and i + 1 < n and sql[i + 1] == "*":
|
|
191
|
+
j = sql.find("*/", i)
|
|
192
|
+
j = n if j == -1 else j + 2
|
|
193
|
+
buf.append(sql[i:j])
|
|
194
|
+
i = j
|
|
195
|
+
continue
|
|
196
|
+
elif c == ";":
|
|
197
|
+
out.append("".join(buf))
|
|
198
|
+
buf = []
|
|
199
|
+
else:
|
|
200
|
+
buf.append(c)
|
|
201
|
+
i += 1
|
|
202
|
+
if buf:
|
|
203
|
+
out.append("".join(buf))
|
|
204
|
+
return out
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _flatten_blocks(
|
|
208
|
+
stmts: list[exp.Expression],
|
|
209
|
+
dialect: str | None,
|
|
210
|
+
_in_block: bool = False,
|
|
211
|
+
_conditional: bool = False,
|
|
212
|
+
) -> list[exp.Expression]:
|
|
213
|
+
"""Statements inside conditional/compound blocks are the file's real
|
|
214
|
+
content: the T-SQL deployment idiom IF NOT EXISTS (...) BEGIN CREATE
|
|
215
|
+
TABLE ... END guards deployment, not identity, and skipping it left
|
|
216
|
+
maintenance_solution with no models at all (holdout round 7). A
|
|
217
|
+
statement the guard parse left as a raw Command (often with the
|
|
218
|
+
block's stray END attached) gets one bounded re-parse.
|
|
219
|
+
|
|
220
|
+
Statements out of an IfBlock branch are marked ripple_conditional:
|
|
221
|
+
they may or may not run, so downstream reaching-definition analysis
|
|
222
|
+
must not let them kill or uniquely supply an earlier definition."""
|
|
223
|
+
out: list[exp.Expression] = []
|
|
224
|
+
for stmt in stmts:
|
|
225
|
+
block_cls = getattr(exp, "Block", None)
|
|
226
|
+
if_cls = getattr(exp, "IfBlock", None)
|
|
227
|
+
if if_cls is not None and isinstance(stmt, if_cls):
|
|
228
|
+
branches = [stmt.args.get("true"), stmt.args.get("false")]
|
|
229
|
+
inner = [b for b in branches if isinstance(b, exp.Expression)]
|
|
230
|
+
out.extend(_flatten_blocks(inner, dialect, _in_block=True, _conditional=True))
|
|
231
|
+
elif block_cls is not None and isinstance(stmt, block_cls):
|
|
232
|
+
out.extend(
|
|
233
|
+
_flatten_blocks(
|
|
234
|
+
list(stmt.expressions or []),
|
|
235
|
+
dialect,
|
|
236
|
+
_in_block=True,
|
|
237
|
+
_conditional=_conditional,
|
|
238
|
+
)
|
|
239
|
+
)
|
|
240
|
+
elif _in_block and isinstance(stmt, exp.Command):
|
|
241
|
+
raw = safe_sql(stmt, dialect)
|
|
242
|
+
if raw is None:
|
|
243
|
+
out.append(stmt)
|
|
244
|
+
continue
|
|
245
|
+
text = re.sub(r"\s*END\s*$", "", raw, flags=re.I)
|
|
246
|
+
try:
|
|
247
|
+
reparsed = sqlglot.parse_one(text, dialect=dialect, error_level=ErrorLevel.IGNORE)
|
|
248
|
+
except Exception:
|
|
249
|
+
reparsed = None
|
|
250
|
+
if reparsed is not None and not isinstance(reparsed, exp.Command):
|
|
251
|
+
out.append(reparsed)
|
|
252
|
+
else:
|
|
253
|
+
out.append(stmt)
|
|
254
|
+
else:
|
|
255
|
+
out.append(stmt)
|
|
256
|
+
if _conditional:
|
|
257
|
+
for stmt in out:
|
|
258
|
+
if stmt is not None:
|
|
259
|
+
stmt.meta["ripple_conditional"] = True
|
|
260
|
+
return out
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def split_statements(sql: str, dialect: str | None) -> list[exp.Expression]:
|
|
264
|
+
"""Every parseable statement in the script, junk skipped, never raising."""
|
|
265
|
+
cleaned = _META_LINE.sub("", sql)
|
|
266
|
+
if (dialect or "").lower() == "tsql":
|
|
267
|
+
cleaned = _rewrite_go_separators(cleaned)
|
|
268
|
+
try:
|
|
269
|
+
parsed = sqlglot.parse(cleaned, dialect=dialect, error_level=ErrorLevel.IGNORE)
|
|
270
|
+
stmts = [p for p in parsed if p is not None]
|
|
271
|
+
if stmts:
|
|
272
|
+
return _flatten_blocks(stmts, dialect)
|
|
273
|
+
except Exception:
|
|
274
|
+
pass
|
|
275
|
+
out: list[exp.Expression] = []
|
|
276
|
+
for part in _naive_split(cleaned):
|
|
277
|
+
if not part.strip():
|
|
278
|
+
continue
|
|
279
|
+
try:
|
|
280
|
+
p = sqlglot.parse_one(part, dialect=dialect, error_level=ErrorLevel.IGNORE)
|
|
281
|
+
if p is not None:
|
|
282
|
+
out.append(p)
|
|
283
|
+
except Exception:
|
|
284
|
+
continue
|
|
285
|
+
return _flatten_blocks(out, dialect)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _table_name(node) -> str | None:
|
|
289
|
+
from ripple.engine.column_ref import temp_marked_name
|
|
290
|
+
|
|
291
|
+
if node is None:
|
|
292
|
+
return None
|
|
293
|
+
if isinstance(node, exp.Table):
|
|
294
|
+
return temp_marked_name(node)
|
|
295
|
+
if isinstance(node, (exp.Schema, exp.Identifier)):
|
|
296
|
+
return (
|
|
297
|
+
temp_marked_name(node) if isinstance(node, exp.Identifier) else _table_name(node.this)
|
|
298
|
+
)
|
|
299
|
+
name = getattr(node, "name", None)
|
|
300
|
+
return name or None
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
_SEARCH_PATH_RE = re.compile(r"search_path\s*(?:=|to)\s*\"?([A-Za-z_][\w$]*)", re.I)
|
|
304
|
+
_SEARCH_PATH_RESET_RE = re.compile(r"reset\s+search_path|search_path\s*(?:=|to)\s*default", re.I)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _qualified_target(node) -> str | None:
|
|
308
|
+
"""The created/inserted relation's name WITH its qualifiers, when the SQL
|
|
309
|
+
writes any. The model keeps the bare name (round 0 proved renaming
|
|
310
|
+
regresses), but the qualified spelling must survive as an alias: openFEC's
|
|
311
|
+
migrations create disclosure.<view>, and holdout round 3 scored 12 correct
|
|
312
|
+
answers as misses because that spelling was recorded nowhere."""
|
|
313
|
+
if isinstance(node, exp.Schema):
|
|
314
|
+
node = node.this
|
|
315
|
+
if isinstance(node, exp.Table):
|
|
316
|
+
if not (node.catalog or node.db):
|
|
317
|
+
return None # bare target: search_path (if any) supplies the schema
|
|
318
|
+
from ripple.engine.column_ref import qualified_table_name
|
|
319
|
+
|
|
320
|
+
qualified = qualified_table_name(node)
|
|
321
|
+
return qualified if qualified != node.name else None
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _referenced_tables(select: exp.Expression) -> set[str]:
|
|
325
|
+
# temp/tablevar markers survive here too: parents recorded bare (a, b)
|
|
326
|
+
# while the models are #a/#b left star chains unresolvable (the
|
|
327
|
+
# review of cycle 7)
|
|
328
|
+
from ripple.engine.column_ref import temp_marked_name
|
|
329
|
+
|
|
330
|
+
return {temp_marked_name(t) for t in select.find_all(exp.Table) if t.name}
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _create_columns(create: exp.Create) -> list[str]:
|
|
334
|
+
"""Column names declared in a CREATE TABLE (...) definition, typed
|
|
335
|
+
(ColumnDef) or identifier-form as in CREATE TABLE x (a) AS ... (the
|
|
336
|
+
review of cycle 6: the VALUES CTAS lost its declared schema)."""
|
|
337
|
+
schema = create.this
|
|
338
|
+
cols: list[str] = []
|
|
339
|
+
if isinstance(schema, exp.Schema):
|
|
340
|
+
for c in schema.expressions:
|
|
341
|
+
if isinstance(c, (exp.ColumnDef, exp.Identifier, exp.Column)):
|
|
342
|
+
cols.append(c.name)
|
|
343
|
+
return cols
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _is_schema_clone(inner) -> bool:
|
|
347
|
+
"""A star-only SELECT that provably writes no rows: WHERE with no column
|
|
348
|
+
reference (0 = 1, false) or LIMIT 0. The loader idiom for cloning a
|
|
349
|
+
table's schema before INSERTing the real rows (usaspending's subaward
|
|
350
|
+
loader, holdout round 2). Only this exact shape may be replaced by a
|
|
351
|
+
later derivation for the same target; a plain star INSERT is real
|
|
352
|
+
lineage and must never be."""
|
|
353
|
+
if not isinstance(inner, exp.Select):
|
|
354
|
+
return False
|
|
355
|
+
selects = inner.selects
|
|
356
|
+
if not selects or not all(
|
|
357
|
+
isinstance(s, exp.Star) or (isinstance(s, exp.Column) and isinstance(s.this, exp.Star))
|
|
358
|
+
for s in selects
|
|
359
|
+
):
|
|
360
|
+
return False
|
|
361
|
+
where = inner.args.get("where")
|
|
362
|
+
if where is not None and next(where.find_all(exp.Column), None) is None:
|
|
363
|
+
return True
|
|
364
|
+
limit = inner.args.get("limit")
|
|
365
|
+
if limit is not None:
|
|
366
|
+
lit = limit.expression
|
|
367
|
+
if isinstance(lit, exp.Literal) and lit.this == "0":
|
|
368
|
+
return True
|
|
369
|
+
return False
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _realias_to_insert_columns(insert: exp.Expression, inner):
|
|
373
|
+
"""INSERT INTO t (a, b) SELECT x, y: the table gains columns a and b, so
|
|
374
|
+
the derivation's projection is re-aliased to the target list. Pairwise,
|
|
375
|
+
only when the target list is all plain identifiers, the lengths match,
|
|
376
|
+
and no star is involved; otherwise the select is kept as written.
|
|
377
|
+
Identifier nodes are copied whole so quoting survives ("MixedCase").
|
|
378
|
+
usaspending's parent_award lost parent_award_id to the un-aliased
|
|
379
|
+
reading (holdout round 2). CREATE TABLE t (a, b) AS SELECT is the same
|
|
380
|
+
shape and takes the same rule (the cycle-6 review). Known
|
|
381
|
+
unhandled shapes, kept as written: UNION inners and postgres INSERT
|
|
382
|
+
INTO t AS alias (cols)."""
|
|
383
|
+
schema = insert.this
|
|
384
|
+
if not isinstance(schema, exp.Schema) or not isinstance(inner, exp.Select):
|
|
385
|
+
return inner
|
|
386
|
+
idents = schema.expressions
|
|
387
|
+
if not idents or not all(isinstance(c, exp.Identifier) for c in idents):
|
|
388
|
+
return inner
|
|
389
|
+
selects = inner.selects
|
|
390
|
+
if len(idents) != len(selects):
|
|
391
|
+
return inner
|
|
392
|
+
# only true star ITEMS (* or t.*) refuse the mapping; an aggregate whose
|
|
393
|
+
# argument is a star, COUNT_BIG(*), is an ordinary expression, and one of
|
|
394
|
+
# them disabled the whole #plan_creation column list (first_responder_kit,
|
|
395
|
+
# holdout round 6)
|
|
396
|
+
if any(
|
|
397
|
+
isinstance(s, exp.Star) or (isinstance(s, exp.Column) and isinstance(s.this, exp.Star))
|
|
398
|
+
for s in selects
|
|
399
|
+
):
|
|
400
|
+
return inner
|
|
401
|
+
inner = inner.copy()
|
|
402
|
+
realiased = []
|
|
403
|
+
for item, ident in zip(inner.selects, idents, strict=True):
|
|
404
|
+
base = item.this if isinstance(item, exp.Alias) else item
|
|
405
|
+
realiased.append(exp.Alias(this=base, alias=ident.copy()))
|
|
406
|
+
inner.set("expressions", realiased)
|
|
407
|
+
return inner
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
# One model discovered inside a script. `sql` is the derivation to analyze (a plain
|
|
411
|
+
# SELECT), empty for a base table. `columns` are declared columns for a base table.
|
|
412
|
+
class ScriptModel:
|
|
413
|
+
__slots__ = (
|
|
414
|
+
"name",
|
|
415
|
+
"sql",
|
|
416
|
+
"parents",
|
|
417
|
+
"columns",
|
|
418
|
+
"is_schema",
|
|
419
|
+
"schema_clone",
|
|
420
|
+
"qualified_name",
|
|
421
|
+
"self_read",
|
|
422
|
+
"is_function",
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
def __init__(
|
|
426
|
+
self,
|
|
427
|
+
name,
|
|
428
|
+
sql="",
|
|
429
|
+
parents=None,
|
|
430
|
+
columns=None,
|
|
431
|
+
is_schema=False,
|
|
432
|
+
schema_clone=False,
|
|
433
|
+
qualified_name=None,
|
|
434
|
+
self_read=False,
|
|
435
|
+
is_function=False,
|
|
436
|
+
):
|
|
437
|
+
self.name = name
|
|
438
|
+
self.sql = sql
|
|
439
|
+
self.parents = parents or set()
|
|
440
|
+
self.columns = columns or []
|
|
441
|
+
self.is_schema = is_schema
|
|
442
|
+
self.schema_clone = schema_clone
|
|
443
|
+
self.qualified_name = qualified_name
|
|
444
|
+
# an UPDATE reads the table it writes; the self edge is prior
|
|
445
|
+
# state, not a CTE resolution loop
|
|
446
|
+
self.self_read = self_read
|
|
447
|
+
# minted from CREATE FUNCTION: a downstream call by this written
|
|
448
|
+
# name binds here instead of degrading as a relation collision
|
|
449
|
+
self.is_function = is_function
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _update_models(
|
|
453
|
+
stmt: exp.Update, dialect: str | None, active_schema: str | None
|
|
454
|
+
) -> list[ScriptModel]:
|
|
455
|
+
"""UPDATE ... SET as derivations, one ScriptModel per written table.
|
|
456
|
+
|
|
457
|
+
nycdb builds every dataset as ALTER TABLE + UPDATE (holdout round 5).
|
|
458
|
+
The shapes come from the the review of PR #28: `UPDATE alias ... FROM
|
|
459
|
+
t alias` writes the aliased table, mysql's UPDATE t1, t2 writes each SET
|
|
460
|
+
qualifier's table, the statement's WITH and WHERE belong to the
|
|
461
|
+
synthesized derivation, and SET (a, b) = (subquery) maps by position.
|
|
462
|
+
"""
|
|
463
|
+
if not isinstance(stmt.this, exp.Table):
|
|
464
|
+
return []
|
|
465
|
+
|
|
466
|
+
scope: dict[str, exp.Table] = {}
|
|
467
|
+
upd_from = stmt.args.get("from_") or stmt.args.get("from")
|
|
468
|
+
if upd_from is not None:
|
|
469
|
+
for t in upd_from.find_all(exp.Table):
|
|
470
|
+
scope.setdefault((t.alias or t.name).lower(), t)
|
|
471
|
+
for join in stmt.this.args.get("joins") or []:
|
|
472
|
+
for t in join.find_all(exp.Table):
|
|
473
|
+
scope.setdefault((t.alias or t.name).lower(), t)
|
|
474
|
+
|
|
475
|
+
primary = stmt.this
|
|
476
|
+
if not (primary.db or primary.catalog):
|
|
477
|
+
primary = scope.get(primary.name.lower(), primary)
|
|
478
|
+
|
|
479
|
+
def _positional_items(lhs: exp.Tuple, rhs: exp.Expression) -> list[exp.Alias]:
|
|
480
|
+
inner = rhs.this if isinstance(rhs, exp.Subquery) else None
|
|
481
|
+
if not isinstance(inner, exp.Select) or len(inner.selects) != len(lhs.expressions):
|
|
482
|
+
return []
|
|
483
|
+
items = []
|
|
484
|
+
for idx, target_col in enumerate(lhs.expressions):
|
|
485
|
+
if not isinstance(target_col, exp.Column):
|
|
486
|
+
continue
|
|
487
|
+
pruned = rhs.copy()
|
|
488
|
+
pruned.this.set("expressions", [inner.selects[idx].copy()])
|
|
489
|
+
items.append(exp.alias_(pruned, target_col.name))
|
|
490
|
+
return items
|
|
491
|
+
|
|
492
|
+
items_by_target: dict[str, tuple[exp.Table, list[exp.Alias]]] = {}
|
|
493
|
+
|
|
494
|
+
def _add(tnode: exp.Table, item: exp.Alias) -> None:
|
|
495
|
+
key = (tnode.alias or tnode.name).lower()
|
|
496
|
+
items_by_target.setdefault(key, (tnode, []))[1].append(item)
|
|
497
|
+
|
|
498
|
+
for eq in stmt.expressions:
|
|
499
|
+
if not isinstance(eq, exp.EQ) or eq.expression is None:
|
|
500
|
+
continue
|
|
501
|
+
lhs = eq.this
|
|
502
|
+
if isinstance(lhs, exp.Column):
|
|
503
|
+
tnode = scope.get(lhs.table.lower(), primary) if lhs.table else primary
|
|
504
|
+
_add(tnode, exp.alias_(eq.expression.copy(), lhs.name))
|
|
505
|
+
elif isinstance(lhs, exp.Tuple):
|
|
506
|
+
for item in _positional_items(lhs, eq.expression):
|
|
507
|
+
_add(primary, item)
|
|
508
|
+
|
|
509
|
+
base_tree = upd_from.this if upd_from is not None else stmt.this
|
|
510
|
+
models: list[ScriptModel] = []
|
|
511
|
+
for tnode, items in items_by_target.values():
|
|
512
|
+
target = _table_name(tnode)
|
|
513
|
+
if not target or not items:
|
|
514
|
+
continue
|
|
515
|
+
tree = base_tree.copy()
|
|
516
|
+
synthesized = exp.select(*items).from_(tree)
|
|
517
|
+
covered = {t.name.lower() for t in tree.find_all(exp.Table)}
|
|
518
|
+
if target.lower() not in covered:
|
|
519
|
+
bare = tnode.copy()
|
|
520
|
+
bare.set("joins", None)
|
|
521
|
+
synthesized = synthesized.join(bare)
|
|
522
|
+
where = stmt.args.get("where")
|
|
523
|
+
if where is not None:
|
|
524
|
+
synthesized.set("where", where.copy())
|
|
525
|
+
with_clause = stmt.args.get("with_") or stmt.args.get("with")
|
|
526
|
+
if with_clause is not None:
|
|
527
|
+
synthesized.set("with_", with_clause.copy())
|
|
528
|
+
synthesized_sql = safe_sql(synthesized, dialect)
|
|
529
|
+
if synthesized_sql is None:
|
|
530
|
+
continue
|
|
531
|
+
models.append(
|
|
532
|
+
ScriptModel(
|
|
533
|
+
name=target,
|
|
534
|
+
sql=synthesized_sql,
|
|
535
|
+
parents=_referenced_tables(synthesized),
|
|
536
|
+
qualified_name=_qualified_target(tnode)
|
|
537
|
+
or (f"{active_schema}.{target}" if active_schema else None),
|
|
538
|
+
self_read=True,
|
|
539
|
+
)
|
|
540
|
+
)
|
|
541
|
+
return models
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def expand_script(
|
|
545
|
+
file_stem: str, sql: str, dialect: str | None, warnings: list[str] | None = None
|
|
546
|
+
) -> list[ScriptModel]:
|
|
547
|
+
"""Turn one .sql file into the models it defines.
|
|
548
|
+
|
|
549
|
+
- CREATE VIEW / CREATE TABLE AS SELECT / INSERT INTO ... SELECT -> a model named
|
|
550
|
+
for the target, analyzed from its SELECT.
|
|
551
|
+
- CREATE TABLE (col defs) -> a base-table model carrying its declared columns
|
|
552
|
+
(the schema, so downstream SELECT * can resolve).
|
|
553
|
+
- a single bare SELECT with no target -> one model named after the file.
|
|
554
|
+
Everything else (DROP, SET, INSERT ... VALUES) contributes nothing.
|
|
555
|
+
`warnings`, when given, collects forfeits the caller should surface.
|
|
556
|
+
"""
|
|
557
|
+
from ripple.engine.preprocess import prepare_sql_for_parse
|
|
558
|
+
|
|
559
|
+
# the same pre-parse encoding the statement path uses; parsing the raw
|
|
560
|
+
# text instead dropped or shredded jinja chain-path DML targets
|
|
561
|
+
# (cycle-13 review, F8)
|
|
562
|
+
sql, _ = prepare_sql_for_parse(sql, dialect or "")
|
|
563
|
+
|
|
564
|
+
models: list[ScriptModel] = []
|
|
565
|
+
if re.search(r"\bRETURNS\s+@[\w$]+\s+(?:as\s+)?TABLE\b", sql, re.I):
|
|
566
|
+
from ripple.engine.tsql_tvf import expand_table_functions
|
|
567
|
+
|
|
568
|
+
sql, tvf_models = expand_table_functions(sql, warnings=warnings)
|
|
569
|
+
models.extend(tvf_models)
|
|
570
|
+
statements = split_statements(sql, dialect)
|
|
571
|
+
bare_selects: list[exp.Expression] = []
|
|
572
|
+
|
|
573
|
+
tsql_scalars = "@" in sql and (dialect or "").lower() == "tsql"
|
|
574
|
+
if tsql_scalars:
|
|
575
|
+
from ripple.engine.tsql_scalar_vars import inline_scalar_params
|
|
576
|
+
|
|
577
|
+
inline_scalar_params(statements, dialect)
|
|
578
|
+
|
|
579
|
+
# SET search_path = disclosure, pg_catalog: everything created bare after
|
|
580
|
+
# it lives in that schema. openFEC's migrations create every view this
|
|
581
|
+
# way, and round 3 scored 12 correct answers as misses because the
|
|
582
|
+
# schema-qualified spelling existed nowhere on the model.
|
|
583
|
+
active_schema: str | None = None
|
|
584
|
+
|
|
585
|
+
for stmt in statements:
|
|
586
|
+
if isinstance(stmt, (exp.Set, exp.Command)):
|
|
587
|
+
# an ungenerable SET (tsql `SET @x -= 1`) forfeits only its own
|
|
588
|
+
# search_path sniff, never the rest of the script
|
|
589
|
+
text = safe_sql(stmt, dialect) or ""
|
|
590
|
+
if _SEARCH_PATH_RESET_RE.search(text):
|
|
591
|
+
# RESET search_path / SET search_path TO DEFAULT: a stale
|
|
592
|
+
# schema must not leak onto later creations
|
|
593
|
+
active_schema = None
|
|
594
|
+
else:
|
|
595
|
+
m = _SEARCH_PATH_RE.search(text)
|
|
596
|
+
if m:
|
|
597
|
+
schema = m.group(1)
|
|
598
|
+
active_schema = None if schema.lower() in ("pg_catalog", "public") else schema
|
|
599
|
+
if isinstance(stmt, exp.Create):
|
|
600
|
+
if "INDEX" in (stmt.kind or "").upper():
|
|
601
|
+
# an index is not a relation: minting a model for it made
|
|
602
|
+
# full_text.sql's stem five-way ambiguous (nycdb, holdout
|
|
603
|
+
# round 5 gap cycle). Substring match: tsql kinds read
|
|
604
|
+
# CLUSTERED INDEX / NONCLUSTERED INDEX (cycle-6 review)
|
|
605
|
+
continue
|
|
606
|
+
target = _table_name(stmt.this)
|
|
607
|
+
# CREATE TABLE x AS ( SELECT ... ): the paren wrapper must not
|
|
608
|
+
# reach the model's SQL, the statement layer refuses it as
|
|
609
|
+
# not-a-SELECT and the whole derivation vanishes (nycdb
|
|
610
|
+
# business_addrs, the real mechanism of round 5's class B)
|
|
611
|
+
inner = stmt.expression # the SELECT for a view / CTAS, else None
|
|
612
|
+
while isinstance(inner, (exp.Subquery, exp.Paren)):
|
|
613
|
+
inner = inner.this
|
|
614
|
+
if target and isinstance(inner, (exp.Select, exp.Union)):
|
|
615
|
+
inner = _realias_to_insert_columns(stmt, inner)
|
|
616
|
+
inner_sql = safe_sql(inner, dialect)
|
|
617
|
+
if inner_sql is None:
|
|
618
|
+
# the derivation exists but cannot be regenerated: the
|
|
619
|
+
# model is forfeited outright. A schema-only stub here
|
|
620
|
+
# would silently swallow downstream reads (cycle-12
|
|
621
|
+
# review, F14).
|
|
622
|
+
if warnings is not None:
|
|
623
|
+
warnings.append(
|
|
624
|
+
f"CREATE {target}: its SELECT body could not be "
|
|
625
|
+
"regenerated; the model is forfeited"
|
|
626
|
+
)
|
|
627
|
+
continue
|
|
628
|
+
models.append(
|
|
629
|
+
ScriptModel(
|
|
630
|
+
name=target,
|
|
631
|
+
sql=inner_sql,
|
|
632
|
+
parents=_referenced_tables(inner),
|
|
633
|
+
columns=_create_columns(stmt),
|
|
634
|
+
schema_clone=_is_schema_clone(inner),
|
|
635
|
+
qualified_name=_qualified_target(stmt.this)
|
|
636
|
+
or (f"{active_schema}.{target}" if active_schema else None),
|
|
637
|
+
)
|
|
638
|
+
)
|
|
639
|
+
elif target:
|
|
640
|
+
models.append(
|
|
641
|
+
ScriptModel(
|
|
642
|
+
name=target,
|
|
643
|
+
columns=_create_columns(stmt),
|
|
644
|
+
is_schema=True,
|
|
645
|
+
qualified_name=_qualified_target(stmt.this)
|
|
646
|
+
or (f"{active_schema}.{target}" if active_schema else None),
|
|
647
|
+
)
|
|
648
|
+
)
|
|
649
|
+
elif isinstance(stmt, exp.Insert):
|
|
650
|
+
target = _table_name(stmt.this)
|
|
651
|
+
inner = stmt.expression
|
|
652
|
+
if target and isinstance(inner, (exp.Select, exp.Union)):
|
|
653
|
+
inner = _realias_to_insert_columns(stmt, inner)
|
|
654
|
+
inner_sql = safe_sql(inner, dialect)
|
|
655
|
+
if inner_sql is None:
|
|
656
|
+
continue
|
|
657
|
+
models.append(
|
|
658
|
+
ScriptModel(
|
|
659
|
+
name=target,
|
|
660
|
+
sql=inner_sql,
|
|
661
|
+
parents=_referenced_tables(inner),
|
|
662
|
+
schema_clone=_is_schema_clone(inner),
|
|
663
|
+
qualified_name=_qualified_target(stmt.this)
|
|
664
|
+
or (f"{active_schema}.{target}" if active_schema else None),
|
|
665
|
+
)
|
|
666
|
+
)
|
|
667
|
+
elif isinstance(stmt, exp.Update):
|
|
668
|
+
# UPDATE t SET col = expr is a derivation of t: nycdb builds
|
|
669
|
+
# every dataset this way (ALTER ... ADD COLUMN + UPDATE), and
|
|
670
|
+
# round 5 scored all eleven of its cases as empty because no
|
|
671
|
+
# model existed. Synthesized as a SELECT so the ordinary
|
|
672
|
+
# engine walks the expressions.
|
|
673
|
+
models.extend(_update_models(stmt, dialect, active_schema))
|
|
674
|
+
elif isinstance(stmt, (exp.Select, exp.Union)):
|
|
675
|
+
# the INTO lives on the leftmost SELECT, including under a set
|
|
676
|
+
# operation (review: SELECT INTO ... UNION ALL ...)
|
|
677
|
+
head = stmt
|
|
678
|
+
while isinstance(head, exp.Union):
|
|
679
|
+
head = head.this
|
|
680
|
+
into = head.args.get("into") if isinstance(head, exp.Select) else None
|
|
681
|
+
target = _table_name(into.this) if into is not None else None
|
|
682
|
+
if target:
|
|
683
|
+
# SELECT ... INTO t is a creation of t (postgres); leaving it
|
|
684
|
+
# a bare select made the target read as a PARENT and left no
|
|
685
|
+
# model to resolve (nyc export.sql, holdout round 4)
|
|
686
|
+
body = stmt.copy()
|
|
687
|
+
body_head = body
|
|
688
|
+
while isinstance(body_head, exp.Union):
|
|
689
|
+
body_head = body_head.this
|
|
690
|
+
body_head.set("into", None)
|
|
691
|
+
body_sql = safe_sql(body, dialect)
|
|
692
|
+
if body_sql is None:
|
|
693
|
+
continue
|
|
694
|
+
models.append(
|
|
695
|
+
ScriptModel(
|
|
696
|
+
name=target,
|
|
697
|
+
sql=body_sql,
|
|
698
|
+
parents=_referenced_tables(body),
|
|
699
|
+
schema_clone=_is_schema_clone(body),
|
|
700
|
+
qualified_name=_qualified_target(into.this)
|
|
701
|
+
or (f"{active_schema}.{target}" if active_schema else None),
|
|
702
|
+
)
|
|
703
|
+
)
|
|
704
|
+
else:
|
|
705
|
+
bare_selects.append(stmt)
|
|
706
|
+
|
|
707
|
+
if tsql_scalars:
|
|
708
|
+
from ripple.engine.tsql_scalar_vars import scalar_value_insert_models
|
|
709
|
+
|
|
710
|
+
models.extend(scalar_value_insert_models(statements, dialect))
|
|
711
|
+
|
|
712
|
+
# a write into an @tablevar is procedure-local scratch, not a relation:
|
|
713
|
+
# minting models for them fabricated @loadedModules/@dm_os_memory_clerks
|
|
714
|
+
# and multiplied the file stem's owners until the column pick refused
|
|
715
|
+
# (sp_BlitzInMemoryOLTP, holdout round 6)
|
|
716
|
+
models = [m for m in models if not m.name.startswith("@")]
|
|
717
|
+
if not models and bare_selects:
|
|
718
|
+
# a plain query file: one model named for the file. The whole file is
|
|
719
|
+
# kept only when the query is the file's sole statement; a CREATE TEMP
|
|
720
|
+
# FUNCTION prelude (crux compliance-rates, holdout round 4) must not
|
|
721
|
+
# ride along, because multi-statement model SQL yields no edges.
|
|
722
|
+
body_sql = sql if len(statements) == 1 else safe_sql(bare_selects[-1], dialect)
|
|
723
|
+
if body_sql is not None:
|
|
724
|
+
models.append(
|
|
725
|
+
ScriptModel(
|
|
726
|
+
name=file_stem, sql=body_sql, parents=_referenced_tables(bare_selects[-1])
|
|
727
|
+
)
|
|
728
|
+
)
|
|
729
|
+
return models
|