ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,729 @@
1
+ """Multi-statement SQL scripts and dumps.
2
+
3
+ Real non-dbt SQL is rarely one statement per file. A Chinook dump or a Postgres
4
+ sample DB is a single file with hundreds to thousands of statements: DROP, CREATE
5
+ TABLE, INSERT, CREATE VIEW, mixed with psql `\\` meta-commands that aren't SQL at
6
+ all. This module splits such a file into statements resiliently (never raising),
7
+ sniffs the dialect from content when there's no dbt adapter to ask, and turns the
8
+ derivation-bearing statements into models with schema.
9
+
10
+ Nothing here fabricates lineage. A statement that will not parse is skipped, and a
11
+ CREATE TABLE with no SELECT contributes only its declared columns (the schema), not
12
+ a made-up edge.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+
19
+ import sqlglot
20
+ from sqlglot import exp
21
+ from sqlglot.errors import ErrorLevel
22
+
23
+ from ripple.engine.safe_gen import safe_sql
24
+
25
+ # psql / mysql-cli meta-commands: whole lines starting with a backslash (\pset, \c,
26
+ # \d, \.), which are client directives, not SQL. Stripped before parsing.
27
+ _META_LINE = re.compile(r"^[ \t]*\\[a-zA-Z.].*$", re.M)
28
+
29
+ # T-SQL's GO is the same class of client directive: a batch separator sqlcmd
30
+ # understands but the server never sees. Left in place it lumps the next
31
+ # batch into one unparseable Command ("GO\nCREATE VIEW ...", fhir_server,
32
+ # holdout round 11). Rewritten to ";" so batch boundaries survive. The full
33
+ # separator grammar: case-insensitive GO, optional count, optional trailing
34
+ # line or block comment (cycle-12 review, F12). Applied per line by the
35
+ # quote/comment-state scan below, never as a blind substitution, so a GO
36
+ # inside a bracketed identifier or string stays part of the name (F1).
37
+ _GO_LINE = re.compile(
38
+ r"^[ \t]*GO(?:[ \t]+\d+)?[ \t]*(?:--[^\r\n]*|/\*.*?\*/[ \t]*)?\r?$",
39
+ re.I,
40
+ )
41
+
42
+
43
+ def _rewrite_go_separators(sql: str) -> str:
44
+ """Replace batch-separator GO lines with ";", tracking quote and comment
45
+ state so quoted identifiers and comments are never rewritten."""
46
+ out: list[str] = []
47
+ state = "" # "" | "'" | '"' | "`" | "[" | "/*"
48
+ for line in sql.split("\n"):
49
+ if not state and _GO_LINE.match(line):
50
+ out.append(";")
51
+ continue
52
+ i, n = 0, len(line)
53
+ while i < n:
54
+ c = line[i]
55
+ if state == "/*":
56
+ if c == "*" and line[i + 1 : i + 2] == "/":
57
+ state = ""
58
+ i += 2
59
+ continue
60
+ i += 1
61
+ continue
62
+ if state == "[":
63
+ if c == "]":
64
+ if line[i + 1 : i + 2] == "]": # ]] escapes ] inside the name
65
+ i += 2
66
+ continue
67
+ state = ""
68
+ i += 1
69
+ continue
70
+ if state:
71
+ if c == state:
72
+ # a doubled quote reads as close-then-reopen; same state
73
+ state = ""
74
+ i += 1
75
+ continue
76
+ if c == "-" and line[i + 1 : i + 2] == "-":
77
+ break # the rest of the line is a comment
78
+ if c == "/" and line[i + 1 : i + 2] == "*":
79
+ state = "/*"
80
+ i += 2
81
+ continue
82
+ if c in "'\"`[":
83
+ state = c
84
+ i += 1
85
+ out.append(line)
86
+ return "\n".join(out)
87
+
88
+
89
+ # Markers that only one dialect writes, weighted by how exclusive they are.
90
+ # A backtick scores for both BigQuery and MySQL because both quote with it;
91
+ # reading it as MySQL alone is what made filecoin-data-portal, a BigQuery
92
+ # project, parse as the wrong dialect.
93
+ _DIALECT_MARKERS: tuple[tuple[str, str, int, int], ...] = (
94
+ # (dialect, pattern, weight, re flags)
95
+ ("mysql", r"\bENGINE\s*=\s*\w+|\bAUTO_INCREMENT\b|\bUNSIGNED\b", 4, re.I),
96
+ ("mysql", r"\bTINYINT\b|\bLONGTEXT\b|\bDATETIME\(\d\)", 2, re.I),
97
+ ("bigquery", r"\bINFORMATION_SCHEMA\b", 3, 0),
98
+ ("bigquery", r"\bSTRUCT\s*<|\bARRAY\s*<", 3, re.I),
99
+ ("bigquery", r"\bSAFE[._]\w+", 3, re.I),
100
+ ("bigquery", r"`[\w-]+\.[\w-]+\.[\w-]+`", 3, 0),
101
+ # types and functions BigQuery alone spells this way
102
+ ("bigquery", r"\bFLOAT64\b|\bBIGNUMERIC\b|\bINT64\b", 3, re.I),
103
+ ("bigquery", r"\bTIMESTAMP_SECONDS\s*\(|\bCOUNTIF\s*\(|\bSAFE_DIVIDE\s*\(", 3, re.I),
104
+ ("bigquery", r"\bGENERATE_(DATE|TIMESTAMP)_ARRAY\s*\(|\bUNNEST\s*\(", 2, re.I),
105
+ # the GO alternative mirrors _GO_LINE's full separator grammar,
106
+ # case-insensitive via a scoped flag (cycle-12 review, F13)
107
+ (
108
+ "tsql",
109
+ r"\bIDENTITY\s*\(|\bNVARCHAR\b|\bGETDATE\s*\("
110
+ r"|(?i:^[ \t]*GO(?:[ \t]+\d+)?[ \t]*(?:--[^\r\n]*|/\*.*?\*/[ \t]*)?\r?$)",
111
+ 4,
112
+ re.M,
113
+ ),
114
+ ("snowflake", r"\bVARIANT\b|\bLATERAL\s+FLATTEN\b|\bQUALIFY\b", 3, re.I),
115
+ # a notebook exported from Databricks announces itself on line one;
116
+ # dbdemos-notebooks was read as tsql without this
117
+ ("databricks", r"^-- Databricks notebook source\b", 6, re.M),
118
+ (
119
+ "databricks",
120
+ r"\bUSING\s+DELTA\b|\bTBLPROPERTIES\b|\bLIVE\s+TABLE\b|\bSTREAMING\s+TABLE\b",
121
+ 3,
122
+ re.I,
123
+ ),
124
+ # stl_/svl_/svv_/stv_ system tables and DISTKEY/SORTKEY exist only on
125
+ # Redshift; dbt-labs/redshift has no profile and guessed snowflake
126
+ ("redshift", r"\b(stl|svl|svv|stv)_\w+", 3, re.I),
127
+ ("redshift", r"\bDISTKEY\b|\bSORTKEY\b|\bDISTSTYLE\b", 4, re.I),
128
+ )
129
+
130
+ # One backtick proves quoting style, not vendor: split evenly so a stronger
131
+ # marker decides.
132
+ _AMBIGUOUS_BACKTICK = ("bigquery", "mysql")
133
+
134
+ _MIN_DIALECT_SCORE = 3
135
+
136
+
137
+ def sniff_dialect(sql: str) -> str | None:
138
+ """Best-effort dialect from the SQL text itself, for repos with no dbt adapter
139
+ to declare one. Returns None when there's no strong signal (caller keeps its
140
+ default and says it assumed). Deliberately conservative: a wrong guess here is
141
+ what makes a real repo misparse.
142
+
143
+ Weighs every marker in the whole sample rather than returning on the first
144
+ hit. The caller assembles a slice of every file on purpose, so truncating
145
+ here threw that away: filecoin-data-portal's BigQuery markers sat past the
146
+ old 20,000 character cut and it was read as postgres.
147
+ """
148
+ if re.search(r"\\(pset|connect|echo|copy|d[itsv]?\b|c\b)", sql):
149
+ return "postgres" # psql directives cannot appear in another dialect
150
+
151
+ scores: dict[str, int] = {}
152
+ for dialect, pattern, weight, flags in _DIALECT_MARKERS:
153
+ if re.search(pattern, sql, flags):
154
+ scores[dialect] = scores.get(dialect, 0) + weight
155
+ if "`" in sql:
156
+ for dialect in _AMBIGUOUS_BACKTICK:
157
+ scores[dialect] = scores.get(dialect, 0) + 1
158
+
159
+ if not scores:
160
+ return None
161
+ best = max(scores, key=lambda d: (scores[d], d))
162
+ if scores[best] < _MIN_DIALECT_SCORE:
163
+ return None
164
+ return best
165
+
166
+
167
+ def _naive_split(sql: str):
168
+ """Split on semicolons that are not inside a string or comment. A fallback for
169
+ when sqlglot's own tokenizer refuses the whole blob."""
170
+ out, buf = [], []
171
+ i, n = 0, len(sql)
172
+ quote = None
173
+ while i < n:
174
+ c = sql[i]
175
+ if quote:
176
+ buf.append(c)
177
+ if c == quote:
178
+ quote = None
179
+ i += 1
180
+ continue
181
+ if c in ("'", '"', "`"):
182
+ quote = c
183
+ buf.append(c)
184
+ elif c == "-" and i + 1 < n and sql[i + 1] == "-":
185
+ j = sql.find("\n", i)
186
+ j = n if j == -1 else j
187
+ buf.append(sql[i:j])
188
+ i = j
189
+ continue
190
+ elif c == "/" and i + 1 < n and sql[i + 1] == "*":
191
+ j = sql.find("*/", i)
192
+ j = n if j == -1 else j + 2
193
+ buf.append(sql[i:j])
194
+ i = j
195
+ continue
196
+ elif c == ";":
197
+ out.append("".join(buf))
198
+ buf = []
199
+ else:
200
+ buf.append(c)
201
+ i += 1
202
+ if buf:
203
+ out.append("".join(buf))
204
+ return out
205
+
206
+
207
+ def _flatten_blocks(
208
+ stmts: list[exp.Expression],
209
+ dialect: str | None,
210
+ _in_block: bool = False,
211
+ _conditional: bool = False,
212
+ ) -> list[exp.Expression]:
213
+ """Statements inside conditional/compound blocks are the file's real
214
+ content: the T-SQL deployment idiom IF NOT EXISTS (...) BEGIN CREATE
215
+ TABLE ... END guards deployment, not identity, and skipping it left
216
+ maintenance_solution with no models at all (holdout round 7). A
217
+ statement the guard parse left as a raw Command (often with the
218
+ block's stray END attached) gets one bounded re-parse.
219
+
220
+ Statements out of an IfBlock branch are marked ripple_conditional:
221
+ they may or may not run, so downstream reaching-definition analysis
222
+ must not let them kill or uniquely supply an earlier definition."""
223
+ out: list[exp.Expression] = []
224
+ for stmt in stmts:
225
+ block_cls = getattr(exp, "Block", None)
226
+ if_cls = getattr(exp, "IfBlock", None)
227
+ if if_cls is not None and isinstance(stmt, if_cls):
228
+ branches = [stmt.args.get("true"), stmt.args.get("false")]
229
+ inner = [b for b in branches if isinstance(b, exp.Expression)]
230
+ out.extend(_flatten_blocks(inner, dialect, _in_block=True, _conditional=True))
231
+ elif block_cls is not None and isinstance(stmt, block_cls):
232
+ out.extend(
233
+ _flatten_blocks(
234
+ list(stmt.expressions or []),
235
+ dialect,
236
+ _in_block=True,
237
+ _conditional=_conditional,
238
+ )
239
+ )
240
+ elif _in_block and isinstance(stmt, exp.Command):
241
+ raw = safe_sql(stmt, dialect)
242
+ if raw is None:
243
+ out.append(stmt)
244
+ continue
245
+ text = re.sub(r"\s*END\s*$", "", raw, flags=re.I)
246
+ try:
247
+ reparsed = sqlglot.parse_one(text, dialect=dialect, error_level=ErrorLevel.IGNORE)
248
+ except Exception:
249
+ reparsed = None
250
+ if reparsed is not None and not isinstance(reparsed, exp.Command):
251
+ out.append(reparsed)
252
+ else:
253
+ out.append(stmt)
254
+ else:
255
+ out.append(stmt)
256
+ if _conditional:
257
+ for stmt in out:
258
+ if stmt is not None:
259
+ stmt.meta["ripple_conditional"] = True
260
+ return out
261
+
262
+
263
+ def split_statements(sql: str, dialect: str | None) -> list[exp.Expression]:
264
+ """Every parseable statement in the script, junk skipped, never raising."""
265
+ cleaned = _META_LINE.sub("", sql)
266
+ if (dialect or "").lower() == "tsql":
267
+ cleaned = _rewrite_go_separators(cleaned)
268
+ try:
269
+ parsed = sqlglot.parse(cleaned, dialect=dialect, error_level=ErrorLevel.IGNORE)
270
+ stmts = [p for p in parsed if p is not None]
271
+ if stmts:
272
+ return _flatten_blocks(stmts, dialect)
273
+ except Exception:
274
+ pass
275
+ out: list[exp.Expression] = []
276
+ for part in _naive_split(cleaned):
277
+ if not part.strip():
278
+ continue
279
+ try:
280
+ p = sqlglot.parse_one(part, dialect=dialect, error_level=ErrorLevel.IGNORE)
281
+ if p is not None:
282
+ out.append(p)
283
+ except Exception:
284
+ continue
285
+ return _flatten_blocks(out, dialect)
286
+
287
+
288
+ def _table_name(node) -> str | None:
289
+ from ripple.engine.column_ref import temp_marked_name
290
+
291
+ if node is None:
292
+ return None
293
+ if isinstance(node, exp.Table):
294
+ return temp_marked_name(node)
295
+ if isinstance(node, (exp.Schema, exp.Identifier)):
296
+ return (
297
+ temp_marked_name(node) if isinstance(node, exp.Identifier) else _table_name(node.this)
298
+ )
299
+ name = getattr(node, "name", None)
300
+ return name or None
301
+
302
+
303
+ _SEARCH_PATH_RE = re.compile(r"search_path\s*(?:=|to)\s*\"?([A-Za-z_][\w$]*)", re.I)
304
+ _SEARCH_PATH_RESET_RE = re.compile(r"reset\s+search_path|search_path\s*(?:=|to)\s*default", re.I)
305
+
306
+
307
+ def _qualified_target(node) -> str | None:
308
+ """The created/inserted relation's name WITH its qualifiers, when the SQL
309
+ writes any. The model keeps the bare name (round 0 proved renaming
310
+ regresses), but the qualified spelling must survive as an alias: openFEC's
311
+ migrations create disclosure.<view>, and holdout round 3 scored 12 correct
312
+ answers as misses because that spelling was recorded nowhere."""
313
+ if isinstance(node, exp.Schema):
314
+ node = node.this
315
+ if isinstance(node, exp.Table):
316
+ if not (node.catalog or node.db):
317
+ return None # bare target: search_path (if any) supplies the schema
318
+ from ripple.engine.column_ref import qualified_table_name
319
+
320
+ qualified = qualified_table_name(node)
321
+ return qualified if qualified != node.name else None
322
+
323
+
324
+ def _referenced_tables(select: exp.Expression) -> set[str]:
325
+ # temp/tablevar markers survive here too: parents recorded bare (a, b)
326
+ # while the models are #a/#b left star chains unresolvable (the
327
+ # review of cycle 7)
328
+ from ripple.engine.column_ref import temp_marked_name
329
+
330
+ return {temp_marked_name(t) for t in select.find_all(exp.Table) if t.name}
331
+
332
+
333
+ def _create_columns(create: exp.Create) -> list[str]:
334
+ """Column names declared in a CREATE TABLE (...) definition, typed
335
+ (ColumnDef) or identifier-form as in CREATE TABLE x (a) AS ... (the
336
+ review of cycle 6: the VALUES CTAS lost its declared schema)."""
337
+ schema = create.this
338
+ cols: list[str] = []
339
+ if isinstance(schema, exp.Schema):
340
+ for c in schema.expressions:
341
+ if isinstance(c, (exp.ColumnDef, exp.Identifier, exp.Column)):
342
+ cols.append(c.name)
343
+ return cols
344
+
345
+
346
+ def _is_schema_clone(inner) -> bool:
347
+ """A star-only SELECT that provably writes no rows: WHERE with no column
348
+ reference (0 = 1, false) or LIMIT 0. The loader idiom for cloning a
349
+ table's schema before INSERTing the real rows (usaspending's subaward
350
+ loader, holdout round 2). Only this exact shape may be replaced by a
351
+ later derivation for the same target; a plain star INSERT is real
352
+ lineage and must never be."""
353
+ if not isinstance(inner, exp.Select):
354
+ return False
355
+ selects = inner.selects
356
+ if not selects or not all(
357
+ isinstance(s, exp.Star) or (isinstance(s, exp.Column) and isinstance(s.this, exp.Star))
358
+ for s in selects
359
+ ):
360
+ return False
361
+ where = inner.args.get("where")
362
+ if where is not None and next(where.find_all(exp.Column), None) is None:
363
+ return True
364
+ limit = inner.args.get("limit")
365
+ if limit is not None:
366
+ lit = limit.expression
367
+ if isinstance(lit, exp.Literal) and lit.this == "0":
368
+ return True
369
+ return False
370
+
371
+
372
+ def _realias_to_insert_columns(insert: exp.Expression, inner):
373
+ """INSERT INTO t (a, b) SELECT x, y: the table gains columns a and b, so
374
+ the derivation's projection is re-aliased to the target list. Pairwise,
375
+ only when the target list is all plain identifiers, the lengths match,
376
+ and no star is involved; otherwise the select is kept as written.
377
+ Identifier nodes are copied whole so quoting survives ("MixedCase").
378
+ usaspending's parent_award lost parent_award_id to the un-aliased
379
+ reading (holdout round 2). CREATE TABLE t (a, b) AS SELECT is the same
380
+ shape and takes the same rule (the cycle-6 review). Known
381
+ unhandled shapes, kept as written: UNION inners and postgres INSERT
382
+ INTO t AS alias (cols)."""
383
+ schema = insert.this
384
+ if not isinstance(schema, exp.Schema) or not isinstance(inner, exp.Select):
385
+ return inner
386
+ idents = schema.expressions
387
+ if not idents or not all(isinstance(c, exp.Identifier) for c in idents):
388
+ return inner
389
+ selects = inner.selects
390
+ if len(idents) != len(selects):
391
+ return inner
392
+ # only true star ITEMS (* or t.*) refuse the mapping; an aggregate whose
393
+ # argument is a star, COUNT_BIG(*), is an ordinary expression, and one of
394
+ # them disabled the whole #plan_creation column list (first_responder_kit,
395
+ # holdout round 6)
396
+ if any(
397
+ isinstance(s, exp.Star) or (isinstance(s, exp.Column) and isinstance(s.this, exp.Star))
398
+ for s in selects
399
+ ):
400
+ return inner
401
+ inner = inner.copy()
402
+ realiased = []
403
+ for item, ident in zip(inner.selects, idents, strict=True):
404
+ base = item.this if isinstance(item, exp.Alias) else item
405
+ realiased.append(exp.Alias(this=base, alias=ident.copy()))
406
+ inner.set("expressions", realiased)
407
+ return inner
408
+
409
+
410
+ # One model discovered inside a script. `sql` is the derivation to analyze (a plain
411
+ # SELECT), empty for a base table. `columns` are declared columns for a base table.
412
+ class ScriptModel:
413
+ __slots__ = (
414
+ "name",
415
+ "sql",
416
+ "parents",
417
+ "columns",
418
+ "is_schema",
419
+ "schema_clone",
420
+ "qualified_name",
421
+ "self_read",
422
+ "is_function",
423
+ )
424
+
425
+ def __init__(
426
+ self,
427
+ name,
428
+ sql="",
429
+ parents=None,
430
+ columns=None,
431
+ is_schema=False,
432
+ schema_clone=False,
433
+ qualified_name=None,
434
+ self_read=False,
435
+ is_function=False,
436
+ ):
437
+ self.name = name
438
+ self.sql = sql
439
+ self.parents = parents or set()
440
+ self.columns = columns or []
441
+ self.is_schema = is_schema
442
+ self.schema_clone = schema_clone
443
+ self.qualified_name = qualified_name
444
+ # an UPDATE reads the table it writes; the self edge is prior
445
+ # state, not a CTE resolution loop
446
+ self.self_read = self_read
447
+ # minted from CREATE FUNCTION: a downstream call by this written
448
+ # name binds here instead of degrading as a relation collision
449
+ self.is_function = is_function
450
+
451
+
452
+ def _update_models(
453
+ stmt: exp.Update, dialect: str | None, active_schema: str | None
454
+ ) -> list[ScriptModel]:
455
+ """UPDATE ... SET as derivations, one ScriptModel per written table.
456
+
457
+ nycdb builds every dataset as ALTER TABLE + UPDATE (holdout round 5).
458
+ The shapes come from the the review of PR #28: `UPDATE alias ... FROM
459
+ t alias` writes the aliased table, mysql's UPDATE t1, t2 writes each SET
460
+ qualifier's table, the statement's WITH and WHERE belong to the
461
+ synthesized derivation, and SET (a, b) = (subquery) maps by position.
462
+ """
463
+ if not isinstance(stmt.this, exp.Table):
464
+ return []
465
+
466
+ scope: dict[str, exp.Table] = {}
467
+ upd_from = stmt.args.get("from_") or stmt.args.get("from")
468
+ if upd_from is not None:
469
+ for t in upd_from.find_all(exp.Table):
470
+ scope.setdefault((t.alias or t.name).lower(), t)
471
+ for join in stmt.this.args.get("joins") or []:
472
+ for t in join.find_all(exp.Table):
473
+ scope.setdefault((t.alias or t.name).lower(), t)
474
+
475
+ primary = stmt.this
476
+ if not (primary.db or primary.catalog):
477
+ primary = scope.get(primary.name.lower(), primary)
478
+
479
+ def _positional_items(lhs: exp.Tuple, rhs: exp.Expression) -> list[exp.Alias]:
480
+ inner = rhs.this if isinstance(rhs, exp.Subquery) else None
481
+ if not isinstance(inner, exp.Select) or len(inner.selects) != len(lhs.expressions):
482
+ return []
483
+ items = []
484
+ for idx, target_col in enumerate(lhs.expressions):
485
+ if not isinstance(target_col, exp.Column):
486
+ continue
487
+ pruned = rhs.copy()
488
+ pruned.this.set("expressions", [inner.selects[idx].copy()])
489
+ items.append(exp.alias_(pruned, target_col.name))
490
+ return items
491
+
492
+ items_by_target: dict[str, tuple[exp.Table, list[exp.Alias]]] = {}
493
+
494
+ def _add(tnode: exp.Table, item: exp.Alias) -> None:
495
+ key = (tnode.alias or tnode.name).lower()
496
+ items_by_target.setdefault(key, (tnode, []))[1].append(item)
497
+
498
+ for eq in stmt.expressions:
499
+ if not isinstance(eq, exp.EQ) or eq.expression is None:
500
+ continue
501
+ lhs = eq.this
502
+ if isinstance(lhs, exp.Column):
503
+ tnode = scope.get(lhs.table.lower(), primary) if lhs.table else primary
504
+ _add(tnode, exp.alias_(eq.expression.copy(), lhs.name))
505
+ elif isinstance(lhs, exp.Tuple):
506
+ for item in _positional_items(lhs, eq.expression):
507
+ _add(primary, item)
508
+
509
+ base_tree = upd_from.this if upd_from is not None else stmt.this
510
+ models: list[ScriptModel] = []
511
+ for tnode, items in items_by_target.values():
512
+ target = _table_name(tnode)
513
+ if not target or not items:
514
+ continue
515
+ tree = base_tree.copy()
516
+ synthesized = exp.select(*items).from_(tree)
517
+ covered = {t.name.lower() for t in tree.find_all(exp.Table)}
518
+ if target.lower() not in covered:
519
+ bare = tnode.copy()
520
+ bare.set("joins", None)
521
+ synthesized = synthesized.join(bare)
522
+ where = stmt.args.get("where")
523
+ if where is not None:
524
+ synthesized.set("where", where.copy())
525
+ with_clause = stmt.args.get("with_") or stmt.args.get("with")
526
+ if with_clause is not None:
527
+ synthesized.set("with_", with_clause.copy())
528
+ synthesized_sql = safe_sql(synthesized, dialect)
529
+ if synthesized_sql is None:
530
+ continue
531
+ models.append(
532
+ ScriptModel(
533
+ name=target,
534
+ sql=synthesized_sql,
535
+ parents=_referenced_tables(synthesized),
536
+ qualified_name=_qualified_target(tnode)
537
+ or (f"{active_schema}.{target}" if active_schema else None),
538
+ self_read=True,
539
+ )
540
+ )
541
+ return models
542
+
543
+
544
+ def expand_script(
545
+ file_stem: str, sql: str, dialect: str | None, warnings: list[str] | None = None
546
+ ) -> list[ScriptModel]:
547
+ """Turn one .sql file into the models it defines.
548
+
549
+ - CREATE VIEW / CREATE TABLE AS SELECT / INSERT INTO ... SELECT -> a model named
550
+ for the target, analyzed from its SELECT.
551
+ - CREATE TABLE (col defs) -> a base-table model carrying its declared columns
552
+ (the schema, so downstream SELECT * can resolve).
553
+ - a single bare SELECT with no target -> one model named after the file.
554
+ Everything else (DROP, SET, INSERT ... VALUES) contributes nothing.
555
+ `warnings`, when given, collects forfeits the caller should surface.
556
+ """
557
+ from ripple.engine.preprocess import prepare_sql_for_parse
558
+
559
+ # the same pre-parse encoding the statement path uses; parsing the raw
560
+ # text instead dropped or shredded jinja chain-path DML targets
561
+ # (cycle-13 review, F8)
562
+ sql, _ = prepare_sql_for_parse(sql, dialect or "")
563
+
564
+ models: list[ScriptModel] = []
565
+ if re.search(r"\bRETURNS\s+@[\w$]+\s+(?:as\s+)?TABLE\b", sql, re.I):
566
+ from ripple.engine.tsql_tvf import expand_table_functions
567
+
568
+ sql, tvf_models = expand_table_functions(sql, warnings=warnings)
569
+ models.extend(tvf_models)
570
+ statements = split_statements(sql, dialect)
571
+ bare_selects: list[exp.Expression] = []
572
+
573
+ tsql_scalars = "@" in sql and (dialect or "").lower() == "tsql"
574
+ if tsql_scalars:
575
+ from ripple.engine.tsql_scalar_vars import inline_scalar_params
576
+
577
+ inline_scalar_params(statements, dialect)
578
+
579
+ # SET search_path = disclosure, pg_catalog: everything created bare after
580
+ # it lives in that schema. openFEC's migrations create every view this
581
+ # way, and round 3 scored 12 correct answers as misses because the
582
+ # schema-qualified spelling existed nowhere on the model.
583
+ active_schema: str | None = None
584
+
585
+ for stmt in statements:
586
+ if isinstance(stmt, (exp.Set, exp.Command)):
587
+ # an ungenerable SET (tsql `SET @x -= 1`) forfeits only its own
588
+ # search_path sniff, never the rest of the script
589
+ text = safe_sql(stmt, dialect) or ""
590
+ if _SEARCH_PATH_RESET_RE.search(text):
591
+ # RESET search_path / SET search_path TO DEFAULT: a stale
592
+ # schema must not leak onto later creations
593
+ active_schema = None
594
+ else:
595
+ m = _SEARCH_PATH_RE.search(text)
596
+ if m:
597
+ schema = m.group(1)
598
+ active_schema = None if schema.lower() in ("pg_catalog", "public") else schema
599
+ if isinstance(stmt, exp.Create):
600
+ if "INDEX" in (stmt.kind or "").upper():
601
+ # an index is not a relation: minting a model for it made
602
+ # full_text.sql's stem five-way ambiguous (nycdb, holdout
603
+ # round 5 gap cycle). Substring match: tsql kinds read
604
+ # CLUSTERED INDEX / NONCLUSTERED INDEX (cycle-6 review)
605
+ continue
606
+ target = _table_name(stmt.this)
607
+ # CREATE TABLE x AS ( SELECT ... ): the paren wrapper must not
608
+ # reach the model's SQL, the statement layer refuses it as
609
+ # not-a-SELECT and the whole derivation vanishes (nycdb
610
+ # business_addrs, the real mechanism of round 5's class B)
611
+ inner = stmt.expression # the SELECT for a view / CTAS, else None
612
+ while isinstance(inner, (exp.Subquery, exp.Paren)):
613
+ inner = inner.this
614
+ if target and isinstance(inner, (exp.Select, exp.Union)):
615
+ inner = _realias_to_insert_columns(stmt, inner)
616
+ inner_sql = safe_sql(inner, dialect)
617
+ if inner_sql is None:
618
+ # the derivation exists but cannot be regenerated: the
619
+ # model is forfeited outright. A schema-only stub here
620
+ # would silently swallow downstream reads (cycle-12
621
+ # review, F14).
622
+ if warnings is not None:
623
+ warnings.append(
624
+ f"CREATE {target}: its SELECT body could not be "
625
+ "regenerated; the model is forfeited"
626
+ )
627
+ continue
628
+ models.append(
629
+ ScriptModel(
630
+ name=target,
631
+ sql=inner_sql,
632
+ parents=_referenced_tables(inner),
633
+ columns=_create_columns(stmt),
634
+ schema_clone=_is_schema_clone(inner),
635
+ qualified_name=_qualified_target(stmt.this)
636
+ or (f"{active_schema}.{target}" if active_schema else None),
637
+ )
638
+ )
639
+ elif target:
640
+ models.append(
641
+ ScriptModel(
642
+ name=target,
643
+ columns=_create_columns(stmt),
644
+ is_schema=True,
645
+ qualified_name=_qualified_target(stmt.this)
646
+ or (f"{active_schema}.{target}" if active_schema else None),
647
+ )
648
+ )
649
+ elif isinstance(stmt, exp.Insert):
650
+ target = _table_name(stmt.this)
651
+ inner = stmt.expression
652
+ if target and isinstance(inner, (exp.Select, exp.Union)):
653
+ inner = _realias_to_insert_columns(stmt, inner)
654
+ inner_sql = safe_sql(inner, dialect)
655
+ if inner_sql is None:
656
+ continue
657
+ models.append(
658
+ ScriptModel(
659
+ name=target,
660
+ sql=inner_sql,
661
+ parents=_referenced_tables(inner),
662
+ schema_clone=_is_schema_clone(inner),
663
+ qualified_name=_qualified_target(stmt.this)
664
+ or (f"{active_schema}.{target}" if active_schema else None),
665
+ )
666
+ )
667
+ elif isinstance(stmt, exp.Update):
668
+ # UPDATE t SET col = expr is a derivation of t: nycdb builds
669
+ # every dataset this way (ALTER ... ADD COLUMN + UPDATE), and
670
+ # round 5 scored all eleven of its cases as empty because no
671
+ # model existed. Synthesized as a SELECT so the ordinary
672
+ # engine walks the expressions.
673
+ models.extend(_update_models(stmt, dialect, active_schema))
674
+ elif isinstance(stmt, (exp.Select, exp.Union)):
675
+ # the INTO lives on the leftmost SELECT, including under a set
676
+ # operation (review: SELECT INTO ... UNION ALL ...)
677
+ head = stmt
678
+ while isinstance(head, exp.Union):
679
+ head = head.this
680
+ into = head.args.get("into") if isinstance(head, exp.Select) else None
681
+ target = _table_name(into.this) if into is not None else None
682
+ if target:
683
+ # SELECT ... INTO t is a creation of t (postgres); leaving it
684
+ # a bare select made the target read as a PARENT and left no
685
+ # model to resolve (nyc export.sql, holdout round 4)
686
+ body = stmt.copy()
687
+ body_head = body
688
+ while isinstance(body_head, exp.Union):
689
+ body_head = body_head.this
690
+ body_head.set("into", None)
691
+ body_sql = safe_sql(body, dialect)
692
+ if body_sql is None:
693
+ continue
694
+ models.append(
695
+ ScriptModel(
696
+ name=target,
697
+ sql=body_sql,
698
+ parents=_referenced_tables(body),
699
+ schema_clone=_is_schema_clone(body),
700
+ qualified_name=_qualified_target(into.this)
701
+ or (f"{active_schema}.{target}" if active_schema else None),
702
+ )
703
+ )
704
+ else:
705
+ bare_selects.append(stmt)
706
+
707
+ if tsql_scalars:
708
+ from ripple.engine.tsql_scalar_vars import scalar_value_insert_models
709
+
710
+ models.extend(scalar_value_insert_models(statements, dialect))
711
+
712
+ # a write into an @tablevar is procedure-local scratch, not a relation:
713
+ # minting models for them fabricated @loadedModules/@dm_os_memory_clerks
714
+ # and multiplied the file stem's owners until the column pick refused
715
+ # (sp_BlitzInMemoryOLTP, holdout round 6)
716
+ models = [m for m in models if not m.name.startswith("@")]
717
+ if not models and bare_selects:
718
+ # a plain query file: one model named for the file. The whole file is
719
+ # kept only when the query is the file's sole statement; a CREATE TEMP
720
+ # FUNCTION prelude (crux compliance-rates, holdout round 4) must not
721
+ # ride along, because multi-statement model SQL yields no edges.
722
+ body_sql = sql if len(statements) == 1 else safe_sql(bare_selects[-1], dialect)
723
+ if body_sql is not None:
724
+ models.append(
725
+ ScriptModel(
726
+ name=file_stem, sql=body_sql, parents=_referenced_tables(bare_selects[-1])
727
+ )
728
+ )
729
+ return models