ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,1038 @@
1
+ """Select-item source extraction.
2
+
3
+ Walks a statement's select list and fills the result map: for every output
4
+ column, which upstream (table, column) pairs feed it, with confidence and
5
+ trust metadata. Handles aliases, stars, qualified stars, script TRANSFORM,
6
+ JSON access, array-expansion aliases, lateral column aliases, ambiguity
7
+ resolution against the warehouse schema, and CTE tracing of each source.
8
+ """
9
+
10
+ import logging
11
+ from typing import Any
12
+
13
+ from sqlglot import exp
14
+
15
+ from ripple.engine.column_ref import (
16
+ column_parts,
17
+ expansion_field_reads,
18
+ in_selector_argument,
19
+ in_window_ordering,
20
+ register_subquery_from_aliases,
21
+ resolve_column_ref,
22
+ scoped_unnest_entry,
23
+ under_subquery_where,
24
+ )
25
+ from ripple.engine.cte_tracing import (
26
+ expand_star_columns,
27
+ star_except_spellings,
28
+ trace_through_ctes,
29
+ )
30
+ from ripple.engine.json_sources import collect_json_sources
31
+ from ripple.engine.safe_gen import safe_sql
32
+ from ripple.engine.schema_qualification import resolve_column_table
33
+ from ripple.engine.tsql_catalog import system_catalog_owner
34
+ from ripple.engine.types import (
35
+ LATERAL_ALIAS_DIALECTS,
36
+ META_CTE_SHADOWED,
37
+ META_NEEDS_EXPANSION,
38
+ META_PREFIX,
39
+ WarehouseColumns,
40
+ )
41
+
42
+ logger = logging.getLogger(__name__)
43
+
44
+ try:
45
+ from sqlglot import exp as _sqlglot_exp
46
+
47
+ _QUERY_TRANSFORM_CLS = getattr(_sqlglot_exp, "QueryTransform", None)
48
+ except ImportError: # pragma: no cover
49
+ _QUERY_TRANSFORM_CLS = None
50
+
51
+
52
+ def _append_column_sources(
53
+ result: dict[str, Any], column: str, entries: list[dict[str, Any]]
54
+ ) -> None:
55
+ """Append entries to result[column], deduped by (table, column)."""
56
+ existing = result.setdefault(column, [])
57
+ seen = {(e.get("table", ""), e.get("column", "")) for e in existing}
58
+ for entry in entries:
59
+ key = (entry.get("table", ""), entry.get("column", ""))
60
+ if key in seen or not (key[0] or key[1]):
61
+ continue
62
+ seen.add(key)
63
+ existing.append(entry)
64
+
65
+
66
+ def _expansion_extra_entries(
67
+ expansion_info: dict[str, Any], expanded_as: str
68
+ ) -> list[dict[str, Any]]:
69
+ """Source entries for the extra columns a computed array expression reads.
70
+
71
+ UNNEST(GENERATE_DATE_ARRAY(start_date, LEAST(end_date, ...))) AS dt: the
72
+ element derives from every argument column, not just the first."""
73
+ entries = []
74
+ for extra in expansion_info.get("extra_columns", []):
75
+ column = extra.get("source_column", "")
76
+ if not column:
77
+ continue
78
+ entries.append(
79
+ {
80
+ "table": extra.get("source_table", ""),
81
+ "column": column,
82
+ "confidence": 0.5,
83
+ "trust_level": "moderate",
84
+ "inferred": False,
85
+ "from_array_expansion": True,
86
+ "expansion_type": expansion_info.get("expansion_type"),
87
+ "source_column_in_array": column,
88
+ "expanded_as": expanded_as,
89
+ }
90
+ )
91
+ return entries
92
+
93
+
94
+ def _field_selective_entries(
95
+ expansion_info: dict[str, Any],
96
+ column: str,
97
+ col: "exp.Column",
98
+ field_path: str | None = None,
99
+ ) -> list[dict[str, Any]] | None:
100
+ """Source entries for a read of one named STRUCT field through an
101
+ array-of-structs expansion; None when the expansion has no field map
102
+ for the read (callers keep the every-column behavior)."""
103
+ selective = expansion_field_reads(expansion_info, col, column, field_path)
104
+ if selective is None:
105
+ return None
106
+ return [
107
+ {
108
+ "table": fs.get("source_table", ""),
109
+ "column": fs.get("source_column", ""),
110
+ "confidence": 0.5,
111
+ "trust_level": "moderate",
112
+ "inferred": False,
113
+ "from_array_expansion": True,
114
+ "expansion_type": expansion_info.get("expansion_type"),
115
+ "source_column_in_array": fs.get("source_column", ""),
116
+ "expanded_as": column,
117
+ }
118
+ for fs in selective
119
+ if fs.get("source_column")
120
+ ]
121
+
122
+
123
+ def process_select_items(
124
+ selects_list: list["exp.Expression"],
125
+ select_for_from: "exp.Expression",
126
+ result: dict[str, Any],
127
+ warnings: list[dict],
128
+ alias_map: dict[str, str],
129
+ all_tables: list[str],
130
+ array_expansion_sources: dict[str, dict[str, str]],
131
+ default_table: str,
132
+ join_anchor: str,
133
+ cte_columns: dict[str, dict[str, list[dict]]],
134
+ cte_collision_names: set[str],
135
+ dialect: str,
136
+ warehouse_columns: WarehouseColumns | None,
137
+ qualified_via_schema: bool,
138
+ ) -> None:
139
+ """Fill result with per-output-column sources; mutates result/warnings."""
140
+
141
+ def _sole_enumerated_cte_owner(column: str) -> str | None:
142
+ """The one relation in scope proven to own the column, or None.
143
+
144
+ Provable only when EVERY relation is a CTE whose outputs are fully
145
+ enumerated (present in cte_columns, no star, not shadowed); a real
146
+ table or star-veiled CTE makes ownership unprovable and keeps the
147
+ ambiguous refusal."""
148
+ owners: list[str] = []
149
+ for t in all_tables:
150
+ entry = cte_columns.get(str(t).lower())
151
+ if entry is None or "*" in entry or entry.get(META_CTE_SHADOWED):
152
+ return None
153
+ if any(
154
+ not str(k).startswith(META_PREFIX) and str(k).lower() == column.lower()
155
+ for k in entry
156
+ ):
157
+ owners.append(t)
158
+ return owners[0] if len(owners) == 1 else None
159
+
160
+ def _lookup_anchor(column: str) -> str:
161
+ """Lookup-shape fallback for an unqualified read; never a CTE
162
+ whose known outputs prove the column absent."""
163
+ if not join_anchor:
164
+ return ""
165
+ anchor_cols = cte_columns.get(join_anchor.lower())
166
+ if anchor_cols is not None and not any(k.lower() == column.lower() for k in anchor_cols):
167
+ return ""
168
+ return join_anchor
169
+
170
+ register_subquery_from_aliases(selects_list, alias_map)
171
+ # Output maps of THIS select's aliased FROM/JOIN subqueries, plus the
172
+ # relations registered at this level (find_all descends into subqueries,
173
+ # so all_tables alone cannot say what sits at this level). Together they
174
+ # support elimination: an unqualified column absent from every derived
175
+ # item's outputs must come from the one real table (patentsview's
176
+ # webtool joins, holdout round 5).
177
+ from ripple.engine.column_ref import (
178
+ alias_is_foldable,
179
+ nested_scope_sole_table,
180
+ owning_select,
181
+ qualified_table_name,
182
+ qualifier_is_quoted,
183
+ unwrap_select,
184
+ )
185
+ from ripple.engine.cte_tracing import (
186
+ extract_select_column_sources as _sub_map,
187
+ )
188
+ from ripple.engine.cte_tracing import (
189
+ extract_set_op_column_sources as _sub_set_op_map,
190
+ )
191
+ from ripple.engine.cte_tracing import (
192
+ lateral_subquery_map as _lateral_subquery_map,
193
+ )
194
+ from ripple.engine.cte_tracing import subquery_column_entries as _subquery_column_entries
195
+
196
+ subquery_output_maps: list[dict] = []
197
+ subquery_maps_by_alias: dict[str, dict] = {}
198
+ subquery_maps_folded: dict[str, dict] = {}
199
+ local_real_tables: list[str] = []
200
+ from ripple.engine.column_ref import table_function_kind as _tf_kind
201
+
202
+ _from = select_for_from.args.get("from_") or select_for_from.args.get("from")
203
+ # nested subqueries belong to their own select; registering them here
204
+ # leaked inner projections into the outer scope (the cycle-8 review)
205
+ _sub_nodes = (
206
+ [sq for sq in _from.find_all(exp.Subquery) if owning_select(sq) is select_for_from]
207
+ if _from is not None
208
+ else []
209
+ )
210
+ if _from is not None:
211
+ for _t in _from.find_all(exp.Table):
212
+ if owning_select(_t) is select_for_from and _tf_kind(_t) != "srf":
213
+ local_real_tables.append(qualified_table_name(_t))
214
+ for _join in select_for_from.find_all(exp.Join):
215
+ if owning_select(_join) is not select_for_from:
216
+ continue
217
+ if isinstance(_join.this, exp.Subquery):
218
+ _sub_nodes.append(_join.this)
219
+ elif isinstance(_join.this, exp.Table) and _tf_kind(_join.this) != "srf":
220
+ local_real_tables.append(qualified_table_name(_join.this))
221
+ for _sq in _sub_nodes:
222
+ _inner = unwrap_select(_sq.this)
223
+ try:
224
+ if isinstance(_inner, exp.Select):
225
+ _map = _sub_map(_inner, dialect)
226
+ elif isinstance(_inner, (exp.Union, exp.Except, exp.Intersect)):
227
+ _map = _sub_set_op_map(_inner, dialect)
228
+ else:
229
+ continue
230
+ except Exception:
231
+ continue
232
+ subquery_output_maps.append(_map)
233
+ if _sq.alias:
234
+ subquery_maps_by_alias[_sq.alias] = _map
235
+ if alias_is_foldable(_sq):
236
+ subquery_maps_folded.setdefault(_sq.alias.lower(), _map)
237
+ else:
238
+ # an UNALIASED from-subquery: its projection IS this scope's
239
+ # source (transfermarkt's dedup wrapper, holdout round 7) but
240
+ # only when it is the SOLE relation; a peer subquery or any
241
+ # other relation makes its outputs a coin flip (review
242
+ # of cycle 8)
243
+ if (
244
+ len([q for q in _sub_nodes if not q.alias]) == 1
245
+ and len(_sub_nodes) == 1
246
+ and not local_real_tables
247
+ ):
248
+ subquery_maps_by_alias.setdefault("", _map)
249
+
250
+ # lateral derived tables (CROSS/OUTER APPLY (SELECT ...) AS x (cols)),
251
+ # in document order so each apply sees the maps of earlier ones
252
+ # (DarkQueries' stacked CROSS APPLY chain, holdout round 8)
253
+ for _join in select_for_from.find_all(exp.Join):
254
+ if owning_select(_join) is not select_for_from:
255
+ continue
256
+ _lat = _join.this
257
+ if not isinstance(_lat, exp.Lateral) or not _lat.alias:
258
+ continue
259
+ try:
260
+ _lat_map = _lateral_subquery_map(
261
+ _lat,
262
+ dialect,
263
+ correlated_aliases=alias_map,
264
+ correlated_maps=subquery_maps_by_alias,
265
+ correlated_tables=all_tables,
266
+ )
267
+ except Exception:
268
+ continue
269
+ if _lat_map is None:
270
+ continue
271
+ subquery_maps_by_alias[_lat.alias] = _lat_map
272
+ if alias_is_foldable(_lat):
273
+ subquery_maps_folded.setdefault(_lat.alias.lower(), _lat_map)
274
+
275
+ def _subquery_map_for(alias: str, quoted: bool) -> dict | None:
276
+ # a quoted reference matches only exactly; an unquoted one folds
277
+ # case against foldable definitions (the cycle-6 review:
278
+ # "Q" and q are different postgres aliases)
279
+ exact = subquery_maps_by_alias.get(str(alias))
280
+ if exact is not None or quoted:
281
+ return exact
282
+ return subquery_maps_folded.get(str(alias).lower())
283
+
284
+ def _derived_expansion_entries(expansion_info: dict, expanded_as: str) -> list[dict] | None:
285
+ """An expansion whose root column is a derived table's output reads
286
+ through that derivation: data.nodes(...) over FROM (SELECT
287
+ CONVERT(XML, target_data) AS data ...) t traces to target_data,
288
+ never to a phantom relation (InvestigateWaits, holdout round 8)."""
289
+ root_table = expansion_info.get("source_table") or ""
290
+ root_column = expansion_info.get("source_column") or ""
291
+ if not root_column:
292
+ return None
293
+ sub_map = _subquery_map_for(root_table, False) if root_table else None
294
+ if sub_map is None:
295
+ if root_table or local_real_tables:
296
+ return None
297
+ owners = [
298
+ m
299
+ for m in subquery_output_maps
300
+ if any(
301
+ k != "*"
302
+ and not str(k).startswith(META_PREFIX)
303
+ and str(k).lower() == root_column.lower()
304
+ for k in m
305
+ )
306
+ ]
307
+ if len(owners) != 1 or any("*" in m for m in subquery_output_maps):
308
+ return None
309
+ sub_map = owners[0]
310
+
311
+ def _chained(chain_map: dict, chain_column: str) -> list[dict]:
312
+ entries = []
313
+ for src in _subquery_column_entries(chain_map, chain_column):
314
+ if not (src.get("table") or src.get("column")):
315
+ continue
316
+ entry = dict(src)
317
+ entry["confidence"] = min(entry.get("confidence", 1.0), 0.5)
318
+ entry["trust_level"] = "moderate"
319
+ entry["from_array_expansion"] = True
320
+ entry["expansion_type"] = expansion_info.get("expansion_type")
321
+ entry["expanded_as"] = expanded_as
322
+ entries.append(entry)
323
+ return entries
324
+
325
+ entries = _chained(sub_map, root_column)
326
+ if not entries:
327
+ return None
328
+ # every argument counts: the extras chain through their own derived
329
+ # tables exactly like the primary (cycle-9 review, F10)
330
+ for extra in expansion_info.get("extra_columns", []):
331
+ extra_column = extra.get("source_column") or ""
332
+ if not extra_column:
333
+ continue
334
+ extra_table = extra.get("source_table") or ""
335
+ extra_map = _subquery_map_for(extra_table, False) if extra_table else None
336
+ if extra_map is not None:
337
+ entries.extend(_chained(extra_map, extra_column))
338
+ else:
339
+ entries.extend(
340
+ _expansion_extra_entries(
341
+ {**expansion_info, "extra_columns": [extra]}, expanded_as
342
+ )
343
+ )
344
+ return entries
345
+
346
+ def _selective_expansion_entries(
347
+ expansion_info: dict, column: str, col: "exp.Column", field_path: str | None
348
+ ) -> list[dict] | None:
349
+ """Field-selective entries with each source chained through its own
350
+ derived table's map. Selectivity runs BEFORE the derived-expansion
351
+ fallback: a named-struct melt over subquery outputs must keep each
352
+ field's own sources, not every input column (cycle-12, F16)."""
353
+ selective = _field_selective_entries(expansion_info, column, col, field_path)
354
+ if selective is None:
355
+ return None
356
+ entries: list[dict] = []
357
+ for entry in selective:
358
+ table = entry.get("table") or ""
359
+ sub_map = _subquery_map_for(table, False) if table else None
360
+ if sub_map is None:
361
+ entries.append(entry)
362
+ continue
363
+ for src in _subquery_column_entries(sub_map, entry.get("column", "")):
364
+ if not (src.get("table") or src.get("column")):
365
+ continue
366
+ chained_entry = dict(src)
367
+ chained_entry["confidence"] = min(chained_entry.get("confidence", 1.0), 0.5)
368
+ chained_entry["trust_level"] = "moderate"
369
+ chained_entry["from_array_expansion"] = True
370
+ chained_entry["expansion_type"] = expansion_info.get("expansion_type")
371
+ chained_entry["expanded_as"] = column
372
+ entries.append(chained_entry)
373
+ return entries
374
+
375
+ # Every name a multi-part column reference could legitimately anchor
376
+ # to: aliases, table names (bare and as written), expansion aliases,
377
+ # inline-subquery aliases. Used to tell struct access (o.customer.name)
378
+ # from qualified table.column (ds.orders.customer) without fabricating
379
+ # tables.
380
+ _known_relations = (
381
+ {str(k).lower() for k in alias_map}
382
+ | {str(v).lower() for v in alias_map.values()}
383
+ | {str(k).lower() for k in array_expansion_sources}
384
+ | {str(k).lower() for k in subquery_maps_by_alias}
385
+ )
386
+
387
+ # Select-item aliases already resolved, in list order, for dialects
388
+ # where a later item may reference an earlier alias (lateral column
389
+ # aliases). The alias is a name of the expression, not a column of
390
+ # the upstream relation.
391
+ lateral_aliases: dict[str, list[dict[str, Any]]] = {}
392
+ allow_lateral = (dialect or "").lower() in LATERAL_ALIAS_DIALECTS
393
+
394
+ for item_idx, select_expr in enumerate(selects_list):
395
+ # Qualified star over a known CTE: rel.* (except ...). Expand
396
+ # through the CTE column map (following star-passthrough chains)
397
+ # and trace each named column; collapsing to '*' would silently
398
+ # drop most of the model's columns.
399
+ if isinstance(select_expr, exp.Column) and isinstance(select_expr.this, exp.Star):
400
+ star = select_expr.this
401
+ qualifier = select_expr.table or ""
402
+ rel = str(alias_map.get(qualifier, qualifier))
403
+ except_cols = star_except_spellings(
404
+ star.args.get("except_") or star.args.get("except") or []
405
+ )
406
+ # rel.* over an inline subquery alias: absorb the subquery's own
407
+ # map. The round-5 fix landed only in the CTE mappers; at the
408
+ # model's own top level the alias still surfaced as a struct
409
+ # column of the inner table (the review of PR #28).
410
+ sub_absorbed = _subquery_map_for(qualifier, qualifier_is_quoted(select_expr))
411
+ if sub_absorbed is not None:
412
+ for name, srcs in sub_absorbed.items():
413
+ if name == "*":
414
+ _append_column_sources(result, "*", srcs)
415
+ result[META_NEEDS_EXPANSION] = True
416
+ elif name.lower() not in {e.lower() for e in except_cols}:
417
+ result.setdefault(name, [dict(s) for s in srcs])
418
+ continue
419
+ expanded, residual = expand_star_columns([rel], cte_columns, except_cols)
420
+ if expanded:
421
+ for name, srcs in expanded.items():
422
+ result.setdefault(name, srcs)
423
+ if residual:
424
+ _append_column_sources(result, "*", residual)
425
+ result[META_NEEDS_EXPANSION] = True
426
+ continue
427
+ if residual:
428
+ # rel.* over an unexpandable relation keeps its wildcard
429
+ # link WITH its exclusions, appended so a second qualified
430
+ # star never overwrites the first (cycle-12, F4/F5)
431
+ _append_column_sources(result, "*", residual)
432
+ result[META_NEEDS_EXPANSION] = True
433
+ continue
434
+ # a nameless qualifier falls through to the ordinary paths
435
+
436
+ is_computed = False
437
+
438
+ if isinstance(select_expr, exp.Alias):
439
+ col_name = select_expr.alias
440
+ source_expr = select_expr.this
441
+ is_computed = not isinstance(source_expr, exp.Column)
442
+ elif isinstance(select_expr, exp.Column):
443
+ col_name = select_expr.name
444
+ source_expr = select_expr
445
+ elif isinstance(select_expr, exp.Star):
446
+ if not local_real_tables and subquery_output_maps:
447
+ # bare star over a scope of only derived tables: the
448
+ # columns are the subqueries' own projections, all of
449
+ # them (transfermarkt's dedup wrapper, holdout round 7;
450
+ # generalized to peers by the cycle-8 review)
451
+ for m in subquery_output_maps:
452
+ for name, srcs in m.items():
453
+ if str(name).startswith("\x00"):
454
+ continue
455
+ if name == "*":
456
+ _append_column_sources(result, "*", [dict(s) for s in srcs])
457
+ result[META_NEEDS_EXPANSION] = True
458
+ else:
459
+ result.setdefault(name, [dict(s) for s in srcs])
460
+ continue
461
+ # A bare star over known CTEs expands through the CTE column
462
+ # map, so named CTE columns become outputs instead of
463
+ # vanishing behind '*' with the CTE name as a phantom source.
464
+ except_cols = star_except_spellings(
465
+ select_expr.args.get("except_") or select_expr.args.get("except") or []
466
+ )
467
+ expanded, residual = expand_star_columns(all_tables, cte_columns, except_cols)
468
+ if expanded:
469
+ for name, srcs in expanded.items():
470
+ result.setdefault(name, srcs)
471
+ if residual:
472
+ _append_column_sources(result, "*", residual)
473
+ result[META_NEEDS_EXPANSION] = True
474
+ elif residual:
475
+ # a pure star chain (b -> a -> real) yields no named columns
476
+ # but the walk still found the real relations; rebuilding
477
+ # from all_tables here resurfaced the CTE itself as a
478
+ # phantom source (the cycle-7 review)
479
+ _append_column_sources(result, "*", residual)
480
+ result[META_NEEDS_EXPANSION] = True
481
+ else:
482
+ result["*"] = [
483
+ {"table": table_name, "column": "*", "confidence": 0.5}
484
+ for table_name in all_tables
485
+ ]
486
+ result[META_NEEDS_EXPANSION] = True
487
+ continue
488
+ elif _QUERY_TRANSFORM_CLS is not None and isinstance(select_expr, _QUERY_TRANSFORM_CLS):
489
+ # SELECT TRANSFORM(...) USING 'script': the mapping through
490
+ # the script is unknowable, but inputs and declared outputs
491
+ # are static. Honest contract: every output depends on every
492
+ # input, all at review_required. Never assume identity, even
493
+ # for 'cat' (ROW FORMAT can reshape fields).
494
+ schema_arg = select_expr.args.get("schema")
495
+ outputs = (
496
+ [s.name for s in schema_arg.expressions]
497
+ if schema_arg is not None and schema_arg.expressions
498
+ else ["key", "value"] # Spark's documented default
499
+ )
500
+ script_node = select_expr.args.get("command_script")
501
+ script = script_node.name if script_node is not None else "?"
502
+ input_sources = []
503
+ input_names = []
504
+ for item in select_expr.expressions or []:
505
+ if isinstance(item, exp.Star):
506
+ for table_name in all_tables:
507
+ input_sources.append(
508
+ {
509
+ "table": table_name,
510
+ "column": "*",
511
+ "confidence": 0.3,
512
+ "trust_level": "review_required",
513
+ "via_script_transform": True,
514
+ "script": script,
515
+ }
516
+ )
517
+ result[META_NEEDS_EXPANSION] = True
518
+ input_names.append("*")
519
+ continue
520
+ for col in item.find_all(exp.Column):
521
+ col_alias = col.table or ""
522
+ table = (
523
+ alias_map.get(col_alias, col_alias) if col_alias else (default_table or "")
524
+ )
525
+ input_sources.append(
526
+ {
527
+ "table": table,
528
+ "column": col.name,
529
+ "confidence": 0.3,
530
+ "trust_level": "review_required",
531
+ "via_script_transform": True,
532
+ "script": script,
533
+ }
534
+ )
535
+ input_names.append(col.name)
536
+ for out in outputs:
537
+ result[out] = [dict(s) for s in input_sources]
538
+ warnings.append(
539
+ {
540
+ "warning_type": "script_transform",
541
+ "message": (
542
+ f"SELECT TRANSFORM pipes rows through external script '{script}'; "
543
+ f"the column mapping through the script is opaque. Each of the "
544
+ f"{len(outputs)} output columns is conservatively linked to all "
545
+ f"{len(input_names)} input columns at review_required."
546
+ ),
547
+ "script": script,
548
+ "output_columns": outputs,
549
+ "input_columns": input_names,
550
+ }
551
+ )
552
+ continue
553
+ else:
554
+ col_sql = safe_sql(select_expr, dialect)
555
+ col_name = col_sql if col_sql is not None and len(col_sql) < 50 else None
556
+ source_expr = select_expr
557
+ is_computed = True
558
+
559
+ if not col_name:
560
+ continue
561
+ if "{" in col_name:
562
+ # a template placeholder is not a real output name; exposing it
563
+ # (or its internal dot encoding) leaked into reports and
564
+ # exported nodes (cycle-12, F19). The output stays, opaque.
565
+ col_name = f"_col_{item_idx}"
566
+
567
+ sources = []
568
+ seen = set()
569
+
570
+ json_columns_handled = collect_json_sources(
571
+ source_expr,
572
+ _known_relations,
573
+ alias_map,
574
+ array_expansion_sources,
575
+ default_table,
576
+ qualified_via_schema,
577
+ sources,
578
+ seen,
579
+ )
580
+
581
+ # qualify() rewrites a read of a table-valued alias (CROSS JOIN
582
+ # jsonb_array_elements_text(roles) AS role ... SELECT role) into a
583
+ # TableColumn node the Column sweep never sees (kingfisher, round 7)
584
+ _tc_cls = getattr(exp, "TableColumn", None)
585
+ if _tc_cls is not None:
586
+ for tc in source_expr.find_all(_tc_cls):
587
+ entry = array_expansion_sources.get(tc.name) or array_expansion_sources.get(
588
+ tc.name.lower()
589
+ )
590
+ if not entry:
591
+ continue
592
+ reads = [(entry.get("source_table", ""), entry.get("source_column", ""))] + [
593
+ (x.get("source_table", ""), x.get("source_column", ""))
594
+ for x in entry.get("extra_columns", [])
595
+ ]
596
+ for key in reads:
597
+ if key not in seen and (key[0] or key[1]):
598
+ seen.add(key)
599
+ sources.append(
600
+ {
601
+ "table": key[0],
602
+ "column": key[1],
603
+ "confidence": 0.5,
604
+ "trust_level": "moderate",
605
+ "inferred": False,
606
+ "from_array_expansion": True,
607
+ "expansion_type": entry.get("expansion_type"),
608
+ }
609
+ )
610
+
611
+ # TODO: this classification block near-duplicates cte_tracing.extract_select_column_sources; merging needs judgment on trust/warning metadata.
612
+ for col in source_expr.find_all(exp.Column):
613
+ if id(col) in json_columns_handled:
614
+ continue # Already handled as part of JSONExtract
615
+ if under_subquery_where(col, source_expr):
616
+ # a nested subquery's WHERE selects rows there; it never
617
+ # feeds the produced value
618
+ continue
619
+ if in_window_ordering(col, source_expr):
620
+ # frames the window, does not flow into its value
621
+ continue
622
+ if in_selector_argument(col, source_expr):
623
+ # picks the arg_max/arg_min row, does not flow into its value
624
+ continue
625
+ alias, column, field_path, certain = resolve_column_ref(col, _known_relations)
626
+
627
+ if "{" in column or (not certain and any("{" in p for p in column_parts(col))):
628
+ # a {placeholder} or jinja chain in column OR qualifier
629
+ # position may format to anything; only relation-position
630
+ # braces carry a citable identity. Anchoring the qualifier
631
+ # as a struct root fabricated a source (cycle-12, F11).
632
+ spelled = ".".join(column_parts(col))
633
+ warnings.append(
634
+ {
635
+ "column": spelled,
636
+ "warning_type": "template_placeholder_expression",
637
+ "trust_level": "review_required",
638
+ "message": f"template placeholder {spelled} in an expression; "
639
+ "its lineage cannot be resolved offline",
640
+ }
641
+ )
642
+ continue
643
+
644
+ source_entry: dict[str, Any] = {
645
+ "column": column,
646
+ "inferred": False,
647
+ }
648
+ if field_path:
649
+ source_entry["field_path"] = field_path
650
+
651
+ if not certain:
652
+ # multi-part reference whose leading identifier matches no
653
+ # known relation: read as a struct root. Anchor it to the
654
+ # UNNEST of its own subquery scope, the anonymous UNNEST
655
+ # of this select, or the sole relation when possible;
656
+ # never emit a confident phantom table.
657
+ anon = scoped_unnest_entry(
658
+ col, select_for_from, alias_map, all_tables
659
+ ) or array_expansion_sources.get("__anonymous_unnest__")
660
+ if anon is not None:
661
+ table = anon.get("source_table", "")
662
+ confidence = 0.5
663
+ trust_level = "moderate"
664
+ source_entry["from_array_expansion"] = True
665
+ source_entry["expansion_type"] = anon.get("expansion_type")
666
+ source_entry["source_column_in_array"] = anon.get("source_column", "")
667
+ if anon.get("source_column"):
668
+ source_entry["expanded_as"] = column
669
+ source_entry["column"] = anon["source_column"]
670
+ elif default_table:
671
+ table = default_table
672
+ confidence = 0.5
673
+ trust_level = "moderate"
674
+ source_entry["inferred"] = True
675
+ source_entry["struct_guess"] = True
676
+ else:
677
+ table = ""
678
+ confidence = 0.4
679
+ trust_level = "review_required"
680
+ source_entry["inferred"] = True
681
+ source_entry["unresolved_reference"] = True
682
+ elif alias:
683
+ if alias in array_expansion_sources:
684
+ expansion_info = array_expansion_sources[alias]
685
+ selective = _selective_expansion_entries(
686
+ expansion_info, column, col, field_path
687
+ )
688
+ if selective is not None:
689
+ for entry in selective:
690
+ key = (entry.get("table", ""), entry.get("column", ""))
691
+ if key not in seen:
692
+ seen.add(key)
693
+ sources.append(entry)
694
+ continue
695
+ chained = _derived_expansion_entries(expansion_info, column)
696
+ if chained:
697
+ for entry in chained:
698
+ chain_key = (entry.get("table", ""), entry.get("column", ""))
699
+ if chain_key not in seen:
700
+ seen.add(chain_key)
701
+ sources.append(entry)
702
+ continue
703
+ # Trace to the actual source table/column
704
+ table = expansion_info.get("source_table", alias)
705
+ source_column = expansion_info.get("source_column", column)
706
+ # Lower confidence: we know the source but can't track row expansion
707
+ confidence = 0.5
708
+ trust_level = "moderate"
709
+ source_entry["from_array_expansion"] = True
710
+ source_entry["expansion_type"] = expansion_info.get("expansion_type")
711
+ source_entry["source_column_in_array"] = source_column
712
+ if source_column:
713
+ # the upstream column IS the array; the alias's
714
+ # output name does not exist on the source table
715
+ source_entry["expanded_as"] = column
716
+ source_entry["column"] = source_column
717
+ if expansion_info.get("json_path"):
718
+ source_entry["json_path"] = expansion_info["json_path"]
719
+ for extra_entry in _expansion_extra_entries(expansion_info, column):
720
+ extra_key = (extra_entry["table"], extra_entry["column"])
721
+ if extra_key not in seen:
722
+ seen.add(extra_key)
723
+ sources.append(extra_entry)
724
+ warnings.append(
725
+ {
726
+ "column": column,
727
+ "warning_type": "array_expansion",
728
+ "message": f"Column '{column}' comes from array expansion ({expansion_info.get('expansion_type')}). "
729
+ f"Source array: {table}.{source_column}",
730
+ "source_table": table,
731
+ "source_column": source_column,
732
+ "expansion_type": expansion_info.get("expansion_type"),
733
+ }
734
+ )
735
+ elif (
736
+ sub_absorbed := _subquery_map_for(alias, qualifier_is_quoted(col))
737
+ ) is not None:
738
+ # the qualifier is an inline-subquery alias: the
739
+ # column's sources are the subquery's own, never a
740
+ # relation named after the alias (pensjon under
741
+ # qualify(), the gap named in PR #28)
742
+ for src in _subquery_column_entries(sub_absorbed, column):
743
+ key = (src.get("table", ""), src.get("column", ""))
744
+ if key not in seen and (key[0] or key[1]):
745
+ seen.add(key)
746
+ sources.append(dict(src))
747
+ continue
748
+ else:
749
+ # Column is explicitly qualified - high confidence
750
+ table = alias_map.get(alias, alias)
751
+ confidence = 1.0
752
+ trust_level = "verified" if qualified_via_schema else "high_confidence"
753
+ elif column in array_expansion_sources:
754
+ # Unqualified column that matches an UNNEST output column
755
+ expansion_info = array_expansion_sources[column]
756
+ selective = _selective_expansion_entries(expansion_info, column, col, field_path)
757
+ if selective is not None:
758
+ for entry in selective:
759
+ key = (entry.get("table", ""), entry.get("column", ""))
760
+ if key not in seen:
761
+ seen.add(key)
762
+ sources.append(entry)
763
+ continue
764
+ chained = _derived_expansion_entries(expansion_info, column)
765
+ if chained:
766
+ for entry in chained:
767
+ chain_key = (entry.get("table", ""), entry.get("column", ""))
768
+ if chain_key not in seen:
769
+ seen.add(chain_key)
770
+ sources.append(entry)
771
+ continue
772
+ table = expansion_info.get("source_table", "")
773
+ source_column = expansion_info.get("source_column", column)
774
+ confidence = 0.5
775
+ trust_level = "moderate"
776
+ source_entry["from_array_expansion"] = True
777
+ source_entry["expansion_type"] = expansion_info.get("expansion_type")
778
+ source_entry["source_column_in_array"] = source_column
779
+ source_entry["is_unnest_column"] = expansion_info.get("is_unnest_column", False)
780
+ if source_column:
781
+ source_entry["expanded_as"] = column
782
+ source_entry["column"] = source_column
783
+ for extra_entry in _expansion_extra_entries(expansion_info, column):
784
+ extra_key = (extra_entry["table"], extra_entry["column"])
785
+ if extra_key not in seen:
786
+ seen.add(extra_key)
787
+ sources.append(extra_entry)
788
+ warnings.append(
789
+ {
790
+ "column": column,
791
+ "warning_type": "array_expansion",
792
+ "message": f"Column '{column}' comes from array expansion ({expansion_info.get('expansion_type')}). "
793
+ f"Source array: {table}.{source_column}",
794
+ "source_table": table,
795
+ "source_column": source_column,
796
+ "expansion_type": expansion_info.get("expansion_type"),
797
+ }
798
+ )
799
+ elif (
800
+ allow_lateral
801
+ and column.lower() in lateral_aliases
802
+ and not (
803
+ warehouse_columns
804
+ and resolve_column_table(column, all_tables, warehouse_columns)[1]
805
+ )
806
+ ):
807
+ # lateral column alias: the name belongs to an earlier
808
+ # select item, not to the upstream relation. When the
809
+ # schema proves a real column of that name exists, the
810
+ # table wins; otherwise resolve to the definition.
811
+ for src in lateral_aliases[column.lower()]:
812
+ entry = dict(src)
813
+ entry["via_lateral_alias"] = True
814
+ key = (entry.get("table", ""), entry.get("column", ""))
815
+ if key not in seen and (key[0] or key[1]):
816
+ seen.add(key)
817
+ sources.append(entry)
818
+ continue
819
+ elif (anon_map := subquery_maps_by_alias.get("")) is not None and any(
820
+ not str(k).startswith("\x00") and k != "*" and str(k).lower() == column.lower()
821
+ for k in anon_map
822
+ ):
823
+ # unqualified read of a column the unaliased from-subquery
824
+ # projects (transfermarkt's `where n = 1` wrapper,
825
+ # holdout round 7)
826
+ for src in _subquery_column_entries(anon_map, column):
827
+ key = (src.get("table", ""), src.get("column", ""))
828
+ if key not in seen and (key[0] or key[1]):
829
+ seen.add(key)
830
+ sources.append(dict(src))
831
+ continue
832
+ elif (
833
+ not local_real_tables
834
+ and all("*" not in m for m in subquery_output_maps)
835
+ and len(
836
+ owner_maps := [
837
+ m
838
+ for m in subquery_output_maps
839
+ if any(
840
+ not str(k).startswith("\x00")
841
+ and k != "*"
842
+ and str(k).lower() == column.lower()
843
+ for k in m
844
+ )
845
+ ]
846
+ )
847
+ == 1
848
+ ):
849
+ # this select reads only derived tables, and exactly one of
850
+ # them projects the column: anchoring to a table found
851
+ # INSIDE the subquery skips the derivation (kingfisher's
852
+ # role expansion read through id_role, holdout round 7)
853
+ for src in _subquery_column_entries(owner_maps[0], column):
854
+ key = (src.get("table", ""), src.get("column", ""))
855
+ if key not in seen and (key[0] or key[1]):
856
+ seen.add(key)
857
+ sources.append(dict(src))
858
+ continue
859
+ elif (nested := nested_scope_sole_table(col, select_for_from)) is not None:
860
+ # the column lives in a subselect with its own FROM; the
861
+ # outer sole table or elimination would anchor it a scope
862
+ # too high (the review of PR #28: scalar subqueries in
863
+ # UPDATE SET expressions)
864
+ table = nested
865
+ if nested:
866
+ confidence = 0.8
867
+ trust_level = "high_confidence"
868
+ else:
869
+ confidence = 0.4
870
+ trust_level = "review_required"
871
+ source_entry["unresolved_reference"] = True
872
+ source_entry["inferred"] = True
873
+ elif default_table:
874
+ # Single table in FROM - inferred but reliable
875
+ table = default_table
876
+ confidence = 0.8
877
+ trust_level = "high_confidence"
878
+ source_entry["inferred"] = True
879
+ elif (
880
+ subquery_output_maps
881
+ and len(local_real_tables) == 1
882
+ and all(
883
+ column.lower() not in {str(k).lower() for k in m} for m in subquery_output_maps
884
+ )
885
+ and all("*" not in m for m in subquery_output_maps)
886
+ ):
887
+ # every derived FROM item enumerates its outputs and none
888
+ # carries this name, so the one real table must. A '*' entry
889
+ # means a subquery's outputs are NOT enumerated: the column
890
+ # may well live behind the star, and eliminating on it
891
+ # fabricated an anchor (the review of PR #28)
892
+ table = local_real_tables[0]
893
+ confidence = 0.6
894
+ trust_level = "moderate"
895
+ source_entry["inferred"] = True
896
+ elif (
897
+ len(all_tables) > 1
898
+ and not (
899
+ warehouse_columns
900
+ and resolve_column_table(column, all_tables, warehouse_columns)[1]
901
+ )
902
+ and (_sys_owner := system_catalog_owner(column, all_tables, dialect))
903
+ ):
904
+ # every relation in scope is a documented sys object and
905
+ # exactly one owns the column (InvestigateWaits, round 8);
906
+ # a supplied schema that resolves the column outranks the
907
+ # built-in catalog (cycle-9 review, F1)
908
+ table = _sys_owner
909
+ confidence = 0.8
910
+ trust_level = "high_confidence"
911
+ source_entry["inferred"] = True
912
+ elif len(all_tables) > 1 and (_owner := _sole_enumerated_cte_owner(column)):
913
+ # SQL resolves an unqualified name to the one relation that
914
+ # can own it. Provable only when every relation in scope is
915
+ # a CTE with fully enumerated outputs and exactly one
916
+ # projects the name (dbt_salesforce's manager_id coalesce
917
+ # across three joined aggregate CTEs, holdout round 7).
918
+ table = _owner
919
+ confidence = 0.8
920
+ trust_level = "high_confidence"
921
+ source_entry["inferred"] = True
922
+ elif len(all_tables) > 1 and warehouse_columns:
923
+ # Multiple tables and warehouse schema available - try to resolve
924
+ resolved_table, was_resolved = resolve_column_table(
925
+ column, all_tables, warehouse_columns
926
+ )
927
+ if was_resolved and resolved_table:
928
+ # Successfully resolved ambiguity via warehouse schema
929
+ table = resolved_table
930
+ confidence = 1.0
931
+ trust_level = "verified"
932
+ source_entry["qualified_via_schema"] = True
933
+ source_entry["inferred"] = True
934
+ elif _lookup_anchor(column):
935
+ table = _lookup_anchor(column)
936
+ confidence = 0.6
937
+ trust_level = "moderate"
938
+ source_entry["inferred"] = True
939
+ else:
940
+ # Could not resolve - mark as ambiguous
941
+ table = ""
942
+ confidence = 0.5
943
+ trust_level = "review_required"
944
+ source_entry["ambiguous_tables"] = all_tables
945
+ source_entry["ambiguity_unresolved"] = True
946
+ warnings.append(
947
+ {
948
+ "column": column,
949
+ "warning_type": "ambiguous_column",
950
+ "message": f"Column '{column}' could come from any of: {', '.join(all_tables)}",
951
+ "candidate_tables": all_tables,
952
+ }
953
+ )
954
+ elif len(all_tables) > 1 and _lookup_anchor(column):
955
+ table = _lookup_anchor(column)
956
+ confidence = 0.6
957
+ trust_level = "moderate"
958
+ source_entry["inferred"] = True
959
+ elif len(all_tables) > 1:
960
+ # Multiple tables but no warehouse schema - ambiguous, needs resolution
961
+ table = ""
962
+ confidence = 0.5
963
+ trust_level = "moderate"
964
+ source_entry["ambiguous_tables"] = all_tables
965
+ source_entry["needs_warehouse_connection"] = True
966
+ warnings.append(
967
+ {
968
+ "column": column,
969
+ "warning_type": "ambiguous_column_no_schema",
970
+ "message": f"Column '{column}' is ambiguous. Connect a warehouse to enable automatic resolution.",
971
+ "candidate_tables": all_tables,
972
+ "hint": "Connect a warehouse to enable automatic column resolution.",
973
+ }
974
+ )
975
+ else:
976
+ # No FROM clause or empty tables list
977
+ table = ""
978
+ confidence = 0.5
979
+ trust_level = "moderate"
980
+
981
+ source_entry["table"] = table
982
+ source_entry["confidence"] = confidence
983
+ source_entry["trust_level"] = trust_level
984
+
985
+ key = (table, column)
986
+ if key not in seen and (table or column):
987
+ seen.add(key)
988
+ sources.append(source_entry)
989
+
990
+ if not sources and is_computed:
991
+ sources = [
992
+ {
993
+ "table": "",
994
+ "column": col_name,
995
+ "computed": True,
996
+ "confidence": 0.3,
997
+ "trust_level": "moderate",
998
+ }
999
+ ]
1000
+
1001
+ # Trace sources through CTEs
1002
+ if sources and cte_columns:
1003
+ traced_sources = []
1004
+ seen_traced = set()
1005
+ for src in sources:
1006
+ if src.get("computed"):
1007
+ traced_sources.append(src)
1008
+ elif src.get("table") and src["table"].lower() in cte_columns:
1009
+ traced = trace_through_ctes(
1010
+ src["column"],
1011
+ src["table"],
1012
+ cte_columns,
1013
+ )
1014
+ for t in traced:
1015
+ key = (t.get("table", ""), t.get("column", ""))
1016
+ if key not in seen_traced:
1017
+ seen_traced.add(key)
1018
+ # Degrade confidence if source CTE collides with real table
1019
+ if cte_collision_names and src["table"].lower() in cte_collision_names:
1020
+ t["confidence"] = min(t.get("confidence", 1.0), 0.5)
1021
+ t["trust_level"] = "review_required"
1022
+ t["cte_collision"] = True
1023
+ # Preserve trust metadata from traced sources
1024
+ elif src.get("qualified_via_schema"):
1025
+ t["qualified_via_schema"] = True
1026
+ t["trust_level"] = "verified"
1027
+ traced_sources.append(t)
1028
+ else:
1029
+ key = (src.get("table", ""), src.get("column", ""))
1030
+ if key not in seen_traced:
1031
+ seen_traced.add(key)
1032
+ traced_sources.append(src)
1033
+ sources = traced_sources if traced_sources else sources
1034
+
1035
+ if isinstance(select_expr, exp.Alias) and sources:
1036
+ lateral_aliases.setdefault(col_name.lower(), sources)
1037
+
1038
+ result[col_name] = sources