ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,1038 @@
|
|
|
1
|
+
"""Select-item source extraction.
|
|
2
|
+
|
|
3
|
+
Walks a statement's select list and fills the result map: for every output
|
|
4
|
+
column, which upstream (table, column) pairs feed it, with confidence and
|
|
5
|
+
trust metadata. Handles aliases, stars, qualified stars, script TRANSFORM,
|
|
6
|
+
JSON access, array-expansion aliases, lateral column aliases, ambiguity
|
|
7
|
+
resolution against the warehouse schema, and CTE tracing of each source.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from sqlglot import exp
|
|
14
|
+
|
|
15
|
+
from ripple.engine.column_ref import (
|
|
16
|
+
column_parts,
|
|
17
|
+
expansion_field_reads,
|
|
18
|
+
in_selector_argument,
|
|
19
|
+
in_window_ordering,
|
|
20
|
+
register_subquery_from_aliases,
|
|
21
|
+
resolve_column_ref,
|
|
22
|
+
scoped_unnest_entry,
|
|
23
|
+
under_subquery_where,
|
|
24
|
+
)
|
|
25
|
+
from ripple.engine.cte_tracing import (
|
|
26
|
+
expand_star_columns,
|
|
27
|
+
star_except_spellings,
|
|
28
|
+
trace_through_ctes,
|
|
29
|
+
)
|
|
30
|
+
from ripple.engine.json_sources import collect_json_sources
|
|
31
|
+
from ripple.engine.safe_gen import safe_sql
|
|
32
|
+
from ripple.engine.schema_qualification import resolve_column_table
|
|
33
|
+
from ripple.engine.tsql_catalog import system_catalog_owner
|
|
34
|
+
from ripple.engine.types import (
|
|
35
|
+
LATERAL_ALIAS_DIALECTS,
|
|
36
|
+
META_CTE_SHADOWED,
|
|
37
|
+
META_NEEDS_EXPANSION,
|
|
38
|
+
META_PREFIX,
|
|
39
|
+
WarehouseColumns,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
logger = logging.getLogger(__name__)
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
from sqlglot import exp as _sqlglot_exp
|
|
46
|
+
|
|
47
|
+
_QUERY_TRANSFORM_CLS = getattr(_sqlglot_exp, "QueryTransform", None)
|
|
48
|
+
except ImportError: # pragma: no cover
|
|
49
|
+
_QUERY_TRANSFORM_CLS = None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _append_column_sources(
|
|
53
|
+
result: dict[str, Any], column: str, entries: list[dict[str, Any]]
|
|
54
|
+
) -> None:
|
|
55
|
+
"""Append entries to result[column], deduped by (table, column)."""
|
|
56
|
+
existing = result.setdefault(column, [])
|
|
57
|
+
seen = {(e.get("table", ""), e.get("column", "")) for e in existing}
|
|
58
|
+
for entry in entries:
|
|
59
|
+
key = (entry.get("table", ""), entry.get("column", ""))
|
|
60
|
+
if key in seen or not (key[0] or key[1]):
|
|
61
|
+
continue
|
|
62
|
+
seen.add(key)
|
|
63
|
+
existing.append(entry)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _expansion_extra_entries(
|
|
67
|
+
expansion_info: dict[str, Any], expanded_as: str
|
|
68
|
+
) -> list[dict[str, Any]]:
|
|
69
|
+
"""Source entries for the extra columns a computed array expression reads.
|
|
70
|
+
|
|
71
|
+
UNNEST(GENERATE_DATE_ARRAY(start_date, LEAST(end_date, ...))) AS dt: the
|
|
72
|
+
element derives from every argument column, not just the first."""
|
|
73
|
+
entries = []
|
|
74
|
+
for extra in expansion_info.get("extra_columns", []):
|
|
75
|
+
column = extra.get("source_column", "")
|
|
76
|
+
if not column:
|
|
77
|
+
continue
|
|
78
|
+
entries.append(
|
|
79
|
+
{
|
|
80
|
+
"table": extra.get("source_table", ""),
|
|
81
|
+
"column": column,
|
|
82
|
+
"confidence": 0.5,
|
|
83
|
+
"trust_level": "moderate",
|
|
84
|
+
"inferred": False,
|
|
85
|
+
"from_array_expansion": True,
|
|
86
|
+
"expansion_type": expansion_info.get("expansion_type"),
|
|
87
|
+
"source_column_in_array": column,
|
|
88
|
+
"expanded_as": expanded_as,
|
|
89
|
+
}
|
|
90
|
+
)
|
|
91
|
+
return entries
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _field_selective_entries(
|
|
95
|
+
expansion_info: dict[str, Any],
|
|
96
|
+
column: str,
|
|
97
|
+
col: "exp.Column",
|
|
98
|
+
field_path: str | None = None,
|
|
99
|
+
) -> list[dict[str, Any]] | None:
|
|
100
|
+
"""Source entries for a read of one named STRUCT field through an
|
|
101
|
+
array-of-structs expansion; None when the expansion has no field map
|
|
102
|
+
for the read (callers keep the every-column behavior)."""
|
|
103
|
+
selective = expansion_field_reads(expansion_info, col, column, field_path)
|
|
104
|
+
if selective is None:
|
|
105
|
+
return None
|
|
106
|
+
return [
|
|
107
|
+
{
|
|
108
|
+
"table": fs.get("source_table", ""),
|
|
109
|
+
"column": fs.get("source_column", ""),
|
|
110
|
+
"confidence": 0.5,
|
|
111
|
+
"trust_level": "moderate",
|
|
112
|
+
"inferred": False,
|
|
113
|
+
"from_array_expansion": True,
|
|
114
|
+
"expansion_type": expansion_info.get("expansion_type"),
|
|
115
|
+
"source_column_in_array": fs.get("source_column", ""),
|
|
116
|
+
"expanded_as": column,
|
|
117
|
+
}
|
|
118
|
+
for fs in selective
|
|
119
|
+
if fs.get("source_column")
|
|
120
|
+
]
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def process_select_items(
|
|
124
|
+
selects_list: list["exp.Expression"],
|
|
125
|
+
select_for_from: "exp.Expression",
|
|
126
|
+
result: dict[str, Any],
|
|
127
|
+
warnings: list[dict],
|
|
128
|
+
alias_map: dict[str, str],
|
|
129
|
+
all_tables: list[str],
|
|
130
|
+
array_expansion_sources: dict[str, dict[str, str]],
|
|
131
|
+
default_table: str,
|
|
132
|
+
join_anchor: str,
|
|
133
|
+
cte_columns: dict[str, dict[str, list[dict]]],
|
|
134
|
+
cte_collision_names: set[str],
|
|
135
|
+
dialect: str,
|
|
136
|
+
warehouse_columns: WarehouseColumns | None,
|
|
137
|
+
qualified_via_schema: bool,
|
|
138
|
+
) -> None:
|
|
139
|
+
"""Fill result with per-output-column sources; mutates result/warnings."""
|
|
140
|
+
|
|
141
|
+
def _sole_enumerated_cte_owner(column: str) -> str | None:
|
|
142
|
+
"""The one relation in scope proven to own the column, or None.
|
|
143
|
+
|
|
144
|
+
Provable only when EVERY relation is a CTE whose outputs are fully
|
|
145
|
+
enumerated (present in cte_columns, no star, not shadowed); a real
|
|
146
|
+
table or star-veiled CTE makes ownership unprovable and keeps the
|
|
147
|
+
ambiguous refusal."""
|
|
148
|
+
owners: list[str] = []
|
|
149
|
+
for t in all_tables:
|
|
150
|
+
entry = cte_columns.get(str(t).lower())
|
|
151
|
+
if entry is None or "*" in entry or entry.get(META_CTE_SHADOWED):
|
|
152
|
+
return None
|
|
153
|
+
if any(
|
|
154
|
+
not str(k).startswith(META_PREFIX) and str(k).lower() == column.lower()
|
|
155
|
+
for k in entry
|
|
156
|
+
):
|
|
157
|
+
owners.append(t)
|
|
158
|
+
return owners[0] if len(owners) == 1 else None
|
|
159
|
+
|
|
160
|
+
def _lookup_anchor(column: str) -> str:
|
|
161
|
+
"""Lookup-shape fallback for an unqualified read; never a CTE
|
|
162
|
+
whose known outputs prove the column absent."""
|
|
163
|
+
if not join_anchor:
|
|
164
|
+
return ""
|
|
165
|
+
anchor_cols = cte_columns.get(join_anchor.lower())
|
|
166
|
+
if anchor_cols is not None and not any(k.lower() == column.lower() for k in anchor_cols):
|
|
167
|
+
return ""
|
|
168
|
+
return join_anchor
|
|
169
|
+
|
|
170
|
+
register_subquery_from_aliases(selects_list, alias_map)
|
|
171
|
+
# Output maps of THIS select's aliased FROM/JOIN subqueries, plus the
|
|
172
|
+
# relations registered at this level (find_all descends into subqueries,
|
|
173
|
+
# so all_tables alone cannot say what sits at this level). Together they
|
|
174
|
+
# support elimination: an unqualified column absent from every derived
|
|
175
|
+
# item's outputs must come from the one real table (patentsview's
|
|
176
|
+
# webtool joins, holdout round 5).
|
|
177
|
+
from ripple.engine.column_ref import (
|
|
178
|
+
alias_is_foldable,
|
|
179
|
+
nested_scope_sole_table,
|
|
180
|
+
owning_select,
|
|
181
|
+
qualified_table_name,
|
|
182
|
+
qualifier_is_quoted,
|
|
183
|
+
unwrap_select,
|
|
184
|
+
)
|
|
185
|
+
from ripple.engine.cte_tracing import (
|
|
186
|
+
extract_select_column_sources as _sub_map,
|
|
187
|
+
)
|
|
188
|
+
from ripple.engine.cte_tracing import (
|
|
189
|
+
extract_set_op_column_sources as _sub_set_op_map,
|
|
190
|
+
)
|
|
191
|
+
from ripple.engine.cte_tracing import (
|
|
192
|
+
lateral_subquery_map as _lateral_subquery_map,
|
|
193
|
+
)
|
|
194
|
+
from ripple.engine.cte_tracing import subquery_column_entries as _subquery_column_entries
|
|
195
|
+
|
|
196
|
+
subquery_output_maps: list[dict] = []
|
|
197
|
+
subquery_maps_by_alias: dict[str, dict] = {}
|
|
198
|
+
subquery_maps_folded: dict[str, dict] = {}
|
|
199
|
+
local_real_tables: list[str] = []
|
|
200
|
+
from ripple.engine.column_ref import table_function_kind as _tf_kind
|
|
201
|
+
|
|
202
|
+
_from = select_for_from.args.get("from_") or select_for_from.args.get("from")
|
|
203
|
+
# nested subqueries belong to their own select; registering them here
|
|
204
|
+
# leaked inner projections into the outer scope (the cycle-8 review)
|
|
205
|
+
_sub_nodes = (
|
|
206
|
+
[sq for sq in _from.find_all(exp.Subquery) if owning_select(sq) is select_for_from]
|
|
207
|
+
if _from is not None
|
|
208
|
+
else []
|
|
209
|
+
)
|
|
210
|
+
if _from is not None:
|
|
211
|
+
for _t in _from.find_all(exp.Table):
|
|
212
|
+
if owning_select(_t) is select_for_from and _tf_kind(_t) != "srf":
|
|
213
|
+
local_real_tables.append(qualified_table_name(_t))
|
|
214
|
+
for _join in select_for_from.find_all(exp.Join):
|
|
215
|
+
if owning_select(_join) is not select_for_from:
|
|
216
|
+
continue
|
|
217
|
+
if isinstance(_join.this, exp.Subquery):
|
|
218
|
+
_sub_nodes.append(_join.this)
|
|
219
|
+
elif isinstance(_join.this, exp.Table) and _tf_kind(_join.this) != "srf":
|
|
220
|
+
local_real_tables.append(qualified_table_name(_join.this))
|
|
221
|
+
for _sq in _sub_nodes:
|
|
222
|
+
_inner = unwrap_select(_sq.this)
|
|
223
|
+
try:
|
|
224
|
+
if isinstance(_inner, exp.Select):
|
|
225
|
+
_map = _sub_map(_inner, dialect)
|
|
226
|
+
elif isinstance(_inner, (exp.Union, exp.Except, exp.Intersect)):
|
|
227
|
+
_map = _sub_set_op_map(_inner, dialect)
|
|
228
|
+
else:
|
|
229
|
+
continue
|
|
230
|
+
except Exception:
|
|
231
|
+
continue
|
|
232
|
+
subquery_output_maps.append(_map)
|
|
233
|
+
if _sq.alias:
|
|
234
|
+
subquery_maps_by_alias[_sq.alias] = _map
|
|
235
|
+
if alias_is_foldable(_sq):
|
|
236
|
+
subquery_maps_folded.setdefault(_sq.alias.lower(), _map)
|
|
237
|
+
else:
|
|
238
|
+
# an UNALIASED from-subquery: its projection IS this scope's
|
|
239
|
+
# source (transfermarkt's dedup wrapper, holdout round 7) but
|
|
240
|
+
# only when it is the SOLE relation; a peer subquery or any
|
|
241
|
+
# other relation makes its outputs a coin flip (review
|
|
242
|
+
# of cycle 8)
|
|
243
|
+
if (
|
|
244
|
+
len([q for q in _sub_nodes if not q.alias]) == 1
|
|
245
|
+
and len(_sub_nodes) == 1
|
|
246
|
+
and not local_real_tables
|
|
247
|
+
):
|
|
248
|
+
subquery_maps_by_alias.setdefault("", _map)
|
|
249
|
+
|
|
250
|
+
# lateral derived tables (CROSS/OUTER APPLY (SELECT ...) AS x (cols)),
|
|
251
|
+
# in document order so each apply sees the maps of earlier ones
|
|
252
|
+
# (DarkQueries' stacked CROSS APPLY chain, holdout round 8)
|
|
253
|
+
for _join in select_for_from.find_all(exp.Join):
|
|
254
|
+
if owning_select(_join) is not select_for_from:
|
|
255
|
+
continue
|
|
256
|
+
_lat = _join.this
|
|
257
|
+
if not isinstance(_lat, exp.Lateral) or not _lat.alias:
|
|
258
|
+
continue
|
|
259
|
+
try:
|
|
260
|
+
_lat_map = _lateral_subquery_map(
|
|
261
|
+
_lat,
|
|
262
|
+
dialect,
|
|
263
|
+
correlated_aliases=alias_map,
|
|
264
|
+
correlated_maps=subquery_maps_by_alias,
|
|
265
|
+
correlated_tables=all_tables,
|
|
266
|
+
)
|
|
267
|
+
except Exception:
|
|
268
|
+
continue
|
|
269
|
+
if _lat_map is None:
|
|
270
|
+
continue
|
|
271
|
+
subquery_maps_by_alias[_lat.alias] = _lat_map
|
|
272
|
+
if alias_is_foldable(_lat):
|
|
273
|
+
subquery_maps_folded.setdefault(_lat.alias.lower(), _lat_map)
|
|
274
|
+
|
|
275
|
+
def _subquery_map_for(alias: str, quoted: bool) -> dict | None:
|
|
276
|
+
# a quoted reference matches only exactly; an unquoted one folds
|
|
277
|
+
# case against foldable definitions (the cycle-6 review:
|
|
278
|
+
# "Q" and q are different postgres aliases)
|
|
279
|
+
exact = subquery_maps_by_alias.get(str(alias))
|
|
280
|
+
if exact is not None or quoted:
|
|
281
|
+
return exact
|
|
282
|
+
return subquery_maps_folded.get(str(alias).lower())
|
|
283
|
+
|
|
284
|
+
def _derived_expansion_entries(expansion_info: dict, expanded_as: str) -> list[dict] | None:
|
|
285
|
+
"""An expansion whose root column is a derived table's output reads
|
|
286
|
+
through that derivation: data.nodes(...) over FROM (SELECT
|
|
287
|
+
CONVERT(XML, target_data) AS data ...) t traces to target_data,
|
|
288
|
+
never to a phantom relation (InvestigateWaits, holdout round 8)."""
|
|
289
|
+
root_table = expansion_info.get("source_table") or ""
|
|
290
|
+
root_column = expansion_info.get("source_column") or ""
|
|
291
|
+
if not root_column:
|
|
292
|
+
return None
|
|
293
|
+
sub_map = _subquery_map_for(root_table, False) if root_table else None
|
|
294
|
+
if sub_map is None:
|
|
295
|
+
if root_table or local_real_tables:
|
|
296
|
+
return None
|
|
297
|
+
owners = [
|
|
298
|
+
m
|
|
299
|
+
for m in subquery_output_maps
|
|
300
|
+
if any(
|
|
301
|
+
k != "*"
|
|
302
|
+
and not str(k).startswith(META_PREFIX)
|
|
303
|
+
and str(k).lower() == root_column.lower()
|
|
304
|
+
for k in m
|
|
305
|
+
)
|
|
306
|
+
]
|
|
307
|
+
if len(owners) != 1 or any("*" in m for m in subquery_output_maps):
|
|
308
|
+
return None
|
|
309
|
+
sub_map = owners[0]
|
|
310
|
+
|
|
311
|
+
def _chained(chain_map: dict, chain_column: str) -> list[dict]:
|
|
312
|
+
entries = []
|
|
313
|
+
for src in _subquery_column_entries(chain_map, chain_column):
|
|
314
|
+
if not (src.get("table") or src.get("column")):
|
|
315
|
+
continue
|
|
316
|
+
entry = dict(src)
|
|
317
|
+
entry["confidence"] = min(entry.get("confidence", 1.0), 0.5)
|
|
318
|
+
entry["trust_level"] = "moderate"
|
|
319
|
+
entry["from_array_expansion"] = True
|
|
320
|
+
entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
321
|
+
entry["expanded_as"] = expanded_as
|
|
322
|
+
entries.append(entry)
|
|
323
|
+
return entries
|
|
324
|
+
|
|
325
|
+
entries = _chained(sub_map, root_column)
|
|
326
|
+
if not entries:
|
|
327
|
+
return None
|
|
328
|
+
# every argument counts: the extras chain through their own derived
|
|
329
|
+
# tables exactly like the primary (cycle-9 review, F10)
|
|
330
|
+
for extra in expansion_info.get("extra_columns", []):
|
|
331
|
+
extra_column = extra.get("source_column") or ""
|
|
332
|
+
if not extra_column:
|
|
333
|
+
continue
|
|
334
|
+
extra_table = extra.get("source_table") or ""
|
|
335
|
+
extra_map = _subquery_map_for(extra_table, False) if extra_table else None
|
|
336
|
+
if extra_map is not None:
|
|
337
|
+
entries.extend(_chained(extra_map, extra_column))
|
|
338
|
+
else:
|
|
339
|
+
entries.extend(
|
|
340
|
+
_expansion_extra_entries(
|
|
341
|
+
{**expansion_info, "extra_columns": [extra]}, expanded_as
|
|
342
|
+
)
|
|
343
|
+
)
|
|
344
|
+
return entries
|
|
345
|
+
|
|
346
|
+
def _selective_expansion_entries(
|
|
347
|
+
expansion_info: dict, column: str, col: "exp.Column", field_path: str | None
|
|
348
|
+
) -> list[dict] | None:
|
|
349
|
+
"""Field-selective entries with each source chained through its own
|
|
350
|
+
derived table's map. Selectivity runs BEFORE the derived-expansion
|
|
351
|
+
fallback: a named-struct melt over subquery outputs must keep each
|
|
352
|
+
field's own sources, not every input column (cycle-12, F16)."""
|
|
353
|
+
selective = _field_selective_entries(expansion_info, column, col, field_path)
|
|
354
|
+
if selective is None:
|
|
355
|
+
return None
|
|
356
|
+
entries: list[dict] = []
|
|
357
|
+
for entry in selective:
|
|
358
|
+
table = entry.get("table") or ""
|
|
359
|
+
sub_map = _subquery_map_for(table, False) if table else None
|
|
360
|
+
if sub_map is None:
|
|
361
|
+
entries.append(entry)
|
|
362
|
+
continue
|
|
363
|
+
for src in _subquery_column_entries(sub_map, entry.get("column", "")):
|
|
364
|
+
if not (src.get("table") or src.get("column")):
|
|
365
|
+
continue
|
|
366
|
+
chained_entry = dict(src)
|
|
367
|
+
chained_entry["confidence"] = min(chained_entry.get("confidence", 1.0), 0.5)
|
|
368
|
+
chained_entry["trust_level"] = "moderate"
|
|
369
|
+
chained_entry["from_array_expansion"] = True
|
|
370
|
+
chained_entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
371
|
+
chained_entry["expanded_as"] = column
|
|
372
|
+
entries.append(chained_entry)
|
|
373
|
+
return entries
|
|
374
|
+
|
|
375
|
+
# Every name a multi-part column reference could legitimately anchor
|
|
376
|
+
# to: aliases, table names (bare and as written), expansion aliases,
|
|
377
|
+
# inline-subquery aliases. Used to tell struct access (o.customer.name)
|
|
378
|
+
# from qualified table.column (ds.orders.customer) without fabricating
|
|
379
|
+
# tables.
|
|
380
|
+
_known_relations = (
|
|
381
|
+
{str(k).lower() for k in alias_map}
|
|
382
|
+
| {str(v).lower() for v in alias_map.values()}
|
|
383
|
+
| {str(k).lower() for k in array_expansion_sources}
|
|
384
|
+
| {str(k).lower() for k in subquery_maps_by_alias}
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
# Select-item aliases already resolved, in list order, for dialects
|
|
388
|
+
# where a later item may reference an earlier alias (lateral column
|
|
389
|
+
# aliases). The alias is a name of the expression, not a column of
|
|
390
|
+
# the upstream relation.
|
|
391
|
+
lateral_aliases: dict[str, list[dict[str, Any]]] = {}
|
|
392
|
+
allow_lateral = (dialect or "").lower() in LATERAL_ALIAS_DIALECTS
|
|
393
|
+
|
|
394
|
+
for item_idx, select_expr in enumerate(selects_list):
|
|
395
|
+
# Qualified star over a known CTE: rel.* (except ...). Expand
|
|
396
|
+
# through the CTE column map (following star-passthrough chains)
|
|
397
|
+
# and trace each named column; collapsing to '*' would silently
|
|
398
|
+
# drop most of the model's columns.
|
|
399
|
+
if isinstance(select_expr, exp.Column) and isinstance(select_expr.this, exp.Star):
|
|
400
|
+
star = select_expr.this
|
|
401
|
+
qualifier = select_expr.table or ""
|
|
402
|
+
rel = str(alias_map.get(qualifier, qualifier))
|
|
403
|
+
except_cols = star_except_spellings(
|
|
404
|
+
star.args.get("except_") or star.args.get("except") or []
|
|
405
|
+
)
|
|
406
|
+
# rel.* over an inline subquery alias: absorb the subquery's own
|
|
407
|
+
# map. The round-5 fix landed only in the CTE mappers; at the
|
|
408
|
+
# model's own top level the alias still surfaced as a struct
|
|
409
|
+
# column of the inner table (the review of PR #28).
|
|
410
|
+
sub_absorbed = _subquery_map_for(qualifier, qualifier_is_quoted(select_expr))
|
|
411
|
+
if sub_absorbed is not None:
|
|
412
|
+
for name, srcs in sub_absorbed.items():
|
|
413
|
+
if name == "*":
|
|
414
|
+
_append_column_sources(result, "*", srcs)
|
|
415
|
+
result[META_NEEDS_EXPANSION] = True
|
|
416
|
+
elif name.lower() not in {e.lower() for e in except_cols}:
|
|
417
|
+
result.setdefault(name, [dict(s) for s in srcs])
|
|
418
|
+
continue
|
|
419
|
+
expanded, residual = expand_star_columns([rel], cte_columns, except_cols)
|
|
420
|
+
if expanded:
|
|
421
|
+
for name, srcs in expanded.items():
|
|
422
|
+
result.setdefault(name, srcs)
|
|
423
|
+
if residual:
|
|
424
|
+
_append_column_sources(result, "*", residual)
|
|
425
|
+
result[META_NEEDS_EXPANSION] = True
|
|
426
|
+
continue
|
|
427
|
+
if residual:
|
|
428
|
+
# rel.* over an unexpandable relation keeps its wildcard
|
|
429
|
+
# link WITH its exclusions, appended so a second qualified
|
|
430
|
+
# star never overwrites the first (cycle-12, F4/F5)
|
|
431
|
+
_append_column_sources(result, "*", residual)
|
|
432
|
+
result[META_NEEDS_EXPANSION] = True
|
|
433
|
+
continue
|
|
434
|
+
# a nameless qualifier falls through to the ordinary paths
|
|
435
|
+
|
|
436
|
+
is_computed = False
|
|
437
|
+
|
|
438
|
+
if isinstance(select_expr, exp.Alias):
|
|
439
|
+
col_name = select_expr.alias
|
|
440
|
+
source_expr = select_expr.this
|
|
441
|
+
is_computed = not isinstance(source_expr, exp.Column)
|
|
442
|
+
elif isinstance(select_expr, exp.Column):
|
|
443
|
+
col_name = select_expr.name
|
|
444
|
+
source_expr = select_expr
|
|
445
|
+
elif isinstance(select_expr, exp.Star):
|
|
446
|
+
if not local_real_tables and subquery_output_maps:
|
|
447
|
+
# bare star over a scope of only derived tables: the
|
|
448
|
+
# columns are the subqueries' own projections, all of
|
|
449
|
+
# them (transfermarkt's dedup wrapper, holdout round 7;
|
|
450
|
+
# generalized to peers by the cycle-8 review)
|
|
451
|
+
for m in subquery_output_maps:
|
|
452
|
+
for name, srcs in m.items():
|
|
453
|
+
if str(name).startswith("\x00"):
|
|
454
|
+
continue
|
|
455
|
+
if name == "*":
|
|
456
|
+
_append_column_sources(result, "*", [dict(s) for s in srcs])
|
|
457
|
+
result[META_NEEDS_EXPANSION] = True
|
|
458
|
+
else:
|
|
459
|
+
result.setdefault(name, [dict(s) for s in srcs])
|
|
460
|
+
continue
|
|
461
|
+
# A bare star over known CTEs expands through the CTE column
|
|
462
|
+
# map, so named CTE columns become outputs instead of
|
|
463
|
+
# vanishing behind '*' with the CTE name as a phantom source.
|
|
464
|
+
except_cols = star_except_spellings(
|
|
465
|
+
select_expr.args.get("except_") or select_expr.args.get("except") or []
|
|
466
|
+
)
|
|
467
|
+
expanded, residual = expand_star_columns(all_tables, cte_columns, except_cols)
|
|
468
|
+
if expanded:
|
|
469
|
+
for name, srcs in expanded.items():
|
|
470
|
+
result.setdefault(name, srcs)
|
|
471
|
+
if residual:
|
|
472
|
+
_append_column_sources(result, "*", residual)
|
|
473
|
+
result[META_NEEDS_EXPANSION] = True
|
|
474
|
+
elif residual:
|
|
475
|
+
# a pure star chain (b -> a -> real) yields no named columns
|
|
476
|
+
# but the walk still found the real relations; rebuilding
|
|
477
|
+
# from all_tables here resurfaced the CTE itself as a
|
|
478
|
+
# phantom source (the cycle-7 review)
|
|
479
|
+
_append_column_sources(result, "*", residual)
|
|
480
|
+
result[META_NEEDS_EXPANSION] = True
|
|
481
|
+
else:
|
|
482
|
+
result["*"] = [
|
|
483
|
+
{"table": table_name, "column": "*", "confidence": 0.5}
|
|
484
|
+
for table_name in all_tables
|
|
485
|
+
]
|
|
486
|
+
result[META_NEEDS_EXPANSION] = True
|
|
487
|
+
continue
|
|
488
|
+
elif _QUERY_TRANSFORM_CLS is not None and isinstance(select_expr, _QUERY_TRANSFORM_CLS):
|
|
489
|
+
# SELECT TRANSFORM(...) USING 'script': the mapping through
|
|
490
|
+
# the script is unknowable, but inputs and declared outputs
|
|
491
|
+
# are static. Honest contract: every output depends on every
|
|
492
|
+
# input, all at review_required. Never assume identity, even
|
|
493
|
+
# for 'cat' (ROW FORMAT can reshape fields).
|
|
494
|
+
schema_arg = select_expr.args.get("schema")
|
|
495
|
+
outputs = (
|
|
496
|
+
[s.name for s in schema_arg.expressions]
|
|
497
|
+
if schema_arg is not None and schema_arg.expressions
|
|
498
|
+
else ["key", "value"] # Spark's documented default
|
|
499
|
+
)
|
|
500
|
+
script_node = select_expr.args.get("command_script")
|
|
501
|
+
script = script_node.name if script_node is not None else "?"
|
|
502
|
+
input_sources = []
|
|
503
|
+
input_names = []
|
|
504
|
+
for item in select_expr.expressions or []:
|
|
505
|
+
if isinstance(item, exp.Star):
|
|
506
|
+
for table_name in all_tables:
|
|
507
|
+
input_sources.append(
|
|
508
|
+
{
|
|
509
|
+
"table": table_name,
|
|
510
|
+
"column": "*",
|
|
511
|
+
"confidence": 0.3,
|
|
512
|
+
"trust_level": "review_required",
|
|
513
|
+
"via_script_transform": True,
|
|
514
|
+
"script": script,
|
|
515
|
+
}
|
|
516
|
+
)
|
|
517
|
+
result[META_NEEDS_EXPANSION] = True
|
|
518
|
+
input_names.append("*")
|
|
519
|
+
continue
|
|
520
|
+
for col in item.find_all(exp.Column):
|
|
521
|
+
col_alias = col.table or ""
|
|
522
|
+
table = (
|
|
523
|
+
alias_map.get(col_alias, col_alias) if col_alias else (default_table or "")
|
|
524
|
+
)
|
|
525
|
+
input_sources.append(
|
|
526
|
+
{
|
|
527
|
+
"table": table,
|
|
528
|
+
"column": col.name,
|
|
529
|
+
"confidence": 0.3,
|
|
530
|
+
"trust_level": "review_required",
|
|
531
|
+
"via_script_transform": True,
|
|
532
|
+
"script": script,
|
|
533
|
+
}
|
|
534
|
+
)
|
|
535
|
+
input_names.append(col.name)
|
|
536
|
+
for out in outputs:
|
|
537
|
+
result[out] = [dict(s) for s in input_sources]
|
|
538
|
+
warnings.append(
|
|
539
|
+
{
|
|
540
|
+
"warning_type": "script_transform",
|
|
541
|
+
"message": (
|
|
542
|
+
f"SELECT TRANSFORM pipes rows through external script '{script}'; "
|
|
543
|
+
f"the column mapping through the script is opaque. Each of the "
|
|
544
|
+
f"{len(outputs)} output columns is conservatively linked to all "
|
|
545
|
+
f"{len(input_names)} input columns at review_required."
|
|
546
|
+
),
|
|
547
|
+
"script": script,
|
|
548
|
+
"output_columns": outputs,
|
|
549
|
+
"input_columns": input_names,
|
|
550
|
+
}
|
|
551
|
+
)
|
|
552
|
+
continue
|
|
553
|
+
else:
|
|
554
|
+
col_sql = safe_sql(select_expr, dialect)
|
|
555
|
+
col_name = col_sql if col_sql is not None and len(col_sql) < 50 else None
|
|
556
|
+
source_expr = select_expr
|
|
557
|
+
is_computed = True
|
|
558
|
+
|
|
559
|
+
if not col_name:
|
|
560
|
+
continue
|
|
561
|
+
if "{" in col_name:
|
|
562
|
+
# a template placeholder is not a real output name; exposing it
|
|
563
|
+
# (or its internal dot encoding) leaked into reports and
|
|
564
|
+
# exported nodes (cycle-12, F19). The output stays, opaque.
|
|
565
|
+
col_name = f"_col_{item_idx}"
|
|
566
|
+
|
|
567
|
+
sources = []
|
|
568
|
+
seen = set()
|
|
569
|
+
|
|
570
|
+
json_columns_handled = collect_json_sources(
|
|
571
|
+
source_expr,
|
|
572
|
+
_known_relations,
|
|
573
|
+
alias_map,
|
|
574
|
+
array_expansion_sources,
|
|
575
|
+
default_table,
|
|
576
|
+
qualified_via_schema,
|
|
577
|
+
sources,
|
|
578
|
+
seen,
|
|
579
|
+
)
|
|
580
|
+
|
|
581
|
+
# qualify() rewrites a read of a table-valued alias (CROSS JOIN
|
|
582
|
+
# jsonb_array_elements_text(roles) AS role ... SELECT role) into a
|
|
583
|
+
# TableColumn node the Column sweep never sees (kingfisher, round 7)
|
|
584
|
+
_tc_cls = getattr(exp, "TableColumn", None)
|
|
585
|
+
if _tc_cls is not None:
|
|
586
|
+
for tc in source_expr.find_all(_tc_cls):
|
|
587
|
+
entry = array_expansion_sources.get(tc.name) or array_expansion_sources.get(
|
|
588
|
+
tc.name.lower()
|
|
589
|
+
)
|
|
590
|
+
if not entry:
|
|
591
|
+
continue
|
|
592
|
+
reads = [(entry.get("source_table", ""), entry.get("source_column", ""))] + [
|
|
593
|
+
(x.get("source_table", ""), x.get("source_column", ""))
|
|
594
|
+
for x in entry.get("extra_columns", [])
|
|
595
|
+
]
|
|
596
|
+
for key in reads:
|
|
597
|
+
if key not in seen and (key[0] or key[1]):
|
|
598
|
+
seen.add(key)
|
|
599
|
+
sources.append(
|
|
600
|
+
{
|
|
601
|
+
"table": key[0],
|
|
602
|
+
"column": key[1],
|
|
603
|
+
"confidence": 0.5,
|
|
604
|
+
"trust_level": "moderate",
|
|
605
|
+
"inferred": False,
|
|
606
|
+
"from_array_expansion": True,
|
|
607
|
+
"expansion_type": entry.get("expansion_type"),
|
|
608
|
+
}
|
|
609
|
+
)
|
|
610
|
+
|
|
611
|
+
# TODO: this classification block near-duplicates cte_tracing.extract_select_column_sources; merging needs judgment on trust/warning metadata.
|
|
612
|
+
for col in source_expr.find_all(exp.Column):
|
|
613
|
+
if id(col) in json_columns_handled:
|
|
614
|
+
continue # Already handled as part of JSONExtract
|
|
615
|
+
if under_subquery_where(col, source_expr):
|
|
616
|
+
# a nested subquery's WHERE selects rows there; it never
|
|
617
|
+
# feeds the produced value
|
|
618
|
+
continue
|
|
619
|
+
if in_window_ordering(col, source_expr):
|
|
620
|
+
# frames the window, does not flow into its value
|
|
621
|
+
continue
|
|
622
|
+
if in_selector_argument(col, source_expr):
|
|
623
|
+
# picks the arg_max/arg_min row, does not flow into its value
|
|
624
|
+
continue
|
|
625
|
+
alias, column, field_path, certain = resolve_column_ref(col, _known_relations)
|
|
626
|
+
|
|
627
|
+
if "{" in column or (not certain and any("{" in p for p in column_parts(col))):
|
|
628
|
+
# a {placeholder} or jinja chain in column OR qualifier
|
|
629
|
+
# position may format to anything; only relation-position
|
|
630
|
+
# braces carry a citable identity. Anchoring the qualifier
|
|
631
|
+
# as a struct root fabricated a source (cycle-12, F11).
|
|
632
|
+
spelled = ".".join(column_parts(col))
|
|
633
|
+
warnings.append(
|
|
634
|
+
{
|
|
635
|
+
"column": spelled,
|
|
636
|
+
"warning_type": "template_placeholder_expression",
|
|
637
|
+
"trust_level": "review_required",
|
|
638
|
+
"message": f"template placeholder {spelled} in an expression; "
|
|
639
|
+
"its lineage cannot be resolved offline",
|
|
640
|
+
}
|
|
641
|
+
)
|
|
642
|
+
continue
|
|
643
|
+
|
|
644
|
+
source_entry: dict[str, Any] = {
|
|
645
|
+
"column": column,
|
|
646
|
+
"inferred": False,
|
|
647
|
+
}
|
|
648
|
+
if field_path:
|
|
649
|
+
source_entry["field_path"] = field_path
|
|
650
|
+
|
|
651
|
+
if not certain:
|
|
652
|
+
# multi-part reference whose leading identifier matches no
|
|
653
|
+
# known relation: read as a struct root. Anchor it to the
|
|
654
|
+
# UNNEST of its own subquery scope, the anonymous UNNEST
|
|
655
|
+
# of this select, or the sole relation when possible;
|
|
656
|
+
# never emit a confident phantom table.
|
|
657
|
+
anon = scoped_unnest_entry(
|
|
658
|
+
col, select_for_from, alias_map, all_tables
|
|
659
|
+
) or array_expansion_sources.get("__anonymous_unnest__")
|
|
660
|
+
if anon is not None:
|
|
661
|
+
table = anon.get("source_table", "")
|
|
662
|
+
confidence = 0.5
|
|
663
|
+
trust_level = "moderate"
|
|
664
|
+
source_entry["from_array_expansion"] = True
|
|
665
|
+
source_entry["expansion_type"] = anon.get("expansion_type")
|
|
666
|
+
source_entry["source_column_in_array"] = anon.get("source_column", "")
|
|
667
|
+
if anon.get("source_column"):
|
|
668
|
+
source_entry["expanded_as"] = column
|
|
669
|
+
source_entry["column"] = anon["source_column"]
|
|
670
|
+
elif default_table:
|
|
671
|
+
table = default_table
|
|
672
|
+
confidence = 0.5
|
|
673
|
+
trust_level = "moderate"
|
|
674
|
+
source_entry["inferred"] = True
|
|
675
|
+
source_entry["struct_guess"] = True
|
|
676
|
+
else:
|
|
677
|
+
table = ""
|
|
678
|
+
confidence = 0.4
|
|
679
|
+
trust_level = "review_required"
|
|
680
|
+
source_entry["inferred"] = True
|
|
681
|
+
source_entry["unresolved_reference"] = True
|
|
682
|
+
elif alias:
|
|
683
|
+
if alias in array_expansion_sources:
|
|
684
|
+
expansion_info = array_expansion_sources[alias]
|
|
685
|
+
selective = _selective_expansion_entries(
|
|
686
|
+
expansion_info, column, col, field_path
|
|
687
|
+
)
|
|
688
|
+
if selective is not None:
|
|
689
|
+
for entry in selective:
|
|
690
|
+
key = (entry.get("table", ""), entry.get("column", ""))
|
|
691
|
+
if key not in seen:
|
|
692
|
+
seen.add(key)
|
|
693
|
+
sources.append(entry)
|
|
694
|
+
continue
|
|
695
|
+
chained = _derived_expansion_entries(expansion_info, column)
|
|
696
|
+
if chained:
|
|
697
|
+
for entry in chained:
|
|
698
|
+
chain_key = (entry.get("table", ""), entry.get("column", ""))
|
|
699
|
+
if chain_key not in seen:
|
|
700
|
+
seen.add(chain_key)
|
|
701
|
+
sources.append(entry)
|
|
702
|
+
continue
|
|
703
|
+
# Trace to the actual source table/column
|
|
704
|
+
table = expansion_info.get("source_table", alias)
|
|
705
|
+
source_column = expansion_info.get("source_column", column)
|
|
706
|
+
# Lower confidence: we know the source but can't track row expansion
|
|
707
|
+
confidence = 0.5
|
|
708
|
+
trust_level = "moderate"
|
|
709
|
+
source_entry["from_array_expansion"] = True
|
|
710
|
+
source_entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
711
|
+
source_entry["source_column_in_array"] = source_column
|
|
712
|
+
if source_column:
|
|
713
|
+
# the upstream column IS the array; the alias's
|
|
714
|
+
# output name does not exist on the source table
|
|
715
|
+
source_entry["expanded_as"] = column
|
|
716
|
+
source_entry["column"] = source_column
|
|
717
|
+
if expansion_info.get("json_path"):
|
|
718
|
+
source_entry["json_path"] = expansion_info["json_path"]
|
|
719
|
+
for extra_entry in _expansion_extra_entries(expansion_info, column):
|
|
720
|
+
extra_key = (extra_entry["table"], extra_entry["column"])
|
|
721
|
+
if extra_key not in seen:
|
|
722
|
+
seen.add(extra_key)
|
|
723
|
+
sources.append(extra_entry)
|
|
724
|
+
warnings.append(
|
|
725
|
+
{
|
|
726
|
+
"column": column,
|
|
727
|
+
"warning_type": "array_expansion",
|
|
728
|
+
"message": f"Column '{column}' comes from array expansion ({expansion_info.get('expansion_type')}). "
|
|
729
|
+
f"Source array: {table}.{source_column}",
|
|
730
|
+
"source_table": table,
|
|
731
|
+
"source_column": source_column,
|
|
732
|
+
"expansion_type": expansion_info.get("expansion_type"),
|
|
733
|
+
}
|
|
734
|
+
)
|
|
735
|
+
elif (
|
|
736
|
+
sub_absorbed := _subquery_map_for(alias, qualifier_is_quoted(col))
|
|
737
|
+
) is not None:
|
|
738
|
+
# the qualifier is an inline-subquery alias: the
|
|
739
|
+
# column's sources are the subquery's own, never a
|
|
740
|
+
# relation named after the alias (pensjon under
|
|
741
|
+
# qualify(), the gap named in PR #28)
|
|
742
|
+
for src in _subquery_column_entries(sub_absorbed, column):
|
|
743
|
+
key = (src.get("table", ""), src.get("column", ""))
|
|
744
|
+
if key not in seen and (key[0] or key[1]):
|
|
745
|
+
seen.add(key)
|
|
746
|
+
sources.append(dict(src))
|
|
747
|
+
continue
|
|
748
|
+
else:
|
|
749
|
+
# Column is explicitly qualified - high confidence
|
|
750
|
+
table = alias_map.get(alias, alias)
|
|
751
|
+
confidence = 1.0
|
|
752
|
+
trust_level = "verified" if qualified_via_schema else "high_confidence"
|
|
753
|
+
elif column in array_expansion_sources:
|
|
754
|
+
# Unqualified column that matches an UNNEST output column
|
|
755
|
+
expansion_info = array_expansion_sources[column]
|
|
756
|
+
selective = _selective_expansion_entries(expansion_info, column, col, field_path)
|
|
757
|
+
if selective is not None:
|
|
758
|
+
for entry in selective:
|
|
759
|
+
key = (entry.get("table", ""), entry.get("column", ""))
|
|
760
|
+
if key not in seen:
|
|
761
|
+
seen.add(key)
|
|
762
|
+
sources.append(entry)
|
|
763
|
+
continue
|
|
764
|
+
chained = _derived_expansion_entries(expansion_info, column)
|
|
765
|
+
if chained:
|
|
766
|
+
for entry in chained:
|
|
767
|
+
chain_key = (entry.get("table", ""), entry.get("column", ""))
|
|
768
|
+
if chain_key not in seen:
|
|
769
|
+
seen.add(chain_key)
|
|
770
|
+
sources.append(entry)
|
|
771
|
+
continue
|
|
772
|
+
table = expansion_info.get("source_table", "")
|
|
773
|
+
source_column = expansion_info.get("source_column", column)
|
|
774
|
+
confidence = 0.5
|
|
775
|
+
trust_level = "moderate"
|
|
776
|
+
source_entry["from_array_expansion"] = True
|
|
777
|
+
source_entry["expansion_type"] = expansion_info.get("expansion_type")
|
|
778
|
+
source_entry["source_column_in_array"] = source_column
|
|
779
|
+
source_entry["is_unnest_column"] = expansion_info.get("is_unnest_column", False)
|
|
780
|
+
if source_column:
|
|
781
|
+
source_entry["expanded_as"] = column
|
|
782
|
+
source_entry["column"] = source_column
|
|
783
|
+
for extra_entry in _expansion_extra_entries(expansion_info, column):
|
|
784
|
+
extra_key = (extra_entry["table"], extra_entry["column"])
|
|
785
|
+
if extra_key not in seen:
|
|
786
|
+
seen.add(extra_key)
|
|
787
|
+
sources.append(extra_entry)
|
|
788
|
+
warnings.append(
|
|
789
|
+
{
|
|
790
|
+
"column": column,
|
|
791
|
+
"warning_type": "array_expansion",
|
|
792
|
+
"message": f"Column '{column}' comes from array expansion ({expansion_info.get('expansion_type')}). "
|
|
793
|
+
f"Source array: {table}.{source_column}",
|
|
794
|
+
"source_table": table,
|
|
795
|
+
"source_column": source_column,
|
|
796
|
+
"expansion_type": expansion_info.get("expansion_type"),
|
|
797
|
+
}
|
|
798
|
+
)
|
|
799
|
+
elif (
|
|
800
|
+
allow_lateral
|
|
801
|
+
and column.lower() in lateral_aliases
|
|
802
|
+
and not (
|
|
803
|
+
warehouse_columns
|
|
804
|
+
and resolve_column_table(column, all_tables, warehouse_columns)[1]
|
|
805
|
+
)
|
|
806
|
+
):
|
|
807
|
+
# lateral column alias: the name belongs to an earlier
|
|
808
|
+
# select item, not to the upstream relation. When the
|
|
809
|
+
# schema proves a real column of that name exists, the
|
|
810
|
+
# table wins; otherwise resolve to the definition.
|
|
811
|
+
for src in lateral_aliases[column.lower()]:
|
|
812
|
+
entry = dict(src)
|
|
813
|
+
entry["via_lateral_alias"] = True
|
|
814
|
+
key = (entry.get("table", ""), entry.get("column", ""))
|
|
815
|
+
if key not in seen and (key[0] or key[1]):
|
|
816
|
+
seen.add(key)
|
|
817
|
+
sources.append(entry)
|
|
818
|
+
continue
|
|
819
|
+
elif (anon_map := subquery_maps_by_alias.get("")) is not None and any(
|
|
820
|
+
not str(k).startswith("\x00") and k != "*" and str(k).lower() == column.lower()
|
|
821
|
+
for k in anon_map
|
|
822
|
+
):
|
|
823
|
+
# unqualified read of a column the unaliased from-subquery
|
|
824
|
+
# projects (transfermarkt's `where n = 1` wrapper,
|
|
825
|
+
# holdout round 7)
|
|
826
|
+
for src in _subquery_column_entries(anon_map, column):
|
|
827
|
+
key = (src.get("table", ""), src.get("column", ""))
|
|
828
|
+
if key not in seen and (key[0] or key[1]):
|
|
829
|
+
seen.add(key)
|
|
830
|
+
sources.append(dict(src))
|
|
831
|
+
continue
|
|
832
|
+
elif (
|
|
833
|
+
not local_real_tables
|
|
834
|
+
and all("*" not in m for m in subquery_output_maps)
|
|
835
|
+
and len(
|
|
836
|
+
owner_maps := [
|
|
837
|
+
m
|
|
838
|
+
for m in subquery_output_maps
|
|
839
|
+
if any(
|
|
840
|
+
not str(k).startswith("\x00")
|
|
841
|
+
and k != "*"
|
|
842
|
+
and str(k).lower() == column.lower()
|
|
843
|
+
for k in m
|
|
844
|
+
)
|
|
845
|
+
]
|
|
846
|
+
)
|
|
847
|
+
== 1
|
|
848
|
+
):
|
|
849
|
+
# this select reads only derived tables, and exactly one of
|
|
850
|
+
# them projects the column: anchoring to a table found
|
|
851
|
+
# INSIDE the subquery skips the derivation (kingfisher's
|
|
852
|
+
# role expansion read through id_role, holdout round 7)
|
|
853
|
+
for src in _subquery_column_entries(owner_maps[0], column):
|
|
854
|
+
key = (src.get("table", ""), src.get("column", ""))
|
|
855
|
+
if key not in seen and (key[0] or key[1]):
|
|
856
|
+
seen.add(key)
|
|
857
|
+
sources.append(dict(src))
|
|
858
|
+
continue
|
|
859
|
+
elif (nested := nested_scope_sole_table(col, select_for_from)) is not None:
|
|
860
|
+
# the column lives in a subselect with its own FROM; the
|
|
861
|
+
# outer sole table or elimination would anchor it a scope
|
|
862
|
+
# too high (the review of PR #28: scalar subqueries in
|
|
863
|
+
# UPDATE SET expressions)
|
|
864
|
+
table = nested
|
|
865
|
+
if nested:
|
|
866
|
+
confidence = 0.8
|
|
867
|
+
trust_level = "high_confidence"
|
|
868
|
+
else:
|
|
869
|
+
confidence = 0.4
|
|
870
|
+
trust_level = "review_required"
|
|
871
|
+
source_entry["unresolved_reference"] = True
|
|
872
|
+
source_entry["inferred"] = True
|
|
873
|
+
elif default_table:
|
|
874
|
+
# Single table in FROM - inferred but reliable
|
|
875
|
+
table = default_table
|
|
876
|
+
confidence = 0.8
|
|
877
|
+
trust_level = "high_confidence"
|
|
878
|
+
source_entry["inferred"] = True
|
|
879
|
+
elif (
|
|
880
|
+
subquery_output_maps
|
|
881
|
+
and len(local_real_tables) == 1
|
|
882
|
+
and all(
|
|
883
|
+
column.lower() not in {str(k).lower() for k in m} for m in subquery_output_maps
|
|
884
|
+
)
|
|
885
|
+
and all("*" not in m for m in subquery_output_maps)
|
|
886
|
+
):
|
|
887
|
+
# every derived FROM item enumerates its outputs and none
|
|
888
|
+
# carries this name, so the one real table must. A '*' entry
|
|
889
|
+
# means a subquery's outputs are NOT enumerated: the column
|
|
890
|
+
# may well live behind the star, and eliminating on it
|
|
891
|
+
# fabricated an anchor (the review of PR #28)
|
|
892
|
+
table = local_real_tables[0]
|
|
893
|
+
confidence = 0.6
|
|
894
|
+
trust_level = "moderate"
|
|
895
|
+
source_entry["inferred"] = True
|
|
896
|
+
elif (
|
|
897
|
+
len(all_tables) > 1
|
|
898
|
+
and not (
|
|
899
|
+
warehouse_columns
|
|
900
|
+
and resolve_column_table(column, all_tables, warehouse_columns)[1]
|
|
901
|
+
)
|
|
902
|
+
and (_sys_owner := system_catalog_owner(column, all_tables, dialect))
|
|
903
|
+
):
|
|
904
|
+
# every relation in scope is a documented sys object and
|
|
905
|
+
# exactly one owns the column (InvestigateWaits, round 8);
|
|
906
|
+
# a supplied schema that resolves the column outranks the
|
|
907
|
+
# built-in catalog (cycle-9 review, F1)
|
|
908
|
+
table = _sys_owner
|
|
909
|
+
confidence = 0.8
|
|
910
|
+
trust_level = "high_confidence"
|
|
911
|
+
source_entry["inferred"] = True
|
|
912
|
+
elif len(all_tables) > 1 and (_owner := _sole_enumerated_cte_owner(column)):
|
|
913
|
+
# SQL resolves an unqualified name to the one relation that
|
|
914
|
+
# can own it. Provable only when every relation in scope is
|
|
915
|
+
# a CTE with fully enumerated outputs and exactly one
|
|
916
|
+
# projects the name (dbt_salesforce's manager_id coalesce
|
|
917
|
+
# across three joined aggregate CTEs, holdout round 7).
|
|
918
|
+
table = _owner
|
|
919
|
+
confidence = 0.8
|
|
920
|
+
trust_level = "high_confidence"
|
|
921
|
+
source_entry["inferred"] = True
|
|
922
|
+
elif len(all_tables) > 1 and warehouse_columns:
|
|
923
|
+
# Multiple tables and warehouse schema available - try to resolve
|
|
924
|
+
resolved_table, was_resolved = resolve_column_table(
|
|
925
|
+
column, all_tables, warehouse_columns
|
|
926
|
+
)
|
|
927
|
+
if was_resolved and resolved_table:
|
|
928
|
+
# Successfully resolved ambiguity via warehouse schema
|
|
929
|
+
table = resolved_table
|
|
930
|
+
confidence = 1.0
|
|
931
|
+
trust_level = "verified"
|
|
932
|
+
source_entry["qualified_via_schema"] = True
|
|
933
|
+
source_entry["inferred"] = True
|
|
934
|
+
elif _lookup_anchor(column):
|
|
935
|
+
table = _lookup_anchor(column)
|
|
936
|
+
confidence = 0.6
|
|
937
|
+
trust_level = "moderate"
|
|
938
|
+
source_entry["inferred"] = True
|
|
939
|
+
else:
|
|
940
|
+
# Could not resolve - mark as ambiguous
|
|
941
|
+
table = ""
|
|
942
|
+
confidence = 0.5
|
|
943
|
+
trust_level = "review_required"
|
|
944
|
+
source_entry["ambiguous_tables"] = all_tables
|
|
945
|
+
source_entry["ambiguity_unresolved"] = True
|
|
946
|
+
warnings.append(
|
|
947
|
+
{
|
|
948
|
+
"column": column,
|
|
949
|
+
"warning_type": "ambiguous_column",
|
|
950
|
+
"message": f"Column '{column}' could come from any of: {', '.join(all_tables)}",
|
|
951
|
+
"candidate_tables": all_tables,
|
|
952
|
+
}
|
|
953
|
+
)
|
|
954
|
+
elif len(all_tables) > 1 and _lookup_anchor(column):
|
|
955
|
+
table = _lookup_anchor(column)
|
|
956
|
+
confidence = 0.6
|
|
957
|
+
trust_level = "moderate"
|
|
958
|
+
source_entry["inferred"] = True
|
|
959
|
+
elif len(all_tables) > 1:
|
|
960
|
+
# Multiple tables but no warehouse schema - ambiguous, needs resolution
|
|
961
|
+
table = ""
|
|
962
|
+
confidence = 0.5
|
|
963
|
+
trust_level = "moderate"
|
|
964
|
+
source_entry["ambiguous_tables"] = all_tables
|
|
965
|
+
source_entry["needs_warehouse_connection"] = True
|
|
966
|
+
warnings.append(
|
|
967
|
+
{
|
|
968
|
+
"column": column,
|
|
969
|
+
"warning_type": "ambiguous_column_no_schema",
|
|
970
|
+
"message": f"Column '{column}' is ambiguous. Connect a warehouse to enable automatic resolution.",
|
|
971
|
+
"candidate_tables": all_tables,
|
|
972
|
+
"hint": "Connect a warehouse to enable automatic column resolution.",
|
|
973
|
+
}
|
|
974
|
+
)
|
|
975
|
+
else:
|
|
976
|
+
# No FROM clause or empty tables list
|
|
977
|
+
table = ""
|
|
978
|
+
confidence = 0.5
|
|
979
|
+
trust_level = "moderate"
|
|
980
|
+
|
|
981
|
+
source_entry["table"] = table
|
|
982
|
+
source_entry["confidence"] = confidence
|
|
983
|
+
source_entry["trust_level"] = trust_level
|
|
984
|
+
|
|
985
|
+
key = (table, column)
|
|
986
|
+
if key not in seen and (table or column):
|
|
987
|
+
seen.add(key)
|
|
988
|
+
sources.append(source_entry)
|
|
989
|
+
|
|
990
|
+
if not sources and is_computed:
|
|
991
|
+
sources = [
|
|
992
|
+
{
|
|
993
|
+
"table": "",
|
|
994
|
+
"column": col_name,
|
|
995
|
+
"computed": True,
|
|
996
|
+
"confidence": 0.3,
|
|
997
|
+
"trust_level": "moderate",
|
|
998
|
+
}
|
|
999
|
+
]
|
|
1000
|
+
|
|
1001
|
+
# Trace sources through CTEs
|
|
1002
|
+
if sources and cte_columns:
|
|
1003
|
+
traced_sources = []
|
|
1004
|
+
seen_traced = set()
|
|
1005
|
+
for src in sources:
|
|
1006
|
+
if src.get("computed"):
|
|
1007
|
+
traced_sources.append(src)
|
|
1008
|
+
elif src.get("table") and src["table"].lower() in cte_columns:
|
|
1009
|
+
traced = trace_through_ctes(
|
|
1010
|
+
src["column"],
|
|
1011
|
+
src["table"],
|
|
1012
|
+
cte_columns,
|
|
1013
|
+
)
|
|
1014
|
+
for t in traced:
|
|
1015
|
+
key = (t.get("table", ""), t.get("column", ""))
|
|
1016
|
+
if key not in seen_traced:
|
|
1017
|
+
seen_traced.add(key)
|
|
1018
|
+
# Degrade confidence if source CTE collides with real table
|
|
1019
|
+
if cte_collision_names and src["table"].lower() in cte_collision_names:
|
|
1020
|
+
t["confidence"] = min(t.get("confidence", 1.0), 0.5)
|
|
1021
|
+
t["trust_level"] = "review_required"
|
|
1022
|
+
t["cte_collision"] = True
|
|
1023
|
+
# Preserve trust metadata from traced sources
|
|
1024
|
+
elif src.get("qualified_via_schema"):
|
|
1025
|
+
t["qualified_via_schema"] = True
|
|
1026
|
+
t["trust_level"] = "verified"
|
|
1027
|
+
traced_sources.append(t)
|
|
1028
|
+
else:
|
|
1029
|
+
key = (src.get("table", ""), src.get("column", ""))
|
|
1030
|
+
if key not in seen_traced:
|
|
1031
|
+
seen_traced.add(key)
|
|
1032
|
+
traced_sources.append(src)
|
|
1033
|
+
sources = traced_sources if traced_sources else sources
|
|
1034
|
+
|
|
1035
|
+
if isinstance(select_expr, exp.Alias) and sources:
|
|
1036
|
+
lateral_aliases.setdefault(col_name.lower(), sources)
|
|
1037
|
+
|
|
1038
|
+
result[col_name] = sources
|