ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,555 @@
|
|
|
1
|
+
"""Unused dependency detection for SQL optimization.
|
|
2
|
+
|
|
3
|
+
This module analyzes SQL to find:
|
|
4
|
+
- Tables that are joined but no columns are actually used from them
|
|
5
|
+
- CTEs that are defined but never referenced
|
|
6
|
+
- Subqueries that don't contribute to the output
|
|
7
|
+
|
|
8
|
+
This helps with:
|
|
9
|
+
1. Fewer upstream dependencies to maintain
|
|
10
|
+
2. Tech debt cleanup - Identify dead code in SQL
|
|
11
|
+
3. Performance - Unnecessary joins waste compute
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Literal
|
|
16
|
+
|
|
17
|
+
import sqlglot
|
|
18
|
+
from sqlglot import exp
|
|
19
|
+
|
|
20
|
+
from ripple.engine.safe_gen import safe_sql
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class TableUsage:
|
|
25
|
+
"""Track how a table is used in a query."""
|
|
26
|
+
|
|
27
|
+
name: str
|
|
28
|
+
alias: str | None
|
|
29
|
+
schema_name: str | None = None
|
|
30
|
+
database: str | None = None
|
|
31
|
+
|
|
32
|
+
# Usage tracking
|
|
33
|
+
columns_selected: list[str] = field(default_factory=list)
|
|
34
|
+
columns_in_where: list[str] = field(default_factory=list)
|
|
35
|
+
columns_in_group_by: list[str] = field(default_factory=list)
|
|
36
|
+
columns_in_order_by: list[str] = field(default_factory=list)
|
|
37
|
+
columns_in_having: list[str] = field(default_factory=list)
|
|
38
|
+
columns_in_join_condition: list[str] = field(default_factory=list)
|
|
39
|
+
|
|
40
|
+
# Join info
|
|
41
|
+
join_type: str | None = None # LEFT, RIGHT, INNER, CROSS, etc.
|
|
42
|
+
is_from_table: bool = False # True if this is the main FROM table
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def identifier(self) -> str:
|
|
46
|
+
"""Return the identifier used in the query (alias if present, else name)."""
|
|
47
|
+
return self.alias or self.name
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def full_name(self) -> str:
|
|
51
|
+
"""Return fully qualified name."""
|
|
52
|
+
parts = []
|
|
53
|
+
if self.database:
|
|
54
|
+
parts.append(self.database)
|
|
55
|
+
if self.schema_name:
|
|
56
|
+
parts.append(self.schema_name)
|
|
57
|
+
parts.append(self.name)
|
|
58
|
+
return ".".join(parts)
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def total_column_uses(self) -> int:
|
|
62
|
+
"""Total number of column references from this table."""
|
|
63
|
+
return (
|
|
64
|
+
len(self.columns_selected)
|
|
65
|
+
+ len(self.columns_in_where)
|
|
66
|
+
+ len(self.columns_in_group_by)
|
|
67
|
+
+ len(self.columns_in_order_by)
|
|
68
|
+
+ len(self.columns_in_having)
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def is_used(self) -> bool:
|
|
73
|
+
"""Check if this table contributes to the query output."""
|
|
74
|
+
# A table is "used" if:
|
|
75
|
+
# 1. It's the main FROM table (always needed)
|
|
76
|
+
# 2. Columns from it are selected
|
|
77
|
+
# 3. Columns from it are used in WHERE/GROUP BY/HAVING/ORDER BY
|
|
78
|
+
if self.is_from_table:
|
|
79
|
+
return True
|
|
80
|
+
return self.total_column_uses > 0
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def only_used_for_join(self) -> bool:
|
|
84
|
+
"""Check if table is only used in join condition, not in output."""
|
|
85
|
+
return (
|
|
86
|
+
not self.is_from_table
|
|
87
|
+
and self.total_column_uses == 0
|
|
88
|
+
and len(self.columns_in_join_condition) > 0
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass
|
|
93
|
+
class CTEUsage:
|
|
94
|
+
"""Track CTE definition and usage."""
|
|
95
|
+
|
|
96
|
+
name: str
|
|
97
|
+
sql: str
|
|
98
|
+
referenced_count: int = 0
|
|
99
|
+
referenced_in: list[str] = field(default_factory=list) # Where it's used
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def is_used(self) -> bool:
|
|
103
|
+
return self.referenced_count > 0
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass
|
|
107
|
+
class UnusedDependency:
|
|
108
|
+
"""A detected unused dependency."""
|
|
109
|
+
|
|
110
|
+
type: Literal["join", "cte", "subquery"]
|
|
111
|
+
name: str
|
|
112
|
+
location: str # Human-readable location in query
|
|
113
|
+
reason: str # Why it's considered unused
|
|
114
|
+
recommendation: str # What to do about it
|
|
115
|
+
impact: str # Potential impact of removal
|
|
116
|
+
confidence: float = 1.0 # How confident we are (0-1)
|
|
117
|
+
columns_in_join_only: list[str] = field(default_factory=list)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass
|
|
121
|
+
class UnusedDepsAnalysisResult:
|
|
122
|
+
"""Result of unused dependency analysis."""
|
|
123
|
+
|
|
124
|
+
unused_joins: list[UnusedDependency] = field(default_factory=list)
|
|
125
|
+
unused_ctes: list[UnusedDependency] = field(default_factory=list)
|
|
126
|
+
table_usage: list[TableUsage] = field(default_factory=list)
|
|
127
|
+
cte_usage: list[CTEUsage] = field(default_factory=list)
|
|
128
|
+
warnings: list[str] = field(default_factory=list)
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def total_issues(self) -> int:
|
|
132
|
+
return len(self.unused_joins) + len(self.unused_ctes)
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def has_issues(self) -> bool:
|
|
136
|
+
return self.total_issues > 0
|
|
137
|
+
|
|
138
|
+
def to_dict(self) -> dict:
|
|
139
|
+
"""Convert to dictionary for JSON serialization."""
|
|
140
|
+
return {
|
|
141
|
+
"total_issues": self.total_issues,
|
|
142
|
+
"has_issues": self.has_issues,
|
|
143
|
+
"unused_joins": [
|
|
144
|
+
{
|
|
145
|
+
"type": dep.type,
|
|
146
|
+
"name": dep.name,
|
|
147
|
+
"location": dep.location,
|
|
148
|
+
"reason": dep.reason,
|
|
149
|
+
"recommendation": dep.recommendation,
|
|
150
|
+
"impact": dep.impact,
|
|
151
|
+
"confidence": dep.confidence,
|
|
152
|
+
"columns_in_join_only": dep.columns_in_join_only,
|
|
153
|
+
}
|
|
154
|
+
for dep in self.unused_joins
|
|
155
|
+
],
|
|
156
|
+
"unused_ctes": [
|
|
157
|
+
{
|
|
158
|
+
"type": dep.type,
|
|
159
|
+
"name": dep.name,
|
|
160
|
+
"location": dep.location,
|
|
161
|
+
"reason": dep.reason,
|
|
162
|
+
"recommendation": dep.recommendation,
|
|
163
|
+
"impact": dep.impact,
|
|
164
|
+
"confidence": dep.confidence,
|
|
165
|
+
}
|
|
166
|
+
for dep in self.unused_ctes
|
|
167
|
+
],
|
|
168
|
+
"table_usage": [
|
|
169
|
+
{
|
|
170
|
+
"name": t.name,
|
|
171
|
+
"alias": t.alias,
|
|
172
|
+
"full_name": t.full_name,
|
|
173
|
+
"join_type": t.join_type,
|
|
174
|
+
"is_from_table": t.is_from_table,
|
|
175
|
+
"columns_selected": t.columns_selected,
|
|
176
|
+
"columns_in_where": t.columns_in_where,
|
|
177
|
+
"columns_in_join_condition": t.columns_in_join_condition,
|
|
178
|
+
"total_uses": t.total_column_uses,
|
|
179
|
+
"is_used": t.is_used,
|
|
180
|
+
"only_used_for_join": t.only_used_for_join,
|
|
181
|
+
}
|
|
182
|
+
for t in self.table_usage
|
|
183
|
+
],
|
|
184
|
+
"cte_usage": [
|
|
185
|
+
{
|
|
186
|
+
"name": c.name,
|
|
187
|
+
"referenced_count": c.referenced_count,
|
|
188
|
+
"referenced_in": c.referenced_in,
|
|
189
|
+
"is_used": c.is_used,
|
|
190
|
+
}
|
|
191
|
+
for c in self.cte_usage
|
|
192
|
+
],
|
|
193
|
+
"warnings": self.warnings,
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class UnusedDependencyDetector:
|
|
198
|
+
"""Detect unused dependencies in SQL queries.
|
|
199
|
+
|
|
200
|
+
This analyzer finds:
|
|
201
|
+
1. JOINed tables where no columns are actually used in the output
|
|
202
|
+
2. CTEs that are defined but never referenced
|
|
203
|
+
3. Tables only used for filtering that could be optimized
|
|
204
|
+
|
|
205
|
+
Usage:
|
|
206
|
+
detector = UnusedDependencyDetector(dialect="snowflake")
|
|
207
|
+
result = detector.analyze(sql)
|
|
208
|
+
for issue in result.unused_joins:
|
|
209
|
+
print(f"Unused join: {issue.name} - {issue.reason}")
|
|
210
|
+
"""
|
|
211
|
+
|
|
212
|
+
def __init__(self, dialect: str = "snowflake"):
|
|
213
|
+
"""Initialize detector.
|
|
214
|
+
|
|
215
|
+
Args:
|
|
216
|
+
dialect: SQL dialect (snowflake, bigquery, databricks, postgres, etc.)
|
|
217
|
+
"""
|
|
218
|
+
self.dialect = dialect
|
|
219
|
+
self._has_unqualified_columns = False
|
|
220
|
+
self._has_select_star = False
|
|
221
|
+
self._has_using_join = False
|
|
222
|
+
self._has_natural_join = False
|
|
223
|
+
self._has_lateral_join = False
|
|
224
|
+
self._has_table_function = False
|
|
225
|
+
|
|
226
|
+
def analyze(self, sql: str) -> UnusedDepsAnalysisResult:
|
|
227
|
+
"""Analyze SQL for unused dependencies.
|
|
228
|
+
|
|
229
|
+
Args:
|
|
230
|
+
sql: The SQL query to analyze
|
|
231
|
+
|
|
232
|
+
Returns:
|
|
233
|
+
UnusedDepsAnalysisResult with detected issues
|
|
234
|
+
"""
|
|
235
|
+
result = UnusedDepsAnalysisResult()
|
|
236
|
+
|
|
237
|
+
try:
|
|
238
|
+
parsed = sqlglot.parse_one(sql, dialect=self.dialect)
|
|
239
|
+
except Exception as e:
|
|
240
|
+
result.warnings.append(f"Failed to parse SQL: {str(e)}")
|
|
241
|
+
return result
|
|
242
|
+
|
|
243
|
+
# Analyze CTEs
|
|
244
|
+
self._analyze_ctes(parsed, result)
|
|
245
|
+
|
|
246
|
+
# Analyze table usage (JOINs)
|
|
247
|
+
self._analyze_table_usage(parsed, result)
|
|
248
|
+
|
|
249
|
+
# Detect unused joins
|
|
250
|
+
self._detect_unused_joins(result)
|
|
251
|
+
|
|
252
|
+
# Detect unused CTEs
|
|
253
|
+
self._detect_unused_ctes(result)
|
|
254
|
+
|
|
255
|
+
if self._has_unqualified_columns:
|
|
256
|
+
result.warnings.append("unqualified_columns")
|
|
257
|
+
if self._has_select_star:
|
|
258
|
+
result.warnings.append("select_star")
|
|
259
|
+
if self._has_using_join:
|
|
260
|
+
result.warnings.append("using_join")
|
|
261
|
+
if self._has_natural_join:
|
|
262
|
+
result.warnings.append("natural_join")
|
|
263
|
+
if self._has_lateral_join:
|
|
264
|
+
result.warnings.append("lateral_join")
|
|
265
|
+
if self._has_table_function:
|
|
266
|
+
result.warnings.append("table_function")
|
|
267
|
+
|
|
268
|
+
return result
|
|
269
|
+
|
|
270
|
+
def _analyze_ctes(self, parsed: exp.Expression, result: UnusedDepsAnalysisResult) -> None:
|
|
271
|
+
"""Analyze CTE definitions and usage."""
|
|
272
|
+
# Find all CTE definitions
|
|
273
|
+
cte_names: dict[str, CTEUsage] = {}
|
|
274
|
+
|
|
275
|
+
for cte in parsed.find_all(exp.CTE):
|
|
276
|
+
cte_name = cte.alias
|
|
277
|
+
cte_sql = (safe_sql(cte.this, self.dialect) or "") if cte.this else ""
|
|
278
|
+
cte_names[cte_name.lower()] = CTEUsage(
|
|
279
|
+
name=cte_name,
|
|
280
|
+
sql=cte_sql[:200] + "..." if len(cte_sql) > 200 else cte_sql,
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
if not cte_names:
|
|
284
|
+
return
|
|
285
|
+
|
|
286
|
+
# Find all table references
|
|
287
|
+
for table in parsed.find_all(exp.Table):
|
|
288
|
+
table_name = table.name.lower()
|
|
289
|
+
|
|
290
|
+
if table_name in cte_names:
|
|
291
|
+
cte_names[table_name].referenced_count += 1
|
|
292
|
+
|
|
293
|
+
# Try to determine where it's referenced
|
|
294
|
+
parent = table.parent
|
|
295
|
+
location = "unknown"
|
|
296
|
+
if parent:
|
|
297
|
+
if isinstance(parent, exp.From):
|
|
298
|
+
location = "FROM clause"
|
|
299
|
+
elif isinstance(parent, exp.Join):
|
|
300
|
+
location = "JOIN"
|
|
301
|
+
elif isinstance(parent, exp.Subquery):
|
|
302
|
+
location = "subquery"
|
|
303
|
+
|
|
304
|
+
cte_names[table_name].referenced_in.append(location)
|
|
305
|
+
|
|
306
|
+
# Also check CTEs referencing other CTEs
|
|
307
|
+
for cte in parsed.find_all(exp.CTE):
|
|
308
|
+
cte_name = cte.alias.lower()
|
|
309
|
+
for inner_table in cte.find_all(exp.Table):
|
|
310
|
+
inner_name = inner_table.name.lower()
|
|
311
|
+
if inner_name in cte_names and inner_name != cte_name:
|
|
312
|
+
cte_names[inner_name].referenced_count += 1
|
|
313
|
+
cte_names[inner_name].referenced_in.append(f"CTE '{cte.alias}'")
|
|
314
|
+
|
|
315
|
+
result.cte_usage = list(cte_names.values())
|
|
316
|
+
|
|
317
|
+
def _analyze_table_usage(
|
|
318
|
+
self, parsed: exp.Expression, result: UnusedDepsAnalysisResult
|
|
319
|
+
) -> None:
|
|
320
|
+
"""Analyze how each table is used in the query."""
|
|
321
|
+
tables: dict[str, TableUsage] = {}
|
|
322
|
+
|
|
323
|
+
# Find all tables in FROM and JOINs
|
|
324
|
+
for from_clause in parsed.find_all(exp.From):
|
|
325
|
+
table_expr = from_clause.this
|
|
326
|
+
if isinstance(table_expr, exp.Table):
|
|
327
|
+
usage = self._create_table_usage(table_expr, is_from=True)
|
|
328
|
+
tables[usage.identifier.lower()] = usage
|
|
329
|
+
|
|
330
|
+
for join in parsed.find_all(exp.Join):
|
|
331
|
+
table_expr = join.this
|
|
332
|
+
if isinstance(table_expr, exp.Table):
|
|
333
|
+
usage = self._create_table_usage(table_expr, is_from=False)
|
|
334
|
+
usage.join_type = join.kind or "INNER"
|
|
335
|
+
tables[usage.identifier.lower()] = usage
|
|
336
|
+
if join.args.get("using"):
|
|
337
|
+
self._has_using_join = True
|
|
338
|
+
if join.args.get("method") == "natural":
|
|
339
|
+
self._has_natural_join = True
|
|
340
|
+
|
|
341
|
+
# Track columns used in join condition
|
|
342
|
+
if join.args.get("on"):
|
|
343
|
+
for col in join.args["on"].find_all(exp.Column):
|
|
344
|
+
col_table = (col.table or "").lower()
|
|
345
|
+
if col_table in tables:
|
|
346
|
+
tables[col_table].columns_in_join_condition.append(col.name)
|
|
347
|
+
|
|
348
|
+
# Detect LATERAL joins and table functions (can't reliably analyze)
|
|
349
|
+
if join.args.get("side") == "LATERAL" or isinstance(table_expr, exp.Lateral):
|
|
350
|
+
self._has_lateral_join = True
|
|
351
|
+
|
|
352
|
+
# Detect table functions (FLATTEN, UNNEST, etc.)
|
|
353
|
+
for _lateral in parsed.find_all(exp.Lateral):
|
|
354
|
+
self._has_lateral_join = True
|
|
355
|
+
for _func in parsed.find_all(exp.Unnest):
|
|
356
|
+
self._has_table_function = True
|
|
357
|
+
for _func in parsed.find_all(exp.Explode):
|
|
358
|
+
self._has_table_function = True
|
|
359
|
+
|
|
360
|
+
if not tables:
|
|
361
|
+
return
|
|
362
|
+
|
|
363
|
+
# Find column usage in different clauses
|
|
364
|
+
self._track_column_usage(parsed, tables, "select", "columns_selected")
|
|
365
|
+
self._track_column_usage(parsed, tables, "where", "columns_in_where")
|
|
366
|
+
self._track_column_usage(parsed, tables, "group", "columns_in_group_by")
|
|
367
|
+
self._track_column_usage(parsed, tables, "order", "columns_in_order_by")
|
|
368
|
+
self._track_column_usage(parsed, tables, "having", "columns_in_having")
|
|
369
|
+
|
|
370
|
+
result.table_usage = list(tables.values())
|
|
371
|
+
|
|
372
|
+
def _create_table_usage(self, table_expr: exp.Table, is_from: bool) -> TableUsage:
|
|
373
|
+
"""Create TableUsage from a table expression."""
|
|
374
|
+
return TableUsage(
|
|
375
|
+
name=table_expr.name,
|
|
376
|
+
alias=table_expr.alias if table_expr.alias else None,
|
|
377
|
+
schema_name=table_expr.db,
|
|
378
|
+
database=table_expr.catalog,
|
|
379
|
+
is_from_table=is_from,
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
def _track_column_usage(
|
|
383
|
+
self,
|
|
384
|
+
parsed: exp.Expression,
|
|
385
|
+
tables: dict[str, TableUsage],
|
|
386
|
+
clause_type: str,
|
|
387
|
+
attr_name: str,
|
|
388
|
+
) -> None:
|
|
389
|
+
"""Track column usage in a specific clause type.
|
|
390
|
+
|
|
391
|
+
IMPORTANT: We need to be precise about what we search.
|
|
392
|
+
For SELECT, we only want the actual selected expressions, not JOINs.
|
|
393
|
+
For WHERE/GROUP/ORDER/HAVING, we want those specific clauses only.
|
|
394
|
+
"""
|
|
395
|
+
for select_stmt in parsed.find_all(exp.Select):
|
|
396
|
+
if clause_type == "select":
|
|
397
|
+
# Only look at the SELECT expressions (not JOINs, WHERE, etc.)
|
|
398
|
+
for select_expr in select_stmt.expressions:
|
|
399
|
+
if isinstance(select_expr, exp.Star):
|
|
400
|
+
self._has_select_star = True
|
|
401
|
+
for table in tables.values():
|
|
402
|
+
col_list = getattr(table, attr_name)
|
|
403
|
+
if "*" not in col_list:
|
|
404
|
+
col_list.append("*")
|
|
405
|
+
continue
|
|
406
|
+
if isinstance(select_expr, exp.Column) and select_expr.name == "*":
|
|
407
|
+
table_name = (select_expr.table or "").lower()
|
|
408
|
+
if table_name and table_name in tables:
|
|
409
|
+
col_list = getattr(tables[table_name], attr_name)
|
|
410
|
+
if "*" not in col_list:
|
|
411
|
+
col_list.append("*")
|
|
412
|
+
else:
|
|
413
|
+
self._has_select_star = True
|
|
414
|
+
continue
|
|
415
|
+
self._extract_columns_from_expr(select_expr, tables, attr_name)
|
|
416
|
+
elif clause_type == "where":
|
|
417
|
+
where_clause = select_stmt.args.get("where")
|
|
418
|
+
if where_clause:
|
|
419
|
+
self._extract_columns_from_expr(where_clause, tables, attr_name)
|
|
420
|
+
elif clause_type == "group":
|
|
421
|
+
group_clause = select_stmt.args.get("group")
|
|
422
|
+
if group_clause:
|
|
423
|
+
self._extract_columns_from_expr(group_clause, tables, attr_name)
|
|
424
|
+
elif clause_type == "order":
|
|
425
|
+
order_clause = select_stmt.args.get("order")
|
|
426
|
+
if order_clause:
|
|
427
|
+
self._extract_columns_from_expr(order_clause, tables, attr_name)
|
|
428
|
+
elif clause_type == "having":
|
|
429
|
+
having_clause = select_stmt.args.get("having")
|
|
430
|
+
if having_clause:
|
|
431
|
+
self._extract_columns_from_expr(having_clause, tables, attr_name)
|
|
432
|
+
|
|
433
|
+
def _extract_columns_from_expr(
|
|
434
|
+
self,
|
|
435
|
+
expr: exp.Expression,
|
|
436
|
+
tables: dict[str, TableUsage],
|
|
437
|
+
attr_name: str,
|
|
438
|
+
) -> None:
|
|
439
|
+
"""Extract column references from an expression and add to table usage."""
|
|
440
|
+
for col in expr.find_all(exp.Column):
|
|
441
|
+
col_table = (col.table or "").lower()
|
|
442
|
+
|
|
443
|
+
# If no table qualifier, we can't reliably attribute the column
|
|
444
|
+
if not col_table:
|
|
445
|
+
self._has_unqualified_columns = True
|
|
446
|
+
continue
|
|
447
|
+
|
|
448
|
+
if col_table in tables:
|
|
449
|
+
col_list = getattr(tables[col_table], attr_name)
|
|
450
|
+
if col.name not in col_list:
|
|
451
|
+
col_list.append(col.name)
|
|
452
|
+
|
|
453
|
+
def _detect_unused_joins(self, result: UnusedDepsAnalysisResult) -> None:
|
|
454
|
+
"""Detect joins that don't contribute to the output."""
|
|
455
|
+
for table in result.table_usage:
|
|
456
|
+
if table.is_from_table:
|
|
457
|
+
continue # Main table is always needed
|
|
458
|
+
|
|
459
|
+
if table.only_used_for_join:
|
|
460
|
+
# Table columns only appear in JOIN condition, not in output
|
|
461
|
+
dep = UnusedDependency(
|
|
462
|
+
type="join",
|
|
463
|
+
name=table.full_name,
|
|
464
|
+
location=f"{table.join_type} JOIN {table.full_name}"
|
|
465
|
+
+ (f" AS {table.alias}" if table.alias else ""),
|
|
466
|
+
reason=f"Columns from this table ({', '.join(table.columns_in_join_condition)}) are only used in the JOIN condition, not in the query output",
|
|
467
|
+
recommendation=(
|
|
468
|
+
"Consider if this JOIN is necessary. If you're only using it for filtering, "
|
|
469
|
+
"an EXISTS subquery or IN clause might be more efficient"
|
|
470
|
+
),
|
|
471
|
+
impact="This table is an upstream dependency but doesn't directly contribute data to the output",
|
|
472
|
+
confidence=0.7, # Lower confidence - might be intentional
|
|
473
|
+
columns_in_join_only=table.columns_in_join_condition,
|
|
474
|
+
)
|
|
475
|
+
result.unused_joins.append(dep)
|
|
476
|
+
continue
|
|
477
|
+
|
|
478
|
+
if not table.is_used:
|
|
479
|
+
# Table is joined but no columns are used anywhere
|
|
480
|
+
dep = UnusedDependency(
|
|
481
|
+
type="join",
|
|
482
|
+
name=table.full_name,
|
|
483
|
+
location=f"{table.join_type} JOIN {table.full_name}"
|
|
484
|
+
+ (f" AS {table.alias}" if table.alias else ""),
|
|
485
|
+
reason="No columns from this table are used in SELECT, WHERE, GROUP BY, or HAVING",
|
|
486
|
+
recommendation=f"Remove the {table.join_type} JOIN to '{table.full_name}' - it adds overhead without contributing to results",
|
|
487
|
+
impact="Removing this join will improve query performance and reduce upstream dependencies",
|
|
488
|
+
confidence=0.95,
|
|
489
|
+
columns_in_join_only=table.columns_in_join_condition,
|
|
490
|
+
)
|
|
491
|
+
result.unused_joins.append(dep)
|
|
492
|
+
|
|
493
|
+
def _detect_unused_ctes(self, result: UnusedDepsAnalysisResult) -> None:
|
|
494
|
+
"""Detect CTEs that are defined but never used."""
|
|
495
|
+
for cte in result.cte_usage:
|
|
496
|
+
if not cte.is_used:
|
|
497
|
+
dep = UnusedDependency(
|
|
498
|
+
type="cte",
|
|
499
|
+
name=cte.name,
|
|
500
|
+
location=f"WITH {cte.name} AS (...)",
|
|
501
|
+
reason="This CTE is defined but never referenced in the query",
|
|
502
|
+
recommendation=f"Remove the CTE '{cte.name}' - it's dead code that adds complexity",
|
|
503
|
+
impact="Removing unused CTEs improves query readability and may improve planning time",
|
|
504
|
+
confidence=1.0, # Very confident - unused CTE is clearly dead code
|
|
505
|
+
)
|
|
506
|
+
result.unused_ctes.append(dep)
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def analyze_sql_for_unused_deps(sql: str, dialect: str = "snowflake") -> UnusedDepsAnalysisResult:
|
|
510
|
+
"""Convenience function to analyze SQL for unused dependencies.
|
|
511
|
+
|
|
512
|
+
Args:
|
|
513
|
+
sql: SQL query to analyze
|
|
514
|
+
dialect: SQL dialect
|
|
515
|
+
|
|
516
|
+
Returns:
|
|
517
|
+
Analysis result with any detected issues
|
|
518
|
+
"""
|
|
519
|
+
detector = UnusedDependencyDetector(dialect=dialect)
|
|
520
|
+
return detector.analyze(sql)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def get_optimization_summary(result: UnusedDepsAnalysisResult) -> str:
|
|
524
|
+
"""Generate a human-readable summary of optimization opportunities.
|
|
525
|
+
|
|
526
|
+
Args:
|
|
527
|
+
result: Analysis result
|
|
528
|
+
|
|
529
|
+
Returns:
|
|
530
|
+
Formatted summary string
|
|
531
|
+
"""
|
|
532
|
+
if not result.has_issues:
|
|
533
|
+
return "No unused dependencies detected."
|
|
534
|
+
|
|
535
|
+
lines = [
|
|
536
|
+
f"Found {result.total_issues} potential optimization(s):",
|
|
537
|
+
"",
|
|
538
|
+
]
|
|
539
|
+
|
|
540
|
+
if result.unused_joins:
|
|
541
|
+
lines.append(f"## Unused JOINs ({len(result.unused_joins)})")
|
|
542
|
+
for dep in result.unused_joins:
|
|
543
|
+
lines.append(f" - {dep.name}")
|
|
544
|
+
lines.append(f" Reason: {dep.reason}")
|
|
545
|
+
lines.append(f" Recommendation: {dep.recommendation}")
|
|
546
|
+
lines.append("")
|
|
547
|
+
|
|
548
|
+
if result.unused_ctes:
|
|
549
|
+
lines.append(f"## Unused CTEs ({len(result.unused_ctes)})")
|
|
550
|
+
for dep in result.unused_ctes:
|
|
551
|
+
lines.append(f" - {dep.name}")
|
|
552
|
+
lines.append(f" Reason: {dep.reason}")
|
|
553
|
+
lines.append("")
|
|
554
|
+
|
|
555
|
+
return "\n".join(lines)
|