ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/engine/jinja.py ADDED
@@ -0,0 +1,282 @@
1
+ """Convert Jinja-templated dbt SQL into parseable SQL.
2
+
3
+ This module handles the transformation of dbt's Jinja-templated SQL into
4
+ standard SQL that can be parsed by SQLGlot for column lineage extraction.
5
+
6
+ Key transformations:
7
+ - {{ ref('model') }} → model
8
+ - {{ source('schema', 'table') }} → schema__table
9
+ - {{ config(...) }} → removed
10
+ - {% if/for/... %} → removed
11
+ - Custom macros → NULL placeholder
12
+
13
+ Security:
14
+ - ReDoS protection via size limits on regex processing
15
+ - Bounded regex patterns to prevent catastrophic backtracking
16
+ """
17
+
18
+ import logging
19
+ import re
20
+ from typing import Final
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+ # Maximum SQL size for regex processing (500KB) - prevents ReDoS on huge files
25
+ MAX_SQL_SIZE_FOR_REGEX: Final[int] = 500 * 1024
26
+
27
+
28
+ def convert_jinja_to_sql(raw_sql: str) -> str:
29
+ """Convert Jinja-templated SQL to parseable SQL.
30
+
31
+ This is the full conversion that handles:
32
+
33
+ Core dbt functions:
34
+ - {{ ref('model') }} → model
35
+ - {{ source('src', 'table') }} → src__table
36
+ - {{ config(...) }} → removed
37
+ - {{ var('x') }} → 'x' (variable reference)
38
+ - {{ this }} → this_model
39
+
40
+ dbt-utils macros:
41
+ - {{ group_by(n) }} → GROUP BY n
42
+ - {{ dbt_utils.star(...) }} → *
43
+ - {{ dbt_utils.surrogate_key(...) }} → md5(...)
44
+
45
+ SQL normalization:
46
+ - Escaped quotes: ''value'' → 'value'
47
+ - Dollar-quoted strings: $$text$$ → 'text'
48
+ - Prepared statement placeholders: $1, ?, :name → 'param'
49
+
50
+ Args:
51
+ raw_sql: SQL with Jinja templates
52
+
53
+ Returns:
54
+ Clean SQL parseable by SQLGlot
55
+
56
+ Security:
57
+ Returns raw SQL unchanged if size exceeds MAX_SQL_SIZE_FOR_REGEX
58
+ to prevent ReDoS attacks.
59
+ """
60
+ if len(raw_sql) > MAX_SQL_SIZE_FOR_REGEX:
61
+ logger.warning(
62
+ f"SQL too large for Jinja cleaning ({len(raw_sql)} bytes > {MAX_SQL_SIZE_FOR_REGEX}). "
63
+ "Returning raw SQL to avoid ReDoS risk."
64
+ )
65
+ return raw_sql
66
+
67
+ cleaned = raw_sql
68
+
69
+ # ==========================================================================
70
+ # SQL NORMALIZATION FOR CROSS-DIALECT PARSING
71
+ # These patterns handle syntax variations across Snowflake, BigQuery,
72
+ # Databricks, Trino, Redshift, Postgres that sqlglot may struggle with.
73
+ # ==========================================================================
74
+
75
+ # 1. Escaped single quotes: ''value'' → 'value'
76
+ # Common in Snowflake/Trino compiled SQL for Jinja variables
77
+ cleaned = re.sub(r"''([^']+)''", r"'\1'", cleaned)
78
+
79
+ # 2. Dollar-quoted strings (PostgreSQL/Redshift): $$text$$ → 'text'
80
+ cleaned = re.sub(r"\$\$([^$]*)\$\$", r"'\1'", cleaned)
81
+ cleaned = re.sub(r"\$(\w+)\$([^$]*)\$\1\$", r"'\2'", cleaned) # Tagged: $tag$text$tag$
82
+
83
+ # 3. Unicode escape sequences: U&'text' → 'text'
84
+ cleaned = re.sub(r"U&'([^']*)'", r"'\1'", cleaned, flags=re.IGNORECASE)
85
+
86
+ # 4. National character strings: N'text' → 'text'
87
+ cleaned = re.sub(r"(?<!\w)N'([^']*)'", r"'\1'", cleaned)
88
+
89
+ # 5. Raw string literals (BigQuery): r'pattern' → 'pattern'
90
+ cleaned = re.sub(r"(?<!\w)[rR]'([^']*)'", r"'\1'", cleaned)
91
+ cleaned = re.sub(r'(?<!\w)[rR]"([^"]*)"', r"'\1'", cleaned)
92
+
93
+ # 6. Byte literals (BigQuery): b'bytes' → 'bytes'
94
+ cleaned = re.sub(r"(?<!\w)[bB]'([^']*)'", r"'\1'", cleaned)
95
+ cleaned = re.sub(r'(?<!\w)[bB]"([^"]*)"', r"'\1'", cleaned)
96
+
97
+ # 7. Prepared statement placeholders → literal placeholder
98
+ cleaned = re.sub(r"\$(\d+)", r"'param\1'", cleaned) # $1, $2 (Postgres)
99
+ cleaned = re.sub(r"(?<!\?)\?(?!\?)", "'param'", cleaned) # ? (JDBC)
100
+ cleaned = re.sub(
101
+ r"(?<!:):(\w+)(?=\s|,|\)|$|AND|OR|=|>|<|\+|-|\*|/)",
102
+ r"'\1'",
103
+ cleaned,
104
+ flags=re.IGNORECASE,
105
+ ) # :name (Oracle)
106
+
107
+ # 8. Interval literals with quoted units
108
+ cleaned = re.sub(r"INTERVAL\s+'(\d+)'\s+(\w+)", r"INTERVAL \1 \2", cleaned, flags=re.IGNORECASE)
109
+
110
+ # ==========================================================================
111
+ # CORE DBT FUNCTIONS
112
+ # ==========================================================================
113
+
114
+ # ref('model') → model
115
+ cleaned = re.sub(r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"\1", cleaned)
116
+
117
+ # source('schema', 'table') → schema__table
118
+ cleaned = re.sub(
119
+ r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
120
+ r"\1__\2",
121
+ cleaned,
122
+ )
123
+
124
+ # config() blocks (can be multiline, nested)
125
+ cleaned = re.sub(r"\{\{\s*config\s*\(.*?\)\s*\}\}", "", cleaned, flags=re.DOTALL)
126
+ cleaned = re.sub(r"\{\{\s*config\([^}]*\}\}\s*\}\}", "", cleaned)
127
+
128
+ # ==========================================================================
129
+ # DBT-UTILS MACROS
130
+ # ==========================================================================
131
+
132
+ cleaned = re.sub(r"\{\{\s*group_by\s*\(\s*(\d+)\s*\)\s*\}\}", r"GROUP BY \1", cleaned)
133
+ cleaned = re.sub(r"\{\{\s*order_by\s*\(\s*(\d+)\s*\)\s*\}\}", r"ORDER BY \1", cleaned)
134
+ cleaned = re.sub(r"\{\{\s*limit\s*\(\s*(\d+)\s*\)\s*\}\}", r"LIMIT \1", cleaned)
135
+
136
+ # dbt_utils.star() → *
137
+ cleaned = re.sub(r"\{\{\s*dbt_utils\.star\s*\([^}]*\)\s*\}\}", "*", cleaned)
138
+
139
+ # dbt_utils.surrogate_key() → md5('key')
140
+ cleaned = re.sub(r"\{\{\s*dbt_utils\.surrogate_key\s*\([^)]*\)\s*\}\}", "md5('key')", cleaned)
141
+ cleaned = re.sub(
142
+ r"\{\{\s*dbt_utils\.generate_surrogate_key\s*\([^)]*\)\s*\}\}",
143
+ "md5('key')",
144
+ cleaned,
145
+ )
146
+
147
+ # dbt_utils.pivot/unpivot → comment placeholder
148
+ cleaned = re.sub(r"\{\{\s*dbt_utils\.pivot\s*\([^)]*\)\s*\}\}", "/* pivot */", cleaned)
149
+ cleaned = re.sub(r"\{\{\s*dbt_utils\.unpivot\s*\([^)]*\)\s*\}\}", "/* unpivot */", cleaned)
150
+
151
+ # dbt_utils.generate_series() → subquery
152
+ cleaned = re.sub(
153
+ r"\{\{\s*dbt_utils\.generate_series\s*\([^)]*\)\s*\}\}",
154
+ "(SELECT 1 AS n)",
155
+ cleaned,
156
+ )
157
+
158
+ # dbt_utils.get_column_values() → tuple
159
+ cleaned = re.sub(
160
+ r"\{\{\s*dbt_utils\.get_column_values\s*\([^)]*\)\s*\}\}",
161
+ "('value1', 'value2')",
162
+ cleaned,
163
+ )
164
+
165
+ # dbt_utils.safe_cast() → TRY_CAST
166
+ cleaned = re.sub(
167
+ r"\{\{\s*dbt_utils\.safe_cast\s*\(\s*['\"]?([^'\"]+)['\"]?\s*,\s*['\"]?([^'\"]+)['\"]?\s*\)\s*\}\}",
168
+ r"TRY_CAST(\1 AS \2)",
169
+ cleaned,
170
+ )
171
+
172
+ # dbt_utils.date_trunc() → DATE_TRUNC
173
+ cleaned = re.sub(
174
+ r"\{\{\s*dbt_utils\.date_trunc\s*\(\s*['\"](\w+)['\"]\s*,\s*([^)]+)\s*\)\s*\}\}",
175
+ r"DATE_TRUNC('\1', \2)",
176
+ cleaned,
177
+ )
178
+
179
+ # Generic dbt_utils.* → NULL
180
+ cleaned = re.sub(r"\{\{\s*dbt_utils\.\w+\s*\([^)]*\)\s*\}\}", "NULL", cleaned)
181
+
182
+ # ==========================================================================
183
+ # OTHER DBT MACROS
184
+ # ==========================================================================
185
+
186
+ # dbt_date macros
187
+ cleaned = re.sub(
188
+ r"\{\{\s*dbt_date\.date_spine\s*\([^)]*\)\s*\}\}",
189
+ "(SELECT CURRENT_DATE AS date_day)",
190
+ cleaned,
191
+ )
192
+ cleaned = re.sub(r"\{\{\s*dbt_date\.\w+\s*\([^)]*\)\s*\}\}", "CURRENT_DATE", cleaned)
193
+
194
+ # fivetran_utils macros
195
+ cleaned = re.sub(
196
+ r"\{\{\s*fivetran_utils\.\w+\s*\([^)]*\)\s*\}\}", "/* fivetran_macro */", cleaned
197
+ )
198
+
199
+ # dbt_expectations macros
200
+ cleaned = re.sub(r"\{\{\s*dbt_expectations\.\w+\s*\([^)]*\)\s*\}\}", "TRUE", cleaned)
201
+
202
+ # ==========================================================================
203
+ # VARIABLE REFERENCES
204
+ # ==========================================================================
205
+
206
+ cleaned = re.sub(r"\{\{\s*var\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"'\1'", cleaned)
207
+ cleaned = re.sub(r"\{\{\s*env_var\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"'\1'", cleaned)
208
+ cleaned = re.sub(r"\{\{\s*this\s*\}\}", "this_model", cleaned)
209
+ cleaned = re.sub(r"\{\{\s*target\.\w+\s*\}\}", "'target_value'", cleaned)
210
+ cleaned = re.sub(r"\{\{\s*run_started_at\s*\}\}", "CURRENT_TIMESTAMP", cleaned)
211
+ cleaned = re.sub(r"\{\{\s*invocation_id\s*\}\}", "'invocation_id'", cleaned)
212
+
213
+ # ==========================================================================
214
+ # CONTROL FLOW AND CLEANUP
215
+ # ==========================================================================
216
+
217
+ # {% ... %} control blocks carry no lineage
218
+ cleaned = re.sub(r"\{%[\s\S]*?%\}", "", cleaned)
219
+
220
+ # Any remaining {{ ... }} → NULL
221
+ cleaned = re.sub(r"\{\{[^}]*\}\}", "NULL", cleaned)
222
+
223
+ # Clean up leftover Jinja fragments
224
+ cleaned = re.sub(r'["\')]*\s*\}\}', "", cleaned)
225
+ cleaned = re.sub(r'\{\{\s*["\(\']*', "", cleaned)
226
+ cleaned = re.sub(r'^\s*["\']+["\')]*\s*$', "", cleaned, flags=re.MULTILINE)
227
+
228
+ # a stripped config block leaves its indentation behind
229
+ cleaned = cleaned.lstrip()
230
+
231
+ return cleaned
232
+
233
+
234
+ def convert_jinja_to_sql_minimal(raw_sql: str) -> str:
235
+ """Minimal Jinja conversion - only guaranteed-safe transformations.
236
+
237
+ Use this as a fallback when full conversion breaks valid SQL.
238
+ Only handles patterns that cannot produce invalid SQL.
239
+
240
+ Safe transformations:
241
+ - {{ ref('model') }} → model
242
+ - {{ source('src', 'table') }} → src__table
243
+ - {{ config(...) }} → removed
244
+ - {% ... %} → removed
245
+ - Other {{ ... }} → NULL
246
+
247
+ Does NOT apply:
248
+ - SQL string literal normalization
249
+ - Prepared statement placeholder handling
250
+ - Dialect-specific pattern fixes
251
+
252
+ Args:
253
+ raw_sql: SQL with Jinja templates
254
+
255
+ Returns:
256
+ Minimally cleaned SQL
257
+ """
258
+ if len(raw_sql) > MAX_SQL_SIZE_FOR_REGEX:
259
+ return raw_sql
260
+
261
+ cleaned = raw_sql
262
+
263
+ # Only the safest Jinja transformations
264
+ cleaned = re.sub(r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", r"\1", cleaned)
265
+ cleaned = re.sub(
266
+ r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
267
+ r"\1__\2",
268
+ cleaned,
269
+ )
270
+ cleaned = re.sub(r"\{\{\s*config\s*\(.*?\)\s*\}\}", "", cleaned, flags=re.DOTALL)
271
+
272
+ # Control flow blocks
273
+ cleaned = re.sub(r"\{%[\s\S]*?%\}", "", cleaned)
274
+
275
+ # Remaining Jinja macros → NULL
276
+ cleaned = re.sub(r"\{\{[^}]*\}\}", "NULL", cleaned)
277
+
278
+ # Minimal cleanup
279
+ cleaned = re.sub(r'["\')]*\s*\}\}', "", cleaned)
280
+ cleaned = re.sub(r'\{\{\s*["\(\']*', "", cleaned)
281
+
282
+ return cleaned
@@ -0,0 +1,241 @@
1
+ """JSON-access source extraction for select items.
2
+
3
+ Handles column references reached through JSON path expressions
4
+ (JSON_EXTRACT, ->, data:field) and bracket syntax (data['key']), preserving
5
+ the JSON path on the emitted source entry.
6
+ """
7
+
8
+ from typing import Any
9
+
10
+ from sqlglot import exp
11
+
12
+ from ripple.engine.column_ref import (
13
+ resolve_column_ref,
14
+ under_subquery_where,
15
+ )
16
+
17
+ # built once: per-column construction showed up in profiles
18
+ # Covers all dialects:
19
+ # - Snowflake: data:field::string (parses to JSONExtract)
20
+ # - BigQuery: JSON_EXTRACT, JSON_EXTRACT_SCALAR, JSON_VALUE
21
+ # - Trino: json_extract, json_extract_scalar
22
+ # - Redshift: json_extract_path_text
23
+ # - Postgres: ->, ->> (JSONExtract), #>, #>> (JSONBExtract)
24
+ # - Databricks: get_json_object
25
+ JSON_EXPR_TYPES: tuple[type, ...] = (
26
+ exp.JSONExtract, # Most dialects: JSON_EXTRACT, ->, data:field
27
+ exp.JSONExtractScalar, # JSON_EXTRACT_SCALAR, ->>, JSON_VALUE
28
+ )
29
+ # the JSONB variants exist only in newer sqlglot
30
+ if hasattr(exp, "JSONBExtract"):
31
+ JSON_EXPR_TYPES = JSON_EXPR_TYPES + (exp.JSONBExtract,)
32
+ if hasattr(exp, "JSONBExtractScalar"):
33
+ JSON_EXPR_TYPES = JSON_EXPR_TYPES + (exp.JSONBExtractScalar,)
34
+
35
+
36
+ def _extract_json_path(json_path_expr: "exp.Expression | None") -> str:
37
+ """Extract JSON path string from various JSON path expression types.
38
+
39
+ Handles:
40
+ - JSONPath expressions: $.field.subfield -> "field.subfield"
41
+ - String literals: '$.path' -> "path"
42
+ - Trino array arguments: 'key1', 'key2' -> "key1.key2"
43
+ - Redshift json_extract_path_text: col, 'key1', 'key2' -> "key1.key2"
44
+
45
+ Returns:
46
+ Extracted path string (without leading $.) or empty string if not extractable.
47
+ """
48
+ if json_path_expr is None:
49
+ return ""
50
+
51
+ if hasattr(json_path_expr, "expressions"):
52
+ path_parts = []
53
+ for path_part in json_path_expr.expressions:
54
+ # Skip JSONPathRoot ($)
55
+ if type(path_part).__name__ == "JSONPathRoot":
56
+ continue
57
+ if hasattr(path_part, "this"):
58
+ part_name = str(path_part.this)
59
+ path_parts.append(part_name)
60
+ elif hasattr(path_part, "name"):
61
+ path_parts.append(path_part.name)
62
+ if path_parts:
63
+ return ".".join(path_parts)
64
+
65
+ if hasattr(json_path_expr, "this"):
66
+ path_str = str(json_path_expr.this)
67
+ # Strip quotes and leading $.
68
+ return path_str.strip("'\"").lstrip("$.")
69
+
70
+ # Fallback: try string representation
71
+ path_str = str(json_path_expr)
72
+ return path_str.strip("'\"").lstrip("$.")
73
+
74
+
75
+ def collect_json_sources(
76
+ source_expr: "exp.Expression",
77
+ known_relations: set[str],
78
+ alias_map: dict[str, str],
79
+ array_expansion_sources: dict[str, dict[str, str]],
80
+ default_table: str,
81
+ qualified_via_schema: bool,
82
+ sources: list[dict[str, Any]],
83
+ seen: set,
84
+ ) -> set[int]:
85
+ """Append JSON-path and bracket-access sources for one select item.
86
+
87
+ Mutates sources/seen in place and returns the ids of Column nodes
88
+ already handled, so the plain column sweep can skip them.
89
+ """
90
+ json_columns_handled = set()
91
+ for json_expr in source_expr.find_all(*JSON_EXPR_TYPES):
92
+ base_col = json_expr.this
93
+ json_path_expr = json_expr.expression
94
+
95
+ # JSON_EXTRACT(JSON_EXTRACT(...)): the inner call owns the column
96
+ while isinstance(base_col, JSON_EXPR_TYPES):
97
+ base_col = base_col.this
98
+
99
+ if isinstance(base_col, exp.Column):
100
+ if under_subquery_where(base_col, source_expr):
101
+ json_columns_handled.add(id(base_col))
102
+ continue
103
+ alias, column, _struct_path, _certain = resolve_column_ref(base_col, known_relations)
104
+
105
+ json_path = _extract_json_path(json_path_expr)
106
+
107
+ # Mark this column as handled via JSON path
108
+ json_columns_handled.add(id(base_col))
109
+
110
+ source_entry: dict[str, Any] = {
111
+ "column": column,
112
+ "inferred": False,
113
+ "json_path": json_path if json_path else None,
114
+ "is_json_access": True,
115
+ }
116
+
117
+ if not alias and column in array_expansion_sources:
118
+ # the expansion alias used bare: elem->>'k' reads the value
119
+ # jsonb_array_elements produced, so the true source is the
120
+ # expanded array column (FAC findings_text, holdout round 4)
121
+ expansion_info = array_expansion_sources[column]
122
+ table = expansion_info.get("source_table", "")
123
+ confidence = 0.5
124
+ trust_level = "moderate"
125
+ source_entry["from_array_expansion"] = True
126
+ source_entry["expansion_type"] = expansion_info.get("expansion_type")
127
+ src_col = expansion_info.get("source_column", column)
128
+ source_entry["source_column_in_array"] = src_col
129
+ if src_col:
130
+ source_entry["expanded_as"] = column
131
+ source_entry["column"] = src_col
132
+ elif alias:
133
+ if alias in array_expansion_sources:
134
+ expansion_info = array_expansion_sources[alias]
135
+ table = expansion_info.get("source_table", alias)
136
+ confidence = 0.5
137
+ trust_level = "moderate"
138
+ source_entry["from_array_expansion"] = True
139
+ source_entry["expansion_type"] = expansion_info.get("expansion_type")
140
+ source_entry["source_column_in_array"] = expansion_info.get(
141
+ "source_column", column
142
+ )
143
+ else:
144
+ table = alias_map.get(alias, alias)
145
+ confidence = 1.0
146
+ trust_level = "verified" if qualified_via_schema else "high_confidence"
147
+ elif default_table:
148
+ table = default_table
149
+ confidence = 0.8
150
+ trust_level = "high_confidence"
151
+ source_entry["inferred"] = True
152
+ else:
153
+ table = ""
154
+ confidence = 0.5
155
+ trust_level = "moderate"
156
+
157
+ source_entry["table"] = table
158
+ source_entry["confidence"] = confidence
159
+ source_entry["trust_level"] = trust_level
160
+
161
+ key = (table, column, json_path)
162
+ if key not in seen:
163
+ seen.add(key)
164
+ sources.append(source_entry)
165
+
166
+ # This pattern is used in Trino, Redshift (SUPER), and some other dialects
167
+ for bracket_expr in source_expr.find_all(exp.Bracket):
168
+ base = bracket_expr.this
169
+ # Skip if already handled via JSON expression
170
+ if any(id(base) in json_columns_handled for base in [bracket_expr.this]):
171
+ continue
172
+
173
+ # Walk up to find the base column
174
+ json_path_parts = []
175
+ while isinstance(base, exp.Bracket):
176
+ if base.expressions:
177
+ key = base.expressions[0]
178
+ if hasattr(key, "this"):
179
+ json_path_parts.insert(0, str(key.this))
180
+ elif hasattr(key, "name"):
181
+ json_path_parts.insert(0, key.name)
182
+ else:
183
+ json_path_parts.insert(0, str(key))
184
+ base = base.this
185
+
186
+ if isinstance(base, exp.Column):
187
+ json_columns_handled.add(id(base))
188
+ if under_subquery_where(base, source_expr):
189
+ continue
190
+ alias, column, _struct_path, _certain = resolve_column_ref(base, known_relations)
191
+ json_path = ".".join(json_path_parts) if json_path_parts else None
192
+
193
+ source_entry: dict[str, Any] = {
194
+ "column": column,
195
+ "inferred": False,
196
+ "json_path": json_path,
197
+ "is_json_access": True,
198
+ }
199
+
200
+ if not alias and column in array_expansion_sources:
201
+ # bare expansion alias (see the JSONExtract branch above)
202
+ expansion_info = array_expansion_sources[column]
203
+ table = expansion_info.get("source_table", "")
204
+ confidence = 0.5
205
+ trust_level = "moderate"
206
+ source_entry["from_array_expansion"] = True
207
+ src_col = expansion_info.get("source_column", column)
208
+ if src_col:
209
+ source_entry["expanded_as"] = column
210
+ source_entry["column"] = src_col
211
+ elif alias:
212
+ if alias in array_expansion_sources:
213
+ expansion_info = array_expansion_sources[alias]
214
+ table = expansion_info.get("source_table", alias)
215
+ confidence = 0.5
216
+ trust_level = "moderate"
217
+ source_entry["from_array_expansion"] = True
218
+ else:
219
+ table = alias_map.get(alias, alias)
220
+ confidence = 1.0
221
+ trust_level = "verified" if qualified_via_schema else "high_confidence"
222
+ elif default_table:
223
+ table = default_table
224
+ confidence = 0.8
225
+ trust_level = "high_confidence"
226
+ source_entry["inferred"] = True
227
+ else:
228
+ table = ""
229
+ confidence = 0.5
230
+ trust_level = "moderate"
231
+
232
+ source_entry["table"] = table
233
+ source_entry["confidence"] = confidence
234
+ source_entry["trust_level"] = trust_level
235
+
236
+ key = (table, column, json_path)
237
+ if key not in seen:
238
+ seen.add(key)
239
+ sources.append(source_entry)
240
+
241
+ return json_columns_handled
@@ -0,0 +1,127 @@
1
+ """Primary-source detection from dbt macro patterns.
2
+
3
+ Regex-only: finds the source() or ref() call that names a model's primary
4
+ data source, before any SQL parsing happens.
5
+ """
6
+
7
+ import logging
8
+
9
+ logger = logging.getLogger(__name__)
10
+
11
+
12
+ def extract_primary_source_from_macro(raw_sql: str) -> dict | None:
13
+ """Extract primary source from staging macro patterns.
14
+
15
+ Handles:
16
+ 1. ANY custom macro wrapping source() or ref() - not just known names
17
+ 2. Direct source()/ref() calls as first statement
18
+ 3. Various whitespace and formatting patterns
19
+ 4. Multi-line macro calls
20
+
21
+ Strategy:
22
+ - First, strip config blocks to find the "main body"
23
+ - Look for source() or ref() as the FIRST reference in the main body
24
+ - This is almost always the primary data source (not WHERE clause refs)
25
+
26
+ Returns:
27
+ {
28
+ "type": "source"|"ref",
29
+ "source": "source_name", # for source type
30
+ "table": "table_name", # for source type
31
+ "ref": "model_name", # for ref type
32
+ "detection_method": "macro_wrapper"|"direct_call"|"first_ref"
33
+ }
34
+ or None if no pattern found
35
+ """
36
+ import re
37
+
38
+ if not raw_sql or not raw_sql.strip():
39
+ return None
40
+
41
+ # Step 1: Strip config block(s) to get to the main SQL body
42
+ config_pattern = re.compile(
43
+ r"\{\{\s*config\s*\([^)]*(?:\([^)]*\)[^)]*)*\)\s*\}\}", re.IGNORECASE | re.DOTALL
44
+ )
45
+ main_body = config_pattern.sub("", raw_sql).strip()
46
+
47
+ # Step 2: Try to find ANY macro that wraps source() or ref()
48
+ # Generic macro wrapping source()
49
+ generic_macro_source = re.compile(
50
+ r"\{\{\s*(\w+)\s*\(\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)",
51
+ re.IGNORECASE | re.DOTALL,
52
+ )
53
+
54
+ match = generic_macro_source.search(main_body)
55
+ if match:
56
+ macro_name = match.group(1).lower()
57
+ skip_macros = {"config", "if", "elif", "else", "endif", "for", "endfor", "set", "do"}
58
+ if macro_name not in skip_macros:
59
+ return {
60
+ "type": "source",
61
+ "source": match.group(2),
62
+ "table": match.group(3),
63
+ "detection_method": "macro_wrapper",
64
+ "macro_name": macro_name,
65
+ }
66
+
67
+ # Generic macro wrapping ref()
68
+ generic_macro_ref = re.compile(
69
+ r"\{\{\s*(\w+)\s*\(\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)", re.IGNORECASE | re.DOTALL
70
+ )
71
+
72
+ match = generic_macro_ref.search(main_body)
73
+ if match:
74
+ macro_name = match.group(1).lower()
75
+ skip_macros = {"config", "if", "elif", "else", "endif", "for", "endfor", "set", "do"}
76
+ if macro_name not in skip_macros:
77
+ return {
78
+ "type": "ref",
79
+ "ref": match.group(2),
80
+ "detection_method": "macro_wrapper",
81
+ "macro_name": macro_name,
82
+ }
83
+
84
+ # Step 3: Look for direct source() or ref() as first Jinja call
85
+ first_source = re.search(
86
+ r"\{\{\s*source\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}",
87
+ main_body,
88
+ re.IGNORECASE,
89
+ )
90
+ first_ref = re.search(
91
+ r"\{\{\s*ref\s*\(\s*['\"]([^'\"]+)['\"]\s*\)\s*\}\}", main_body, re.IGNORECASE
92
+ )
93
+
94
+ where_pos = main_body.upper().find("WHERE")
95
+ if where_pos == -1:
96
+ where_pos = len(main_body)
97
+
98
+ candidates = []
99
+ if first_source and first_source.start() < where_pos:
100
+ candidates.append(
101
+ (
102
+ first_source.start(),
103
+ {
104
+ "type": "source",
105
+ "source": first_source.group(1),
106
+ "table": first_source.group(2),
107
+ "detection_method": "direct_call",
108
+ },
109
+ )
110
+ )
111
+ if first_ref and first_ref.start() < where_pos:
112
+ candidates.append(
113
+ (
114
+ first_ref.start(),
115
+ {
116
+ "type": "ref",
117
+ "ref": first_ref.group(1),
118
+ "detection_method": "direct_call",
119
+ },
120
+ )
121
+ )
122
+
123
+ if candidates:
124
+ candidates.sort(key=lambda x: x[0])
125
+ return candidates[0][1]
126
+
127
+ return None