ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/graph.py
ADDED
|
@@ -0,0 +1,1499 @@
|
|
|
1
|
+
"""Cross-model column lineage graph.
|
|
2
|
+
|
|
3
|
+
Runs the engine over every model in a Project (parents before children) and
|
|
4
|
+
stitches per-statement lineage into one graph that can answer:
|
|
5
|
+
|
|
6
|
+
- breaks(model.column): everything downstream that would be affected
|
|
7
|
+
- trace(model.column): where a column's value comes from, hop by hop
|
|
8
|
+
- to_dict(): the full graph for the canvas, MCP, and CI bot
|
|
9
|
+
|
|
10
|
+
Honesty model, in order of what can go wrong:
|
|
11
|
+
- every edge carries confidence (0..1) and a trust label
|
|
12
|
+
(verified / high_confidence / moderate / review_required)
|
|
13
|
+
- trust is always the WORST of: the engine's label, the label implied by
|
|
14
|
+
confidence, how the source table's name resolved, and the evidence tier
|
|
15
|
+
(compiled manifest vs raw Jinja vs plain files)
|
|
16
|
+
- a model that yields no lineage still appears, connected to its declared
|
|
17
|
+
parents by wildcard edges at review_required, so nothing downstream of a
|
|
18
|
+
failed model is ever hidden
|
|
19
|
+
- an unknown model or column raises with suggestions; it never returns an
|
|
20
|
+
empty result that reads as "safe"
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import difflib
|
|
26
|
+
import logging
|
|
27
|
+
from collections import defaultdict, deque
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
|
|
30
|
+
from ripple.engine.budget import TIMED_OUT, budget_seconds, max_sql_bytes, run_within_budget
|
|
31
|
+
from ripple.engine.dispatch import extract_lineage_complete
|
|
32
|
+
from ripple.engine.jinja import convert_jinja_to_sql
|
|
33
|
+
from ripple.engine.preprocess import prepare_sql_for_parse
|
|
34
|
+
from ripple.loaders.identity import _name_suffixes
|
|
35
|
+
from ripple.loaders.identity import strip_placeholders as _strip_placeholders
|
|
36
|
+
from ripple.project import Model, Project
|
|
37
|
+
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
# worst first; every combinator takes the minimum index
|
|
41
|
+
TRUST_ORDER = ["review_required", "moderate", "high_confidence", "verified"]
|
|
42
|
+
|
|
43
|
+
# evidence mode -> maximum trust an edge can claim
|
|
44
|
+
EVIDENCE_CAP = {
|
|
45
|
+
"manifest": "verified",
|
|
46
|
+
"raw_jinja": "high_confidence",
|
|
47
|
+
"plain_sql": "high_confidence",
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
WILDCARD = "*"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _trust_from_confidence(confidence: float) -> str:
|
|
54
|
+
if confidence >= 1.0:
|
|
55
|
+
return "verified"
|
|
56
|
+
if confidence >= 0.8:
|
|
57
|
+
return "high_confidence"
|
|
58
|
+
if confidence >= 0.5:
|
|
59
|
+
return "moderate"
|
|
60
|
+
return "review_required"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _worst_trust(*labels: str) -> str:
|
|
64
|
+
return min(labels, key=TRUST_ORDER.index)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class UnknownTarget(LookupError):
|
|
68
|
+
def __init__(self, message: str, suggestions: list[str]):
|
|
69
|
+
super().__init__(message)
|
|
70
|
+
self.suggestions = suggestions
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True)
|
|
74
|
+
class Edge:
|
|
75
|
+
src_model: str
|
|
76
|
+
src_column: str
|
|
77
|
+
dst_model: str
|
|
78
|
+
dst_column: str
|
|
79
|
+
confidence: float
|
|
80
|
+
trust: str
|
|
81
|
+
kind: str # "value" | "filter" | "join" | "window"
|
|
82
|
+
reason: str = "" # why trust was downgraded, "" if clean
|
|
83
|
+
|
|
84
|
+
def key(self) -> tuple:
|
|
85
|
+
return (self.src_model, self.src_column, self.dst_model, self.dst_column, self.kind)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class ModelReport:
|
|
90
|
+
name: str
|
|
91
|
+
status: str = "failed" # "ok" | "star_only" | "fallback" | "failed" | "no_sql" | "timed_out"
|
|
92
|
+
columns: list[str] = field(default_factory=list)
|
|
93
|
+
error: str | None = None
|
|
94
|
+
warnings: list[str] = field(default_factory=list)
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def parsed(self) -> bool:
|
|
98
|
+
return self.status in ("ok", "star_only")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class LineageGraph:
|
|
103
|
+
project: Project
|
|
104
|
+
edges: list[Edge] = field(default_factory=list)
|
|
105
|
+
reports: dict[str, ModelReport] = field(default_factory=dict)
|
|
106
|
+
_jinja_env: object = None
|
|
107
|
+
_candidates: dict[str, list[str]] = field(default_factory=dict)
|
|
108
|
+
_all_candidates: dict[str, list[str]] = field(default_factory=dict)
|
|
109
|
+
_stem_aliases: set[str] = field(default_factory=set)
|
|
110
|
+
_models_by_name: dict = None
|
|
111
|
+
_reads_cache: dict = None
|
|
112
|
+
_down: dict = None
|
|
113
|
+
_down_by_model: dict = None
|
|
114
|
+
_up: dict = None
|
|
115
|
+
# (dst_model, src_model) -> lowercased except-list of that star edge
|
|
116
|
+
_star_excepts: dict = field(default_factory=dict)
|
|
117
|
+
# lowered names of function-minted models (TVFs); call sites bind these
|
|
118
|
+
_function_relations: set = field(default_factory=set)
|
|
119
|
+
|
|
120
|
+
# ---- build ----
|
|
121
|
+
|
|
122
|
+
@classmethod
|
|
123
|
+
def build(cls, project: Project) -> LineageGraph:
|
|
124
|
+
graph = cls(project=project)
|
|
125
|
+
graph._candidates = project.name_candidates
|
|
126
|
+
graph._all_candidates = project.all_name_candidates
|
|
127
|
+
graph._stem_aliases = project.stem_alias_names
|
|
128
|
+
real_tables = set(graph._candidates.keys())
|
|
129
|
+
# a call of a minted TVF by its written name binds that model
|
|
130
|
+
# instead of degrading as a function/relation collision (F11)
|
|
131
|
+
graph._function_relations = {
|
|
132
|
+
m.name.lower() for m in project.models if getattr(m, "is_function", False)
|
|
133
|
+
}
|
|
134
|
+
if any(m.evidence != "manifest" for m in project.models):
|
|
135
|
+
from ripple.render import build_project_environment
|
|
136
|
+
|
|
137
|
+
graph._jinja_env = build_project_environment(
|
|
138
|
+
project.macro_sources,
|
|
139
|
+
root_package=project.package_name,
|
|
140
|
+
dialect=project.dialect,
|
|
141
|
+
project_vars=project.jinja_vars,
|
|
142
|
+
)
|
|
143
|
+
for model in graph._topological(project.models):
|
|
144
|
+
graph._add_model(model, real_tables)
|
|
145
|
+
graph._fallback_edges()
|
|
146
|
+
graph._add_downstream()
|
|
147
|
+
graph._dedupe()
|
|
148
|
+
graph._index()
|
|
149
|
+
return graph
|
|
150
|
+
|
|
151
|
+
def _topological(self, models: list[Model]) -> list[Model]:
|
|
152
|
+
"""Parents before children, so a child's SELECT * can expand through
|
|
153
|
+
the columns its parents were just found to have. Cycles and unknown
|
|
154
|
+
parents fall back to input order at the end."""
|
|
155
|
+
by_name = {m.name: m for m in models}
|
|
156
|
+
state: dict[str, int] = {} # 1 = on stack, 2 = done
|
|
157
|
+
order: list[Model] = []
|
|
158
|
+
for root in models:
|
|
159
|
+
stack: list[tuple[str, bool]] = [(root.name, False)]
|
|
160
|
+
while stack:
|
|
161
|
+
name, children_done = stack.pop()
|
|
162
|
+
if state.get(name) == 2:
|
|
163
|
+
continue
|
|
164
|
+
model = by_name.get(name)
|
|
165
|
+
if model is None:
|
|
166
|
+
state[name] = 2
|
|
167
|
+
continue
|
|
168
|
+
if children_done:
|
|
169
|
+
state[name] = 2
|
|
170
|
+
order.append(model)
|
|
171
|
+
continue
|
|
172
|
+
if state.get(name) == 1: # cycle; settle in stack order
|
|
173
|
+
continue
|
|
174
|
+
state[name] = 1
|
|
175
|
+
stack.append((name, True))
|
|
176
|
+
for parent in model.declared_parents:
|
|
177
|
+
# every candidate, not just the first: a name shared by
|
|
178
|
+
# dialect variants must not leave the real one unordered
|
|
179
|
+
for candidate in self._all_candidates.get(parent.lower(), ()):
|
|
180
|
+
if not state.get(candidate):
|
|
181
|
+
stack.append((candidate, False))
|
|
182
|
+
return order
|
|
183
|
+
|
|
184
|
+
def _parent_schema(self, model: Model) -> dict[str, dict[str, str]]:
|
|
185
|
+
"""The project as its own schema: every known column of this model's
|
|
186
|
+
parents, keyed by each name the SQL might use for them."""
|
|
187
|
+
if self._models_by_name is None:
|
|
188
|
+
self._models_by_name = {
|
|
189
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
190
|
+
}
|
|
191
|
+
schema: dict[str, dict[str, str]] = {}
|
|
192
|
+
bare_claims: dict[str, dict[str, dict[str, str]]] = {}
|
|
193
|
+
for parent in model.declared_parents:
|
|
194
|
+
# binding stays enabled-only, but a model's own star expansion may
|
|
195
|
+
# read a disabled parent's columns (dbt disables whole branches)
|
|
196
|
+
resolved = self._candidates.get(parent.lower()) or self._all_candidates.get(
|
|
197
|
+
parent.lower()
|
|
198
|
+
)
|
|
199
|
+
if "." not in parent:
|
|
200
|
+
# the declared parent is bare; the SQL may have written it
|
|
201
|
+
# qualified, and an ingested table answers only to that
|
|
202
|
+
# spelling when another model owns the bare name (Mozilla's
|
|
203
|
+
# mdn_fred.events_stream beside a model named events_stream)
|
|
204
|
+
for read in self._reads_as_written(model) or ():
|
|
205
|
+
if "." in read and read.split(".")[-1] == parent.lower():
|
|
206
|
+
for c in self._candidates.get(read, ()):
|
|
207
|
+
if c not in (resolved or []):
|
|
208
|
+
resolved = [*(resolved or []), c]
|
|
209
|
+
if not resolved:
|
|
210
|
+
continue
|
|
211
|
+
# a bare parent shared by two relations (Mozilla's telemetry
|
|
212
|
+
# view over telemetry_derived, same table name) is settled by
|
|
213
|
+
# the spelling the SQL wrote, exactly as edges are; the old
|
|
214
|
+
# first-candidate pick stays only when nothing disambiguates
|
|
215
|
+
bound = [c for c in resolved if self._parent_binds(model, parent, c)]
|
|
216
|
+
chosen = bound[0] if len(bound) == 1 else resolved[0]
|
|
217
|
+
if not self._parent_binds(model, parent, chosen):
|
|
218
|
+
# refused spelling: supplying the bare relation's columns
|
|
219
|
+
# would fabricate them onto the external (cycle-11, F3)
|
|
220
|
+
continue
|
|
221
|
+
report = self.reports.get(chosen)
|
|
222
|
+
parent_model = self._models_by_name.get(chosen)
|
|
223
|
+
if report is not None and report.status == "ok":
|
|
224
|
+
known = report.columns
|
|
225
|
+
elif report is None and parent_model is not None:
|
|
226
|
+
# a source never gets a report; ingested warehouse schemas
|
|
227
|
+
# give it declared columns, which are as good as analyzed ones
|
|
228
|
+
known = parent_model.declared_columns
|
|
229
|
+
else:
|
|
230
|
+
continue
|
|
231
|
+
if not known:
|
|
232
|
+
continue
|
|
233
|
+
columns = {c: "unknown" for c in known if c != WILDCARD}
|
|
234
|
+
names = {chosen, *(parent_model.aliases if parent_model else set())}
|
|
235
|
+
for name in names:
|
|
236
|
+
schema[name] = columns
|
|
237
|
+
bare_claims.setdefault(chosen.split(".")[-1].lower(), {}).setdefault(chosen, columns)
|
|
238
|
+
# the engine expands a star by the bare table name; a qualified
|
|
239
|
+
# parent whose bare spelling another model owns (Mozilla's
|
|
240
|
+
# mdn_fred.events_stream beside a model named events_stream) still
|
|
241
|
+
# gets it HERE, per model, unless two of this model's own parents
|
|
242
|
+
# share it, which would be a guess
|
|
243
|
+
for bare, owners in bare_claims.items():
|
|
244
|
+
if len(owners) == 1 and bare not in schema:
|
|
245
|
+
schema[bare] = next(iter(owners.values()))
|
|
246
|
+
return schema
|
|
247
|
+
|
|
248
|
+
def _add_model(self, model: Model, real_tables: set[str]) -> None:
|
|
249
|
+
report = ModelReport(name=model.name)
|
|
250
|
+
self.reports[model.name] = report
|
|
251
|
+
if model.load_error:
|
|
252
|
+
report.status = "timed_out"
|
|
253
|
+
report.error = model.load_error
|
|
254
|
+
return
|
|
255
|
+
if not model.sql.strip():
|
|
256
|
+
if model.declared_columns:
|
|
257
|
+
# a base table from CREATE TABLE: no derivation, but its columns
|
|
258
|
+
# are known, so downstream SELECT * resolves against them
|
|
259
|
+
report.columns = list(model.declared_columns)
|
|
260
|
+
report.status = "ok"
|
|
261
|
+
else:
|
|
262
|
+
report.status = "no_sql"
|
|
263
|
+
report.error = "no SQL"
|
|
264
|
+
return
|
|
265
|
+
self._analyze_sql(model, model.sql, real_tables, report, primary=True)
|
|
266
|
+
# a table built by CREATE plus UPDATEs is one model; each further
|
|
267
|
+
# statement adds its lineage to the union (nycdb, PR #28 gap 3)
|
|
268
|
+
for extra in model.extra_sqls:
|
|
269
|
+
self._analyze_sql(model, extra, real_tables, report, primary=False)
|
|
270
|
+
|
|
271
|
+
def _analyze_sql(
|
|
272
|
+
self,
|
|
273
|
+
model: Model,
|
|
274
|
+
sql: str,
|
|
275
|
+
real_tables: set[str],
|
|
276
|
+
report: ModelReport,
|
|
277
|
+
primary: bool,
|
|
278
|
+
) -> None:
|
|
279
|
+
edge_start = len(self.edges)
|
|
280
|
+
limit = max_sql_bytes()
|
|
281
|
+
if limit and len(sql) > limit:
|
|
282
|
+
message = (
|
|
283
|
+
f"statement is {len(sql) / 1e6:.1f}MB, over the {limit / 1e6:g}MB analysis "
|
|
284
|
+
"limit; generated SQL this size wedges the parser for minutes "
|
|
285
|
+
"(RIPPLE_MAX_SQL_BYTES to raise, 0 to disable)"
|
|
286
|
+
)
|
|
287
|
+
if primary:
|
|
288
|
+
report.status = "timed_out"
|
|
289
|
+
report.error = f"{message}: {model.path}"
|
|
290
|
+
else:
|
|
291
|
+
report.warnings.append(message)
|
|
292
|
+
return
|
|
293
|
+
# computed before the worker: it caches onto self, and the worker must
|
|
294
|
+
# stay pure so an abandoned one cannot race the build
|
|
295
|
+
warehouse_columns = self._parent_schema(model) if model.declared_parents else None
|
|
296
|
+
dialect = model.dialect or self.project.dialect
|
|
297
|
+
|
|
298
|
+
def analyze():
|
|
299
|
+
body = sql
|
|
300
|
+
jinja_cleaner = None
|
|
301
|
+
if model.evidence != "manifest" and ("{{" in body or "{%" in body):
|
|
302
|
+
try: # tier b: real Jinja render beats regex stripping
|
|
303
|
+
from ripple.render import render_dbt_sql
|
|
304
|
+
|
|
305
|
+
declared = (
|
|
306
|
+
model.jinja_vars
|
|
307
|
+
if model.jinja_vars is not None
|
|
308
|
+
else self.project.jinja_vars
|
|
309
|
+
)
|
|
310
|
+
body = render_dbt_sql(
|
|
311
|
+
body, model.name, env=self._jinja_env, jinja_vars=declared
|
|
312
|
+
)
|
|
313
|
+
# rendered SQL has no Jinja left; re-cleaning it corrupts
|
|
314
|
+
# regex literals and $-quoted strings in the output
|
|
315
|
+
except Exception:
|
|
316
|
+
jinja_cleaner = convert_jinja_to_sql # tier c: regex cleaner on the raw SQL
|
|
317
|
+
extracted = extract_lineage_complete(
|
|
318
|
+
body,
|
|
319
|
+
dialect=dialect,
|
|
320
|
+
real_tables=real_tables,
|
|
321
|
+
warehouse_columns=warehouse_columns or None,
|
|
322
|
+
clean_jinja_func=jinja_cleaner,
|
|
323
|
+
function_relations=self._function_relations or None,
|
|
324
|
+
# a CTE named after this model shadows only itself; without
|
|
325
|
+
# this, dbt's `with X as (...) select * from X` house style
|
|
326
|
+
# marks every correct pass-through edge review_required
|
|
327
|
+
self_names={model.name, *model.aliases, *model.stem_aliases},
|
|
328
|
+
)
|
|
329
|
+
# the same pre-parse pipeline the engine used, or these reparses
|
|
330
|
+
# drift and drop edges (cycle-10 review, F8)
|
|
331
|
+
prepared, _ = prepare_sql_for_parse(body, dialect, jinja_cleaner)
|
|
332
|
+
return extracted, prepared, self._alias_map(prepared, dialect=dialect)
|
|
333
|
+
|
|
334
|
+
# the loader's budget caps parse time, but a file can parse fast and
|
|
335
|
+
# wedge HERE: a generated 3.7MB one-liner parsed in 13s, then spent
|
|
336
|
+
# 139s inside qualify_tables. Same budget, same honest report.
|
|
337
|
+
try:
|
|
338
|
+
outcome = run_within_budget(model.name, analyze)
|
|
339
|
+
except Exception as e: # a model that won't parse is a report, not a crash
|
|
340
|
+
if primary:
|
|
341
|
+
report.error = f"{type(e).__name__}: {e}"
|
|
342
|
+
else:
|
|
343
|
+
report.warnings.append(f"additional statement failed: {type(e).__name__}: {e}")
|
|
344
|
+
return
|
|
345
|
+
if outcome is TIMED_OUT:
|
|
346
|
+
message = (
|
|
347
|
+
f"analysis exceeded the {budget_seconds():g}s per-file budget: "
|
|
348
|
+
f"{model.path} (RIPPLE_PARSE_BUDGET_S to raise)"
|
|
349
|
+
)
|
|
350
|
+
if primary:
|
|
351
|
+
report.status = "timed_out"
|
|
352
|
+
report.error = message
|
|
353
|
+
else:
|
|
354
|
+
report.warnings.append(message)
|
|
355
|
+
return
|
|
356
|
+
result, prepared_sql, alias_map = outcome
|
|
357
|
+
|
|
358
|
+
columns = [c.lower() for c in result.contributing]
|
|
359
|
+
if primary:
|
|
360
|
+
report.columns = columns
|
|
361
|
+
else:
|
|
362
|
+
report.columns.extend(c for c in columns if c not in report.columns)
|
|
363
|
+
real_columns = [c for c in columns if c != WILDCARD and c != "null"]
|
|
364
|
+
if real_columns:
|
|
365
|
+
if not primary and report.status == "star_only":
|
|
366
|
+
# the state must read like any partially-starred model:
|
|
367
|
+
# status ok, no stale error (the cycle-6 review)
|
|
368
|
+
report.error = None
|
|
369
|
+
report.status = "ok"
|
|
370
|
+
elif primary:
|
|
371
|
+
if columns:
|
|
372
|
+
report.status = "star_only"
|
|
373
|
+
report.error = "only SELECT * survived; column-level lineage unavailable"
|
|
374
|
+
else:
|
|
375
|
+
report.error = "no columns extracted"
|
|
376
|
+
return
|
|
377
|
+
new_warnings = [str(w.get("warning", w)) for w in result.warnings]
|
|
378
|
+
if primary:
|
|
379
|
+
report.warnings = new_warnings
|
|
380
|
+
else:
|
|
381
|
+
report.warnings.extend(new_warnings)
|
|
382
|
+
|
|
383
|
+
# project-inferred schema is good evidence, not warehouse-verified truth
|
|
384
|
+
cap = EVIDENCE_CAP.get(model.evidence, "high_confidence")
|
|
385
|
+
if result.qualified_via_schema:
|
|
386
|
+
cap = _worst_trust(cap, "high_confidence")
|
|
387
|
+
|
|
388
|
+
for out_column, upstream_columns in result.contributing.items():
|
|
389
|
+
for src in upstream_columns:
|
|
390
|
+
before = len(self.edges)
|
|
391
|
+
self._append_edge(
|
|
392
|
+
model,
|
|
393
|
+
out_column.lower(),
|
|
394
|
+
src.get("table"),
|
|
395
|
+
src.get("column"),
|
|
396
|
+
src.get("confidence", 0.5),
|
|
397
|
+
src.get("trust_level"),
|
|
398
|
+
"value",
|
|
399
|
+
cap,
|
|
400
|
+
)
|
|
401
|
+
excepted = src.get("except_columns")
|
|
402
|
+
if excepted and out_column == WILDCARD:
|
|
403
|
+
# spellings kept as normalized upstream (quoted
|
|
404
|
+
# mixed-case stays exact, F20); same-pair excepts merge
|
|
405
|
+
# by union so a column excluded in any union branch
|
|
406
|
+
# refuses (F7)
|
|
407
|
+
for e in self.edges[before:]:
|
|
408
|
+
if e.src_column == WILDCARD and e.dst_column == WILDCARD:
|
|
409
|
+
pair = (e.dst_model, e.src_model)
|
|
410
|
+
self._star_excepts[pair] = self._star_excepts.get(
|
|
411
|
+
pair, frozenset()
|
|
412
|
+
) | frozenset(excepted)
|
|
413
|
+
|
|
414
|
+
# An unqualified filter/join column ("where status = ...") can still be
|
|
415
|
+
# attributed when THIS statement reads from exactly one upstream
|
|
416
|
+
# relation, and a query-local alias ("FROM stg_orders o ... WHERE
|
|
417
|
+
# o.status") resolves through the statement's own alias map. Only this
|
|
418
|
+
# statement's edges count: borrowing an earlier statement's parents
|
|
419
|
+
# claimed the CTAS source wrote the UPDATE's filter column (the
|
|
420
|
+
# review of cycle 6).
|
|
421
|
+
value_parents = {
|
|
422
|
+
e.src_model
|
|
423
|
+
for e in self.edges[edge_start:]
|
|
424
|
+
if e.dst_model == model.name and e.kind == "value"
|
|
425
|
+
}
|
|
426
|
+
sole_parent = next(iter(value_parents)) if len(value_parents) == 1 else None
|
|
427
|
+
if sole_parent is None and not value_parents and not primary:
|
|
428
|
+
relations = self._statement_relations(prepared_sql, model.dialect)
|
|
429
|
+
resolved = {
|
|
430
|
+
found[0]
|
|
431
|
+
for name in relations
|
|
432
|
+
if (found := self._candidates.get(name)) and len(found) == 1
|
|
433
|
+
}
|
|
434
|
+
if len(resolved) == 1:
|
|
435
|
+
sole_parent = next(iter(resolved))
|
|
436
|
+
|
|
437
|
+
def resolve_relation(table: str | None) -> str | None:
|
|
438
|
+
if not table:
|
|
439
|
+
return sole_parent
|
|
440
|
+
return alias_map.get(table.lower(), table)
|
|
441
|
+
|
|
442
|
+
for filt in result.filter_columns:
|
|
443
|
+
table = resolve_relation(filt.table)
|
|
444
|
+
confidence = 0.8 if filt.table else 0.6
|
|
445
|
+
self._append_edge(
|
|
446
|
+
model, "(row filter)", table, filt.column, confidence, None, "filter", cap
|
|
447
|
+
)
|
|
448
|
+
for key in result.window_keys:
|
|
449
|
+
table = resolve_relation(key.table)
|
|
450
|
+
confidence = 0.8 if key.table else 0.6
|
|
451
|
+
self._append_edge(
|
|
452
|
+
model, "(window key)", table, key.column, confidence, None, "window", cap
|
|
453
|
+
)
|
|
454
|
+
for join in result.join_keys:
|
|
455
|
+
for table, column in (
|
|
456
|
+
(join.left_table, join.left_column),
|
|
457
|
+
(join.right_table, join.right_column),
|
|
458
|
+
):
|
|
459
|
+
self._append_edge(
|
|
460
|
+
model,
|
|
461
|
+
"(join key)",
|
|
462
|
+
resolve_relation(table),
|
|
463
|
+
column,
|
|
464
|
+
0.8 if table else 0.6,
|
|
465
|
+
None,
|
|
466
|
+
"join",
|
|
467
|
+
cap,
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
# A star chain from an unexpanded upstream (source with no declared
|
|
471
|
+
# columns) still proves the columns this model itself references in
|
|
472
|
+
# predicates or join keys: they must exist upstream and SELECT * passes
|
|
473
|
+
# them through, so they get real value edges instead of vanishing.
|
|
474
|
+
star_parents = {
|
|
475
|
+
e.src_model
|
|
476
|
+
for e in self.edges
|
|
477
|
+
if e.dst_model == model.name
|
|
478
|
+
and e.kind == "value"
|
|
479
|
+
and e.src_column == WILDCARD
|
|
480
|
+
and e.dst_column == WILDCARD
|
|
481
|
+
}
|
|
482
|
+
if star_parents:
|
|
483
|
+
proven = {(f.table, f.column) for f in result.filter_columns}
|
|
484
|
+
for join in result.join_keys:
|
|
485
|
+
proven.add((join.left_table, join.left_column))
|
|
486
|
+
proven.add((join.right_table, join.right_column))
|
|
487
|
+
for table, column in sorted(
|
|
488
|
+
(p for p in proven if p[1]), key=lambda p: (p[0] or "", p[1])
|
|
489
|
+
):
|
|
490
|
+
if column.lower() in report.columns:
|
|
491
|
+
continue # already attributed explicitly
|
|
492
|
+
candidates, _ = self._resolve_name(
|
|
493
|
+
resolve_relation(table), model.declared_parents, model.path
|
|
494
|
+
)
|
|
495
|
+
for src_model in candidates:
|
|
496
|
+
if src_model not in star_parents:
|
|
497
|
+
continue
|
|
498
|
+
self.edges.append(
|
|
499
|
+
Edge(
|
|
500
|
+
src_model=src_model,
|
|
501
|
+
src_column=column.lower(),
|
|
502
|
+
dst_model=model.name,
|
|
503
|
+
dst_column=column.lower(),
|
|
504
|
+
confidence=0.6,
|
|
505
|
+
trust=_worst_trust("moderate", cap),
|
|
506
|
+
kind="value",
|
|
507
|
+
reason="column proven by this model's own reference; "
|
|
508
|
+
"passed through SELECT *",
|
|
509
|
+
)
|
|
510
|
+
)
|
|
511
|
+
if column.lower() not in report.columns:
|
|
512
|
+
report.columns.append(column.lower())
|
|
513
|
+
|
|
514
|
+
def _append_edge(
|
|
515
|
+
self,
|
|
516
|
+
model: Model,
|
|
517
|
+
out_column: str,
|
|
518
|
+
src_table: str | None,
|
|
519
|
+
src_column: str | None,
|
|
520
|
+
confidence: float,
|
|
521
|
+
engine_trust: str | None,
|
|
522
|
+
kind: str,
|
|
523
|
+
cap: str,
|
|
524
|
+
) -> None:
|
|
525
|
+
if not src_column:
|
|
526
|
+
return
|
|
527
|
+
candidates, quality = self._resolve_name(src_table, model.declared_parents, model.path)
|
|
528
|
+
if candidates == [model.name] and quality != "exact":
|
|
529
|
+
# a model cannot be its own upstream: when a QUALIFIED reference's
|
|
530
|
+
# suffix is the model's own name (pg_attribute.sql reading
|
|
531
|
+
# pg_catalog.pg_attribute, the staging-wrapper pattern), the SQL
|
|
532
|
+
# named an external relation, and dropping it as a self-loop
|
|
533
|
+
# erases the model's entire lineage
|
|
534
|
+
wrapped = (src_table or "").replace('"', "").replace("`", "").lower()
|
|
535
|
+
if "." in wrapped:
|
|
536
|
+
candidates, quality = [wrapped], "unknown"
|
|
537
|
+
confidence = min(confidence, 0.4)
|
|
538
|
+
reason = ""
|
|
539
|
+
if not candidates:
|
|
540
|
+
if not src_table:
|
|
541
|
+
return
|
|
542
|
+
cleaned = src_table.replace('"', "").replace("`", "").lower()
|
|
543
|
+
if kind in ("filter", "join", "window") and "." not in cleaned:
|
|
544
|
+
# a query-local alias (FROM orders o) we couldn't resolve to a
|
|
545
|
+
# model; inventing a phantom node named "o" would be a lie
|
|
546
|
+
report = self.reports.get(model.name)
|
|
547
|
+
if report is not None:
|
|
548
|
+
report.warnings.append(
|
|
549
|
+
f"could not attribute a {kind} on '{cleaned}.{src_column}' to a model"
|
|
550
|
+
)
|
|
551
|
+
return
|
|
552
|
+
# traced to a table this project doesn't own (external table, CTE
|
|
553
|
+
# remnant). Keep it, but never at confident trust.
|
|
554
|
+
candidates = [cleaned]
|
|
555
|
+
confidence = min(confidence, 0.4)
|
|
556
|
+
disabled = self._all_candidates.get(cleaned) or self._all_candidates.get(
|
|
557
|
+
cleaned.split(".")[-1]
|
|
558
|
+
)
|
|
559
|
+
if disabled:
|
|
560
|
+
# snowplow ships one copy of a model per warehouse and dbt
|
|
561
|
+
# disables all but the target's; a read of the bare name is
|
|
562
|
+
# not an external table and no ingest resolves it
|
|
563
|
+
reason = (
|
|
564
|
+
f"names only disabled models ({len(disabled)}); "
|
|
565
|
+
"enable one or set the dbt target"
|
|
566
|
+
)
|
|
567
|
+
else:
|
|
568
|
+
reason = "source table not found in this project"
|
|
569
|
+
quality = "unknown"
|
|
570
|
+
elif quality == "ambiguous":
|
|
571
|
+
reason = f"name matches {len(candidates)} models; edge added for each"
|
|
572
|
+
elif quality == "suffix":
|
|
573
|
+
reason = "matched by table name only; schema differs or is missing"
|
|
574
|
+
elif quality == "unknown":
|
|
575
|
+
reason = "source table not found in this project"
|
|
576
|
+
trust = _worst_trust(
|
|
577
|
+
engine_trust or "verified",
|
|
578
|
+
_trust_from_confidence(confidence),
|
|
579
|
+
cap,
|
|
580
|
+
"review_required" if quality in ("ambiguous", "suffix", "unknown") else "verified",
|
|
581
|
+
)
|
|
582
|
+
if trust == "review_required" and not reason and len(candidates) == 1:
|
|
583
|
+
# the engine downgraded a read of a column the declared schema
|
|
584
|
+
# lacks, silently; on balboa an ingest of truncated column lists
|
|
585
|
+
# took review links from 17 to 106 with no reason on any of them
|
|
586
|
+
if self._models_by_name is None:
|
|
587
|
+
self._models_by_name = {
|
|
588
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
589
|
+
}
|
|
590
|
+
source = self._models_by_name.get(candidates[0])
|
|
591
|
+
declared = {c.lower() for c in source.declared_columns} if source else set()
|
|
592
|
+
if declared and src_column.lower() not in declared and src_column != WILDCARD:
|
|
593
|
+
reason = (
|
|
594
|
+
f"'{src_column}' is not in the declared columns of {candidates[0]}; "
|
|
595
|
+
"the declared or ingested column list may be incomplete"
|
|
596
|
+
)
|
|
597
|
+
for src_model in candidates:
|
|
598
|
+
if src_model == model.name and not model.self_read:
|
|
599
|
+
continue # self-loop from CTE resolution
|
|
600
|
+
if (
|
|
601
|
+
src_model == model.name
|
|
602
|
+
and kind == "value"
|
|
603
|
+
and src_column.lower() == out_column.lower()
|
|
604
|
+
):
|
|
605
|
+
# 'names depends on its own prior names' is a tautology, not
|
|
606
|
+
# lineage; only CROSS-column prior-state edges (address <-
|
|
607
|
+
# house) carry information (nycdb business_addrs)
|
|
608
|
+
continue
|
|
609
|
+
if src_model == model.name and not reason:
|
|
610
|
+
reason = "prior state of the same table"
|
|
611
|
+
self.edges.append(
|
|
612
|
+
Edge(
|
|
613
|
+
src_model=src_model,
|
|
614
|
+
src_column=src_column.lower(),
|
|
615
|
+
dst_model=model.name,
|
|
616
|
+
dst_column=out_column if out_column.startswith("(") else out_column.lower(),
|
|
617
|
+
confidence=round(confidence, 2),
|
|
618
|
+
trust=trust,
|
|
619
|
+
kind=kind,
|
|
620
|
+
reason=reason,
|
|
621
|
+
)
|
|
622
|
+
)
|
|
623
|
+
|
|
624
|
+
def _add_downstream(self) -> None:
|
|
625
|
+
"""Append the declared-as-code BI edges (dbt metrics, exposures) so a
|
|
626
|
+
model column traces to the metric and dashboard it feeds. These are
|
|
627
|
+
ordinary edges with a trust label, so breaks() reaches them for free."""
|
|
628
|
+
conf = {"verified": 1.0, "high_confidence": 0.85, "moderate": 0.6, "review_required": 0.4}
|
|
629
|
+
for d in getattr(self.project, "downstream", []):
|
|
630
|
+
cands = self._candidates.get(d.src_model.lower())
|
|
631
|
+
src = cands[0] if cands else d.src_model
|
|
632
|
+
self.edges.append(
|
|
633
|
+
Edge(
|
|
634
|
+
src_model=src,
|
|
635
|
+
src_column=WILDCARD if d.src_column == "*" else d.src_column,
|
|
636
|
+
dst_model=d.dst_model,
|
|
637
|
+
dst_column=WILDCARD if d.dst_column == "*" else d.dst_column,
|
|
638
|
+
confidence=conf.get(d.trust, 0.6),
|
|
639
|
+
trust=d.trust,
|
|
640
|
+
kind="value",
|
|
641
|
+
reason=d.reason,
|
|
642
|
+
)
|
|
643
|
+
)
|
|
644
|
+
|
|
645
|
+
def _statement_relations(self, sql: str, dialect: str | None = None) -> set[str]:
|
|
646
|
+
"""Lowercased real relations one statement reads (tables minus its
|
|
647
|
+
own CTE names). The anchor for an extra statement's unqualified
|
|
648
|
+
filter columns when it produced no value edges of its own."""
|
|
649
|
+
import sqlglot
|
|
650
|
+
from sqlglot import exp
|
|
651
|
+
|
|
652
|
+
from ripple.engine.column_ref import temp_marked_name
|
|
653
|
+
|
|
654
|
+
try:
|
|
655
|
+
parsed = sqlglot.parse_one(sql, dialect=dialect or self.project.dialect)
|
|
656
|
+
except Exception:
|
|
657
|
+
return set()
|
|
658
|
+
ctes = {c.alias_or_name.lower() for c in parsed.find_all(exp.CTE)}
|
|
659
|
+
return {
|
|
660
|
+
temp_marked_name(t).lower()
|
|
661
|
+
for t in parsed.find_all(exp.Table)
|
|
662
|
+
if t.name and t.name.lower() not in ctes
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
def _alias_map(self, sql: str, dialect: str | None = None) -> dict[str, str]:
|
|
666
|
+
"""Query-local alias -> relation name, resolving single-source CTEs
|
|
667
|
+
through to their base table. 'FROM stg_orders o JOIN pay p' where pay
|
|
668
|
+
is a CTE reading stg_payments yields {o: stg_orders, p: stg_payments}.
|
|
669
|
+
"""
|
|
670
|
+
import sqlglot
|
|
671
|
+
from sqlglot import exp
|
|
672
|
+
|
|
673
|
+
try:
|
|
674
|
+
parsed = sqlglot.parse_one(sql, dialect=dialect or self.project.dialect)
|
|
675
|
+
except Exception:
|
|
676
|
+
return {}
|
|
677
|
+
|
|
678
|
+
from ripple.engine.column_ref import qualified_table_name
|
|
679
|
+
from ripple.engine.preprocess import restore_jinja_dots
|
|
680
|
+
|
|
681
|
+
def qualified(t: exp.Table) -> str:
|
|
682
|
+
# qualifiers kept for the same reason as scope.register_table: a
|
|
683
|
+
# filter/join/window column on a model named after the table it
|
|
684
|
+
# wraps must not resolve back into the model and vanish;
|
|
685
|
+
# placeholder qualifiers dropped for the same reason as
|
|
686
|
+
# column_ref.qualified_table_name
|
|
687
|
+
return qualified_table_name(t).lower()
|
|
688
|
+
|
|
689
|
+
cte_sources: dict[str, set[tuple[str, str]]] = {}
|
|
690
|
+
cte_names = set()
|
|
691
|
+
for cte in parsed.find_all(exp.CTE):
|
|
692
|
+
name = restore_jinja_dots(cte.alias_or_name.lower())
|
|
693
|
+
cte_names.add(name)
|
|
694
|
+
cte_sources[name] = {
|
|
695
|
+
(restore_jinja_dots(t.name.lower()), qualified(t))
|
|
696
|
+
for t in cte.this.find_all(exp.Table)
|
|
697
|
+
}
|
|
698
|
+
aliases: dict[str, str] = {}
|
|
699
|
+
for table in parsed.find_all(exp.Table):
|
|
700
|
+
alias = restore_jinja_dots((table.alias_or_name or "").lower())
|
|
701
|
+
name = restore_jinja_dots(table.name.lower())
|
|
702
|
+
if not alias or alias == name:
|
|
703
|
+
full = qualified(table)
|
|
704
|
+
if full != name and name not in cte_names:
|
|
705
|
+
aliases.setdefault(name, full)
|
|
706
|
+
continue
|
|
707
|
+
target = qualified(table)
|
|
708
|
+
# a CTE alias resolves through the CTE when it reads one real table
|
|
709
|
+
if name in cte_names:
|
|
710
|
+
real = {q for (bare, q) in cte_sources.get(name, set()) if bare not in cte_names}
|
|
711
|
+
if len(real) == 1:
|
|
712
|
+
target = next(iter(real))
|
|
713
|
+
else:
|
|
714
|
+
continue # multi-source CTE: leave unresolved, warn path handles it
|
|
715
|
+
aliases[alias] = target
|
|
716
|
+
return aliases
|
|
717
|
+
|
|
718
|
+
def _resolve_name(
|
|
719
|
+
self, table: str | None, scope_parents: set[str], ref_path: str = ""
|
|
720
|
+
) -> tuple[list[str], str]:
|
|
721
|
+
"""Resolve a table name from SQL to project models.
|
|
722
|
+
|
|
723
|
+
Returns (candidates, quality) where quality is one of exact / scoped /
|
|
724
|
+
ambiguous / suffix / unknown. Ambiguity returns ALL candidates; the
|
|
725
|
+
caller emits an edge per candidate at review_required rather than
|
|
726
|
+
silently binding one.
|
|
727
|
+
"""
|
|
728
|
+
if not table:
|
|
729
|
+
return [], "unknown"
|
|
730
|
+
cleaned = table.replace('"', "").replace("`", "").lower()
|
|
731
|
+
scope = {
|
|
732
|
+
resolved[0] for p in scope_parents if (resolved := self._candidates.get(p.lower()))
|
|
733
|
+
}
|
|
734
|
+
candidates = self._candidates.get(cleaned)
|
|
735
|
+
if not candidates and "{" in cleaned:
|
|
736
|
+
# an ingest names the relation without its deploy-time qualifier;
|
|
737
|
+
# a templated NAME comes back unchanged and stays unresolved
|
|
738
|
+
stripped = _strip_placeholders(cleaned)
|
|
739
|
+
if stripped and "{" not in stripped:
|
|
740
|
+
if self._models_by_name is None:
|
|
741
|
+
self._models_by_name = {
|
|
742
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
743
|
+
}
|
|
744
|
+
# only a declared or ingested table answers to the bare
|
|
745
|
+
# spelling; a model with SQL of the same name is a different
|
|
746
|
+
# relation ({{params.dataset_name_raw}}.blocks is not the
|
|
747
|
+
# derived blocks model, round 11)
|
|
748
|
+
candidates = [
|
|
749
|
+
c
|
|
750
|
+
for c in self._candidates.get(stripped, [])
|
|
751
|
+
if not (
|
|
752
|
+
self._models_by_name.get(c) or Model(name="", sql="", path="")
|
|
753
|
+
).sql.strip()
|
|
754
|
+
] or None
|
|
755
|
+
if candidates:
|
|
756
|
+
if len(candidates) == 1:
|
|
757
|
+
return [candidates[0]], "exact"
|
|
758
|
+
# locality before the scope shortcut: sql-dir declared parents are
|
|
759
|
+
# the bare names themselves, so scope would collapse to the first
|
|
760
|
+
# candidate and bind another dump's table (no-op outside sql-dir)
|
|
761
|
+
local = self._prefer_local(candidates, ref_path)
|
|
762
|
+
if len(local) == 1:
|
|
763
|
+
return local, "scoped"
|
|
764
|
+
in_scope = [c for c in local if c in scope]
|
|
765
|
+
if len(in_scope) == 1:
|
|
766
|
+
return in_scope, "scoped"
|
|
767
|
+
return (in_scope or local), "ambiguous"
|
|
768
|
+
last = cleaned.split(".")[-1]
|
|
769
|
+
if last != cleaned:
|
|
770
|
+
candidates = self._candidates.get(last)
|
|
771
|
+
if candidates:
|
|
772
|
+
candidates = [c for c in candidates if self._may_fold_qualified_read(c, cleaned)]
|
|
773
|
+
if candidates:
|
|
774
|
+
local = self._prefer_local(candidates, ref_path)
|
|
775
|
+
if len(local) == 1:
|
|
776
|
+
return local, "scoped"
|
|
777
|
+
in_scope = [c for c in local if c in scope]
|
|
778
|
+
if len(in_scope) == 1:
|
|
779
|
+
return in_scope, "scoped"
|
|
780
|
+
return (in_scope or local), "suffix"
|
|
781
|
+
return [], "unknown"
|
|
782
|
+
|
|
783
|
+
def _may_fold_qualified_read(self, name: str, read: str) -> bool:
|
|
784
|
+
"""In plain SQL the file is the spelling authority: a relation created
|
|
785
|
+
bare never owns a schema-qualified read, and qualified spellings must
|
|
786
|
+
agree (round-5 refuse-to-fold, cycle-6 agreement rule). dbt schema
|
|
787
|
+
qualifiers are deploy artifacts, so dbt-evidence models keep folding.
|
|
788
|
+
|
|
789
|
+
A dotted filename stem counts as an agreeing qualified spelling: the
|
|
790
|
+
author named eicu_crd.patient.sql with the qualifier, so a read of
|
|
791
|
+
eicu_crd.patient folds to its relation deliberately (cycle-11 review,
|
|
792
|
+
F2: pinned as the intended design, not a bypass)."""
|
|
793
|
+
if self._models_by_name is None:
|
|
794
|
+
self._models_by_name = {
|
|
795
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
796
|
+
}
|
|
797
|
+
model = self._models_by_name.get(name)
|
|
798
|
+
if model is None or model.evidence != "plain_sql":
|
|
799
|
+
return True
|
|
800
|
+
spellings = {
|
|
801
|
+
s.lower() for s in (model.name, *model.aliases, *model.stem_aliases) if "." in s
|
|
802
|
+
}
|
|
803
|
+
read_suffixes = _name_suffixes(read)
|
|
804
|
+
return any(s in read_suffixes or read in _name_suffixes(s) for s in spellings)
|
|
805
|
+
|
|
806
|
+
def _reads_as_written(self, model: Model) -> frozenset[str] | None:
|
|
807
|
+
"""Lowercased as-written relation spellings in a model's SQL, or None
|
|
808
|
+
when nothing parsed (no evidence, so no refusal downstream)."""
|
|
809
|
+
if self._reads_cache is None:
|
|
810
|
+
self._reads_cache = {}
|
|
811
|
+
cached = self._reads_cache.get(model.uid, "unset")
|
|
812
|
+
if cached != "unset":
|
|
813
|
+
return cached
|
|
814
|
+
import sqlglot
|
|
815
|
+
from sqlglot import exp
|
|
816
|
+
|
|
817
|
+
from ripple.engine.column_ref import qualified_table_name
|
|
818
|
+
|
|
819
|
+
reads: set[str] = set()
|
|
820
|
+
parsed_any = False
|
|
821
|
+
for sql in (model.sql, *model.extra_sqls):
|
|
822
|
+
if not sql.strip():
|
|
823
|
+
continue
|
|
824
|
+
try:
|
|
825
|
+
parsed = sqlglot.parse_one(sql, read=model.dialect or self.project.dialect)
|
|
826
|
+
except Exception:
|
|
827
|
+
continue
|
|
828
|
+
parsed_any = True
|
|
829
|
+
for t in parsed.find_all(exp.Table):
|
|
830
|
+
if t.name:
|
|
831
|
+
reads.add((qualified_table_name(t) or t.name).lower())
|
|
832
|
+
result = frozenset(reads) if parsed_any else None
|
|
833
|
+
self._reads_cache[model.uid] = result
|
|
834
|
+
return result
|
|
835
|
+
|
|
836
|
+
def _parent_binds(self, model: Model, parent: str, resolved: str) -> bool:
|
|
837
|
+
"""The spelling-authority rule applied to a DECLARED parent link.
|
|
838
|
+
|
|
839
|
+
Declared parents are recorded bare, so the sites that resolve them
|
|
840
|
+
directly (parent schemas, fallback edges, unresolved reporting) were
|
|
841
|
+
bypassing _may_fold_qualified_read: a SELECT * FROM eicu_crd.patient
|
|
842
|
+
inherited the bare-created patient's columns and fallback edges
|
|
843
|
+
reconnected it (cycle-11 review, F3/F4). The read spellings in the
|
|
844
|
+
model's own SQL are the evidence; the parent binds only when some
|
|
845
|
+
read of it may fold to the resolved relation."""
|
|
846
|
+
if self._models_by_name is None:
|
|
847
|
+
self._models_by_name = {
|
|
848
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
849
|
+
}
|
|
850
|
+
target = self._models_by_name.get(resolved)
|
|
851
|
+
if target is None or target.evidence != "plain_sql":
|
|
852
|
+
return True
|
|
853
|
+
reads = self._reads_as_written(model)
|
|
854
|
+
if reads is None:
|
|
855
|
+
return True
|
|
856
|
+
names = {s.lower() for s in (target.name, *target.aliases, *target.stem_aliases)}
|
|
857
|
+
bare = parent.lower().split(".")[-1]
|
|
858
|
+
relevant = [r for r in reads if r == parent.lower() or r.split(".")[-1] == bare]
|
|
859
|
+
if not relevant:
|
|
860
|
+
return True
|
|
861
|
+
return any(r in names or self._may_fold_qualified_read(resolved, r) for r in relevant)
|
|
862
|
+
|
|
863
|
+
def _refused_read_of(self, model: Model, parent: str) -> str | None:
|
|
864
|
+
"""The as-written qualified spelling behind a refused parent link, so
|
|
865
|
+
the external is reported as the SQL cites it, not as the bare name."""
|
|
866
|
+
reads = self._reads_as_written(model) or frozenset()
|
|
867
|
+
bare = parent.lower().split(".")[-1]
|
|
868
|
+
dotted = sorted(r for r in reads if "." in r and r.split(".")[-1] == bare)
|
|
869
|
+
return dotted[0] if dotted else None
|
|
870
|
+
|
|
871
|
+
def _prefer_local(self, candidates: list[str], ref_path: str) -> list[str]:
|
|
872
|
+
"""Same file first, then same subdirectory: in a sql dir one dump is
|
|
873
|
+
one database, so a reference never silently binds another dump's file."""
|
|
874
|
+
if len(candidates) < 2 or not ref_path or self.project.mode != "sql-dir":
|
|
875
|
+
return candidates
|
|
876
|
+
if self._models_by_name is None:
|
|
877
|
+
self._models_by_name = {
|
|
878
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
def path_of(name: str) -> str:
|
|
882
|
+
model = self._models_by_name.get(name)
|
|
883
|
+
return model.path if model else ""
|
|
884
|
+
|
|
885
|
+
same_file = [c for c in candidates if path_of(c) == ref_path]
|
|
886
|
+
if same_file:
|
|
887
|
+
return same_file
|
|
888
|
+
top = ref_path.split("/", 1)[0]
|
|
889
|
+
same_dir = [c for c in candidates if path_of(c) and path_of(c).split("/", 1)[0] == top]
|
|
890
|
+
return same_dir or candidates
|
|
891
|
+
|
|
892
|
+
def _fallback_edges(self) -> None:
|
|
893
|
+
"""A model that yielded nothing still sits between its declared parents
|
|
894
|
+
and everything that reads it. Wildcard edges keep that path visible;
|
|
895
|
+
review_required says exactly how much to trust it."""
|
|
896
|
+
for model in self.project.models:
|
|
897
|
+
report = self.reports.get(model.name)
|
|
898
|
+
if report is None or report.status not in ("failed", "star_only"):
|
|
899
|
+
continue
|
|
900
|
+
connected = False
|
|
901
|
+
for parent in model.declared_parents:
|
|
902
|
+
resolved = self._candidates.get(parent.lower())
|
|
903
|
+
if not resolved:
|
|
904
|
+
continue
|
|
905
|
+
# a bare declared parent that resolves to the model itself is
|
|
906
|
+
# the self-name-collision shape; a wildcard self-edge would
|
|
907
|
+
# assert the model feeds itself. A refused spelling must not
|
|
908
|
+
# reconnect here either: the analysis path already kept the
|
|
909
|
+
# as-written external edge (cycle-11, F4)
|
|
910
|
+
non_self = [
|
|
911
|
+
c for c in resolved if c != model.name and self._parent_binds(model, parent, c)
|
|
912
|
+
]
|
|
913
|
+
if not non_self:
|
|
914
|
+
continue
|
|
915
|
+
self.edges.append(
|
|
916
|
+
Edge(
|
|
917
|
+
src_model=non_self[0],
|
|
918
|
+
src_column=WILDCARD,
|
|
919
|
+
dst_model=model.name,
|
|
920
|
+
dst_column=WILDCARD,
|
|
921
|
+
confidence=0.3,
|
|
922
|
+
trust="review_required",
|
|
923
|
+
kind="value",
|
|
924
|
+
reason="model could not be analyzed; linked via declared dependency",
|
|
925
|
+
)
|
|
926
|
+
)
|
|
927
|
+
connected = True
|
|
928
|
+
if connected and report.status == "failed":
|
|
929
|
+
report.status = "fallback"
|
|
930
|
+
|
|
931
|
+
def _dedupe(self) -> None:
|
|
932
|
+
best: dict[tuple, Edge] = {}
|
|
933
|
+
for edge in self.edges:
|
|
934
|
+
existing = best.get(edge.key())
|
|
935
|
+
if existing is None or edge.confidence > existing.confidence:
|
|
936
|
+
best[edge.key()] = edge
|
|
937
|
+
self.edges = list(best.values())
|
|
938
|
+
|
|
939
|
+
def _index(self) -> None:
|
|
940
|
+
self._down = defaultdict(list)
|
|
941
|
+
self._down_by_model = defaultdict(list)
|
|
942
|
+
self._up = defaultdict(list)
|
|
943
|
+
for edge in self.edges:
|
|
944
|
+
self._down[(edge.src_model, edge.src_column)].append(edge)
|
|
945
|
+
self._down_by_model[edge.src_model].append(edge)
|
|
946
|
+
self._up[(edge.dst_model, edge.dst_column)].append(edge)
|
|
947
|
+
|
|
948
|
+
# ---- queries ----
|
|
949
|
+
|
|
950
|
+
def _model_owns_column(self, name: str, column: str) -> bool:
|
|
951
|
+
if self._down.get((name, column)) or self._up.get((name, column)):
|
|
952
|
+
return True
|
|
953
|
+
report = self.reports.get(name)
|
|
954
|
+
if report and any(c.lower() == column for c in report.columns):
|
|
955
|
+
return True
|
|
956
|
+
# sources count: a declared source schema is ownership evidence for
|
|
957
|
+
# multi-relation star resolution (cycle-12, F9)
|
|
958
|
+
for m in [*self.project.models, *self.project.sources]:
|
|
959
|
+
if m.name == name:
|
|
960
|
+
return any(c.lower() == column for c in m.declared_columns)
|
|
961
|
+
return False
|
|
962
|
+
|
|
963
|
+
def _pick_stem_claimant(self, canonical: list[str], column: str) -> str | None:
|
|
964
|
+
"""The one stem claimant the asked column picks, or None to refuse.
|
|
965
|
+
|
|
966
|
+
A claimant whose outputs are unenumerated (a star passthrough) may
|
|
967
|
+
well own the column too; picking the visible owner past it would
|
|
968
|
+
answer with false confidence (the cycle-6 review, the
|
|
969
|
+
elimination rule's wildcard guard applied to the pick). The
|
|
970
|
+
benchmark harness resolves case models through this same method."""
|
|
971
|
+
wanted = column.lower()
|
|
972
|
+
owners = [c for c in canonical if self._model_owns_column(c, wanted)]
|
|
973
|
+
veiled = [
|
|
974
|
+
c
|
|
975
|
+
for c in canonical
|
|
976
|
+
if c not in owners
|
|
977
|
+
and (report := self.reports.get(c)) is not None
|
|
978
|
+
and WILDCARD in report.columns
|
|
979
|
+
]
|
|
980
|
+
if len(owners) == 1 and not veiled:
|
|
981
|
+
return owners[0]
|
|
982
|
+
return None
|
|
983
|
+
|
|
984
|
+
def _also_matches(self, asked: str, chosen: str) -> list[str]:
|
|
985
|
+
"""The other models a spelling names when it was settled by being one
|
|
986
|
+
model's own name; the answer says so instead of picking silently."""
|
|
987
|
+
others = [c for c in self._candidates.get(asked.lower(), []) if c != chosen]
|
|
988
|
+
return sorted(self._askable_name(c) for c in others)
|
|
989
|
+
|
|
990
|
+
def _askable_name(self, canonical: str) -> str:
|
|
991
|
+
"""The spelling a person would type for a model: its shortest dotted
|
|
992
|
+
alias (telemetry_derived.events_v1), never the path-mangled canonical
|
|
993
|
+
name a collision assigned (sql__moz-fx...__events_v1__view__events_v1)."""
|
|
994
|
+
if self._models_by_name is None:
|
|
995
|
+
self._models_by_name = {
|
|
996
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
997
|
+
}
|
|
998
|
+
target = self._models_by_name.get(canonical)
|
|
999
|
+
dotted = sorted((a for a in (target.aliases if target else ()) if "." in a), key=len)
|
|
1000
|
+
return dotted[0] if dotted else canonical
|
|
1001
|
+
|
|
1002
|
+
def _external_targets(self) -> dict[str, str]:
|
|
1003
|
+
"""lowercased spelling -> as-written name of every table the graph
|
|
1004
|
+
reads but does not define, so a question about one still answers."""
|
|
1005
|
+
if getattr(self, "_externals", None) is None:
|
|
1006
|
+
self._externals = {
|
|
1007
|
+
e.src_model.lower(): e.src_model
|
|
1008
|
+
for e in self.edges
|
|
1009
|
+
if e.src_model not in self.reports
|
|
1010
|
+
}
|
|
1011
|
+
return self._externals
|
|
1012
|
+
|
|
1013
|
+
def _require_target(self, model: str, column: str) -> tuple[str, str]:
|
|
1014
|
+
canonical = self._candidates.get(model.lower())
|
|
1015
|
+
if not canonical:
|
|
1016
|
+
# a dbt-disabled model is still askable: its own analysis exists
|
|
1017
|
+
canonical = self._all_candidates.get(model.lower())
|
|
1018
|
+
if not canonical:
|
|
1019
|
+
external = self._external_targets().get(model.lower())
|
|
1020
|
+
if external is not None:
|
|
1021
|
+
canonical = [external]
|
|
1022
|
+
if not canonical:
|
|
1023
|
+
known = list(self.reports.keys())
|
|
1024
|
+
close = difflib.get_close_matches(model, known, n=3)
|
|
1025
|
+
raise UnknownTarget(f"no model named '{model}'", close)
|
|
1026
|
+
if len(canonical) > 1:
|
|
1027
|
+
# a spelling that IS one model's own name is that model, even
|
|
1028
|
+
# when another model carries it as an alias (postgresDBSamples:
|
|
1029
|
+
# base table Employee beside a path-qualified derived employee)
|
|
1030
|
+
exact = [c for c in canonical if c.lower() == model.lower()]
|
|
1031
|
+
if len(exact) == 1:
|
|
1032
|
+
canonical = exact
|
|
1033
|
+
if len(canonical) > 1:
|
|
1034
|
+
# a file stem naming several relations is DELIBERATELY ambiguous
|
|
1035
|
+
# and only until the column picks one (webtool_tables.invCount,
|
|
1036
|
+
# holdout round 5). Any other shared spelling, like two files
|
|
1037
|
+
# creating the same table, is a real collision: picking whichever
|
|
1038
|
+
# owns the asked column would hand the user the wrong model's
|
|
1039
|
+
# blast radius silently (the review of PR #28)
|
|
1040
|
+
picked = (
|
|
1041
|
+
self._pick_stem_claimant(canonical, column)
|
|
1042
|
+
if model.lower() in self._stem_aliases
|
|
1043
|
+
else None
|
|
1044
|
+
)
|
|
1045
|
+
if picked is None:
|
|
1046
|
+
raise UnknownTarget(
|
|
1047
|
+
f"'{model}' names {len(canonical)} different models; ask by full name",
|
|
1048
|
+
sorted(self._askable_name(c) for c in canonical),
|
|
1049
|
+
)
|
|
1050
|
+
canonical = [picked]
|
|
1051
|
+
name = canonical[0]
|
|
1052
|
+
report = self.reports.get(name)
|
|
1053
|
+
column = column.lower()
|
|
1054
|
+
if report is None or (report.status != "ok" and not report.columns):
|
|
1055
|
+
# outside the project, or defined without columns (CREATE TABLE
|
|
1056
|
+
# ... LIKE in redshift-utils): the only columns Ripple knows are
|
|
1057
|
+
# the ones its models read. "0 impacted, complete" for a column
|
|
1058
|
+
# it never saw read as a clean bill of health (balboa walk)
|
|
1059
|
+
seen = sorted({c for (m, c) in (self._down or {}) if m == name and c != WILDCARD})
|
|
1060
|
+
if column not in seen:
|
|
1061
|
+
raise UnknownTarget(
|
|
1062
|
+
f"'{name}' is outside this project; Ripple only knows the "
|
|
1063
|
+
f"columns its models read from it",
|
|
1064
|
+
seen, # the full list: an agent copies it into ingest_schema, and
|
|
1065
|
+
# mattermost's telemetry event table is read on 200+ columns
|
|
1066
|
+
)
|
|
1067
|
+
if report and report.status == "ok" and column not in report.columns:
|
|
1068
|
+
has_edges = bool(self._down.get((name, column)) or self._up.get((name, column)))
|
|
1069
|
+
if not has_edges:
|
|
1070
|
+
close = difflib.get_close_matches(column, report.columns, n=3)
|
|
1071
|
+
raise UnknownTarget(f"'{name}' has no column '{column}'", close)
|
|
1072
|
+
return name, column
|
|
1073
|
+
|
|
1074
|
+
def breaks(self, model: str, column: str, max_depth: int = 25) -> dict:
|
|
1075
|
+
"""Everything downstream of model.column.
|
|
1076
|
+
|
|
1077
|
+
Trust is path-aware: a hit reached only through an uncertain hop is
|
|
1078
|
+
reported at that path's worst trust, not the last edge's. Row-level
|
|
1079
|
+
impact (this column used in a downstream WHERE/JOIN) is reported per
|
|
1080
|
+
model rather than fanned out, so counts stay column-true.
|
|
1081
|
+
"""
|
|
1082
|
+
name, column = self._require_target(model, column)
|
|
1083
|
+
also = self._also_matches(model, name)
|
|
1084
|
+
# worst path trust each node has been expanded with; a node is
|
|
1085
|
+
# re-expanded when a WORSE path reaches it, so downgrades propagate
|
|
1086
|
+
# to everything downstream of a convergence point (fixed point,
|
|
1087
|
+
# each node expands at most len(TRUST_ORDER) times)
|
|
1088
|
+
expanded_at: dict[tuple[str, str], str] = {}
|
|
1089
|
+
seen: dict[tuple[str, str], dict] = {}
|
|
1090
|
+
row_impact: dict[str, dict] = {}
|
|
1091
|
+
truncated = False
|
|
1092
|
+
frontier: deque = deque([((name, column), 0, "verified")])
|
|
1093
|
+
while frontier:
|
|
1094
|
+
node, depth, path_trust = frontier.popleft()
|
|
1095
|
+
previous = expanded_at.get(node)
|
|
1096
|
+
if previous is not None and TRUST_ORDER.index(path_trust) >= TRUST_ORDER.index(
|
|
1097
|
+
previous
|
|
1098
|
+
):
|
|
1099
|
+
continue
|
|
1100
|
+
if depth >= max_depth:
|
|
1101
|
+
truncated = True
|
|
1102
|
+
continue
|
|
1103
|
+
expanded_at[node] = path_trust
|
|
1104
|
+
node_model, node_column = node
|
|
1105
|
+
outgoing = list(self._down.get(node, []))
|
|
1106
|
+
if node_column == WILDCARD:
|
|
1107
|
+
# a wildcard node means "some unknown column of this model":
|
|
1108
|
+
# anything reading any of its columns might be affected
|
|
1109
|
+
outgoing = self._down_by_model.get(node_model, [])
|
|
1110
|
+
else:
|
|
1111
|
+
outgoing += self._down.get((node_model, WILDCARD), [])
|
|
1112
|
+
for edge in outgoing:
|
|
1113
|
+
through_wildcard = WILDCARD in (edge.src_column, edge.dst_column, node_column)
|
|
1114
|
+
trust = _worst_trust(
|
|
1115
|
+
path_trust, edge.trust, *(["review_required"] if through_wildcard else [])
|
|
1116
|
+
)
|
|
1117
|
+
if edge.kind in ("filter", "join", "window"):
|
|
1118
|
+
entry = row_impact.setdefault(
|
|
1119
|
+
edge.dst_model,
|
|
1120
|
+
{"model": edge.dst_model, "uses": [], "trust": trust},
|
|
1121
|
+
)
|
|
1122
|
+
entry["uses"].append(f"{edge.src_model}.{edge.src_column} ({edge.kind})")
|
|
1123
|
+
entry["trust"] = _worst_trust(entry["trust"], trust)
|
|
1124
|
+
continue
|
|
1125
|
+
hit = {
|
|
1126
|
+
"model": edge.dst_model,
|
|
1127
|
+
"column": edge.dst_column,
|
|
1128
|
+
"via": f"{edge.src_model}.{edge.src_column}",
|
|
1129
|
+
"kind": edge.kind,
|
|
1130
|
+
"trust": trust,
|
|
1131
|
+
"edge_trust": edge.trust,
|
|
1132
|
+
"confidence": edge.confidence,
|
|
1133
|
+
"depth": depth + 1,
|
|
1134
|
+
}
|
|
1135
|
+
if edge.reason:
|
|
1136
|
+
hit["reason"] = edge.reason
|
|
1137
|
+
key = (edge.dst_model, edge.dst_column)
|
|
1138
|
+
previous = seen.get(key)
|
|
1139
|
+
if previous is None:
|
|
1140
|
+
seen[key] = hit
|
|
1141
|
+
else:
|
|
1142
|
+
# a second path to the same column never upgrades trust,
|
|
1143
|
+
# and an uncertain path is never hidden by a confident one
|
|
1144
|
+
previous["trust"] = _worst_trust(previous["trust"], trust)
|
|
1145
|
+
if not edge.dst_column.startswith("("):
|
|
1146
|
+
frontier.append(((edge.dst_model, edge.dst_column), depth + 1, trust))
|
|
1147
|
+
# stable: same-model hits keep the order the walk met them in (select order)
|
|
1148
|
+
hits = sorted(
|
|
1149
|
+
seen.values(),
|
|
1150
|
+
key=lambda h: (h["depth"], TRUST_ORDER[::-1].index(h["edge_trust"]), h["model"]),
|
|
1151
|
+
)
|
|
1152
|
+
by_model: dict[str, list[dict]] = defaultdict(list)
|
|
1153
|
+
for hit in hits:
|
|
1154
|
+
# the numeric score is a branch constant, not a probability;
|
|
1155
|
+
# public output carries the categorical label and the reason
|
|
1156
|
+
hit.pop("confidence", None)
|
|
1157
|
+
by_model[hit["model"]].append(hit)
|
|
1158
|
+
return {
|
|
1159
|
+
"source": {"model": name, "column": column},
|
|
1160
|
+
"impacted_columns": len(hits),
|
|
1161
|
+
"impacted_models": len(by_model),
|
|
1162
|
+
"review_required": sum(1 for h in hits if h["trust"] == "review_required"),
|
|
1163
|
+
"by_model": dict(by_model),
|
|
1164
|
+
"row_level_impact": sorted(row_impact.values(), key=lambda r: r["model"]),
|
|
1165
|
+
"row_impacted_models": len(row_impact),
|
|
1166
|
+
"complete": not truncated,
|
|
1167
|
+
**({"truncated_at_depth": max_depth} if truncated else {}),
|
|
1168
|
+
**({"also_matches": also} if also else {}),
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
def breaks_all(self, model: str, max_depth: int = 25) -> dict:
|
|
1172
|
+
"""Blast radius of the whole model changing at once (a predicate or
|
|
1173
|
+
join change alters every row, so every output column is a source)."""
|
|
1174
|
+
canonical = self._candidates.get(model.lower())
|
|
1175
|
+
if not canonical:
|
|
1176
|
+
raise UnknownTarget(
|
|
1177
|
+
f"no model named '{model}'",
|
|
1178
|
+
difflib.get_close_matches(model, list(self.reports), n=3),
|
|
1179
|
+
)
|
|
1180
|
+
name = canonical[0]
|
|
1181
|
+
report = self.reports.get(name)
|
|
1182
|
+
columns = [c for c in (report.columns if report else []) if c != WILDCARD]
|
|
1183
|
+
merged: dict = {"impacted": {}, "row": {}}
|
|
1184
|
+
complete = True
|
|
1185
|
+
for column in columns or [WILDCARD]:
|
|
1186
|
+
try:
|
|
1187
|
+
result = self.breaks(name, column, max_depth=max_depth)
|
|
1188
|
+
except UnknownTarget:
|
|
1189
|
+
continue
|
|
1190
|
+
complete = complete and result.get("complete", True)
|
|
1191
|
+
for hits in result["by_model"].values():
|
|
1192
|
+
for hit in hits:
|
|
1193
|
+
key = (hit["model"], hit["column"])
|
|
1194
|
+
existing = merged["impacted"].get(key)
|
|
1195
|
+
if existing is None:
|
|
1196
|
+
merged["impacted"][key] = hit
|
|
1197
|
+
else:
|
|
1198
|
+
existing["trust"] = _worst_trust(existing["trust"], hit["trust"])
|
|
1199
|
+
for row in result["row_level_impact"]:
|
|
1200
|
+
merged["row"][row["model"]] = row
|
|
1201
|
+
hits = list(merged["impacted"].values())
|
|
1202
|
+
by_model: dict[str, list[dict]] = defaultdict(list)
|
|
1203
|
+
for hit in hits:
|
|
1204
|
+
by_model[hit["model"]].append(hit)
|
|
1205
|
+
return {
|
|
1206
|
+
"source": {"model": name, "column": "(all columns)"},
|
|
1207
|
+
"impacted_columns": len(hits),
|
|
1208
|
+
"impacted_models": len(by_model),
|
|
1209
|
+
"review_required": sum(1 for h in hits if h["trust"] == "review_required"),
|
|
1210
|
+
"by_model": dict(by_model),
|
|
1211
|
+
"row_level_impact": sorted(merged["row"].values(), key=lambda r: r["model"]),
|
|
1212
|
+
"row_impacted_models": len(merged["row"]),
|
|
1213
|
+
"complete": complete,
|
|
1214
|
+
}
|
|
1215
|
+
|
|
1216
|
+
def trace(self, model: str, column: str, max_depth: int = 25) -> dict:
|
|
1217
|
+
"""Where model.column comes from, hop by hop. Trust accumulates along
|
|
1218
|
+
the path from the target, same contract as breaks()."""
|
|
1219
|
+
name, column = self._require_target(model, column)
|
|
1220
|
+
expanded_at: dict[tuple[str, str], str] = {}
|
|
1221
|
+
hops: dict[tuple, dict] = {}
|
|
1222
|
+
truncated = False
|
|
1223
|
+
frontier: deque = deque([((name, column), 0, "verified")])
|
|
1224
|
+
while frontier:
|
|
1225
|
+
node, depth, path_trust = frontier.popleft()
|
|
1226
|
+
previous = expanded_at.get(node)
|
|
1227
|
+
if previous is not None and TRUST_ORDER.index(path_trust) >= TRUST_ORDER.index(
|
|
1228
|
+
previous
|
|
1229
|
+
):
|
|
1230
|
+
continue
|
|
1231
|
+
if depth >= max_depth:
|
|
1232
|
+
truncated = True
|
|
1233
|
+
continue
|
|
1234
|
+
expanded_at[node] = path_trust
|
|
1235
|
+
node_model, node_column = node
|
|
1236
|
+
incoming = list(self._up.get(node, []))
|
|
1237
|
+
if node_column != WILDCARD:
|
|
1238
|
+
incoming += self._up.get((node_model, WILDCARD), [])
|
|
1239
|
+
for edge in incoming:
|
|
1240
|
+
if edge.kind != "value":
|
|
1241
|
+
continue
|
|
1242
|
+
trust = _worst_trust(path_trust, edge.trust)
|
|
1243
|
+
key = (edge.src_model, edge.src_column, edge.dst_model, edge.dst_column)
|
|
1244
|
+
existing = hops.get(key)
|
|
1245
|
+
if existing is None:
|
|
1246
|
+
hop = {
|
|
1247
|
+
"model": edge.src_model,
|
|
1248
|
+
"column": edge.src_column,
|
|
1249
|
+
"feeds": f"{edge.dst_model}.{edge.dst_column}",
|
|
1250
|
+
"trust": trust,
|
|
1251
|
+
"edge_trust": edge.trust,
|
|
1252
|
+
"confidence": edge.confidence,
|
|
1253
|
+
"depth": depth + 1,
|
|
1254
|
+
}
|
|
1255
|
+
if edge.reason:
|
|
1256
|
+
hop["reason"] = edge.reason
|
|
1257
|
+
hops[key] = hop
|
|
1258
|
+
else:
|
|
1259
|
+
existing["trust"] = _worst_trust(existing["trust"], trust)
|
|
1260
|
+
frontier.append(((edge.src_model, edge.src_column), depth + 1, trust))
|
|
1261
|
+
ordered = sorted(hops.values(), key=lambda h: (h["depth"], h["model"], h["column"]))
|
|
1262
|
+
for hop in ordered:
|
|
1263
|
+
hop.pop("confidence", None)
|
|
1264
|
+
return {
|
|
1265
|
+
"target": {"model": name, "column": column},
|
|
1266
|
+
"upstream": ordered,
|
|
1267
|
+
"complete": not truncated,
|
|
1268
|
+
**({"truncated_at_depth": max_depth} if truncated else {}),
|
|
1269
|
+
}
|
|
1270
|
+
|
|
1271
|
+
def resolve_star_column(self, model: str, column: str) -> list[tuple[str, str]]:
|
|
1272
|
+
"""Resolve a named column through the model's star projection:
|
|
1273
|
+
[(source_relation, column)] when the star provably carries it,
|
|
1274
|
+
[] when it is excepted or ownership is unknowable."""
|
|
1275
|
+
from ripple.star_resolution import resolve_star_column
|
|
1276
|
+
|
|
1277
|
+
return resolve_star_column(self, model, column)
|
|
1278
|
+
|
|
1279
|
+
# ---- export ----
|
|
1280
|
+
|
|
1281
|
+
def stats(self) -> dict:
|
|
1282
|
+
by_status = defaultdict(list)
|
|
1283
|
+
for report in self.reports.values():
|
|
1284
|
+
by_status[report.status].append(report.name)
|
|
1285
|
+
failed = {
|
|
1286
|
+
name: self.reports[name].error
|
|
1287
|
+
for name in [
|
|
1288
|
+
*by_status.get("failed", []),
|
|
1289
|
+
*by_status.get("fallback", []),
|
|
1290
|
+
*by_status.get("timed_out", []),
|
|
1291
|
+
]
|
|
1292
|
+
}
|
|
1293
|
+
return {
|
|
1294
|
+
"mode": self.project.mode,
|
|
1295
|
+
"dialect": self.project.dialect,
|
|
1296
|
+
"models": len(self.project.models),
|
|
1297
|
+
"sources": len(self.project.sources),
|
|
1298
|
+
"ok": len(by_status.get("ok", [])),
|
|
1299
|
+
"star_only": len(by_status.get("star_only", [])),
|
|
1300
|
+
"fallback": len(by_status.get("fallback", [])),
|
|
1301
|
+
"timed_out": len(by_status.get("timed_out", [])),
|
|
1302
|
+
"failed": len(by_status.get("failed", [])) + len(by_status.get("no_sql", [])),
|
|
1303
|
+
"failure_reasons": dict(list(failed.items())[:20]),
|
|
1304
|
+
"edges": len(self.edges),
|
|
1305
|
+
"review_required_edges": sum(1 for e in self.edges if e.trust == "review_required"),
|
|
1306
|
+
"verified_edges": sum(1 for e in self.edges if e.trust == "verified"),
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
def unresolved_tables(self) -> list[dict]:
|
|
1310
|
+
"""External relations referenced by models but with no known columns:
|
|
1311
|
+
names that resolve to nothing, plus sources and base tables whose
|
|
1312
|
+
columns nobody has declared. Sorted by how much coverage they block."""
|
|
1313
|
+
if self._models_by_name is None:
|
|
1314
|
+
self._models_by_name = {
|
|
1315
|
+
m.name: m for m in [*self.project.models, *self.project.sources]
|
|
1316
|
+
}
|
|
1317
|
+
import re
|
|
1318
|
+
|
|
1319
|
+
cte_re = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]*)\s+as\s*\(", re.I)
|
|
1320
|
+
entries: dict[str, dict] = {}
|
|
1321
|
+
for model in self.project.models:
|
|
1322
|
+
report = self.reports.get(model.name)
|
|
1323
|
+
status = report.status if report else "failed"
|
|
1324
|
+
own_ctes = {m.lower() for m in cte_re.findall(model.sql)} if model.sql else set()
|
|
1325
|
+
for parent in model.declared_parents:
|
|
1326
|
+
if "{" in parent or any(c.isspace() for c in parent):
|
|
1327
|
+
continue # jinja leftover, not an ingestable table name
|
|
1328
|
+
if parent.lower() in own_ctes:
|
|
1329
|
+
continue # a CTE of this model's own script, not a table
|
|
1330
|
+
resolved = self._candidates.get(parent.lower()) or self._all_candidates.get(
|
|
1331
|
+
parent.lower()
|
|
1332
|
+
)
|
|
1333
|
+
as_written = None
|
|
1334
|
+
if resolved and not self._parent_binds(model, parent, resolved[0]):
|
|
1335
|
+
# refused spelling: the read names an external relation,
|
|
1336
|
+
# reported as the SQL cites it (cycle-11, F3)
|
|
1337
|
+
name = self._refused_read_of(model, parent) or parent.lower()
|
|
1338
|
+
if "{" in name:
|
|
1339
|
+
stripped = _strip_placeholders(name)
|
|
1340
|
+
if "{" not in stripped:
|
|
1341
|
+
as_written, name = name, stripped or parent.lower()
|
|
1342
|
+
known = self._candidates.get(name.lower(), [])
|
|
1343
|
+
if len(known) == 1 and (
|
|
1344
|
+
(self.reports.get(known[0]) or ModelReport(name="")).status == "ok"
|
|
1345
|
+
or (
|
|
1346
|
+
self._models_by_name.get(known[0]) or Model(name="", sql="", path="")
|
|
1347
|
+
).declared_columns
|
|
1348
|
+
):
|
|
1349
|
+
continue # the spelling the SQL wrote was ingested; nothing to do
|
|
1350
|
+
qualified = name if "." in name else None
|
|
1351
|
+
elif resolved:
|
|
1352
|
+
target = self._models_by_name.get(resolved[0])
|
|
1353
|
+
target_report = self.reports.get(resolved[0])
|
|
1354
|
+
if target_report is not None and target_report.status == "ok":
|
|
1355
|
+
continue
|
|
1356
|
+
if target is None or target.sql.strip() or target.declared_columns:
|
|
1357
|
+
continue # a real model that failed is not a schema gap
|
|
1358
|
+
name = resolved[0]
|
|
1359
|
+
qualified = max((a for a in target.aliases if "." in a), key=len, default=None)
|
|
1360
|
+
else:
|
|
1361
|
+
name = parent.lower()
|
|
1362
|
+
qualified = name if "." in name else None
|
|
1363
|
+
as_written = None
|
|
1364
|
+
if qualified is None:
|
|
1365
|
+
# the declared parent is bare; the SQL may have
|
|
1366
|
+
# written it qualified, and that spelling is the one
|
|
1367
|
+
# an ingest must use (balboa's slow_query reads
|
|
1368
|
+
# snowflake_sample_data.tpch_sf1.region)
|
|
1369
|
+
written = [
|
|
1370
|
+
_strip_placeholders(r)
|
|
1371
|
+
for r in (self._reads_as_written(model) or ())
|
|
1372
|
+
if r.split(".")[-1] == name
|
|
1373
|
+
]
|
|
1374
|
+
written = [w for w in written if "." in w]
|
|
1375
|
+
qualified = min(written, key=len) if written else None
|
|
1376
|
+
as_written = next(
|
|
1377
|
+
(
|
|
1378
|
+
r
|
|
1379
|
+
for r in (self._reads_as_written(model) or ())
|
|
1380
|
+
if "{" in r and r.split(".")[-1] == name
|
|
1381
|
+
),
|
|
1382
|
+
None,
|
|
1383
|
+
)
|
|
1384
|
+
entry = entries.setdefault(
|
|
1385
|
+
name,
|
|
1386
|
+
{
|
|
1387
|
+
"table": name,
|
|
1388
|
+
**({"qualified_name": qualified} if qualified else {}),
|
|
1389
|
+
**({"as_written": as_written} if as_written else {}),
|
|
1390
|
+
"referencing_models": 0,
|
|
1391
|
+
"blocked_models": 0,
|
|
1392
|
+
"statuses": {},
|
|
1393
|
+
},
|
|
1394
|
+
)
|
|
1395
|
+
entry["referencing_models"] += 1
|
|
1396
|
+
if status in ("star_only", "failed", "fallback", "no_sql"):
|
|
1397
|
+
entry["blocked_models"] += 1
|
|
1398
|
+
entry["statuses"][status] = entry["statuses"].get(status, 0) + 1
|
|
1399
|
+
# tables only the engine saw: a raw table read by name inside a dbt
|
|
1400
|
+
# model is no declared parent, yet its edges say "not found"
|
|
1401
|
+
# (balboa's _airbyte_raw_zip_coordinates)
|
|
1402
|
+
seen_pairs: set[tuple[str, str]] = set()
|
|
1403
|
+
for edge in self.edges:
|
|
1404
|
+
if edge.src_model in self.reports or "not found" not in edge.reason:
|
|
1405
|
+
continue # disabled-only names carry their own reason and are skipped
|
|
1406
|
+
written = edge.src_model.lower()
|
|
1407
|
+
key = _strip_placeholders(written) if "{" in written else written
|
|
1408
|
+
if not key or "{" in key or any(ch.isspace() for ch in key):
|
|
1409
|
+
continue # a templated name or jinja remnant: cited as written, not ingestable
|
|
1410
|
+
if key in entries or (key, edge.dst_model) in seen_pairs:
|
|
1411
|
+
continue
|
|
1412
|
+
seen_pairs.add((key, edge.dst_model))
|
|
1413
|
+
entry = entries.setdefault(
|
|
1414
|
+
key,
|
|
1415
|
+
{
|
|
1416
|
+
"table": key,
|
|
1417
|
+
**({"qualified_name": key} if "." in key else {}),
|
|
1418
|
+
**({"as_written": written} if written != key else {}),
|
|
1419
|
+
"referencing_models": 0,
|
|
1420
|
+
"blocked_models": 0,
|
|
1421
|
+
"statuses": {},
|
|
1422
|
+
},
|
|
1423
|
+
)
|
|
1424
|
+
entry["referencing_models"] += 1
|
|
1425
|
+
status = self.reports[edge.dst_model].status if edge.dst_model in self.reports else "ok"
|
|
1426
|
+
entry["statuses"][status] = entry["statuses"].get(status, 0) + 1
|
|
1427
|
+
return sorted(
|
|
1428
|
+
entries.values(),
|
|
1429
|
+
key=lambda e: (-e["blocked_models"], -e["referencing_models"], e["table"]),
|
|
1430
|
+
)
|
|
1431
|
+
|
|
1432
|
+
def suggest_target(self) -> dict | None:
|
|
1433
|
+
"""The column with the widest blast radius, for a copy-pasteable
|
|
1434
|
+
first command. Direct fanout shortlists; transitive reach decides."""
|
|
1435
|
+
shortlist = []
|
|
1436
|
+
for (src_model, src_column), edges in (self._down or {}).items():
|
|
1437
|
+
if src_column == WILDCARD or src_column.startswith("("):
|
|
1438
|
+
continue
|
|
1439
|
+
report = self.reports.get(src_model)
|
|
1440
|
+
# only models this project owns: breaks() cannot target an external
|
|
1441
|
+
# table, and on filecoin-data-portal the 197 highest-fanout columns
|
|
1442
|
+
# were external, so the top-8 probe never reached a real model and
|
|
1443
|
+
# the suggestion came back empty
|
|
1444
|
+
if report is None or report.status != "ok":
|
|
1445
|
+
continue
|
|
1446
|
+
fanout = sum(1 for e in edges if e.kind == "value")
|
|
1447
|
+
if fanout:
|
|
1448
|
+
shortlist.append((fanout, src_model, src_column))
|
|
1449
|
+
shortlist.sort(reverse=True)
|
|
1450
|
+
# a path-qualified collision name (sql__2024__cookies__cookies) is
|
|
1451
|
+
# not something a person types; prefer a spelling they can, and fall
|
|
1452
|
+
# back to the mangled names only when nothing else has reach
|
|
1453
|
+
# (HTTP Archive's almanac)
|
|
1454
|
+
askable = [t for t in shortlist if self._askable_name(t[1]) == t[1] and "__" not in t[1]]
|
|
1455
|
+
shortlist = askable or shortlist
|
|
1456
|
+
best = None
|
|
1457
|
+
for _, model, column in shortlist[:8]:
|
|
1458
|
+
try:
|
|
1459
|
+
reach = self.breaks(model, column)["impacted_columns"]
|
|
1460
|
+
except LookupError:
|
|
1461
|
+
continue
|
|
1462
|
+
if best is None or reach > best["fanout"]:
|
|
1463
|
+
best = {"model": model, "column": column, "fanout": reach}
|
|
1464
|
+
return best
|
|
1465
|
+
|
|
1466
|
+
def to_dict(self) -> dict:
|
|
1467
|
+
nodes = []
|
|
1468
|
+
for m in self.project.sources:
|
|
1469
|
+
nodes.append(
|
|
1470
|
+
{"name": m.name, "id": m.uid, "type": "source", "columns": m.declared_columns}
|
|
1471
|
+
)
|
|
1472
|
+
for m in self.project.models:
|
|
1473
|
+
report = self.reports.get(m.name)
|
|
1474
|
+
nodes.append(
|
|
1475
|
+
{
|
|
1476
|
+
"name": m.name,
|
|
1477
|
+
"id": m.uid,
|
|
1478
|
+
"type": "model",
|
|
1479
|
+
"path": m.path,
|
|
1480
|
+
"status": report.status if report else "failed",
|
|
1481
|
+
"parsed": bool(report and report.parsed),
|
|
1482
|
+
"columns": report.columns if report else [],
|
|
1483
|
+
**({"warnings": report.warnings[:10]} if report and report.warnings else {}),
|
|
1484
|
+
}
|
|
1485
|
+
)
|
|
1486
|
+
return {
|
|
1487
|
+
"stats": self.stats(),
|
|
1488
|
+
"nodes": nodes,
|
|
1489
|
+
"edges": [
|
|
1490
|
+
{
|
|
1491
|
+
"src": f"{e.src_model}.{e.src_column}",
|
|
1492
|
+
"dst": f"{e.dst_model}.{e.dst_column}",
|
|
1493
|
+
"kind": e.kind,
|
|
1494
|
+
"trust": e.trust,
|
|
1495
|
+
**({"reason": e.reason} if e.reason else {}),
|
|
1496
|
+
}
|
|
1497
|
+
for e in self.edges
|
|
1498
|
+
],
|
|
1499
|
+
}
|