ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/graph.py ADDED
@@ -0,0 +1,1499 @@
1
+ """Cross-model column lineage graph.
2
+
3
+ Runs the engine over every model in a Project (parents before children) and
4
+ stitches per-statement lineage into one graph that can answer:
5
+
6
+ - breaks(model.column): everything downstream that would be affected
7
+ - trace(model.column): where a column's value comes from, hop by hop
8
+ - to_dict(): the full graph for the canvas, MCP, and CI bot
9
+
10
+ Honesty model, in order of what can go wrong:
11
+ - every edge carries confidence (0..1) and a trust label
12
+ (verified / high_confidence / moderate / review_required)
13
+ - trust is always the WORST of: the engine's label, the label implied by
14
+ confidence, how the source table's name resolved, and the evidence tier
15
+ (compiled manifest vs raw Jinja vs plain files)
16
+ - a model that yields no lineage still appears, connected to its declared
17
+ parents by wildcard edges at review_required, so nothing downstream of a
18
+ failed model is ever hidden
19
+ - an unknown model or column raises with suggestions; it never returns an
20
+ empty result that reads as "safe"
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import difflib
26
+ import logging
27
+ from collections import defaultdict, deque
28
+ from dataclasses import dataclass, field
29
+
30
+ from ripple.engine.budget import TIMED_OUT, budget_seconds, max_sql_bytes, run_within_budget
31
+ from ripple.engine.dispatch import extract_lineage_complete
32
+ from ripple.engine.jinja import convert_jinja_to_sql
33
+ from ripple.engine.preprocess import prepare_sql_for_parse
34
+ from ripple.loaders.identity import _name_suffixes
35
+ from ripple.loaders.identity import strip_placeholders as _strip_placeholders
36
+ from ripple.project import Model, Project
37
+
38
+ logger = logging.getLogger(__name__)
39
+
40
+ # worst first; every combinator takes the minimum index
41
+ TRUST_ORDER = ["review_required", "moderate", "high_confidence", "verified"]
42
+
43
+ # evidence mode -> maximum trust an edge can claim
44
+ EVIDENCE_CAP = {
45
+ "manifest": "verified",
46
+ "raw_jinja": "high_confidence",
47
+ "plain_sql": "high_confidence",
48
+ }
49
+
50
+ WILDCARD = "*"
51
+
52
+
53
+ def _trust_from_confidence(confidence: float) -> str:
54
+ if confidence >= 1.0:
55
+ return "verified"
56
+ if confidence >= 0.8:
57
+ return "high_confidence"
58
+ if confidence >= 0.5:
59
+ return "moderate"
60
+ return "review_required"
61
+
62
+
63
+ def _worst_trust(*labels: str) -> str:
64
+ return min(labels, key=TRUST_ORDER.index)
65
+
66
+
67
+ class UnknownTarget(LookupError):
68
+ def __init__(self, message: str, suggestions: list[str]):
69
+ super().__init__(message)
70
+ self.suggestions = suggestions
71
+
72
+
73
+ @dataclass(frozen=True)
74
+ class Edge:
75
+ src_model: str
76
+ src_column: str
77
+ dst_model: str
78
+ dst_column: str
79
+ confidence: float
80
+ trust: str
81
+ kind: str # "value" | "filter" | "join" | "window"
82
+ reason: str = "" # why trust was downgraded, "" if clean
83
+
84
+ def key(self) -> tuple:
85
+ return (self.src_model, self.src_column, self.dst_model, self.dst_column, self.kind)
86
+
87
+
88
+ @dataclass
89
+ class ModelReport:
90
+ name: str
91
+ status: str = "failed" # "ok" | "star_only" | "fallback" | "failed" | "no_sql" | "timed_out"
92
+ columns: list[str] = field(default_factory=list)
93
+ error: str | None = None
94
+ warnings: list[str] = field(default_factory=list)
95
+
96
+ @property
97
+ def parsed(self) -> bool:
98
+ return self.status in ("ok", "star_only")
99
+
100
+
101
+ @dataclass
102
+ class LineageGraph:
103
+ project: Project
104
+ edges: list[Edge] = field(default_factory=list)
105
+ reports: dict[str, ModelReport] = field(default_factory=dict)
106
+ _jinja_env: object = None
107
+ _candidates: dict[str, list[str]] = field(default_factory=dict)
108
+ _all_candidates: dict[str, list[str]] = field(default_factory=dict)
109
+ _stem_aliases: set[str] = field(default_factory=set)
110
+ _models_by_name: dict = None
111
+ _reads_cache: dict = None
112
+ _down: dict = None
113
+ _down_by_model: dict = None
114
+ _up: dict = None
115
+ # (dst_model, src_model) -> lowercased except-list of that star edge
116
+ _star_excepts: dict = field(default_factory=dict)
117
+ # lowered names of function-minted models (TVFs); call sites bind these
118
+ _function_relations: set = field(default_factory=set)
119
+
120
+ # ---- build ----
121
+
122
+ @classmethod
123
+ def build(cls, project: Project) -> LineageGraph:
124
+ graph = cls(project=project)
125
+ graph._candidates = project.name_candidates
126
+ graph._all_candidates = project.all_name_candidates
127
+ graph._stem_aliases = project.stem_alias_names
128
+ real_tables = set(graph._candidates.keys())
129
+ # a call of a minted TVF by its written name binds that model
130
+ # instead of degrading as a function/relation collision (F11)
131
+ graph._function_relations = {
132
+ m.name.lower() for m in project.models if getattr(m, "is_function", False)
133
+ }
134
+ if any(m.evidence != "manifest" for m in project.models):
135
+ from ripple.render import build_project_environment
136
+
137
+ graph._jinja_env = build_project_environment(
138
+ project.macro_sources,
139
+ root_package=project.package_name,
140
+ dialect=project.dialect,
141
+ project_vars=project.jinja_vars,
142
+ )
143
+ for model in graph._topological(project.models):
144
+ graph._add_model(model, real_tables)
145
+ graph._fallback_edges()
146
+ graph._add_downstream()
147
+ graph._dedupe()
148
+ graph._index()
149
+ return graph
150
+
151
+ def _topological(self, models: list[Model]) -> list[Model]:
152
+ """Parents before children, so a child's SELECT * can expand through
153
+ the columns its parents were just found to have. Cycles and unknown
154
+ parents fall back to input order at the end."""
155
+ by_name = {m.name: m for m in models}
156
+ state: dict[str, int] = {} # 1 = on stack, 2 = done
157
+ order: list[Model] = []
158
+ for root in models:
159
+ stack: list[tuple[str, bool]] = [(root.name, False)]
160
+ while stack:
161
+ name, children_done = stack.pop()
162
+ if state.get(name) == 2:
163
+ continue
164
+ model = by_name.get(name)
165
+ if model is None:
166
+ state[name] = 2
167
+ continue
168
+ if children_done:
169
+ state[name] = 2
170
+ order.append(model)
171
+ continue
172
+ if state.get(name) == 1: # cycle; settle in stack order
173
+ continue
174
+ state[name] = 1
175
+ stack.append((name, True))
176
+ for parent in model.declared_parents:
177
+ # every candidate, not just the first: a name shared by
178
+ # dialect variants must not leave the real one unordered
179
+ for candidate in self._all_candidates.get(parent.lower(), ()):
180
+ if not state.get(candidate):
181
+ stack.append((candidate, False))
182
+ return order
183
+
184
+ def _parent_schema(self, model: Model) -> dict[str, dict[str, str]]:
185
+ """The project as its own schema: every known column of this model's
186
+ parents, keyed by each name the SQL might use for them."""
187
+ if self._models_by_name is None:
188
+ self._models_by_name = {
189
+ m.name: m for m in [*self.project.models, *self.project.sources]
190
+ }
191
+ schema: dict[str, dict[str, str]] = {}
192
+ bare_claims: dict[str, dict[str, dict[str, str]]] = {}
193
+ for parent in model.declared_parents:
194
+ # binding stays enabled-only, but a model's own star expansion may
195
+ # read a disabled parent's columns (dbt disables whole branches)
196
+ resolved = self._candidates.get(parent.lower()) or self._all_candidates.get(
197
+ parent.lower()
198
+ )
199
+ if "." not in parent:
200
+ # the declared parent is bare; the SQL may have written it
201
+ # qualified, and an ingested table answers only to that
202
+ # spelling when another model owns the bare name (Mozilla's
203
+ # mdn_fred.events_stream beside a model named events_stream)
204
+ for read in self._reads_as_written(model) or ():
205
+ if "." in read and read.split(".")[-1] == parent.lower():
206
+ for c in self._candidates.get(read, ()):
207
+ if c not in (resolved or []):
208
+ resolved = [*(resolved or []), c]
209
+ if not resolved:
210
+ continue
211
+ # a bare parent shared by two relations (Mozilla's telemetry
212
+ # view over telemetry_derived, same table name) is settled by
213
+ # the spelling the SQL wrote, exactly as edges are; the old
214
+ # first-candidate pick stays only when nothing disambiguates
215
+ bound = [c for c in resolved if self._parent_binds(model, parent, c)]
216
+ chosen = bound[0] if len(bound) == 1 else resolved[0]
217
+ if not self._parent_binds(model, parent, chosen):
218
+ # refused spelling: supplying the bare relation's columns
219
+ # would fabricate them onto the external (cycle-11, F3)
220
+ continue
221
+ report = self.reports.get(chosen)
222
+ parent_model = self._models_by_name.get(chosen)
223
+ if report is not None and report.status == "ok":
224
+ known = report.columns
225
+ elif report is None and parent_model is not None:
226
+ # a source never gets a report; ingested warehouse schemas
227
+ # give it declared columns, which are as good as analyzed ones
228
+ known = parent_model.declared_columns
229
+ else:
230
+ continue
231
+ if not known:
232
+ continue
233
+ columns = {c: "unknown" for c in known if c != WILDCARD}
234
+ names = {chosen, *(parent_model.aliases if parent_model else set())}
235
+ for name in names:
236
+ schema[name] = columns
237
+ bare_claims.setdefault(chosen.split(".")[-1].lower(), {}).setdefault(chosen, columns)
238
+ # the engine expands a star by the bare table name; a qualified
239
+ # parent whose bare spelling another model owns (Mozilla's
240
+ # mdn_fred.events_stream beside a model named events_stream) still
241
+ # gets it HERE, per model, unless two of this model's own parents
242
+ # share it, which would be a guess
243
+ for bare, owners in bare_claims.items():
244
+ if len(owners) == 1 and bare not in schema:
245
+ schema[bare] = next(iter(owners.values()))
246
+ return schema
247
+
248
+ def _add_model(self, model: Model, real_tables: set[str]) -> None:
249
+ report = ModelReport(name=model.name)
250
+ self.reports[model.name] = report
251
+ if model.load_error:
252
+ report.status = "timed_out"
253
+ report.error = model.load_error
254
+ return
255
+ if not model.sql.strip():
256
+ if model.declared_columns:
257
+ # a base table from CREATE TABLE: no derivation, but its columns
258
+ # are known, so downstream SELECT * resolves against them
259
+ report.columns = list(model.declared_columns)
260
+ report.status = "ok"
261
+ else:
262
+ report.status = "no_sql"
263
+ report.error = "no SQL"
264
+ return
265
+ self._analyze_sql(model, model.sql, real_tables, report, primary=True)
266
+ # a table built by CREATE plus UPDATEs is one model; each further
267
+ # statement adds its lineage to the union (nycdb, PR #28 gap 3)
268
+ for extra in model.extra_sqls:
269
+ self._analyze_sql(model, extra, real_tables, report, primary=False)
270
+
271
+ def _analyze_sql(
272
+ self,
273
+ model: Model,
274
+ sql: str,
275
+ real_tables: set[str],
276
+ report: ModelReport,
277
+ primary: bool,
278
+ ) -> None:
279
+ edge_start = len(self.edges)
280
+ limit = max_sql_bytes()
281
+ if limit and len(sql) > limit:
282
+ message = (
283
+ f"statement is {len(sql) / 1e6:.1f}MB, over the {limit / 1e6:g}MB analysis "
284
+ "limit; generated SQL this size wedges the parser for minutes "
285
+ "(RIPPLE_MAX_SQL_BYTES to raise, 0 to disable)"
286
+ )
287
+ if primary:
288
+ report.status = "timed_out"
289
+ report.error = f"{message}: {model.path}"
290
+ else:
291
+ report.warnings.append(message)
292
+ return
293
+ # computed before the worker: it caches onto self, and the worker must
294
+ # stay pure so an abandoned one cannot race the build
295
+ warehouse_columns = self._parent_schema(model) if model.declared_parents else None
296
+ dialect = model.dialect or self.project.dialect
297
+
298
+ def analyze():
299
+ body = sql
300
+ jinja_cleaner = None
301
+ if model.evidence != "manifest" and ("{{" in body or "{%" in body):
302
+ try: # tier b: real Jinja render beats regex stripping
303
+ from ripple.render import render_dbt_sql
304
+
305
+ declared = (
306
+ model.jinja_vars
307
+ if model.jinja_vars is not None
308
+ else self.project.jinja_vars
309
+ )
310
+ body = render_dbt_sql(
311
+ body, model.name, env=self._jinja_env, jinja_vars=declared
312
+ )
313
+ # rendered SQL has no Jinja left; re-cleaning it corrupts
314
+ # regex literals and $-quoted strings in the output
315
+ except Exception:
316
+ jinja_cleaner = convert_jinja_to_sql # tier c: regex cleaner on the raw SQL
317
+ extracted = extract_lineage_complete(
318
+ body,
319
+ dialect=dialect,
320
+ real_tables=real_tables,
321
+ warehouse_columns=warehouse_columns or None,
322
+ clean_jinja_func=jinja_cleaner,
323
+ function_relations=self._function_relations or None,
324
+ # a CTE named after this model shadows only itself; without
325
+ # this, dbt's `with X as (...) select * from X` house style
326
+ # marks every correct pass-through edge review_required
327
+ self_names={model.name, *model.aliases, *model.stem_aliases},
328
+ )
329
+ # the same pre-parse pipeline the engine used, or these reparses
330
+ # drift and drop edges (cycle-10 review, F8)
331
+ prepared, _ = prepare_sql_for_parse(body, dialect, jinja_cleaner)
332
+ return extracted, prepared, self._alias_map(prepared, dialect=dialect)
333
+
334
+ # the loader's budget caps parse time, but a file can parse fast and
335
+ # wedge HERE: a generated 3.7MB one-liner parsed in 13s, then spent
336
+ # 139s inside qualify_tables. Same budget, same honest report.
337
+ try:
338
+ outcome = run_within_budget(model.name, analyze)
339
+ except Exception as e: # a model that won't parse is a report, not a crash
340
+ if primary:
341
+ report.error = f"{type(e).__name__}: {e}"
342
+ else:
343
+ report.warnings.append(f"additional statement failed: {type(e).__name__}: {e}")
344
+ return
345
+ if outcome is TIMED_OUT:
346
+ message = (
347
+ f"analysis exceeded the {budget_seconds():g}s per-file budget: "
348
+ f"{model.path} (RIPPLE_PARSE_BUDGET_S to raise)"
349
+ )
350
+ if primary:
351
+ report.status = "timed_out"
352
+ report.error = message
353
+ else:
354
+ report.warnings.append(message)
355
+ return
356
+ result, prepared_sql, alias_map = outcome
357
+
358
+ columns = [c.lower() for c in result.contributing]
359
+ if primary:
360
+ report.columns = columns
361
+ else:
362
+ report.columns.extend(c for c in columns if c not in report.columns)
363
+ real_columns = [c for c in columns if c != WILDCARD and c != "null"]
364
+ if real_columns:
365
+ if not primary and report.status == "star_only":
366
+ # the state must read like any partially-starred model:
367
+ # status ok, no stale error (the cycle-6 review)
368
+ report.error = None
369
+ report.status = "ok"
370
+ elif primary:
371
+ if columns:
372
+ report.status = "star_only"
373
+ report.error = "only SELECT * survived; column-level lineage unavailable"
374
+ else:
375
+ report.error = "no columns extracted"
376
+ return
377
+ new_warnings = [str(w.get("warning", w)) for w in result.warnings]
378
+ if primary:
379
+ report.warnings = new_warnings
380
+ else:
381
+ report.warnings.extend(new_warnings)
382
+
383
+ # project-inferred schema is good evidence, not warehouse-verified truth
384
+ cap = EVIDENCE_CAP.get(model.evidence, "high_confidence")
385
+ if result.qualified_via_schema:
386
+ cap = _worst_trust(cap, "high_confidence")
387
+
388
+ for out_column, upstream_columns in result.contributing.items():
389
+ for src in upstream_columns:
390
+ before = len(self.edges)
391
+ self._append_edge(
392
+ model,
393
+ out_column.lower(),
394
+ src.get("table"),
395
+ src.get("column"),
396
+ src.get("confidence", 0.5),
397
+ src.get("trust_level"),
398
+ "value",
399
+ cap,
400
+ )
401
+ excepted = src.get("except_columns")
402
+ if excepted and out_column == WILDCARD:
403
+ # spellings kept as normalized upstream (quoted
404
+ # mixed-case stays exact, F20); same-pair excepts merge
405
+ # by union so a column excluded in any union branch
406
+ # refuses (F7)
407
+ for e in self.edges[before:]:
408
+ if e.src_column == WILDCARD and e.dst_column == WILDCARD:
409
+ pair = (e.dst_model, e.src_model)
410
+ self._star_excepts[pair] = self._star_excepts.get(
411
+ pair, frozenset()
412
+ ) | frozenset(excepted)
413
+
414
+ # An unqualified filter/join column ("where status = ...") can still be
415
+ # attributed when THIS statement reads from exactly one upstream
416
+ # relation, and a query-local alias ("FROM stg_orders o ... WHERE
417
+ # o.status") resolves through the statement's own alias map. Only this
418
+ # statement's edges count: borrowing an earlier statement's parents
419
+ # claimed the CTAS source wrote the UPDATE's filter column (the
420
+ # review of cycle 6).
421
+ value_parents = {
422
+ e.src_model
423
+ for e in self.edges[edge_start:]
424
+ if e.dst_model == model.name and e.kind == "value"
425
+ }
426
+ sole_parent = next(iter(value_parents)) if len(value_parents) == 1 else None
427
+ if sole_parent is None and not value_parents and not primary:
428
+ relations = self._statement_relations(prepared_sql, model.dialect)
429
+ resolved = {
430
+ found[0]
431
+ for name in relations
432
+ if (found := self._candidates.get(name)) and len(found) == 1
433
+ }
434
+ if len(resolved) == 1:
435
+ sole_parent = next(iter(resolved))
436
+
437
+ def resolve_relation(table: str | None) -> str | None:
438
+ if not table:
439
+ return sole_parent
440
+ return alias_map.get(table.lower(), table)
441
+
442
+ for filt in result.filter_columns:
443
+ table = resolve_relation(filt.table)
444
+ confidence = 0.8 if filt.table else 0.6
445
+ self._append_edge(
446
+ model, "(row filter)", table, filt.column, confidence, None, "filter", cap
447
+ )
448
+ for key in result.window_keys:
449
+ table = resolve_relation(key.table)
450
+ confidence = 0.8 if key.table else 0.6
451
+ self._append_edge(
452
+ model, "(window key)", table, key.column, confidence, None, "window", cap
453
+ )
454
+ for join in result.join_keys:
455
+ for table, column in (
456
+ (join.left_table, join.left_column),
457
+ (join.right_table, join.right_column),
458
+ ):
459
+ self._append_edge(
460
+ model,
461
+ "(join key)",
462
+ resolve_relation(table),
463
+ column,
464
+ 0.8 if table else 0.6,
465
+ None,
466
+ "join",
467
+ cap,
468
+ )
469
+
470
+ # A star chain from an unexpanded upstream (source with no declared
471
+ # columns) still proves the columns this model itself references in
472
+ # predicates or join keys: they must exist upstream and SELECT * passes
473
+ # them through, so they get real value edges instead of vanishing.
474
+ star_parents = {
475
+ e.src_model
476
+ for e in self.edges
477
+ if e.dst_model == model.name
478
+ and e.kind == "value"
479
+ and e.src_column == WILDCARD
480
+ and e.dst_column == WILDCARD
481
+ }
482
+ if star_parents:
483
+ proven = {(f.table, f.column) for f in result.filter_columns}
484
+ for join in result.join_keys:
485
+ proven.add((join.left_table, join.left_column))
486
+ proven.add((join.right_table, join.right_column))
487
+ for table, column in sorted(
488
+ (p for p in proven if p[1]), key=lambda p: (p[0] or "", p[1])
489
+ ):
490
+ if column.lower() in report.columns:
491
+ continue # already attributed explicitly
492
+ candidates, _ = self._resolve_name(
493
+ resolve_relation(table), model.declared_parents, model.path
494
+ )
495
+ for src_model in candidates:
496
+ if src_model not in star_parents:
497
+ continue
498
+ self.edges.append(
499
+ Edge(
500
+ src_model=src_model,
501
+ src_column=column.lower(),
502
+ dst_model=model.name,
503
+ dst_column=column.lower(),
504
+ confidence=0.6,
505
+ trust=_worst_trust("moderate", cap),
506
+ kind="value",
507
+ reason="column proven by this model's own reference; "
508
+ "passed through SELECT *",
509
+ )
510
+ )
511
+ if column.lower() not in report.columns:
512
+ report.columns.append(column.lower())
513
+
514
+ def _append_edge(
515
+ self,
516
+ model: Model,
517
+ out_column: str,
518
+ src_table: str | None,
519
+ src_column: str | None,
520
+ confidence: float,
521
+ engine_trust: str | None,
522
+ kind: str,
523
+ cap: str,
524
+ ) -> None:
525
+ if not src_column:
526
+ return
527
+ candidates, quality = self._resolve_name(src_table, model.declared_parents, model.path)
528
+ if candidates == [model.name] and quality != "exact":
529
+ # a model cannot be its own upstream: when a QUALIFIED reference's
530
+ # suffix is the model's own name (pg_attribute.sql reading
531
+ # pg_catalog.pg_attribute, the staging-wrapper pattern), the SQL
532
+ # named an external relation, and dropping it as a self-loop
533
+ # erases the model's entire lineage
534
+ wrapped = (src_table or "").replace('"', "").replace("`", "").lower()
535
+ if "." in wrapped:
536
+ candidates, quality = [wrapped], "unknown"
537
+ confidence = min(confidence, 0.4)
538
+ reason = ""
539
+ if not candidates:
540
+ if not src_table:
541
+ return
542
+ cleaned = src_table.replace('"', "").replace("`", "").lower()
543
+ if kind in ("filter", "join", "window") and "." not in cleaned:
544
+ # a query-local alias (FROM orders o) we couldn't resolve to a
545
+ # model; inventing a phantom node named "o" would be a lie
546
+ report = self.reports.get(model.name)
547
+ if report is not None:
548
+ report.warnings.append(
549
+ f"could not attribute a {kind} on '{cleaned}.{src_column}' to a model"
550
+ )
551
+ return
552
+ # traced to a table this project doesn't own (external table, CTE
553
+ # remnant). Keep it, but never at confident trust.
554
+ candidates = [cleaned]
555
+ confidence = min(confidence, 0.4)
556
+ disabled = self._all_candidates.get(cleaned) or self._all_candidates.get(
557
+ cleaned.split(".")[-1]
558
+ )
559
+ if disabled:
560
+ # snowplow ships one copy of a model per warehouse and dbt
561
+ # disables all but the target's; a read of the bare name is
562
+ # not an external table and no ingest resolves it
563
+ reason = (
564
+ f"names only disabled models ({len(disabled)}); "
565
+ "enable one or set the dbt target"
566
+ )
567
+ else:
568
+ reason = "source table not found in this project"
569
+ quality = "unknown"
570
+ elif quality == "ambiguous":
571
+ reason = f"name matches {len(candidates)} models; edge added for each"
572
+ elif quality == "suffix":
573
+ reason = "matched by table name only; schema differs or is missing"
574
+ elif quality == "unknown":
575
+ reason = "source table not found in this project"
576
+ trust = _worst_trust(
577
+ engine_trust or "verified",
578
+ _trust_from_confidence(confidence),
579
+ cap,
580
+ "review_required" if quality in ("ambiguous", "suffix", "unknown") else "verified",
581
+ )
582
+ if trust == "review_required" and not reason and len(candidates) == 1:
583
+ # the engine downgraded a read of a column the declared schema
584
+ # lacks, silently; on balboa an ingest of truncated column lists
585
+ # took review links from 17 to 106 with no reason on any of them
586
+ if self._models_by_name is None:
587
+ self._models_by_name = {
588
+ m.name: m for m in [*self.project.models, *self.project.sources]
589
+ }
590
+ source = self._models_by_name.get(candidates[0])
591
+ declared = {c.lower() for c in source.declared_columns} if source else set()
592
+ if declared and src_column.lower() not in declared and src_column != WILDCARD:
593
+ reason = (
594
+ f"'{src_column}' is not in the declared columns of {candidates[0]}; "
595
+ "the declared or ingested column list may be incomplete"
596
+ )
597
+ for src_model in candidates:
598
+ if src_model == model.name and not model.self_read:
599
+ continue # self-loop from CTE resolution
600
+ if (
601
+ src_model == model.name
602
+ and kind == "value"
603
+ and src_column.lower() == out_column.lower()
604
+ ):
605
+ # 'names depends on its own prior names' is a tautology, not
606
+ # lineage; only CROSS-column prior-state edges (address <-
607
+ # house) carry information (nycdb business_addrs)
608
+ continue
609
+ if src_model == model.name and not reason:
610
+ reason = "prior state of the same table"
611
+ self.edges.append(
612
+ Edge(
613
+ src_model=src_model,
614
+ src_column=src_column.lower(),
615
+ dst_model=model.name,
616
+ dst_column=out_column if out_column.startswith("(") else out_column.lower(),
617
+ confidence=round(confidence, 2),
618
+ trust=trust,
619
+ kind=kind,
620
+ reason=reason,
621
+ )
622
+ )
623
+
624
+ def _add_downstream(self) -> None:
625
+ """Append the declared-as-code BI edges (dbt metrics, exposures) so a
626
+ model column traces to the metric and dashboard it feeds. These are
627
+ ordinary edges with a trust label, so breaks() reaches them for free."""
628
+ conf = {"verified": 1.0, "high_confidence": 0.85, "moderate": 0.6, "review_required": 0.4}
629
+ for d in getattr(self.project, "downstream", []):
630
+ cands = self._candidates.get(d.src_model.lower())
631
+ src = cands[0] if cands else d.src_model
632
+ self.edges.append(
633
+ Edge(
634
+ src_model=src,
635
+ src_column=WILDCARD if d.src_column == "*" else d.src_column,
636
+ dst_model=d.dst_model,
637
+ dst_column=WILDCARD if d.dst_column == "*" else d.dst_column,
638
+ confidence=conf.get(d.trust, 0.6),
639
+ trust=d.trust,
640
+ kind="value",
641
+ reason=d.reason,
642
+ )
643
+ )
644
+
645
+ def _statement_relations(self, sql: str, dialect: str | None = None) -> set[str]:
646
+ """Lowercased real relations one statement reads (tables minus its
647
+ own CTE names). The anchor for an extra statement's unqualified
648
+ filter columns when it produced no value edges of its own."""
649
+ import sqlglot
650
+ from sqlglot import exp
651
+
652
+ from ripple.engine.column_ref import temp_marked_name
653
+
654
+ try:
655
+ parsed = sqlglot.parse_one(sql, dialect=dialect or self.project.dialect)
656
+ except Exception:
657
+ return set()
658
+ ctes = {c.alias_or_name.lower() for c in parsed.find_all(exp.CTE)}
659
+ return {
660
+ temp_marked_name(t).lower()
661
+ for t in parsed.find_all(exp.Table)
662
+ if t.name and t.name.lower() not in ctes
663
+ }
664
+
665
+ def _alias_map(self, sql: str, dialect: str | None = None) -> dict[str, str]:
666
+ """Query-local alias -> relation name, resolving single-source CTEs
667
+ through to their base table. 'FROM stg_orders o JOIN pay p' where pay
668
+ is a CTE reading stg_payments yields {o: stg_orders, p: stg_payments}.
669
+ """
670
+ import sqlglot
671
+ from sqlglot import exp
672
+
673
+ try:
674
+ parsed = sqlglot.parse_one(sql, dialect=dialect or self.project.dialect)
675
+ except Exception:
676
+ return {}
677
+
678
+ from ripple.engine.column_ref import qualified_table_name
679
+ from ripple.engine.preprocess import restore_jinja_dots
680
+
681
+ def qualified(t: exp.Table) -> str:
682
+ # qualifiers kept for the same reason as scope.register_table: a
683
+ # filter/join/window column on a model named after the table it
684
+ # wraps must not resolve back into the model and vanish;
685
+ # placeholder qualifiers dropped for the same reason as
686
+ # column_ref.qualified_table_name
687
+ return qualified_table_name(t).lower()
688
+
689
+ cte_sources: dict[str, set[tuple[str, str]]] = {}
690
+ cte_names = set()
691
+ for cte in parsed.find_all(exp.CTE):
692
+ name = restore_jinja_dots(cte.alias_or_name.lower())
693
+ cte_names.add(name)
694
+ cte_sources[name] = {
695
+ (restore_jinja_dots(t.name.lower()), qualified(t))
696
+ for t in cte.this.find_all(exp.Table)
697
+ }
698
+ aliases: dict[str, str] = {}
699
+ for table in parsed.find_all(exp.Table):
700
+ alias = restore_jinja_dots((table.alias_or_name or "").lower())
701
+ name = restore_jinja_dots(table.name.lower())
702
+ if not alias or alias == name:
703
+ full = qualified(table)
704
+ if full != name and name not in cte_names:
705
+ aliases.setdefault(name, full)
706
+ continue
707
+ target = qualified(table)
708
+ # a CTE alias resolves through the CTE when it reads one real table
709
+ if name in cte_names:
710
+ real = {q for (bare, q) in cte_sources.get(name, set()) if bare not in cte_names}
711
+ if len(real) == 1:
712
+ target = next(iter(real))
713
+ else:
714
+ continue # multi-source CTE: leave unresolved, warn path handles it
715
+ aliases[alias] = target
716
+ return aliases
717
+
718
+ def _resolve_name(
719
+ self, table: str | None, scope_parents: set[str], ref_path: str = ""
720
+ ) -> tuple[list[str], str]:
721
+ """Resolve a table name from SQL to project models.
722
+
723
+ Returns (candidates, quality) where quality is one of exact / scoped /
724
+ ambiguous / suffix / unknown. Ambiguity returns ALL candidates; the
725
+ caller emits an edge per candidate at review_required rather than
726
+ silently binding one.
727
+ """
728
+ if not table:
729
+ return [], "unknown"
730
+ cleaned = table.replace('"', "").replace("`", "").lower()
731
+ scope = {
732
+ resolved[0] for p in scope_parents if (resolved := self._candidates.get(p.lower()))
733
+ }
734
+ candidates = self._candidates.get(cleaned)
735
+ if not candidates and "{" in cleaned:
736
+ # an ingest names the relation without its deploy-time qualifier;
737
+ # a templated NAME comes back unchanged and stays unresolved
738
+ stripped = _strip_placeholders(cleaned)
739
+ if stripped and "{" not in stripped:
740
+ if self._models_by_name is None:
741
+ self._models_by_name = {
742
+ m.name: m for m in [*self.project.models, *self.project.sources]
743
+ }
744
+ # only a declared or ingested table answers to the bare
745
+ # spelling; a model with SQL of the same name is a different
746
+ # relation ({{params.dataset_name_raw}}.blocks is not the
747
+ # derived blocks model, round 11)
748
+ candidates = [
749
+ c
750
+ for c in self._candidates.get(stripped, [])
751
+ if not (
752
+ self._models_by_name.get(c) or Model(name="", sql="", path="")
753
+ ).sql.strip()
754
+ ] or None
755
+ if candidates:
756
+ if len(candidates) == 1:
757
+ return [candidates[0]], "exact"
758
+ # locality before the scope shortcut: sql-dir declared parents are
759
+ # the bare names themselves, so scope would collapse to the first
760
+ # candidate and bind another dump's table (no-op outside sql-dir)
761
+ local = self._prefer_local(candidates, ref_path)
762
+ if len(local) == 1:
763
+ return local, "scoped"
764
+ in_scope = [c for c in local if c in scope]
765
+ if len(in_scope) == 1:
766
+ return in_scope, "scoped"
767
+ return (in_scope or local), "ambiguous"
768
+ last = cleaned.split(".")[-1]
769
+ if last != cleaned:
770
+ candidates = self._candidates.get(last)
771
+ if candidates:
772
+ candidates = [c for c in candidates if self._may_fold_qualified_read(c, cleaned)]
773
+ if candidates:
774
+ local = self._prefer_local(candidates, ref_path)
775
+ if len(local) == 1:
776
+ return local, "scoped"
777
+ in_scope = [c for c in local if c in scope]
778
+ if len(in_scope) == 1:
779
+ return in_scope, "scoped"
780
+ return (in_scope or local), "suffix"
781
+ return [], "unknown"
782
+
783
+ def _may_fold_qualified_read(self, name: str, read: str) -> bool:
784
+ """In plain SQL the file is the spelling authority: a relation created
785
+ bare never owns a schema-qualified read, and qualified spellings must
786
+ agree (round-5 refuse-to-fold, cycle-6 agreement rule). dbt schema
787
+ qualifiers are deploy artifacts, so dbt-evidence models keep folding.
788
+
789
+ A dotted filename stem counts as an agreeing qualified spelling: the
790
+ author named eicu_crd.patient.sql with the qualifier, so a read of
791
+ eicu_crd.patient folds to its relation deliberately (cycle-11 review,
792
+ F2: pinned as the intended design, not a bypass)."""
793
+ if self._models_by_name is None:
794
+ self._models_by_name = {
795
+ m.name: m for m in [*self.project.models, *self.project.sources]
796
+ }
797
+ model = self._models_by_name.get(name)
798
+ if model is None or model.evidence != "plain_sql":
799
+ return True
800
+ spellings = {
801
+ s.lower() for s in (model.name, *model.aliases, *model.stem_aliases) if "." in s
802
+ }
803
+ read_suffixes = _name_suffixes(read)
804
+ return any(s in read_suffixes or read in _name_suffixes(s) for s in spellings)
805
+
806
+ def _reads_as_written(self, model: Model) -> frozenset[str] | None:
807
+ """Lowercased as-written relation spellings in a model's SQL, or None
808
+ when nothing parsed (no evidence, so no refusal downstream)."""
809
+ if self._reads_cache is None:
810
+ self._reads_cache = {}
811
+ cached = self._reads_cache.get(model.uid, "unset")
812
+ if cached != "unset":
813
+ return cached
814
+ import sqlglot
815
+ from sqlglot import exp
816
+
817
+ from ripple.engine.column_ref import qualified_table_name
818
+
819
+ reads: set[str] = set()
820
+ parsed_any = False
821
+ for sql in (model.sql, *model.extra_sqls):
822
+ if not sql.strip():
823
+ continue
824
+ try:
825
+ parsed = sqlglot.parse_one(sql, read=model.dialect or self.project.dialect)
826
+ except Exception:
827
+ continue
828
+ parsed_any = True
829
+ for t in parsed.find_all(exp.Table):
830
+ if t.name:
831
+ reads.add((qualified_table_name(t) or t.name).lower())
832
+ result = frozenset(reads) if parsed_any else None
833
+ self._reads_cache[model.uid] = result
834
+ return result
835
+
836
+ def _parent_binds(self, model: Model, parent: str, resolved: str) -> bool:
837
+ """The spelling-authority rule applied to a DECLARED parent link.
838
+
839
+ Declared parents are recorded bare, so the sites that resolve them
840
+ directly (parent schemas, fallback edges, unresolved reporting) were
841
+ bypassing _may_fold_qualified_read: a SELECT * FROM eicu_crd.patient
842
+ inherited the bare-created patient's columns and fallback edges
843
+ reconnected it (cycle-11 review, F3/F4). The read spellings in the
844
+ model's own SQL are the evidence; the parent binds only when some
845
+ read of it may fold to the resolved relation."""
846
+ if self._models_by_name is None:
847
+ self._models_by_name = {
848
+ m.name: m for m in [*self.project.models, *self.project.sources]
849
+ }
850
+ target = self._models_by_name.get(resolved)
851
+ if target is None or target.evidence != "plain_sql":
852
+ return True
853
+ reads = self._reads_as_written(model)
854
+ if reads is None:
855
+ return True
856
+ names = {s.lower() for s in (target.name, *target.aliases, *target.stem_aliases)}
857
+ bare = parent.lower().split(".")[-1]
858
+ relevant = [r for r in reads if r == parent.lower() or r.split(".")[-1] == bare]
859
+ if not relevant:
860
+ return True
861
+ return any(r in names or self._may_fold_qualified_read(resolved, r) for r in relevant)
862
+
863
+ def _refused_read_of(self, model: Model, parent: str) -> str | None:
864
+ """The as-written qualified spelling behind a refused parent link, so
865
+ the external is reported as the SQL cites it, not as the bare name."""
866
+ reads = self._reads_as_written(model) or frozenset()
867
+ bare = parent.lower().split(".")[-1]
868
+ dotted = sorted(r for r in reads if "." in r and r.split(".")[-1] == bare)
869
+ return dotted[0] if dotted else None
870
+
871
+ def _prefer_local(self, candidates: list[str], ref_path: str) -> list[str]:
872
+ """Same file first, then same subdirectory: in a sql dir one dump is
873
+ one database, so a reference never silently binds another dump's file."""
874
+ if len(candidates) < 2 or not ref_path or self.project.mode != "sql-dir":
875
+ return candidates
876
+ if self._models_by_name is None:
877
+ self._models_by_name = {
878
+ m.name: m for m in [*self.project.models, *self.project.sources]
879
+ }
880
+
881
+ def path_of(name: str) -> str:
882
+ model = self._models_by_name.get(name)
883
+ return model.path if model else ""
884
+
885
+ same_file = [c for c in candidates if path_of(c) == ref_path]
886
+ if same_file:
887
+ return same_file
888
+ top = ref_path.split("/", 1)[0]
889
+ same_dir = [c for c in candidates if path_of(c) and path_of(c).split("/", 1)[0] == top]
890
+ return same_dir or candidates
891
+
892
+ def _fallback_edges(self) -> None:
893
+ """A model that yielded nothing still sits between its declared parents
894
+ and everything that reads it. Wildcard edges keep that path visible;
895
+ review_required says exactly how much to trust it."""
896
+ for model in self.project.models:
897
+ report = self.reports.get(model.name)
898
+ if report is None or report.status not in ("failed", "star_only"):
899
+ continue
900
+ connected = False
901
+ for parent in model.declared_parents:
902
+ resolved = self._candidates.get(parent.lower())
903
+ if not resolved:
904
+ continue
905
+ # a bare declared parent that resolves to the model itself is
906
+ # the self-name-collision shape; a wildcard self-edge would
907
+ # assert the model feeds itself. A refused spelling must not
908
+ # reconnect here either: the analysis path already kept the
909
+ # as-written external edge (cycle-11, F4)
910
+ non_self = [
911
+ c for c in resolved if c != model.name and self._parent_binds(model, parent, c)
912
+ ]
913
+ if not non_self:
914
+ continue
915
+ self.edges.append(
916
+ Edge(
917
+ src_model=non_self[0],
918
+ src_column=WILDCARD,
919
+ dst_model=model.name,
920
+ dst_column=WILDCARD,
921
+ confidence=0.3,
922
+ trust="review_required",
923
+ kind="value",
924
+ reason="model could not be analyzed; linked via declared dependency",
925
+ )
926
+ )
927
+ connected = True
928
+ if connected and report.status == "failed":
929
+ report.status = "fallback"
930
+
931
+ def _dedupe(self) -> None:
932
+ best: dict[tuple, Edge] = {}
933
+ for edge in self.edges:
934
+ existing = best.get(edge.key())
935
+ if existing is None or edge.confidence > existing.confidence:
936
+ best[edge.key()] = edge
937
+ self.edges = list(best.values())
938
+
939
+ def _index(self) -> None:
940
+ self._down = defaultdict(list)
941
+ self._down_by_model = defaultdict(list)
942
+ self._up = defaultdict(list)
943
+ for edge in self.edges:
944
+ self._down[(edge.src_model, edge.src_column)].append(edge)
945
+ self._down_by_model[edge.src_model].append(edge)
946
+ self._up[(edge.dst_model, edge.dst_column)].append(edge)
947
+
948
+ # ---- queries ----
949
+
950
+ def _model_owns_column(self, name: str, column: str) -> bool:
951
+ if self._down.get((name, column)) or self._up.get((name, column)):
952
+ return True
953
+ report = self.reports.get(name)
954
+ if report and any(c.lower() == column for c in report.columns):
955
+ return True
956
+ # sources count: a declared source schema is ownership evidence for
957
+ # multi-relation star resolution (cycle-12, F9)
958
+ for m in [*self.project.models, *self.project.sources]:
959
+ if m.name == name:
960
+ return any(c.lower() == column for c in m.declared_columns)
961
+ return False
962
+
963
+ def _pick_stem_claimant(self, canonical: list[str], column: str) -> str | None:
964
+ """The one stem claimant the asked column picks, or None to refuse.
965
+
966
+ A claimant whose outputs are unenumerated (a star passthrough) may
967
+ well own the column too; picking the visible owner past it would
968
+ answer with false confidence (the cycle-6 review, the
969
+ elimination rule's wildcard guard applied to the pick). The
970
+ benchmark harness resolves case models through this same method."""
971
+ wanted = column.lower()
972
+ owners = [c for c in canonical if self._model_owns_column(c, wanted)]
973
+ veiled = [
974
+ c
975
+ for c in canonical
976
+ if c not in owners
977
+ and (report := self.reports.get(c)) is not None
978
+ and WILDCARD in report.columns
979
+ ]
980
+ if len(owners) == 1 and not veiled:
981
+ return owners[0]
982
+ return None
983
+
984
+ def _also_matches(self, asked: str, chosen: str) -> list[str]:
985
+ """The other models a spelling names when it was settled by being one
986
+ model's own name; the answer says so instead of picking silently."""
987
+ others = [c for c in self._candidates.get(asked.lower(), []) if c != chosen]
988
+ return sorted(self._askable_name(c) for c in others)
989
+
990
+ def _askable_name(self, canonical: str) -> str:
991
+ """The spelling a person would type for a model: its shortest dotted
992
+ alias (telemetry_derived.events_v1), never the path-mangled canonical
993
+ name a collision assigned (sql__moz-fx...__events_v1__view__events_v1)."""
994
+ if self._models_by_name is None:
995
+ self._models_by_name = {
996
+ m.name: m for m in [*self.project.models, *self.project.sources]
997
+ }
998
+ target = self._models_by_name.get(canonical)
999
+ dotted = sorted((a for a in (target.aliases if target else ()) if "." in a), key=len)
1000
+ return dotted[0] if dotted else canonical
1001
+
1002
+ def _external_targets(self) -> dict[str, str]:
1003
+ """lowercased spelling -> as-written name of every table the graph
1004
+ reads but does not define, so a question about one still answers."""
1005
+ if getattr(self, "_externals", None) is None:
1006
+ self._externals = {
1007
+ e.src_model.lower(): e.src_model
1008
+ for e in self.edges
1009
+ if e.src_model not in self.reports
1010
+ }
1011
+ return self._externals
1012
+
1013
+ def _require_target(self, model: str, column: str) -> tuple[str, str]:
1014
+ canonical = self._candidates.get(model.lower())
1015
+ if not canonical:
1016
+ # a dbt-disabled model is still askable: its own analysis exists
1017
+ canonical = self._all_candidates.get(model.lower())
1018
+ if not canonical:
1019
+ external = self._external_targets().get(model.lower())
1020
+ if external is not None:
1021
+ canonical = [external]
1022
+ if not canonical:
1023
+ known = list(self.reports.keys())
1024
+ close = difflib.get_close_matches(model, known, n=3)
1025
+ raise UnknownTarget(f"no model named '{model}'", close)
1026
+ if len(canonical) > 1:
1027
+ # a spelling that IS one model's own name is that model, even
1028
+ # when another model carries it as an alias (postgresDBSamples:
1029
+ # base table Employee beside a path-qualified derived employee)
1030
+ exact = [c for c in canonical if c.lower() == model.lower()]
1031
+ if len(exact) == 1:
1032
+ canonical = exact
1033
+ if len(canonical) > 1:
1034
+ # a file stem naming several relations is DELIBERATELY ambiguous
1035
+ # and only until the column picks one (webtool_tables.invCount,
1036
+ # holdout round 5). Any other shared spelling, like two files
1037
+ # creating the same table, is a real collision: picking whichever
1038
+ # owns the asked column would hand the user the wrong model's
1039
+ # blast radius silently (the review of PR #28)
1040
+ picked = (
1041
+ self._pick_stem_claimant(canonical, column)
1042
+ if model.lower() in self._stem_aliases
1043
+ else None
1044
+ )
1045
+ if picked is None:
1046
+ raise UnknownTarget(
1047
+ f"'{model}' names {len(canonical)} different models; ask by full name",
1048
+ sorted(self._askable_name(c) for c in canonical),
1049
+ )
1050
+ canonical = [picked]
1051
+ name = canonical[0]
1052
+ report = self.reports.get(name)
1053
+ column = column.lower()
1054
+ if report is None or (report.status != "ok" and not report.columns):
1055
+ # outside the project, or defined without columns (CREATE TABLE
1056
+ # ... LIKE in redshift-utils): the only columns Ripple knows are
1057
+ # the ones its models read. "0 impacted, complete" for a column
1058
+ # it never saw read as a clean bill of health (balboa walk)
1059
+ seen = sorted({c for (m, c) in (self._down or {}) if m == name and c != WILDCARD})
1060
+ if column not in seen:
1061
+ raise UnknownTarget(
1062
+ f"'{name}' is outside this project; Ripple only knows the "
1063
+ f"columns its models read from it",
1064
+ seen, # the full list: an agent copies it into ingest_schema, and
1065
+ # mattermost's telemetry event table is read on 200+ columns
1066
+ )
1067
+ if report and report.status == "ok" and column not in report.columns:
1068
+ has_edges = bool(self._down.get((name, column)) or self._up.get((name, column)))
1069
+ if not has_edges:
1070
+ close = difflib.get_close_matches(column, report.columns, n=3)
1071
+ raise UnknownTarget(f"'{name}' has no column '{column}'", close)
1072
+ return name, column
1073
+
1074
+ def breaks(self, model: str, column: str, max_depth: int = 25) -> dict:
1075
+ """Everything downstream of model.column.
1076
+
1077
+ Trust is path-aware: a hit reached only through an uncertain hop is
1078
+ reported at that path's worst trust, not the last edge's. Row-level
1079
+ impact (this column used in a downstream WHERE/JOIN) is reported per
1080
+ model rather than fanned out, so counts stay column-true.
1081
+ """
1082
+ name, column = self._require_target(model, column)
1083
+ also = self._also_matches(model, name)
1084
+ # worst path trust each node has been expanded with; a node is
1085
+ # re-expanded when a WORSE path reaches it, so downgrades propagate
1086
+ # to everything downstream of a convergence point (fixed point,
1087
+ # each node expands at most len(TRUST_ORDER) times)
1088
+ expanded_at: dict[tuple[str, str], str] = {}
1089
+ seen: dict[tuple[str, str], dict] = {}
1090
+ row_impact: dict[str, dict] = {}
1091
+ truncated = False
1092
+ frontier: deque = deque([((name, column), 0, "verified")])
1093
+ while frontier:
1094
+ node, depth, path_trust = frontier.popleft()
1095
+ previous = expanded_at.get(node)
1096
+ if previous is not None and TRUST_ORDER.index(path_trust) >= TRUST_ORDER.index(
1097
+ previous
1098
+ ):
1099
+ continue
1100
+ if depth >= max_depth:
1101
+ truncated = True
1102
+ continue
1103
+ expanded_at[node] = path_trust
1104
+ node_model, node_column = node
1105
+ outgoing = list(self._down.get(node, []))
1106
+ if node_column == WILDCARD:
1107
+ # a wildcard node means "some unknown column of this model":
1108
+ # anything reading any of its columns might be affected
1109
+ outgoing = self._down_by_model.get(node_model, [])
1110
+ else:
1111
+ outgoing += self._down.get((node_model, WILDCARD), [])
1112
+ for edge in outgoing:
1113
+ through_wildcard = WILDCARD in (edge.src_column, edge.dst_column, node_column)
1114
+ trust = _worst_trust(
1115
+ path_trust, edge.trust, *(["review_required"] if through_wildcard else [])
1116
+ )
1117
+ if edge.kind in ("filter", "join", "window"):
1118
+ entry = row_impact.setdefault(
1119
+ edge.dst_model,
1120
+ {"model": edge.dst_model, "uses": [], "trust": trust},
1121
+ )
1122
+ entry["uses"].append(f"{edge.src_model}.{edge.src_column} ({edge.kind})")
1123
+ entry["trust"] = _worst_trust(entry["trust"], trust)
1124
+ continue
1125
+ hit = {
1126
+ "model": edge.dst_model,
1127
+ "column": edge.dst_column,
1128
+ "via": f"{edge.src_model}.{edge.src_column}",
1129
+ "kind": edge.kind,
1130
+ "trust": trust,
1131
+ "edge_trust": edge.trust,
1132
+ "confidence": edge.confidence,
1133
+ "depth": depth + 1,
1134
+ }
1135
+ if edge.reason:
1136
+ hit["reason"] = edge.reason
1137
+ key = (edge.dst_model, edge.dst_column)
1138
+ previous = seen.get(key)
1139
+ if previous is None:
1140
+ seen[key] = hit
1141
+ else:
1142
+ # a second path to the same column never upgrades trust,
1143
+ # and an uncertain path is never hidden by a confident one
1144
+ previous["trust"] = _worst_trust(previous["trust"], trust)
1145
+ if not edge.dst_column.startswith("("):
1146
+ frontier.append(((edge.dst_model, edge.dst_column), depth + 1, trust))
1147
+ # stable: same-model hits keep the order the walk met them in (select order)
1148
+ hits = sorted(
1149
+ seen.values(),
1150
+ key=lambda h: (h["depth"], TRUST_ORDER[::-1].index(h["edge_trust"]), h["model"]),
1151
+ )
1152
+ by_model: dict[str, list[dict]] = defaultdict(list)
1153
+ for hit in hits:
1154
+ # the numeric score is a branch constant, not a probability;
1155
+ # public output carries the categorical label and the reason
1156
+ hit.pop("confidence", None)
1157
+ by_model[hit["model"]].append(hit)
1158
+ return {
1159
+ "source": {"model": name, "column": column},
1160
+ "impacted_columns": len(hits),
1161
+ "impacted_models": len(by_model),
1162
+ "review_required": sum(1 for h in hits if h["trust"] == "review_required"),
1163
+ "by_model": dict(by_model),
1164
+ "row_level_impact": sorted(row_impact.values(), key=lambda r: r["model"]),
1165
+ "row_impacted_models": len(row_impact),
1166
+ "complete": not truncated,
1167
+ **({"truncated_at_depth": max_depth} if truncated else {}),
1168
+ **({"also_matches": also} if also else {}),
1169
+ }
1170
+
1171
+ def breaks_all(self, model: str, max_depth: int = 25) -> dict:
1172
+ """Blast radius of the whole model changing at once (a predicate or
1173
+ join change alters every row, so every output column is a source)."""
1174
+ canonical = self._candidates.get(model.lower())
1175
+ if not canonical:
1176
+ raise UnknownTarget(
1177
+ f"no model named '{model}'",
1178
+ difflib.get_close_matches(model, list(self.reports), n=3),
1179
+ )
1180
+ name = canonical[0]
1181
+ report = self.reports.get(name)
1182
+ columns = [c for c in (report.columns if report else []) if c != WILDCARD]
1183
+ merged: dict = {"impacted": {}, "row": {}}
1184
+ complete = True
1185
+ for column in columns or [WILDCARD]:
1186
+ try:
1187
+ result = self.breaks(name, column, max_depth=max_depth)
1188
+ except UnknownTarget:
1189
+ continue
1190
+ complete = complete and result.get("complete", True)
1191
+ for hits in result["by_model"].values():
1192
+ for hit in hits:
1193
+ key = (hit["model"], hit["column"])
1194
+ existing = merged["impacted"].get(key)
1195
+ if existing is None:
1196
+ merged["impacted"][key] = hit
1197
+ else:
1198
+ existing["trust"] = _worst_trust(existing["trust"], hit["trust"])
1199
+ for row in result["row_level_impact"]:
1200
+ merged["row"][row["model"]] = row
1201
+ hits = list(merged["impacted"].values())
1202
+ by_model: dict[str, list[dict]] = defaultdict(list)
1203
+ for hit in hits:
1204
+ by_model[hit["model"]].append(hit)
1205
+ return {
1206
+ "source": {"model": name, "column": "(all columns)"},
1207
+ "impacted_columns": len(hits),
1208
+ "impacted_models": len(by_model),
1209
+ "review_required": sum(1 for h in hits if h["trust"] == "review_required"),
1210
+ "by_model": dict(by_model),
1211
+ "row_level_impact": sorted(merged["row"].values(), key=lambda r: r["model"]),
1212
+ "row_impacted_models": len(merged["row"]),
1213
+ "complete": complete,
1214
+ }
1215
+
1216
+ def trace(self, model: str, column: str, max_depth: int = 25) -> dict:
1217
+ """Where model.column comes from, hop by hop. Trust accumulates along
1218
+ the path from the target, same contract as breaks()."""
1219
+ name, column = self._require_target(model, column)
1220
+ expanded_at: dict[tuple[str, str], str] = {}
1221
+ hops: dict[tuple, dict] = {}
1222
+ truncated = False
1223
+ frontier: deque = deque([((name, column), 0, "verified")])
1224
+ while frontier:
1225
+ node, depth, path_trust = frontier.popleft()
1226
+ previous = expanded_at.get(node)
1227
+ if previous is not None and TRUST_ORDER.index(path_trust) >= TRUST_ORDER.index(
1228
+ previous
1229
+ ):
1230
+ continue
1231
+ if depth >= max_depth:
1232
+ truncated = True
1233
+ continue
1234
+ expanded_at[node] = path_trust
1235
+ node_model, node_column = node
1236
+ incoming = list(self._up.get(node, []))
1237
+ if node_column != WILDCARD:
1238
+ incoming += self._up.get((node_model, WILDCARD), [])
1239
+ for edge in incoming:
1240
+ if edge.kind != "value":
1241
+ continue
1242
+ trust = _worst_trust(path_trust, edge.trust)
1243
+ key = (edge.src_model, edge.src_column, edge.dst_model, edge.dst_column)
1244
+ existing = hops.get(key)
1245
+ if existing is None:
1246
+ hop = {
1247
+ "model": edge.src_model,
1248
+ "column": edge.src_column,
1249
+ "feeds": f"{edge.dst_model}.{edge.dst_column}",
1250
+ "trust": trust,
1251
+ "edge_trust": edge.trust,
1252
+ "confidence": edge.confidence,
1253
+ "depth": depth + 1,
1254
+ }
1255
+ if edge.reason:
1256
+ hop["reason"] = edge.reason
1257
+ hops[key] = hop
1258
+ else:
1259
+ existing["trust"] = _worst_trust(existing["trust"], trust)
1260
+ frontier.append(((edge.src_model, edge.src_column), depth + 1, trust))
1261
+ ordered = sorted(hops.values(), key=lambda h: (h["depth"], h["model"], h["column"]))
1262
+ for hop in ordered:
1263
+ hop.pop("confidence", None)
1264
+ return {
1265
+ "target": {"model": name, "column": column},
1266
+ "upstream": ordered,
1267
+ "complete": not truncated,
1268
+ **({"truncated_at_depth": max_depth} if truncated else {}),
1269
+ }
1270
+
1271
+ def resolve_star_column(self, model: str, column: str) -> list[tuple[str, str]]:
1272
+ """Resolve a named column through the model's star projection:
1273
+ [(source_relation, column)] when the star provably carries it,
1274
+ [] when it is excepted or ownership is unknowable."""
1275
+ from ripple.star_resolution import resolve_star_column
1276
+
1277
+ return resolve_star_column(self, model, column)
1278
+
1279
+ # ---- export ----
1280
+
1281
+ def stats(self) -> dict:
1282
+ by_status = defaultdict(list)
1283
+ for report in self.reports.values():
1284
+ by_status[report.status].append(report.name)
1285
+ failed = {
1286
+ name: self.reports[name].error
1287
+ for name in [
1288
+ *by_status.get("failed", []),
1289
+ *by_status.get("fallback", []),
1290
+ *by_status.get("timed_out", []),
1291
+ ]
1292
+ }
1293
+ return {
1294
+ "mode": self.project.mode,
1295
+ "dialect": self.project.dialect,
1296
+ "models": len(self.project.models),
1297
+ "sources": len(self.project.sources),
1298
+ "ok": len(by_status.get("ok", [])),
1299
+ "star_only": len(by_status.get("star_only", [])),
1300
+ "fallback": len(by_status.get("fallback", [])),
1301
+ "timed_out": len(by_status.get("timed_out", [])),
1302
+ "failed": len(by_status.get("failed", [])) + len(by_status.get("no_sql", [])),
1303
+ "failure_reasons": dict(list(failed.items())[:20]),
1304
+ "edges": len(self.edges),
1305
+ "review_required_edges": sum(1 for e in self.edges if e.trust == "review_required"),
1306
+ "verified_edges": sum(1 for e in self.edges if e.trust == "verified"),
1307
+ }
1308
+
1309
+ def unresolved_tables(self) -> list[dict]:
1310
+ """External relations referenced by models but with no known columns:
1311
+ names that resolve to nothing, plus sources and base tables whose
1312
+ columns nobody has declared. Sorted by how much coverage they block."""
1313
+ if self._models_by_name is None:
1314
+ self._models_by_name = {
1315
+ m.name: m for m in [*self.project.models, *self.project.sources]
1316
+ }
1317
+ import re
1318
+
1319
+ cte_re = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]*)\s+as\s*\(", re.I)
1320
+ entries: dict[str, dict] = {}
1321
+ for model in self.project.models:
1322
+ report = self.reports.get(model.name)
1323
+ status = report.status if report else "failed"
1324
+ own_ctes = {m.lower() for m in cte_re.findall(model.sql)} if model.sql else set()
1325
+ for parent in model.declared_parents:
1326
+ if "{" in parent or any(c.isspace() for c in parent):
1327
+ continue # jinja leftover, not an ingestable table name
1328
+ if parent.lower() in own_ctes:
1329
+ continue # a CTE of this model's own script, not a table
1330
+ resolved = self._candidates.get(parent.lower()) or self._all_candidates.get(
1331
+ parent.lower()
1332
+ )
1333
+ as_written = None
1334
+ if resolved and not self._parent_binds(model, parent, resolved[0]):
1335
+ # refused spelling: the read names an external relation,
1336
+ # reported as the SQL cites it (cycle-11, F3)
1337
+ name = self._refused_read_of(model, parent) or parent.lower()
1338
+ if "{" in name:
1339
+ stripped = _strip_placeholders(name)
1340
+ if "{" not in stripped:
1341
+ as_written, name = name, stripped or parent.lower()
1342
+ known = self._candidates.get(name.lower(), [])
1343
+ if len(known) == 1 and (
1344
+ (self.reports.get(known[0]) or ModelReport(name="")).status == "ok"
1345
+ or (
1346
+ self._models_by_name.get(known[0]) or Model(name="", sql="", path="")
1347
+ ).declared_columns
1348
+ ):
1349
+ continue # the spelling the SQL wrote was ingested; nothing to do
1350
+ qualified = name if "." in name else None
1351
+ elif resolved:
1352
+ target = self._models_by_name.get(resolved[0])
1353
+ target_report = self.reports.get(resolved[0])
1354
+ if target_report is not None and target_report.status == "ok":
1355
+ continue
1356
+ if target is None or target.sql.strip() or target.declared_columns:
1357
+ continue # a real model that failed is not a schema gap
1358
+ name = resolved[0]
1359
+ qualified = max((a for a in target.aliases if "." in a), key=len, default=None)
1360
+ else:
1361
+ name = parent.lower()
1362
+ qualified = name if "." in name else None
1363
+ as_written = None
1364
+ if qualified is None:
1365
+ # the declared parent is bare; the SQL may have
1366
+ # written it qualified, and that spelling is the one
1367
+ # an ingest must use (balboa's slow_query reads
1368
+ # snowflake_sample_data.tpch_sf1.region)
1369
+ written = [
1370
+ _strip_placeholders(r)
1371
+ for r in (self._reads_as_written(model) or ())
1372
+ if r.split(".")[-1] == name
1373
+ ]
1374
+ written = [w for w in written if "." in w]
1375
+ qualified = min(written, key=len) if written else None
1376
+ as_written = next(
1377
+ (
1378
+ r
1379
+ for r in (self._reads_as_written(model) or ())
1380
+ if "{" in r and r.split(".")[-1] == name
1381
+ ),
1382
+ None,
1383
+ )
1384
+ entry = entries.setdefault(
1385
+ name,
1386
+ {
1387
+ "table": name,
1388
+ **({"qualified_name": qualified} if qualified else {}),
1389
+ **({"as_written": as_written} if as_written else {}),
1390
+ "referencing_models": 0,
1391
+ "blocked_models": 0,
1392
+ "statuses": {},
1393
+ },
1394
+ )
1395
+ entry["referencing_models"] += 1
1396
+ if status in ("star_only", "failed", "fallback", "no_sql"):
1397
+ entry["blocked_models"] += 1
1398
+ entry["statuses"][status] = entry["statuses"].get(status, 0) + 1
1399
+ # tables only the engine saw: a raw table read by name inside a dbt
1400
+ # model is no declared parent, yet its edges say "not found"
1401
+ # (balboa's _airbyte_raw_zip_coordinates)
1402
+ seen_pairs: set[tuple[str, str]] = set()
1403
+ for edge in self.edges:
1404
+ if edge.src_model in self.reports or "not found" not in edge.reason:
1405
+ continue # disabled-only names carry their own reason and are skipped
1406
+ written = edge.src_model.lower()
1407
+ key = _strip_placeholders(written) if "{" in written else written
1408
+ if not key or "{" in key or any(ch.isspace() for ch in key):
1409
+ continue # a templated name or jinja remnant: cited as written, not ingestable
1410
+ if key in entries or (key, edge.dst_model) in seen_pairs:
1411
+ continue
1412
+ seen_pairs.add((key, edge.dst_model))
1413
+ entry = entries.setdefault(
1414
+ key,
1415
+ {
1416
+ "table": key,
1417
+ **({"qualified_name": key} if "." in key else {}),
1418
+ **({"as_written": written} if written != key else {}),
1419
+ "referencing_models": 0,
1420
+ "blocked_models": 0,
1421
+ "statuses": {},
1422
+ },
1423
+ )
1424
+ entry["referencing_models"] += 1
1425
+ status = self.reports[edge.dst_model].status if edge.dst_model in self.reports else "ok"
1426
+ entry["statuses"][status] = entry["statuses"].get(status, 0) + 1
1427
+ return sorted(
1428
+ entries.values(),
1429
+ key=lambda e: (-e["blocked_models"], -e["referencing_models"], e["table"]),
1430
+ )
1431
+
1432
+ def suggest_target(self) -> dict | None:
1433
+ """The column with the widest blast radius, for a copy-pasteable
1434
+ first command. Direct fanout shortlists; transitive reach decides."""
1435
+ shortlist = []
1436
+ for (src_model, src_column), edges in (self._down or {}).items():
1437
+ if src_column == WILDCARD or src_column.startswith("("):
1438
+ continue
1439
+ report = self.reports.get(src_model)
1440
+ # only models this project owns: breaks() cannot target an external
1441
+ # table, and on filecoin-data-portal the 197 highest-fanout columns
1442
+ # were external, so the top-8 probe never reached a real model and
1443
+ # the suggestion came back empty
1444
+ if report is None or report.status != "ok":
1445
+ continue
1446
+ fanout = sum(1 for e in edges if e.kind == "value")
1447
+ if fanout:
1448
+ shortlist.append((fanout, src_model, src_column))
1449
+ shortlist.sort(reverse=True)
1450
+ # a path-qualified collision name (sql__2024__cookies__cookies) is
1451
+ # not something a person types; prefer a spelling they can, and fall
1452
+ # back to the mangled names only when nothing else has reach
1453
+ # (HTTP Archive's almanac)
1454
+ askable = [t for t in shortlist if self._askable_name(t[1]) == t[1] and "__" not in t[1]]
1455
+ shortlist = askable or shortlist
1456
+ best = None
1457
+ for _, model, column in shortlist[:8]:
1458
+ try:
1459
+ reach = self.breaks(model, column)["impacted_columns"]
1460
+ except LookupError:
1461
+ continue
1462
+ if best is None or reach > best["fanout"]:
1463
+ best = {"model": model, "column": column, "fanout": reach}
1464
+ return best
1465
+
1466
+ def to_dict(self) -> dict:
1467
+ nodes = []
1468
+ for m in self.project.sources:
1469
+ nodes.append(
1470
+ {"name": m.name, "id": m.uid, "type": "source", "columns": m.declared_columns}
1471
+ )
1472
+ for m in self.project.models:
1473
+ report = self.reports.get(m.name)
1474
+ nodes.append(
1475
+ {
1476
+ "name": m.name,
1477
+ "id": m.uid,
1478
+ "type": "model",
1479
+ "path": m.path,
1480
+ "status": report.status if report else "failed",
1481
+ "parsed": bool(report and report.parsed),
1482
+ "columns": report.columns if report else [],
1483
+ **({"warnings": report.warnings[:10]} if report and report.warnings else {}),
1484
+ }
1485
+ )
1486
+ return {
1487
+ "stats": self.stats(),
1488
+ "nodes": nodes,
1489
+ "edges": [
1490
+ {
1491
+ "src": f"{e.src_model}.{e.src_column}",
1492
+ "dst": f"{e.dst_model}.{e.dst_column}",
1493
+ "kind": e.kind,
1494
+ "trust": e.trust,
1495
+ **({"reason": e.reason} if e.reason else {}),
1496
+ }
1497
+ for e in self.edges
1498
+ ],
1499
+ }