ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/ci.py ADDED
@@ -0,0 +1,422 @@
1
+ """CI mode: what does this change break?
2
+
3
+ `ripple ci --base origin/main` compares every changed model against the base
4
+ ref, works out which output columns changed lineage (added, removed, or
5
+ re-derived), and prints the union blast radius as a PR-ready markdown comment.
6
+
7
+ Runs entirely in the repo's own CI with the repo's own checkout. No server,
8
+ no token, nothing leaves the runner.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import dataclasses
14
+ import json
15
+ import subprocess
16
+ import sys
17
+ from pathlib import Path
18
+
19
+ from ripple.answer import (
20
+ DASHBOARD_PREFIXES,
21
+ breaks_lines,
22
+ ci_answers,
23
+ count_hits,
24
+ noun_for,
25
+ plain_text,
26
+ row_impact_kind,
27
+ where_phrase,
28
+ )
29
+ from ripple.ci_signature import CTE_EXPRESSIONS_KEY, PREDICATES_KEY, _lineage_signature
30
+ from ripple.graph import LineageGraph, UnknownTarget
31
+ from ripple.project import load_project
32
+
33
+
34
+ def _git(args: list[str], cwd: Path) -> str:
35
+ result = subprocess.run(["git", *args], cwd=cwd, capture_output=True, text=True, check=False)
36
+ if result.returncode != 0:
37
+ raise RuntimeError(f"git {' '.join(args)}: {result.stderr.strip()}")
38
+ return result.stdout
39
+
40
+
41
+ def changed_sql_files(base: str, cwd: Path) -> tuple[list[str], Path]:
42
+ """Changed .sql paths (relative to the git toplevel) and the toplevel itself.
43
+
44
+ Union of committed changes since the merge-base (the PR diff) and any
45
+ uncommitted worktree changes, so local runs work too.
46
+ """
47
+ toplevel = Path(_git(["rev-parse", "--show-toplevel"], cwd).strip())
48
+ committed = _git(["diff", "--name-only", f"{base}...HEAD"], cwd).splitlines()
49
+ worktree = _git(["diff", "--name-only", base], cwd).splitlines()
50
+ seen: list[str] = []
51
+ for f in [*committed, *worktree]:
52
+ if f.endswith(".sql") and f not in seen:
53
+ seen.append(f)
54
+ return seen, toplevel
55
+
56
+
57
+ def analyze(base: str, path: str = ".", dialect: str | None = None) -> dict:
58
+ project = load_project(path, dialect=dialect)
59
+ return analyze_graph(base, project, LineageGraph.build(project))
60
+
61
+
62
+ def analyze_graph(base: str, project, graph: LineageGraph) -> dict:
63
+ """The report for one loaded project and its graph, so a caller that
64
+ also writes the answer page builds the graph once."""
65
+ root = project.root
66
+ changed, toplevel = changed_sql_files(base, root)
67
+ models_by_abspath = {(root / m.path).resolve(): m for m in project.models if m.path}
68
+ findings = []
69
+ base_sql: dict[str, str] = {}
70
+ pending_removed: list[tuple[dict, str, str]] = []
71
+ for file_path in changed:
72
+ model = models_by_abspath.get((toplevel / file_path).resolve())
73
+ try:
74
+ old_sql = _git(["show", f"{base}:{file_path}"], root)
75
+ except RuntimeError:
76
+ if model is None:
77
+ continue # not a model we can say anything about
78
+ old_sql = "" # new model
79
+ if model is None:
80
+ findings.append(_missing_model_finding(file_path, toplevel, graph))
81
+ continue
82
+ if old_sql:
83
+ base_sql[model.name] = old_sql
84
+ finding = _model_finding(graph, project.dialect, model, old_sql)
85
+ findings.append(finding)
86
+ pending_removed.extend(
87
+ (entry, model.name, entry["column"])
88
+ for entry in finding["columns"]
89
+ if entry["state"] == "removed"
90
+ )
91
+
92
+ if pending_removed:
93
+ _fill_removed_blasts(project, base_sql, pending_removed)
94
+ return _report(base, findings)
95
+
96
+
97
+ def _missing_model_finding(file_path: str, toplevel: Path, graph: LineageGraph) -> dict:
98
+ """The changed file is absent from the current project: deleted, or it
99
+ still exists but no longer reads as a model (garbage SQL)."""
100
+ name = Path(file_path).stem
101
+ still_referenced = sorted({e.dst_model for e in graph.edges if e.src_model == name})
102
+ gone = not (toplevel / file_path).exists()
103
+ finding = {
104
+ "model": name,
105
+ "status": "deleted" if gone else "unparseable",
106
+ "columns": [],
107
+ "still_referenced_by": still_referenced,
108
+ }
109
+ if not gone:
110
+ finding["error"] = "changed file no longer reads as a model"
111
+ return finding
112
+
113
+
114
+ def _model_finding(graph: LineageGraph, dialect: str, model, old_sql: str) -> dict:
115
+ try:
116
+ old_sig = _lineage_signature(old_sql, dialect) if old_sql else {}
117
+ new_sig = _lineage_signature(model.sql, dialect)
118
+ except Exception as e:
119
+ return {"model": model.name, "status": "unparseable", "error": str(e), "columns": []}
120
+ touched = sorted(
121
+ column for column in {*old_sig, *new_sig} if old_sig.get(column) != new_sig.get(column)
122
+ )
123
+ columns = []
124
+ for column in touched:
125
+ entry = _column_finding(graph, model.name, column, old_sig, new_sig, bool(old_sql))
126
+ if entry is not None:
127
+ columns.append(entry)
128
+ return {
129
+ "model": model.name,
130
+ "status": "new" if not old_sql else "changed",
131
+ "columns": columns,
132
+ }
133
+
134
+
135
+ def _column_finding(
136
+ graph: LineageGraph,
137
+ model_name: str,
138
+ column: str,
139
+ old_sig: dict,
140
+ new_sig: dict,
141
+ has_base: bool,
142
+ ) -> dict | None:
143
+ state = (
144
+ "added" if column not in old_sig else "removed" if column not in new_sig else "rederived"
145
+ )
146
+ blast = None
147
+ if column == PREDICATES_KEY:
148
+ if not has_base:
149
+ return None # a new model has no predicate diff to report
150
+ # a row-filter change affects every output of this model
151
+ state = "rows_refiltered"
152
+ blast = graph.breaks_all(model_name)
153
+ elif column == CTE_EXPRESSIONS_KEY:
154
+ if not has_base:
155
+ return None
156
+ # a formula changed inside a CTE: outputs derive differently
157
+ state = "reworked"
158
+ blast = graph.breaks_all(model_name)
159
+ elif state == "removed":
160
+ # the column and its edges are gone from the current graph, so asking
161
+ # it says "nothing downstream" about the one change most likely to
162
+ # break consumers; the real blast lives in the base state and is
163
+ # filled in by _fill_removed_blasts
164
+ blast = None
165
+ elif state != "added":
166
+ try:
167
+ blast = graph.breaks(model_name, column)
168
+ except UnknownTarget:
169
+ blast = None
170
+ return {"column": column, "state": state, **_blast_fields(blast)}
171
+
172
+
173
+ def _blast_fields(blast: dict | None) -> dict:
174
+ if not blast:
175
+ return {
176
+ "impacted_columns": 0,
177
+ "impacted_models": 0,
178
+ "dashboard_numbers": 0,
179
+ "review_required": 0,
180
+ "by_model": {},
181
+ "row_level_models": [],
182
+ "row_level_impact": [],
183
+ "truncated": False,
184
+ }
185
+ columns, models, dashboards = count_hits(blast["by_model"])
186
+ rows = blast.get("row_level_impact", [])
187
+ fields = {
188
+ "impacted_columns": columns,
189
+ "impacted_models": models,
190
+ "dashboard_numbers": dashboards,
191
+ "review_required": blast["review_required"],
192
+ "by_model": blast["by_model"],
193
+ "row_level_models": sorted({r["model"] for r in rows}),
194
+ "row_level_impact": rows,
195
+ "truncated": not blast.get("complete", True),
196
+ }
197
+ if blast.get("truncated_at_depth"):
198
+ fields["truncated_at_depth"] = blast["truncated_at_depth"]
199
+ return fields
200
+
201
+
202
+ def _fill_removed_blasts(project, base_sql: dict[str, str], pending: list) -> None:
203
+ """One extra graph build, only when a change removes columns."""
204
+ base_models = [
205
+ dataclasses.replace(m, sql=base_sql[m.name]) if m.name in base_sql else m
206
+ for m in project.models
207
+ ]
208
+ base_graph = LineageGraph.build(dataclasses.replace(project, models=base_models))
209
+ for entry, model_name, column in pending:
210
+ try:
211
+ blast = base_graph.breaks(model_name, column)
212
+ except UnknownTarget:
213
+ continue # nothing referenced it at base either
214
+ entry.update(_blast_fields(blast))
215
+
216
+
217
+ def _report(base: str, findings: list[dict]) -> dict:
218
+ report = {
219
+ "base": base,
220
+ "changed_models": len(findings),
221
+ "impacted_columns": sum(c["impacted_columns"] for f in findings for c in f["columns"]),
222
+ "dashboard_numbers": sum(c["dashboard_numbers"] for f in findings for c in f["columns"]),
223
+ "review_required": sum(c["review_required"] for f in findings for c in f["columns"]),
224
+ "row_level_models": len(
225
+ {m for f in findings for c in f["columns"] for m in c.get("row_level_models", [])}
226
+ ),
227
+ "deleted_still_referenced": [
228
+ f["model"]
229
+ for f in findings
230
+ if f["status"] == "deleted" and f.get("still_referenced_by")
231
+ ],
232
+ "unparseable": [f["model"] for f in findings if f["status"] == "unparseable"],
233
+ "findings": findings,
234
+ }
235
+ report["rebuild"] = rebuild_tokens(report)
236
+ return report
237
+
238
+
239
+ def write_change_page(report: dict, project, graph: LineageGraph, path: Path) -> Path | None:
240
+ """One self-contained page for the whole change: a view per changed
241
+ column that reaches anything, the rebuild list, and the graph embedded
242
+ so a reader can ask the next question. None when nothing changed."""
243
+ from ripple import answer_page
244
+
245
+ answers = ci_answers(report, noun_for(project.mode))
246
+ if not answers:
247
+ return None
248
+ views = [{"answer": a, "text": plain_text(breaks_lines(a, full=True))} for a in answers]
249
+ info = answer_page.project_info(project, Path(project.root))
250
+ change = {
251
+ "base": report["base"],
252
+ "changed_models": report["changed_models"],
253
+ "impacted_columns": report["impacted_columns"],
254
+ "dashboard_numbers": report["dashboard_numbers"],
255
+ "review_required": report["review_required"],
256
+ "rebuild": report["rebuild"],
257
+ }
258
+ path.parent.mkdir(parents=True, exist_ok=True)
259
+ path.write_text(
260
+ answer_page.render(
261
+ views[0]["answer"],
262
+ views[0]["text"],
263
+ info,
264
+ answer_page.graph_export(graph),
265
+ views=views,
266
+ change=change,
267
+ ),
268
+ encoding="utf-8",
269
+ )
270
+ return path
271
+
272
+
273
+ def rebuild_tokens(report: dict) -> list[str]:
274
+ """The models to rebuild after this change, as dbt --select tokens.
275
+
276
+ dbt's own state:modified+ rebuilds every descendant of any changed file;
277
+ column lineage knows which descendants actually derive differently, so
278
+ the enumerated set is usually much smaller. The one rule: over-select,
279
+ never under-select. Enumeration is only allowed when the analysis saw
280
+ everything; an unparseable diff, a depth-capped blast, a changed row
281
+ filter, or a deleted model's readers get dbt's descendant closure
282
+ (`model+`) instead. A missed rebuild ships stale data; a spare rebuild
283
+ only costs compute.
284
+ """
285
+ exact: set[str] = set()
286
+ widened: set[str] = set()
287
+ for finding in report["findings"]:
288
+ model = finding["model"]
289
+ if finding["status"] == "deleted":
290
+ # its readers will fail or change; their descendants follow
291
+ widened.update(finding.get("still_referenced_by", []))
292
+ continue
293
+ if finding["status"] == "unparseable":
294
+ widened.add(model)
295
+ widened.update(finding.get("still_referenced_by", []))
296
+ continue
297
+ exact.add(model)
298
+ for column in finding["columns"]:
299
+ if column.get("truncated"):
300
+ widened.add(model)
301
+ # metric:/semantic:/exposure: readers are dashboard numbers, not
302
+ # models dbt can build
303
+ exact.update(
304
+ name
305
+ for name in column.get("by_model", {})
306
+ if not name.startswith(DASHBOARD_PREFIXES)
307
+ )
308
+ # a changed row set changes every output of the filtering model
309
+ widened.update(column.get("row_level_models", []))
310
+ exact -= widened
311
+ return sorted(exact) + sorted(f"{name}+" for name in widened)
312
+
313
+
314
+ _ROW_VERBS = {"filter": "filters rows in", "join": "joins rows in", "window": "frames a window in"}
315
+
316
+
317
+ def _row_level_phrase(col: dict) -> str:
318
+ """The readers whose every row this column decides, as "filters rows in `paid`"."""
319
+ rows = col.get("row_level_impact") or [{"model": m} for m in col.get("row_level_models", [])]
320
+ by_verb: dict[str, list[str]] = {}
321
+ for row in rows:
322
+ by_verb.setdefault(_ROW_VERBS[row_impact_kind(row)], []).append(f"`{row['model']}`")
323
+ return "; ".join(f"{verb} {', '.join(names)}" for verb, names in by_verb.items())
324
+
325
+
326
+ def to_markdown(report: dict) -> str:
327
+ lines = ["### Ripple"]
328
+ if not report["findings"]:
329
+ lines.append("No model changes detected.")
330
+ return "\n".join(lines)
331
+ hit = f"**{report['impacted_columns']}** downstream columns"
332
+ if report.get("dashboard_numbers"):
333
+ hit += f" and **{report['dashboard_numbers']}** dashboard numbers"
334
+ if report.get("row_level_models"):
335
+ hit += f" and every row of **{report['row_level_models']}** model(s)"
336
+ lines.append(f"{hit} affected across {report['changed_models']} changed model(s).")
337
+ if report["review_required"]:
338
+ lines.append(
339
+ f"{report['review_required']} could not be fully verified. Check those by hand."
340
+ )
341
+ lines.append("")
342
+ for finding in report["findings"]:
343
+ if finding["status"] == "unparseable":
344
+ lines.append(f"- `{finding['model']}`: could not parse ({finding['error'][:120]})")
345
+ continue
346
+ if finding["status"] == "deleted":
347
+ refs = finding.get("still_referenced_by", [])
348
+ note = f", still referenced by {', '.join(f'`{r}`' for r in refs[:5])}" if refs else ""
349
+ lines.append(f"- `{finding['model']}` deleted{note}")
350
+ continue
351
+ for col in finding["columns"]:
352
+ if col["state"] in ("rows_refiltered", "reworked"):
353
+ what = (
354
+ "row filter changed"
355
+ if col["state"] == "rows_refiltered"
356
+ else "internal logic changed"
357
+ )
358
+ head = f"- `{finding['model']}` {what}"
359
+ rows = _row_level_phrase(col)
360
+ if col["impacted_columns"] or col.get("dashboard_numbers"):
361
+ where = where_phrase(col["impacted_models"], col.get("dashboard_numbers", 0))
362
+ head += f": {col['impacted_columns']} columns in {where} derive differently"
363
+ head += f"; {rows}" if rows else ""
364
+ elif rows:
365
+ head += f": {rows}"
366
+ lines.append(head)
367
+ continue
368
+ head = f"- `{finding['model']}.{col['column']}` ({col['state']})"
369
+ rows = _row_level_phrase(col)
370
+ if col["impacted_columns"] or col.get("dashboard_numbers"):
371
+ top = sorted(col["by_model"].items(), key=lambda kv: -len(kv[1]))[:3]
372
+ names = ", ".join(f"`{name}`" for name, _ in top)
373
+ more = len(col["by_model"]) - len(top)
374
+ suffix = f" and {more} more" if more > 0 else ""
375
+ where = where_phrase(col["impacted_models"], col.get("dashboard_numbers", 0))
376
+ head += f": {col['impacted_columns']} columns in {where} ({names}{suffix})"
377
+ if col["review_required"]:
378
+ head += f", {col['review_required']} need review"
379
+ head += f"; {rows}" if rows else ""
380
+ elif rows:
381
+ head += f": {rows}"
382
+ elif col["state"] != "added":
383
+ head += ": nothing downstream"
384
+ lines.append(head)
385
+ if report.get("rebuild"):
386
+ lines.append("")
387
+ lines.append(f"Rebuild: `{' '.join(report['rebuild'])}`")
388
+ return "\n".join(lines)
389
+
390
+
391
+ def run(
392
+ base: str,
393
+ path: str,
394
+ dialect: str | None,
395
+ as_json: bool,
396
+ fail_on: str,
397
+ select: str | None = None,
398
+ html: str | None = None,
399
+ ) -> int:
400
+ project = load_project(path, dialect=dialect)
401
+ graph = LineageGraph.build(project)
402
+ report = analyze_graph(base, project, graph)
403
+ if html:
404
+ written = write_change_page(report, project, graph, Path(html))
405
+ if written:
406
+ print(f"answer page: {written}", file=sys.stderr)
407
+ if select == "dbt":
408
+ # tokens only, pipeable: dbt build --select "$(ripple ci --select dbt)"
409
+ if report["rebuild"]:
410
+ print(" ".join(report["rebuild"]))
411
+ return 0
412
+ print(json.dumps(report, indent=2) if as_json else to_markdown(report))
413
+ if fail_on == "breaks" and (
414
+ report["impacted_columns"]
415
+ or report["dashboard_numbers"]
416
+ or report["row_level_models"]
417
+ or report["deleted_still_referenced"]
418
+ ):
419
+ return 1
420
+ if fail_on == "review" and (report["review_required"] or report["unparseable"]):
421
+ return 1
422
+ return 0