ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/cli.py ADDED
@@ -0,0 +1,733 @@
1
+ """Ripple CLI.
2
+
3
+ ripple summary of the current project
4
+ ripple breaks model.column what a change to this column affects
5
+ ripple trace model.column where this column's value comes from
6
+ ripple graph [-o FILE] full lineage graph as JSON
7
+ ripple unresolved external tables blocking coverage
8
+ ripple ingest-schema FILE add warehouse column lists (JSON/CSV, - for stdin)
9
+ ripple ingest-usage FILE add a query-history export (what actually ran)
10
+ ripple collect-usage [PLAT] run your own warehouse CLI and ingest what ran
11
+ ripple usage what ran and what didn't, from that history
12
+ ripple mcp run as an MCP server (stdio)
13
+ ripple doctor check that this install can serve MCP clients
14
+ ripple serve [model.column] the answer page, live: ask in a browser, see coverage
15
+
16
+ Output stays quiet: counts and names first, detail on request (--json).
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import json
23
+ import sys
24
+ from pathlib import Path
25
+
26
+ from ripple.answer import (
27
+ PREVIEW_COLUMNS,
28
+ attach,
29
+ breaks_lines,
30
+ noun_for,
31
+ plain_text,
32
+ trace_lines,
33
+ )
34
+
35
+ DIM = "\033[2m"
36
+ BOLD = "\033[1m"
37
+ AMBER = "\033[33m"
38
+ RESET = "\033[0m"
39
+
40
+
41
+ def _style(text: str, *codes: str) -> str:
42
+ if not sys.stdout.isatty():
43
+ return text
44
+ return "".join(codes) + text + RESET
45
+
46
+
47
+ SHARED_NAME_EXAMPLES = 3
48
+
49
+
50
+ def _fold_shared_names(pairs: list[tuple[str, int]]) -> list[tuple[str, int]]:
51
+ """Three shared-name examples and a count, never the whole list.
52
+
53
+ HTTP Archive's almanac keeps one query per chapter per year, so 841 names
54
+ are shared across year folders; each printed its own line, 841 lines
55
+ between the answer and the next step. The full list stays in --json."""
56
+ shared = [pair for pair in pairs if "share the name" in pair[0]]
57
+ if len(shared) <= SHARED_NAME_EXAMPLES:
58
+ return pairs
59
+ others = [pair for pair in pairs if "share the name" not in pair[0]]
60
+ hidden = len(shared) - SHARED_NAME_EXAMPLES
61
+ note = (
62
+ f"...and {hidden} more names shared by several files, each qualified by "
63
+ "path (full list: ripple --json)"
64
+ )
65
+ return others + shared[:SHARED_NAME_EXAMPLES] + [(note, 1)]
66
+
67
+
68
+ def _collapse(messages) -> list[tuple[str, int]]:
69
+ """Distinct messages in first-seen order, with how many times each ran."""
70
+ counts: dict[str, int] = {}
71
+ for message in messages:
72
+ counts[message] = counts.get(message, 0) + 1
73
+ return list(counts.items())
74
+
75
+
76
+ def _configure_logging(verbose: bool) -> None:
77
+ """Per-file parse failures are counted in the summary and detailed in
78
+ --json, so without --verbose they must not reach stderr. Unconfigured,
79
+ Python's last-resort handler prints every warning, which on a large repo
80
+ buries the answer under hundreds of lines and reads as a crash.
81
+
82
+ Set on the root logger, not on "ripple": most of the volume comes from
83
+ sqlglot, and quieting only our own loggers left 298 of 301 lines on a
84
+ 7,439-model repo. Any library we add later is covered by the same line.
85
+ """
86
+ import logging
87
+
88
+ level = logging.DEBUG if verbose else logging.ERROR
89
+ logging.basicConfig(level=level, stream=sys.stderr, format="%(message)s")
90
+ logging.getLogger().setLevel(level)
91
+
92
+
93
+ def _note(message: str) -> None:
94
+ """Progress goes to stderr so piping stdout stays clean."""
95
+ print(_style(message, DIM), file=sys.stderr)
96
+
97
+
98
+ def _load_graph(args):
99
+ import time
100
+
101
+ from ripple.graph import LineageGraph
102
+ from ripple.project import find_project_root, load_project
103
+
104
+ root = find_project_root(Path(args.path))
105
+ use_cache = not getattr(args, "no_cache", False)
106
+ if use_cache:
107
+ from ripple import cache
108
+
109
+ cached = cache.load(root, args.dialect)
110
+ if cached is not None:
111
+ return cached.project, cached
112
+ quiet = getattr(args, "json", False)
113
+ if not quiet:
114
+ # a large repo takes minutes here with nothing to show for it; saying
115
+ # so is the difference between "working" and "hung"
116
+ _note(f"Reading SQL in {root} (first run; later questions use the cache)...")
117
+ started = time.monotonic()
118
+ project = load_project(args.path, dialect=args.dialect)
119
+ graph = LineageGraph.build(project)
120
+ if use_cache:
121
+ from ripple import cache
122
+
123
+ cache.store(graph, root, args.dialect)
124
+ if not quiet:
125
+ elapsed = time.monotonic() - started
126
+ cached_note = "cached, so the next question is instant" if use_cache else "not cached"
127
+ _note(f"Indexed {len(project.models)} models in {elapsed:.0f}s ({cached_note}).\n")
128
+ return project, graph
129
+
130
+
131
+ def _parse_target(target: str) -> tuple[str, str]:
132
+ if "." not in target:
133
+ sys.exit(f"Expected model.column, got '{target}'")
134
+ model, _, column = target.rpartition(".")
135
+ return model, column
136
+
137
+
138
+ MODE_WORDS = {
139
+ "dbt-manifest": "compiled dbt manifest",
140
+ "dbt-raw": "reading raw model SQL",
141
+ "dbt-monorepo": "several dbt projects",
142
+ "sql-dir": "plain SQL files",
143
+ }
144
+
145
+
146
+ def _dominant_cause(project, graph, stats) -> str:
147
+ """Why the review count is what it is, and the command that shrinks it.
148
+
149
+ "426 links need review" told filecoin-data-portal's reader nothing
150
+ actionable. But blame must be counted, not guessed: on jaffle-shop this
151
+ line blamed 6 external tables, the user closed all six, and the count
152
+ moved from 18 to 18. A cause is only named with the number of review
153
+ links it actually accounts for.
154
+ """
155
+ guessed = any("Assuming" in w and "dialect" in w for w in project.warnings)
156
+ if guessed and stats["failed"]:
157
+ return (
158
+ f"the dialect was guessed as {stats['dialect']} and "
159
+ f"{stats['failed']} models would not parse: try --dialect"
160
+ )
161
+ review = [e for e in graph.edges if e.trust == "review_required"]
162
+ external = sum(1 for e in review if "not found in this project" in e.reason)
163
+ if external:
164
+ share = f"all {external}" if external == len(review) else f"{external} of the {len(review)}"
165
+ return f"{share} trace to tables defined outside the project: ripple unresolved"
166
+ if stats["failed"] or stats["star_only"]:
167
+ return "some models could not be fully parsed: ripple --json for which"
168
+ return ""
169
+
170
+
171
+ def cmd_summary(args) -> None:
172
+ project, graph = _load_graph(args)
173
+ stats = graph.stats()
174
+ if args.json:
175
+ print(json.dumps({**stats, "warnings": list(project.warnings)}, indent=2))
176
+ return
177
+ mode = MODE_WORDS.get(stats["mode"], stats["mode"])
178
+ noun = noun_for(stats["mode"])
179
+ print(
180
+ f"{_style(str(stats['models']), BOLD)} {noun}s, {stats['sources']} sources "
181
+ f"{_style('(' + mode + ' · ' + stats['dialect'] + ')', DIM)}"
182
+ )
183
+ print(f"{stats['ok']} analyzed in full, {stats['edges']} column links")
184
+ partial = stats["star_only"] + stats["fallback"] + stats["failed"] + stats["timed_out"]
185
+ if partial:
186
+ print(
187
+ _style(f"{partial} {noun}s partially covered or skipped (details: ripple --json)", DIM)
188
+ )
189
+ if stats["review_required_edges"]:
190
+ print(_style(f"{stats['review_required_edges']} links need review", AMBER))
191
+ cause = _dominant_cause(project, graph, stats)
192
+ if cause:
193
+ print(_style(f" {cause}", DIM))
194
+ for warning, count in _fold_shared_names(_collapse(project.warnings)):
195
+ # one line per distinct warning: a 16-project monorepo used to print
196
+ # the same "no compiled manifest" sentence 16 times
197
+ suffix = f" (x{count})" if count > 1 else ""
198
+ print(_style(warning + suffix, DIM))
199
+ print(_style("\n" + _next_step(project, graph), DIM))
200
+
201
+
202
+ def _next_step(project, graph) -> str:
203
+ """One runnable command, always.
204
+
205
+ A project where nothing has downstream fanout used to end on collision
206
+ warnings with no next move, which is where a new reader stops. sqlmesh-examples
207
+ and dagster-open-platform both landed there.
208
+ """
209
+ suggestion = graph.suggest_target()
210
+ if suggestion:
211
+ return (
212
+ f"try: ripple breaks {suggestion['model']}.{suggestion['column']}"
213
+ f" (feeds {suggestion['fanout']} downstream)"
214
+ )
215
+ if not project.models:
216
+ return f"No SQL found in {project.root}. Run inside a dbt project or SQL directory."
217
+ widest = max(
218
+ project.models,
219
+ key=lambda m: len(getattr(graph.reports.get(m.name), "columns", []) or []),
220
+ )
221
+ return f"try: ripple columns {widest.name} (nothing here has downstream reach yet)"
222
+
223
+
224
+ def _run_target_query(graph, method: str, model: str, column: str, max_depth: int = 25) -> dict:
225
+ from ripple.graph import UnknownTarget
226
+
227
+ try:
228
+ return getattr(graph, method)(model, column, max_depth=max(1, max_depth))
229
+ except UnknownTarget as e:
230
+ hint = f" Did you mean: {', '.join(e.suggestions)}?" if e.suggestions else ""
231
+ print(f"{e}.{hint}", file=sys.stderr)
232
+ sys.exit(2)
233
+
234
+
235
+ LINE_STYLES = {"plain": None, "bold": BOLD, "dim": DIM, "warn": AMBER}
236
+
237
+
238
+ def _print_lines(lines: list[list[tuple[str, str]]]) -> None:
239
+ for segments in lines:
240
+ print(
241
+ "".join(
242
+ _style(text, LINE_STYLES[role]) if LINE_STYLES[role] else text
243
+ for text, role in segments
244
+ )
245
+ )
246
+
247
+
248
+ def cmd_breaks(args) -> None:
249
+ project, graph = _load_graph(args)
250
+ model, column = _parse_target(args.target)
251
+ result = _run_target_query(graph, "breaks", model, column, max_depth=args.depth)
252
+ result = attach(result, "breaks", noun_for(project.mode))
253
+ if args.json:
254
+ print(json.dumps(result, indent=2))
255
+ return
256
+ answer = result["answer"]
257
+ _print_lines(breaks_lines(answer, full=args.full))
258
+ _page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
259
+
260
+
261
+ def cmd_trace(args) -> None:
262
+ project, graph = _load_graph(args)
263
+ model, column = _parse_target(args.target)
264
+ result = _run_target_query(graph, "trace", model, column, max_depth=args.depth)
265
+ result = attach(result, "trace", noun_for(project.mode))
266
+ if args.json:
267
+ print(json.dumps(result, indent=2))
268
+ return
269
+ answer = result["answer"]
270
+ _print_lines(trace_lines(answer))
271
+ _page_door(args, project, graph, answer, result["text"])
272
+
273
+
274
+ def _page_door(args, project, graph, answer: dict, text: str) -> None:
275
+ """--html writes the answer page where asked; --open writes it under
276
+ .ripple/answers and hands it to a browser, or prints the path when no
277
+ browser will take it (a sandbox, an SSH session)."""
278
+ if not (getattr(args, "html", None) or getattr(args, "open", False)):
279
+ return
280
+ from ripple import answer_page
281
+
282
+ path = answer_page.write_for(
283
+ graph, Path(project.root), answer, text, Path(args.html) if args.html else None
284
+ )
285
+ if args.open and answer_page.open_in_browser(path):
286
+ print(_style(f"opened {path}", DIM))
287
+ else:
288
+ print(_style(f"answer page: {path}", DIM))
289
+
290
+
291
+ def cmd_models(args) -> None:
292
+ _, graph = _load_graph(args)
293
+ rows = []
294
+ for name, report in sorted(graph.reports.items()):
295
+ rows.append(
296
+ {
297
+ "name": name,
298
+ "columns": len([c for c in report.columns if c != "*"]),
299
+ "status": report.status,
300
+ }
301
+ )
302
+ if args.json:
303
+ print(json.dumps(rows, indent=2))
304
+ return
305
+ for row in rows:
306
+ note = (
307
+ "" if row["status"] == "ok" else _style(f" ({row['status'].replace('_', ' ')})", DIM)
308
+ )
309
+ print(f"{row['name']} {_style('· ' + str(row['columns']) + ' columns', DIM)}{note}")
310
+
311
+
312
+ def cmd_columns(args) -> None:
313
+ from ripple.graph import UnknownTarget
314
+
315
+ _, graph = _load_graph(args)
316
+ try:
317
+ name, _ = graph._require_target(args.model, "")
318
+ except UnknownTarget as e:
319
+ if "no column" not in str(e):
320
+ hint = f" Did you mean: {', '.join(e.suggestions)}?" if e.suggestions else ""
321
+ print(f"{e}.{hint}", file=sys.stderr)
322
+ sys.exit(2)
323
+ name = graph._candidates[args.model.lower()][0]
324
+ report = graph.reports[name]
325
+ columns = [c for c in report.columns if c != "*"]
326
+ if args.json:
327
+ print(json.dumps({"model": name, "columns": columns, "status": report.status}, indent=2))
328
+ return
329
+ print(f"{name} {_style('· ' + report.status.replace('_', ' '), DIM)}")
330
+ for column in columns:
331
+ fanout = sum(1 for e in graph._down.get((name, column), []) if e.kind == "value")
332
+ mark = _style(f" → feeds {fanout}", DIM) if fanout else ""
333
+ print(f" {column}{mark}")
334
+
335
+
336
+ def cmd_graph(args) -> None:
337
+ _, graph = _load_graph(args)
338
+ payload = json.dumps(graph.to_dict(), indent=2)
339
+ if args.output:
340
+ Path(args.output).write_text(payload, encoding="utf-8")
341
+ print(f"wrote {args.output}")
342
+ else:
343
+ print(payload)
344
+
345
+
346
+ def cmd_unresolved(args) -> None:
347
+ _, graph = _load_graph(args)
348
+ unresolved = graph.unresolved_tables()
349
+ if args.json:
350
+ print(json.dumps({"unresolved": unresolved, "total": len(unresolved)}, indent=2))
351
+ return
352
+ if not unresolved:
353
+ print("Every referenced table resolves; nothing to ingest.")
354
+ return
355
+ print(f"{_style(str(len(unresolved)), BOLD)} external tables with unknown columns:")
356
+ for row in unresolved:
357
+ blocked = (
358
+ _style(f" blocks {row['blocked_models']}", AMBER) if row["blocked_models"] else ""
359
+ )
360
+ print(
361
+ f" {row['table']} "
362
+ f"{_style('· referenced by ' + str(row['referencing_models']), DIM)}{blocked}"
363
+ )
364
+ print(
365
+ _style(
366
+ "\nfetch their columns from your warehouse, then: ripple ingest-schema cols.csv",
367
+ DIM,
368
+ )
369
+ )
370
+
371
+
372
+ def cmd_ingest_schema(args) -> None:
373
+ from ripple.project import find_project_root
374
+ from ripple.schemas import merge_schemas, normalize_tables, parse_csv
375
+
376
+ text = (
377
+ sys.stdin.read()
378
+ if args.file == "-"
379
+ else Path(args.file).read_text(errors="replace", encoding="utf-8")
380
+ )
381
+ stripped = text.lstrip()
382
+ try:
383
+ if stripped.startswith(("{", "[")):
384
+ payload = json.loads(text)
385
+ tables = normalize_tables({"rows": payload} if isinstance(payload, list) else payload)
386
+ else:
387
+ tables = parse_csv(text)
388
+ except (json.JSONDecodeError, ValueError) as e:
389
+ sys.exit(f"Could not read {args.file}: {e}")
390
+ root = find_project_root(Path(args.path))
391
+ from ripple import cache
392
+
393
+ before = None if getattr(args, "no_cache", False) else cache.load(root, args.dialect)
394
+ before_review = before.stats()["review_required_edges"] if before else None
395
+ delta = merge_schemas(root, tables)
396
+ if args.json:
397
+ print(json.dumps(delta, indent=2))
398
+ return
399
+ changed = [*delta["tables_added"], *delta["tables_updated"]]
400
+ if not changed:
401
+ print("Nothing new; schemas already known.")
402
+ return
403
+ print(
404
+ f"{_style(str(delta['columns_added']), BOLD)} columns across "
405
+ f"{len(changed)} tables written to {delta['path']}"
406
+ )
407
+ for name in changed:
408
+ print(f" {name}")
409
+ # the promise of doing this work is a smaller review count; show whether
410
+ # it moved, including an honest "still N" (jaffle went 18 to 18 because
411
+ # its review links had a different cause)
412
+ _, graph = _load_graph(args)
413
+ after_review = graph.stats()["review_required_edges"]
414
+ if before_review is None or before_review != after_review:
415
+ origin = f"{before_review} → " if before_review is not None else ""
416
+ print(_style(f"links needing review: {origin}{after_review}", DIM))
417
+ else:
418
+ print(
419
+ _style(
420
+ f"links needing review: still {before_review}; these have another "
421
+ "cause (ripple --json shows each edge's reason)",
422
+ DIM,
423
+ )
424
+ )
425
+
426
+
427
+ def cmd_ci(args) -> None:
428
+ from ripple.ci import run
429
+
430
+ sys.exit(
431
+ run(args.base, args.path, args.dialect, args.json, args.fail_on, args.select, args.html)
432
+ )
433
+
434
+
435
+ def cmd_ingest_usage(args) -> None:
436
+ from ripple.project import find_project_root
437
+ from ripple.usage import ingest_file
438
+ from ripple.usage.ingest import write_store
439
+
440
+ project, _ = _load_graph(args)
441
+ try:
442
+ store = ingest_file(
443
+ args.file,
444
+ [m.name for m in project.models],
445
+ project.dialect,
446
+ model_aliases={m.name: set(m.aliases) for m in project.models},
447
+ )
448
+ except (OSError, ValueError) as e:
449
+ sys.exit(str(e))
450
+ write_store(find_project_root(Path(args.path)), store)
451
+ if args.json:
452
+ print(json.dumps(store, indent=2))
453
+ return
454
+ stats = store["statements"]
455
+ touched = stats["touched_tables"]
456
+ rate = f", {100.0 * stats['matched'] / touched:.0f}% matched this project" if touched else ""
457
+ print(f"Ingested {stats['total']:,} statements{rate}.")
458
+ print(_style("\ntry: ripple usage", DIM))
459
+
460
+
461
+ def cmd_collect_usage(args) -> None:
462
+ from ripple.usage.cli import run_collect
463
+
464
+ run_collect(args, _load_graph)
465
+
466
+
467
+ def cmd_usage(args) -> None:
468
+ from ripple.project import find_project_root
469
+ from ripple.usage import render, render_empty
470
+ from ripple.usage.ingest import models_fingerprint, read_store
471
+
472
+ store = read_store(find_project_root(Path(args.path)))
473
+ if store is None:
474
+ print(render_empty())
475
+ return
476
+ warning = None
477
+ recorded = store.get("models_fingerprint")
478
+ if recorded:
479
+ current = _current_model_names(args)
480
+ if current is not None and models_fingerprint(current) != recorded:
481
+ warning = (
482
+ "the project's models changed since this usage was ingested; "
483
+ "per-model counts reflect the old set. Re-run: ripple collect-usage "
484
+ "(or ingest-usage)."
485
+ )
486
+ if args.json:
487
+ print(json.dumps({**store, "warning": warning} if warning else store, indent=2))
488
+ return
489
+ if warning:
490
+ print(_style(warning, AMBER) + "\n")
491
+ print(render(store))
492
+
493
+
494
+ def _current_model_names(args) -> list[str] | None:
495
+ """Today's model names, as cheaply as truth allows: the cached graph
496
+ when one exists, a project load (no graph build) otherwise."""
497
+ from ripple import cache
498
+ from ripple.project import find_project_root, load_project
499
+
500
+ root = find_project_root(Path(args.path))
501
+ cached = None if getattr(args, "no_cache", False) else cache.load(root, args.dialect)
502
+ if cached is not None:
503
+ return [m.name for m in cached.project.models]
504
+ try:
505
+ return [m.name for m in load_project(args.path, dialect=args.dialect).models]
506
+ except Exception:
507
+ return None
508
+
509
+
510
+ def cmd_mcp(args) -> None:
511
+ from ripple.mcp_server import run_stdio
512
+
513
+ run_stdio(path=args.path, dialect=args.dialect)
514
+
515
+
516
+ def cmd_doctor(args) -> None:
517
+ from ripple.doctor import run
518
+
519
+ sys.exit(run(as_json=args.json))
520
+
521
+
522
+ def cmd_serve(args) -> None:
523
+ from ripple.server import serve
524
+
525
+ serve(path=args.path, dialect=args.dialect, port=args.port, target=args.target)
526
+
527
+
528
+ def add_common(parser: argparse.ArgumentParser, suppress: bool = False) -> None:
529
+ # SUPPRESS keeps subparser defaults from clobbering flags given before the subcommand.
530
+ def dflt(value):
531
+ return argparse.SUPPRESS if suppress else value
532
+
533
+ parser.add_argument("--path", default=dflt("."), help="project directory (default: current)")
534
+ parser.add_argument(
535
+ "--dialect", default=dflt(None), help="SQL dialect override (snowflake, bigquery, ...)"
536
+ )
537
+ parser.add_argument(
538
+ "--json", action="store_true", default=dflt(False), help="machine-readable output"
539
+ )
540
+ parser.add_argument(
541
+ "--no-cache",
542
+ action="store_true",
543
+ default=dflt(False),
544
+ help="rebuild instead of using the graph cache",
545
+ )
546
+ parser.add_argument(
547
+ "--verbose",
548
+ action="store_true",
549
+ default=dflt(False),
550
+ help="show per-file parse failures instead of counting them",
551
+ )
552
+
553
+
554
+ def _page_flags(parser) -> None:
555
+ parser.add_argument(
556
+ "--html", metavar="PATH", help="also write the answer as a self-contained page"
557
+ )
558
+ parser.add_argument(
559
+ "--open",
560
+ action="store_true",
561
+ help="write the answer page under .ripple/answers and open it in your browser",
562
+ )
563
+
564
+
565
+ def utf8_streams() -> None:
566
+ """A pipe out of ripple carries UTF-8 on every OS. Windows defaults a
567
+ pipe to the ANSI code page, which cannot encode the arrows in the text
568
+ and is not what the program on the other end expects. A console keeps
569
+ its own encoding and shows a placeholder for what it cannot draw."""
570
+ for stream in (sys.stdin, sys.stdout, sys.stderr):
571
+ if stream is None or not hasattr(stream, "reconfigure"):
572
+ continue
573
+ if stream.isatty():
574
+ stream.reconfigure(errors="replace")
575
+ else:
576
+ stream.reconfigure(encoding="utf-8", errors="replace")
577
+
578
+
579
+ def main(argv: list[str] | None = None) -> None:
580
+ utf8_streams()
581
+ common = argparse.ArgumentParser(add_help=False)
582
+ add_common(common, suppress=True)
583
+
584
+ parser = argparse.ArgumentParser(
585
+ prog="ripple",
586
+ description="See what a SQL change breaks before you merge.",
587
+ )
588
+ add_common(parser)
589
+ try:
590
+ from importlib.metadata import version
591
+
592
+ parser.add_argument(
593
+ "--version", action="version", version=f"ripple {version('ripple-sql')}"
594
+ )
595
+ except Exception:
596
+ pass
597
+ sub = parser.add_subparsers(dest="command")
598
+
599
+ sub.add_parser("models", help="list models with column counts", parents=[common])
600
+ p_columns = sub.add_parser(
601
+ "columns", help="list a model's columns and their fanout", parents=[common]
602
+ )
603
+ p_columns.add_argument("model")
604
+
605
+ p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
606
+ p_breaks.add_argument("target", help="model.column")
607
+ p_breaks.add_argument(
608
+ "--full",
609
+ action="store_true",
610
+ help=f"list every affected column, not the first {PREVIEW_COLUMNS} per model",
611
+ )
612
+ p_breaks.add_argument(
613
+ "--depth", type=int, default=25, help="follow lineage this many hops (default 25)"
614
+ )
615
+ _page_flags(p_breaks)
616
+
617
+ p_trace = sub.add_parser("trace", help="where this column comes from", parents=[common])
618
+ p_trace.add_argument("target", help="model.column")
619
+ p_trace.add_argument(
620
+ "--depth", type=int, default=25, help="follow lineage this many hops (default 25)"
621
+ )
622
+ _page_flags(p_trace)
623
+
624
+ p_graph = sub.add_parser("graph", help="export the lineage graph as JSON", parents=[common])
625
+ p_graph.add_argument("-o", "--output", default=None)
626
+
627
+ p_ci = sub.add_parser(
628
+ "ci", help="blast radius of the models changed since a git ref", parents=[common]
629
+ )
630
+ p_ci.add_argument("--base", default="origin/main", help="git ref to compare against")
631
+ p_ci.add_argument(
632
+ "--fail-on",
633
+ choices=["none", "breaks", "review"],
634
+ default="none",
635
+ help="exit nonzero if the change breaks things (breaks) or has unverified edges (review)",
636
+ )
637
+ p_ci.add_argument(
638
+ "--html",
639
+ metavar="PATH",
640
+ help="also write the change as a self-contained answer page, one view per changed column",
641
+ )
642
+ p_ci.add_argument(
643
+ "--select",
644
+ choices=["dbt"],
645
+ default=None,
646
+ help="print only the models to rebuild, as a dbt --select string: "
647
+ 'dbt build --select "$(ripple ci --select dbt)". Exits 0; anything '
648
+ "uncertain widens to model+ rather than being skipped",
649
+ )
650
+
651
+ sub.add_parser(
652
+ "unresolved", help="external tables blocking coverage (unknown columns)", parents=[common]
653
+ )
654
+ p_ingest = sub.add_parser(
655
+ "ingest-schema",
656
+ help="add warehouse column lists for external tables (JSON or CSV)",
657
+ parents=[common],
658
+ )
659
+ p_ingest.add_argument("file", help="JSON/CSV file with table columns, or - for stdin")
660
+
661
+ p_ingest_usage = sub.add_parser(
662
+ "ingest-usage",
663
+ help="add a query-history export: what actually ran (JSONL or JSON array)",
664
+ parents=[common],
665
+ )
666
+ p_ingest_usage.add_argument("file", help="query-history export with a query_text field")
667
+
668
+ p_collect = sub.add_parser(
669
+ "collect-usage",
670
+ help="run your own warehouse CLI (snow, bq, databricks) and ingest what ran",
671
+ parents=[common],
672
+ )
673
+ p_collect.add_argument(
674
+ "platform",
675
+ nargs="?",
676
+ choices=["snowflake", "bigquery", "databricks"],
677
+ help="omit to see what's installed and configured",
678
+ )
679
+ p_collect.add_argument(
680
+ "--connection",
681
+ default=None,
682
+ help="connection or profile name (for bigquery: the project id)",
683
+ )
684
+ p_collect.add_argument("--days", type=int, default=7, help="window size (default 7)")
685
+ p_collect.add_argument(
686
+ "--scope",
687
+ choices=["mine", "account"],
688
+ default="mine",
689
+ help="mine = your queries, no permission; account = everyone's, may need a grant",
690
+ )
691
+ p_collect.add_argument("--region", default="us", help="BigQuery INFORMATION_SCHEMA region")
692
+ p_collect.add_argument(
693
+ "--timeout", type=float, default=60.0, help="seconds before the CLI call is abandoned"
694
+ )
695
+
696
+ sub.add_parser(
697
+ "usage", help="what ran and what didn't, from the ingested history", parents=[common]
698
+ )
699
+
700
+ sub.add_parser("mcp", help="run as an MCP server over stdio", parents=[common])
701
+
702
+ sub.add_parser("doctor", help="check that this install can serve MCP clients", parents=[common])
703
+
704
+ p_serve = sub.add_parser(
705
+ "serve", help="the answer page, live: ask in a browser, see coverage", parents=[common]
706
+ )
707
+ p_serve.add_argument("target", nargs="?", help="open on this question (model.column)")
708
+ p_serve.add_argument("--port", type=int, default=8722)
709
+
710
+ args = parser.parse_args(argv)
711
+ _configure_logging(getattr(args, "verbose", False))
712
+ handlers = {
713
+ None: cmd_summary,
714
+ "models": cmd_models,
715
+ "columns": cmd_columns,
716
+ "breaks": cmd_breaks,
717
+ "trace": cmd_trace,
718
+ "graph": cmd_graph,
719
+ "ci": cmd_ci,
720
+ "unresolved": cmd_unresolved,
721
+ "ingest-schema": cmd_ingest_schema,
722
+ "ingest-usage": cmd_ingest_usage,
723
+ "collect-usage": cmd_collect_usage,
724
+ "usage": cmd_usage,
725
+ "mcp": cmd_mcp,
726
+ "doctor": cmd_doctor,
727
+ "serve": cmd_serve,
728
+ }
729
+ handlers[args.command](args)
730
+
731
+
732
+ if __name__ == "__main__":
733
+ main()