ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/cli.py
ADDED
|
@@ -0,0 +1,733 @@
|
|
|
1
|
+
"""Ripple CLI.
|
|
2
|
+
|
|
3
|
+
ripple summary of the current project
|
|
4
|
+
ripple breaks model.column what a change to this column affects
|
|
5
|
+
ripple trace model.column where this column's value comes from
|
|
6
|
+
ripple graph [-o FILE] full lineage graph as JSON
|
|
7
|
+
ripple unresolved external tables blocking coverage
|
|
8
|
+
ripple ingest-schema FILE add warehouse column lists (JSON/CSV, - for stdin)
|
|
9
|
+
ripple ingest-usage FILE add a query-history export (what actually ran)
|
|
10
|
+
ripple collect-usage [PLAT] run your own warehouse CLI and ingest what ran
|
|
11
|
+
ripple usage what ran and what didn't, from that history
|
|
12
|
+
ripple mcp run as an MCP server (stdio)
|
|
13
|
+
ripple doctor check that this install can serve MCP clients
|
|
14
|
+
ripple serve [model.column] the answer page, live: ask in a browser, see coverage
|
|
15
|
+
|
|
16
|
+
Output stays quiet: counts and names first, detail on request (--json).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import sys
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
from ripple.answer import (
|
|
27
|
+
PREVIEW_COLUMNS,
|
|
28
|
+
attach,
|
|
29
|
+
breaks_lines,
|
|
30
|
+
noun_for,
|
|
31
|
+
plain_text,
|
|
32
|
+
trace_lines,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
DIM = "\033[2m"
|
|
36
|
+
BOLD = "\033[1m"
|
|
37
|
+
AMBER = "\033[33m"
|
|
38
|
+
RESET = "\033[0m"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _style(text: str, *codes: str) -> str:
|
|
42
|
+
if not sys.stdout.isatty():
|
|
43
|
+
return text
|
|
44
|
+
return "".join(codes) + text + RESET
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
SHARED_NAME_EXAMPLES = 3
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _fold_shared_names(pairs: list[tuple[str, int]]) -> list[tuple[str, int]]:
|
|
51
|
+
"""Three shared-name examples and a count, never the whole list.
|
|
52
|
+
|
|
53
|
+
HTTP Archive's almanac keeps one query per chapter per year, so 841 names
|
|
54
|
+
are shared across year folders; each printed its own line, 841 lines
|
|
55
|
+
between the answer and the next step. The full list stays in --json."""
|
|
56
|
+
shared = [pair for pair in pairs if "share the name" in pair[0]]
|
|
57
|
+
if len(shared) <= SHARED_NAME_EXAMPLES:
|
|
58
|
+
return pairs
|
|
59
|
+
others = [pair for pair in pairs if "share the name" not in pair[0]]
|
|
60
|
+
hidden = len(shared) - SHARED_NAME_EXAMPLES
|
|
61
|
+
note = (
|
|
62
|
+
f"...and {hidden} more names shared by several files, each qualified by "
|
|
63
|
+
"path (full list: ripple --json)"
|
|
64
|
+
)
|
|
65
|
+
return others + shared[:SHARED_NAME_EXAMPLES] + [(note, 1)]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _collapse(messages) -> list[tuple[str, int]]:
|
|
69
|
+
"""Distinct messages in first-seen order, with how many times each ran."""
|
|
70
|
+
counts: dict[str, int] = {}
|
|
71
|
+
for message in messages:
|
|
72
|
+
counts[message] = counts.get(message, 0) + 1
|
|
73
|
+
return list(counts.items())
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _configure_logging(verbose: bool) -> None:
|
|
77
|
+
"""Per-file parse failures are counted in the summary and detailed in
|
|
78
|
+
--json, so without --verbose they must not reach stderr. Unconfigured,
|
|
79
|
+
Python's last-resort handler prints every warning, which on a large repo
|
|
80
|
+
buries the answer under hundreds of lines and reads as a crash.
|
|
81
|
+
|
|
82
|
+
Set on the root logger, not on "ripple": most of the volume comes from
|
|
83
|
+
sqlglot, and quieting only our own loggers left 298 of 301 lines on a
|
|
84
|
+
7,439-model repo. Any library we add later is covered by the same line.
|
|
85
|
+
"""
|
|
86
|
+
import logging
|
|
87
|
+
|
|
88
|
+
level = logging.DEBUG if verbose else logging.ERROR
|
|
89
|
+
logging.basicConfig(level=level, stream=sys.stderr, format="%(message)s")
|
|
90
|
+
logging.getLogger().setLevel(level)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _note(message: str) -> None:
|
|
94
|
+
"""Progress goes to stderr so piping stdout stays clean."""
|
|
95
|
+
print(_style(message, DIM), file=sys.stderr)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _load_graph(args):
|
|
99
|
+
import time
|
|
100
|
+
|
|
101
|
+
from ripple.graph import LineageGraph
|
|
102
|
+
from ripple.project import find_project_root, load_project
|
|
103
|
+
|
|
104
|
+
root = find_project_root(Path(args.path))
|
|
105
|
+
use_cache = not getattr(args, "no_cache", False)
|
|
106
|
+
if use_cache:
|
|
107
|
+
from ripple import cache
|
|
108
|
+
|
|
109
|
+
cached = cache.load(root, args.dialect)
|
|
110
|
+
if cached is not None:
|
|
111
|
+
return cached.project, cached
|
|
112
|
+
quiet = getattr(args, "json", False)
|
|
113
|
+
if not quiet:
|
|
114
|
+
# a large repo takes minutes here with nothing to show for it; saying
|
|
115
|
+
# so is the difference between "working" and "hung"
|
|
116
|
+
_note(f"Reading SQL in {root} (first run; later questions use the cache)...")
|
|
117
|
+
started = time.monotonic()
|
|
118
|
+
project = load_project(args.path, dialect=args.dialect)
|
|
119
|
+
graph = LineageGraph.build(project)
|
|
120
|
+
if use_cache:
|
|
121
|
+
from ripple import cache
|
|
122
|
+
|
|
123
|
+
cache.store(graph, root, args.dialect)
|
|
124
|
+
if not quiet:
|
|
125
|
+
elapsed = time.monotonic() - started
|
|
126
|
+
cached_note = "cached, so the next question is instant" if use_cache else "not cached"
|
|
127
|
+
_note(f"Indexed {len(project.models)} models in {elapsed:.0f}s ({cached_note}).\n")
|
|
128
|
+
return project, graph
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _parse_target(target: str) -> tuple[str, str]:
|
|
132
|
+
if "." not in target:
|
|
133
|
+
sys.exit(f"Expected model.column, got '{target}'")
|
|
134
|
+
model, _, column = target.rpartition(".")
|
|
135
|
+
return model, column
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
MODE_WORDS = {
|
|
139
|
+
"dbt-manifest": "compiled dbt manifest",
|
|
140
|
+
"dbt-raw": "reading raw model SQL",
|
|
141
|
+
"dbt-monorepo": "several dbt projects",
|
|
142
|
+
"sql-dir": "plain SQL files",
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _dominant_cause(project, graph, stats) -> str:
|
|
147
|
+
"""Why the review count is what it is, and the command that shrinks it.
|
|
148
|
+
|
|
149
|
+
"426 links need review" told filecoin-data-portal's reader nothing
|
|
150
|
+
actionable. But blame must be counted, not guessed: on jaffle-shop this
|
|
151
|
+
line blamed 6 external tables, the user closed all six, and the count
|
|
152
|
+
moved from 18 to 18. A cause is only named with the number of review
|
|
153
|
+
links it actually accounts for.
|
|
154
|
+
"""
|
|
155
|
+
guessed = any("Assuming" in w and "dialect" in w for w in project.warnings)
|
|
156
|
+
if guessed and stats["failed"]:
|
|
157
|
+
return (
|
|
158
|
+
f"the dialect was guessed as {stats['dialect']} and "
|
|
159
|
+
f"{stats['failed']} models would not parse: try --dialect"
|
|
160
|
+
)
|
|
161
|
+
review = [e for e in graph.edges if e.trust == "review_required"]
|
|
162
|
+
external = sum(1 for e in review if "not found in this project" in e.reason)
|
|
163
|
+
if external:
|
|
164
|
+
share = f"all {external}" if external == len(review) else f"{external} of the {len(review)}"
|
|
165
|
+
return f"{share} trace to tables defined outside the project: ripple unresolved"
|
|
166
|
+
if stats["failed"] or stats["star_only"]:
|
|
167
|
+
return "some models could not be fully parsed: ripple --json for which"
|
|
168
|
+
return ""
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def cmd_summary(args) -> None:
|
|
172
|
+
project, graph = _load_graph(args)
|
|
173
|
+
stats = graph.stats()
|
|
174
|
+
if args.json:
|
|
175
|
+
print(json.dumps({**stats, "warnings": list(project.warnings)}, indent=2))
|
|
176
|
+
return
|
|
177
|
+
mode = MODE_WORDS.get(stats["mode"], stats["mode"])
|
|
178
|
+
noun = noun_for(stats["mode"])
|
|
179
|
+
print(
|
|
180
|
+
f"{_style(str(stats['models']), BOLD)} {noun}s, {stats['sources']} sources "
|
|
181
|
+
f"{_style('(' + mode + ' · ' + stats['dialect'] + ')', DIM)}"
|
|
182
|
+
)
|
|
183
|
+
print(f"{stats['ok']} analyzed in full, {stats['edges']} column links")
|
|
184
|
+
partial = stats["star_only"] + stats["fallback"] + stats["failed"] + stats["timed_out"]
|
|
185
|
+
if partial:
|
|
186
|
+
print(
|
|
187
|
+
_style(f"{partial} {noun}s partially covered or skipped (details: ripple --json)", DIM)
|
|
188
|
+
)
|
|
189
|
+
if stats["review_required_edges"]:
|
|
190
|
+
print(_style(f"{stats['review_required_edges']} links need review", AMBER))
|
|
191
|
+
cause = _dominant_cause(project, graph, stats)
|
|
192
|
+
if cause:
|
|
193
|
+
print(_style(f" {cause}", DIM))
|
|
194
|
+
for warning, count in _fold_shared_names(_collapse(project.warnings)):
|
|
195
|
+
# one line per distinct warning: a 16-project monorepo used to print
|
|
196
|
+
# the same "no compiled manifest" sentence 16 times
|
|
197
|
+
suffix = f" (x{count})" if count > 1 else ""
|
|
198
|
+
print(_style(warning + suffix, DIM))
|
|
199
|
+
print(_style("\n" + _next_step(project, graph), DIM))
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _next_step(project, graph) -> str:
|
|
203
|
+
"""One runnable command, always.
|
|
204
|
+
|
|
205
|
+
A project where nothing has downstream fanout used to end on collision
|
|
206
|
+
warnings with no next move, which is where a new reader stops. sqlmesh-examples
|
|
207
|
+
and dagster-open-platform both landed there.
|
|
208
|
+
"""
|
|
209
|
+
suggestion = graph.suggest_target()
|
|
210
|
+
if suggestion:
|
|
211
|
+
return (
|
|
212
|
+
f"try: ripple breaks {suggestion['model']}.{suggestion['column']}"
|
|
213
|
+
f" (feeds {suggestion['fanout']} downstream)"
|
|
214
|
+
)
|
|
215
|
+
if not project.models:
|
|
216
|
+
return f"No SQL found in {project.root}. Run inside a dbt project or SQL directory."
|
|
217
|
+
widest = max(
|
|
218
|
+
project.models,
|
|
219
|
+
key=lambda m: len(getattr(graph.reports.get(m.name), "columns", []) or []),
|
|
220
|
+
)
|
|
221
|
+
return f"try: ripple columns {widest.name} (nothing here has downstream reach yet)"
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _run_target_query(graph, method: str, model: str, column: str, max_depth: int = 25) -> dict:
|
|
225
|
+
from ripple.graph import UnknownTarget
|
|
226
|
+
|
|
227
|
+
try:
|
|
228
|
+
return getattr(graph, method)(model, column, max_depth=max(1, max_depth))
|
|
229
|
+
except UnknownTarget as e:
|
|
230
|
+
hint = f" Did you mean: {', '.join(e.suggestions)}?" if e.suggestions else ""
|
|
231
|
+
print(f"{e}.{hint}", file=sys.stderr)
|
|
232
|
+
sys.exit(2)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
LINE_STYLES = {"plain": None, "bold": BOLD, "dim": DIM, "warn": AMBER}
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _print_lines(lines: list[list[tuple[str, str]]]) -> None:
|
|
239
|
+
for segments in lines:
|
|
240
|
+
print(
|
|
241
|
+
"".join(
|
|
242
|
+
_style(text, LINE_STYLES[role]) if LINE_STYLES[role] else text
|
|
243
|
+
for text, role in segments
|
|
244
|
+
)
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def cmd_breaks(args) -> None:
|
|
249
|
+
project, graph = _load_graph(args)
|
|
250
|
+
model, column = _parse_target(args.target)
|
|
251
|
+
result = _run_target_query(graph, "breaks", model, column, max_depth=args.depth)
|
|
252
|
+
result = attach(result, "breaks", noun_for(project.mode))
|
|
253
|
+
if args.json:
|
|
254
|
+
print(json.dumps(result, indent=2))
|
|
255
|
+
return
|
|
256
|
+
answer = result["answer"]
|
|
257
|
+
_print_lines(breaks_lines(answer, full=args.full))
|
|
258
|
+
_page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def cmd_trace(args) -> None:
|
|
262
|
+
project, graph = _load_graph(args)
|
|
263
|
+
model, column = _parse_target(args.target)
|
|
264
|
+
result = _run_target_query(graph, "trace", model, column, max_depth=args.depth)
|
|
265
|
+
result = attach(result, "trace", noun_for(project.mode))
|
|
266
|
+
if args.json:
|
|
267
|
+
print(json.dumps(result, indent=2))
|
|
268
|
+
return
|
|
269
|
+
answer = result["answer"]
|
|
270
|
+
_print_lines(trace_lines(answer))
|
|
271
|
+
_page_door(args, project, graph, answer, result["text"])
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _page_door(args, project, graph, answer: dict, text: str) -> None:
|
|
275
|
+
"""--html writes the answer page where asked; --open writes it under
|
|
276
|
+
.ripple/answers and hands it to a browser, or prints the path when no
|
|
277
|
+
browser will take it (a sandbox, an SSH session)."""
|
|
278
|
+
if not (getattr(args, "html", None) or getattr(args, "open", False)):
|
|
279
|
+
return
|
|
280
|
+
from ripple import answer_page
|
|
281
|
+
|
|
282
|
+
path = answer_page.write_for(
|
|
283
|
+
graph, Path(project.root), answer, text, Path(args.html) if args.html else None
|
|
284
|
+
)
|
|
285
|
+
if args.open and answer_page.open_in_browser(path):
|
|
286
|
+
print(_style(f"opened {path}", DIM))
|
|
287
|
+
else:
|
|
288
|
+
print(_style(f"answer page: {path}", DIM))
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def cmd_models(args) -> None:
|
|
292
|
+
_, graph = _load_graph(args)
|
|
293
|
+
rows = []
|
|
294
|
+
for name, report in sorted(graph.reports.items()):
|
|
295
|
+
rows.append(
|
|
296
|
+
{
|
|
297
|
+
"name": name,
|
|
298
|
+
"columns": len([c for c in report.columns if c != "*"]),
|
|
299
|
+
"status": report.status,
|
|
300
|
+
}
|
|
301
|
+
)
|
|
302
|
+
if args.json:
|
|
303
|
+
print(json.dumps(rows, indent=2))
|
|
304
|
+
return
|
|
305
|
+
for row in rows:
|
|
306
|
+
note = (
|
|
307
|
+
"" if row["status"] == "ok" else _style(f" ({row['status'].replace('_', ' ')})", DIM)
|
|
308
|
+
)
|
|
309
|
+
print(f"{row['name']} {_style('· ' + str(row['columns']) + ' columns', DIM)}{note}")
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def cmd_columns(args) -> None:
|
|
313
|
+
from ripple.graph import UnknownTarget
|
|
314
|
+
|
|
315
|
+
_, graph = _load_graph(args)
|
|
316
|
+
try:
|
|
317
|
+
name, _ = graph._require_target(args.model, "")
|
|
318
|
+
except UnknownTarget as e:
|
|
319
|
+
if "no column" not in str(e):
|
|
320
|
+
hint = f" Did you mean: {', '.join(e.suggestions)}?" if e.suggestions else ""
|
|
321
|
+
print(f"{e}.{hint}", file=sys.stderr)
|
|
322
|
+
sys.exit(2)
|
|
323
|
+
name = graph._candidates[args.model.lower()][0]
|
|
324
|
+
report = graph.reports[name]
|
|
325
|
+
columns = [c for c in report.columns if c != "*"]
|
|
326
|
+
if args.json:
|
|
327
|
+
print(json.dumps({"model": name, "columns": columns, "status": report.status}, indent=2))
|
|
328
|
+
return
|
|
329
|
+
print(f"{name} {_style('· ' + report.status.replace('_', ' '), DIM)}")
|
|
330
|
+
for column in columns:
|
|
331
|
+
fanout = sum(1 for e in graph._down.get((name, column), []) if e.kind == "value")
|
|
332
|
+
mark = _style(f" → feeds {fanout}", DIM) if fanout else ""
|
|
333
|
+
print(f" {column}{mark}")
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def cmd_graph(args) -> None:
|
|
337
|
+
_, graph = _load_graph(args)
|
|
338
|
+
payload = json.dumps(graph.to_dict(), indent=2)
|
|
339
|
+
if args.output:
|
|
340
|
+
Path(args.output).write_text(payload, encoding="utf-8")
|
|
341
|
+
print(f"wrote {args.output}")
|
|
342
|
+
else:
|
|
343
|
+
print(payload)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def cmd_unresolved(args) -> None:
|
|
347
|
+
_, graph = _load_graph(args)
|
|
348
|
+
unresolved = graph.unresolved_tables()
|
|
349
|
+
if args.json:
|
|
350
|
+
print(json.dumps({"unresolved": unresolved, "total": len(unresolved)}, indent=2))
|
|
351
|
+
return
|
|
352
|
+
if not unresolved:
|
|
353
|
+
print("Every referenced table resolves; nothing to ingest.")
|
|
354
|
+
return
|
|
355
|
+
print(f"{_style(str(len(unresolved)), BOLD)} external tables with unknown columns:")
|
|
356
|
+
for row in unresolved:
|
|
357
|
+
blocked = (
|
|
358
|
+
_style(f" blocks {row['blocked_models']}", AMBER) if row["blocked_models"] else ""
|
|
359
|
+
)
|
|
360
|
+
print(
|
|
361
|
+
f" {row['table']} "
|
|
362
|
+
f"{_style('· referenced by ' + str(row['referencing_models']), DIM)}{blocked}"
|
|
363
|
+
)
|
|
364
|
+
print(
|
|
365
|
+
_style(
|
|
366
|
+
"\nfetch their columns from your warehouse, then: ripple ingest-schema cols.csv",
|
|
367
|
+
DIM,
|
|
368
|
+
)
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def cmd_ingest_schema(args) -> None:
|
|
373
|
+
from ripple.project import find_project_root
|
|
374
|
+
from ripple.schemas import merge_schemas, normalize_tables, parse_csv
|
|
375
|
+
|
|
376
|
+
text = (
|
|
377
|
+
sys.stdin.read()
|
|
378
|
+
if args.file == "-"
|
|
379
|
+
else Path(args.file).read_text(errors="replace", encoding="utf-8")
|
|
380
|
+
)
|
|
381
|
+
stripped = text.lstrip()
|
|
382
|
+
try:
|
|
383
|
+
if stripped.startswith(("{", "[")):
|
|
384
|
+
payload = json.loads(text)
|
|
385
|
+
tables = normalize_tables({"rows": payload} if isinstance(payload, list) else payload)
|
|
386
|
+
else:
|
|
387
|
+
tables = parse_csv(text)
|
|
388
|
+
except (json.JSONDecodeError, ValueError) as e:
|
|
389
|
+
sys.exit(f"Could not read {args.file}: {e}")
|
|
390
|
+
root = find_project_root(Path(args.path))
|
|
391
|
+
from ripple import cache
|
|
392
|
+
|
|
393
|
+
before = None if getattr(args, "no_cache", False) else cache.load(root, args.dialect)
|
|
394
|
+
before_review = before.stats()["review_required_edges"] if before else None
|
|
395
|
+
delta = merge_schemas(root, tables)
|
|
396
|
+
if args.json:
|
|
397
|
+
print(json.dumps(delta, indent=2))
|
|
398
|
+
return
|
|
399
|
+
changed = [*delta["tables_added"], *delta["tables_updated"]]
|
|
400
|
+
if not changed:
|
|
401
|
+
print("Nothing new; schemas already known.")
|
|
402
|
+
return
|
|
403
|
+
print(
|
|
404
|
+
f"{_style(str(delta['columns_added']), BOLD)} columns across "
|
|
405
|
+
f"{len(changed)} tables written to {delta['path']}"
|
|
406
|
+
)
|
|
407
|
+
for name in changed:
|
|
408
|
+
print(f" {name}")
|
|
409
|
+
# the promise of doing this work is a smaller review count; show whether
|
|
410
|
+
# it moved, including an honest "still N" (jaffle went 18 to 18 because
|
|
411
|
+
# its review links had a different cause)
|
|
412
|
+
_, graph = _load_graph(args)
|
|
413
|
+
after_review = graph.stats()["review_required_edges"]
|
|
414
|
+
if before_review is None or before_review != after_review:
|
|
415
|
+
origin = f"{before_review} → " if before_review is not None else ""
|
|
416
|
+
print(_style(f"links needing review: {origin}{after_review}", DIM))
|
|
417
|
+
else:
|
|
418
|
+
print(
|
|
419
|
+
_style(
|
|
420
|
+
f"links needing review: still {before_review}; these have another "
|
|
421
|
+
"cause (ripple --json shows each edge's reason)",
|
|
422
|
+
DIM,
|
|
423
|
+
)
|
|
424
|
+
)
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def cmd_ci(args) -> None:
|
|
428
|
+
from ripple.ci import run
|
|
429
|
+
|
|
430
|
+
sys.exit(
|
|
431
|
+
run(args.base, args.path, args.dialect, args.json, args.fail_on, args.select, args.html)
|
|
432
|
+
)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def cmd_ingest_usage(args) -> None:
|
|
436
|
+
from ripple.project import find_project_root
|
|
437
|
+
from ripple.usage import ingest_file
|
|
438
|
+
from ripple.usage.ingest import write_store
|
|
439
|
+
|
|
440
|
+
project, _ = _load_graph(args)
|
|
441
|
+
try:
|
|
442
|
+
store = ingest_file(
|
|
443
|
+
args.file,
|
|
444
|
+
[m.name for m in project.models],
|
|
445
|
+
project.dialect,
|
|
446
|
+
model_aliases={m.name: set(m.aliases) for m in project.models},
|
|
447
|
+
)
|
|
448
|
+
except (OSError, ValueError) as e:
|
|
449
|
+
sys.exit(str(e))
|
|
450
|
+
write_store(find_project_root(Path(args.path)), store)
|
|
451
|
+
if args.json:
|
|
452
|
+
print(json.dumps(store, indent=2))
|
|
453
|
+
return
|
|
454
|
+
stats = store["statements"]
|
|
455
|
+
touched = stats["touched_tables"]
|
|
456
|
+
rate = f", {100.0 * stats['matched'] / touched:.0f}% matched this project" if touched else ""
|
|
457
|
+
print(f"Ingested {stats['total']:,} statements{rate}.")
|
|
458
|
+
print(_style("\ntry: ripple usage", DIM))
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def cmd_collect_usage(args) -> None:
|
|
462
|
+
from ripple.usage.cli import run_collect
|
|
463
|
+
|
|
464
|
+
run_collect(args, _load_graph)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def cmd_usage(args) -> None:
|
|
468
|
+
from ripple.project import find_project_root
|
|
469
|
+
from ripple.usage import render, render_empty
|
|
470
|
+
from ripple.usage.ingest import models_fingerprint, read_store
|
|
471
|
+
|
|
472
|
+
store = read_store(find_project_root(Path(args.path)))
|
|
473
|
+
if store is None:
|
|
474
|
+
print(render_empty())
|
|
475
|
+
return
|
|
476
|
+
warning = None
|
|
477
|
+
recorded = store.get("models_fingerprint")
|
|
478
|
+
if recorded:
|
|
479
|
+
current = _current_model_names(args)
|
|
480
|
+
if current is not None and models_fingerprint(current) != recorded:
|
|
481
|
+
warning = (
|
|
482
|
+
"the project's models changed since this usage was ingested; "
|
|
483
|
+
"per-model counts reflect the old set. Re-run: ripple collect-usage "
|
|
484
|
+
"(or ingest-usage)."
|
|
485
|
+
)
|
|
486
|
+
if args.json:
|
|
487
|
+
print(json.dumps({**store, "warning": warning} if warning else store, indent=2))
|
|
488
|
+
return
|
|
489
|
+
if warning:
|
|
490
|
+
print(_style(warning, AMBER) + "\n")
|
|
491
|
+
print(render(store))
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def _current_model_names(args) -> list[str] | None:
|
|
495
|
+
"""Today's model names, as cheaply as truth allows: the cached graph
|
|
496
|
+
when one exists, a project load (no graph build) otherwise."""
|
|
497
|
+
from ripple import cache
|
|
498
|
+
from ripple.project import find_project_root, load_project
|
|
499
|
+
|
|
500
|
+
root = find_project_root(Path(args.path))
|
|
501
|
+
cached = None if getattr(args, "no_cache", False) else cache.load(root, args.dialect)
|
|
502
|
+
if cached is not None:
|
|
503
|
+
return [m.name for m in cached.project.models]
|
|
504
|
+
try:
|
|
505
|
+
return [m.name for m in load_project(args.path, dialect=args.dialect).models]
|
|
506
|
+
except Exception:
|
|
507
|
+
return None
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def cmd_mcp(args) -> None:
|
|
511
|
+
from ripple.mcp_server import run_stdio
|
|
512
|
+
|
|
513
|
+
run_stdio(path=args.path, dialect=args.dialect)
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
def cmd_doctor(args) -> None:
|
|
517
|
+
from ripple.doctor import run
|
|
518
|
+
|
|
519
|
+
sys.exit(run(as_json=args.json))
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def cmd_serve(args) -> None:
|
|
523
|
+
from ripple.server import serve
|
|
524
|
+
|
|
525
|
+
serve(path=args.path, dialect=args.dialect, port=args.port, target=args.target)
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def add_common(parser: argparse.ArgumentParser, suppress: bool = False) -> None:
|
|
529
|
+
# SUPPRESS keeps subparser defaults from clobbering flags given before the subcommand.
|
|
530
|
+
def dflt(value):
|
|
531
|
+
return argparse.SUPPRESS if suppress else value
|
|
532
|
+
|
|
533
|
+
parser.add_argument("--path", default=dflt("."), help="project directory (default: current)")
|
|
534
|
+
parser.add_argument(
|
|
535
|
+
"--dialect", default=dflt(None), help="SQL dialect override (snowflake, bigquery, ...)"
|
|
536
|
+
)
|
|
537
|
+
parser.add_argument(
|
|
538
|
+
"--json", action="store_true", default=dflt(False), help="machine-readable output"
|
|
539
|
+
)
|
|
540
|
+
parser.add_argument(
|
|
541
|
+
"--no-cache",
|
|
542
|
+
action="store_true",
|
|
543
|
+
default=dflt(False),
|
|
544
|
+
help="rebuild instead of using the graph cache",
|
|
545
|
+
)
|
|
546
|
+
parser.add_argument(
|
|
547
|
+
"--verbose",
|
|
548
|
+
action="store_true",
|
|
549
|
+
default=dflt(False),
|
|
550
|
+
help="show per-file parse failures instead of counting them",
|
|
551
|
+
)
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def _page_flags(parser) -> None:
|
|
555
|
+
parser.add_argument(
|
|
556
|
+
"--html", metavar="PATH", help="also write the answer as a self-contained page"
|
|
557
|
+
)
|
|
558
|
+
parser.add_argument(
|
|
559
|
+
"--open",
|
|
560
|
+
action="store_true",
|
|
561
|
+
help="write the answer page under .ripple/answers and open it in your browser",
|
|
562
|
+
)
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def utf8_streams() -> None:
|
|
566
|
+
"""A pipe out of ripple carries UTF-8 on every OS. Windows defaults a
|
|
567
|
+
pipe to the ANSI code page, which cannot encode the arrows in the text
|
|
568
|
+
and is not what the program on the other end expects. A console keeps
|
|
569
|
+
its own encoding and shows a placeholder for what it cannot draw."""
|
|
570
|
+
for stream in (sys.stdin, sys.stdout, sys.stderr):
|
|
571
|
+
if stream is None or not hasattr(stream, "reconfigure"):
|
|
572
|
+
continue
|
|
573
|
+
if stream.isatty():
|
|
574
|
+
stream.reconfigure(errors="replace")
|
|
575
|
+
else:
|
|
576
|
+
stream.reconfigure(encoding="utf-8", errors="replace")
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def main(argv: list[str] | None = None) -> None:
|
|
580
|
+
utf8_streams()
|
|
581
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
582
|
+
add_common(common, suppress=True)
|
|
583
|
+
|
|
584
|
+
parser = argparse.ArgumentParser(
|
|
585
|
+
prog="ripple",
|
|
586
|
+
description="See what a SQL change breaks before you merge.",
|
|
587
|
+
)
|
|
588
|
+
add_common(parser)
|
|
589
|
+
try:
|
|
590
|
+
from importlib.metadata import version
|
|
591
|
+
|
|
592
|
+
parser.add_argument(
|
|
593
|
+
"--version", action="version", version=f"ripple {version('ripple-sql')}"
|
|
594
|
+
)
|
|
595
|
+
except Exception:
|
|
596
|
+
pass
|
|
597
|
+
sub = parser.add_subparsers(dest="command")
|
|
598
|
+
|
|
599
|
+
sub.add_parser("models", help="list models with column counts", parents=[common])
|
|
600
|
+
p_columns = sub.add_parser(
|
|
601
|
+
"columns", help="list a model's columns and their fanout", parents=[common]
|
|
602
|
+
)
|
|
603
|
+
p_columns.add_argument("model")
|
|
604
|
+
|
|
605
|
+
p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
|
|
606
|
+
p_breaks.add_argument("target", help="model.column")
|
|
607
|
+
p_breaks.add_argument(
|
|
608
|
+
"--full",
|
|
609
|
+
action="store_true",
|
|
610
|
+
help=f"list every affected column, not the first {PREVIEW_COLUMNS} per model",
|
|
611
|
+
)
|
|
612
|
+
p_breaks.add_argument(
|
|
613
|
+
"--depth", type=int, default=25, help="follow lineage this many hops (default 25)"
|
|
614
|
+
)
|
|
615
|
+
_page_flags(p_breaks)
|
|
616
|
+
|
|
617
|
+
p_trace = sub.add_parser("trace", help="where this column comes from", parents=[common])
|
|
618
|
+
p_trace.add_argument("target", help="model.column")
|
|
619
|
+
p_trace.add_argument(
|
|
620
|
+
"--depth", type=int, default=25, help="follow lineage this many hops (default 25)"
|
|
621
|
+
)
|
|
622
|
+
_page_flags(p_trace)
|
|
623
|
+
|
|
624
|
+
p_graph = sub.add_parser("graph", help="export the lineage graph as JSON", parents=[common])
|
|
625
|
+
p_graph.add_argument("-o", "--output", default=None)
|
|
626
|
+
|
|
627
|
+
p_ci = sub.add_parser(
|
|
628
|
+
"ci", help="blast radius of the models changed since a git ref", parents=[common]
|
|
629
|
+
)
|
|
630
|
+
p_ci.add_argument("--base", default="origin/main", help="git ref to compare against")
|
|
631
|
+
p_ci.add_argument(
|
|
632
|
+
"--fail-on",
|
|
633
|
+
choices=["none", "breaks", "review"],
|
|
634
|
+
default="none",
|
|
635
|
+
help="exit nonzero if the change breaks things (breaks) or has unverified edges (review)",
|
|
636
|
+
)
|
|
637
|
+
p_ci.add_argument(
|
|
638
|
+
"--html",
|
|
639
|
+
metavar="PATH",
|
|
640
|
+
help="also write the change as a self-contained answer page, one view per changed column",
|
|
641
|
+
)
|
|
642
|
+
p_ci.add_argument(
|
|
643
|
+
"--select",
|
|
644
|
+
choices=["dbt"],
|
|
645
|
+
default=None,
|
|
646
|
+
help="print only the models to rebuild, as a dbt --select string: "
|
|
647
|
+
'dbt build --select "$(ripple ci --select dbt)". Exits 0; anything '
|
|
648
|
+
"uncertain widens to model+ rather than being skipped",
|
|
649
|
+
)
|
|
650
|
+
|
|
651
|
+
sub.add_parser(
|
|
652
|
+
"unresolved", help="external tables blocking coverage (unknown columns)", parents=[common]
|
|
653
|
+
)
|
|
654
|
+
p_ingest = sub.add_parser(
|
|
655
|
+
"ingest-schema",
|
|
656
|
+
help="add warehouse column lists for external tables (JSON or CSV)",
|
|
657
|
+
parents=[common],
|
|
658
|
+
)
|
|
659
|
+
p_ingest.add_argument("file", help="JSON/CSV file with table columns, or - for stdin")
|
|
660
|
+
|
|
661
|
+
p_ingest_usage = sub.add_parser(
|
|
662
|
+
"ingest-usage",
|
|
663
|
+
help="add a query-history export: what actually ran (JSONL or JSON array)",
|
|
664
|
+
parents=[common],
|
|
665
|
+
)
|
|
666
|
+
p_ingest_usage.add_argument("file", help="query-history export with a query_text field")
|
|
667
|
+
|
|
668
|
+
p_collect = sub.add_parser(
|
|
669
|
+
"collect-usage",
|
|
670
|
+
help="run your own warehouse CLI (snow, bq, databricks) and ingest what ran",
|
|
671
|
+
parents=[common],
|
|
672
|
+
)
|
|
673
|
+
p_collect.add_argument(
|
|
674
|
+
"platform",
|
|
675
|
+
nargs="?",
|
|
676
|
+
choices=["snowflake", "bigquery", "databricks"],
|
|
677
|
+
help="omit to see what's installed and configured",
|
|
678
|
+
)
|
|
679
|
+
p_collect.add_argument(
|
|
680
|
+
"--connection",
|
|
681
|
+
default=None,
|
|
682
|
+
help="connection or profile name (for bigquery: the project id)",
|
|
683
|
+
)
|
|
684
|
+
p_collect.add_argument("--days", type=int, default=7, help="window size (default 7)")
|
|
685
|
+
p_collect.add_argument(
|
|
686
|
+
"--scope",
|
|
687
|
+
choices=["mine", "account"],
|
|
688
|
+
default="mine",
|
|
689
|
+
help="mine = your queries, no permission; account = everyone's, may need a grant",
|
|
690
|
+
)
|
|
691
|
+
p_collect.add_argument("--region", default="us", help="BigQuery INFORMATION_SCHEMA region")
|
|
692
|
+
p_collect.add_argument(
|
|
693
|
+
"--timeout", type=float, default=60.0, help="seconds before the CLI call is abandoned"
|
|
694
|
+
)
|
|
695
|
+
|
|
696
|
+
sub.add_parser(
|
|
697
|
+
"usage", help="what ran and what didn't, from the ingested history", parents=[common]
|
|
698
|
+
)
|
|
699
|
+
|
|
700
|
+
sub.add_parser("mcp", help="run as an MCP server over stdio", parents=[common])
|
|
701
|
+
|
|
702
|
+
sub.add_parser("doctor", help="check that this install can serve MCP clients", parents=[common])
|
|
703
|
+
|
|
704
|
+
p_serve = sub.add_parser(
|
|
705
|
+
"serve", help="the answer page, live: ask in a browser, see coverage", parents=[common]
|
|
706
|
+
)
|
|
707
|
+
p_serve.add_argument("target", nargs="?", help="open on this question (model.column)")
|
|
708
|
+
p_serve.add_argument("--port", type=int, default=8722)
|
|
709
|
+
|
|
710
|
+
args = parser.parse_args(argv)
|
|
711
|
+
_configure_logging(getattr(args, "verbose", False))
|
|
712
|
+
handlers = {
|
|
713
|
+
None: cmd_summary,
|
|
714
|
+
"models": cmd_models,
|
|
715
|
+
"columns": cmd_columns,
|
|
716
|
+
"breaks": cmd_breaks,
|
|
717
|
+
"trace": cmd_trace,
|
|
718
|
+
"graph": cmd_graph,
|
|
719
|
+
"ci": cmd_ci,
|
|
720
|
+
"unresolved": cmd_unresolved,
|
|
721
|
+
"ingest-schema": cmd_ingest_schema,
|
|
722
|
+
"ingest-usage": cmd_ingest_usage,
|
|
723
|
+
"collect-usage": cmd_collect_usage,
|
|
724
|
+
"usage": cmd_usage,
|
|
725
|
+
"mcp": cmd_mcp,
|
|
726
|
+
"doctor": cmd_doctor,
|
|
727
|
+
"serve": cmd_serve,
|
|
728
|
+
}
|
|
729
|
+
handlers[args.command](args)
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
if __name__ == "__main__":
|
|
733
|
+
main()
|