ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/mcp_server.py
ADDED
|
@@ -0,0 +1,600 @@
|
|
|
1
|
+
"""MCP server over stdio.
|
|
2
|
+
|
|
3
|
+
Deliberately small: nine deterministic tools, no prompts, no resources.
|
|
4
|
+
The MCP client (Claude Code, Cursor, Codex, ...) does the reasoning; Ripple
|
|
5
|
+
answers lineage questions exactly and admits what it can't resolve.
|
|
6
|
+
|
|
7
|
+
Hand-rolled JSON-RPC rather than an SDK dependency: the protocol surface
|
|
8
|
+
we need is tiny and the tool must stay a one-dependency install.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import difflib
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
import threading
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from ripple import answer_page
|
|
21
|
+
from ripple.answer import attach, breaks_lines, noun_for, plain_text
|
|
22
|
+
|
|
23
|
+
PROTOCOL_VERSION = "2025-06-18"
|
|
24
|
+
# Versions whose stdio framing and tools/* surface match what this server
|
|
25
|
+
# implements. Anything else (including future revisions) gets PROTOCOL_VERSION
|
|
26
|
+
# back, per spec, instead of a blind echo that claims semantics we don't have.
|
|
27
|
+
SUPPORTED_PROTOCOL_VERSIONS = {"2024-11-05", "2025-03-26", PROTOCOL_VERSION}
|
|
28
|
+
|
|
29
|
+
DEPTH_ARG = {
|
|
30
|
+
"type": "integer",
|
|
31
|
+
"description": (
|
|
32
|
+
"how many hops to follow (default 25). Raise it when the result says truncated_at_depth."
|
|
33
|
+
),
|
|
34
|
+
}
|
|
35
|
+
PAGE_ARG = {
|
|
36
|
+
"type": "boolean",
|
|
37
|
+
"description": (
|
|
38
|
+
"also write a self-contained HTML answer page under .ripple/answers "
|
|
39
|
+
"and return its path as html_path, for a person to open in a browser. "
|
|
40
|
+
"Nothing is sent anywhere."
|
|
41
|
+
),
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
TOOLS = [
|
|
45
|
+
{
|
|
46
|
+
"name": "lineage_summary",
|
|
47
|
+
"description": (
|
|
48
|
+
"Overview of the SQL project's column lineage: model count, parse rate, "
|
|
49
|
+
"edge count, and how many edges are uncertain (review_required). "
|
|
50
|
+
"Call this first."
|
|
51
|
+
),
|
|
52
|
+
"inputSchema": {
|
|
53
|
+
"type": "object",
|
|
54
|
+
"properties": {"refresh": {"type": "boolean", "description": "Re-scan the project"}},
|
|
55
|
+
},
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"name": "breaks",
|
|
59
|
+
"description": (
|
|
60
|
+
"What breaks if this column changes: every downstream column, model, filter "
|
|
61
|
+
"and join that depends on it, each with a trust label. Edges marked "
|
|
62
|
+
"review_required could not be fully verified; treat them as 'check by hand'."
|
|
63
|
+
),
|
|
64
|
+
"inputSchema": {
|
|
65
|
+
"type": "object",
|
|
66
|
+
"properties": {
|
|
67
|
+
"target": {
|
|
68
|
+
"type": "string",
|
|
69
|
+
"description": "model.column, e.g. stg_payments.amount",
|
|
70
|
+
},
|
|
71
|
+
"depth": DEPTH_ARG,
|
|
72
|
+
"page": PAGE_ARG,
|
|
73
|
+
},
|
|
74
|
+
"required": ["target"],
|
|
75
|
+
},
|
|
76
|
+
},
|
|
77
|
+
{
|
|
78
|
+
"name": "trace",
|
|
79
|
+
"description": "Where a column's value comes from, hop by hop, back to sources.",
|
|
80
|
+
"inputSchema": {
|
|
81
|
+
"type": "object",
|
|
82
|
+
"properties": {
|
|
83
|
+
"target": {"type": "string", "description": "model.column"},
|
|
84
|
+
"depth": DEPTH_ARG,
|
|
85
|
+
"page": PAGE_ARG,
|
|
86
|
+
},
|
|
87
|
+
"required": ["target"],
|
|
88
|
+
},
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"name": "model_columns",
|
|
92
|
+
"description": "List a model's output columns (use to find the exact column name before calling breaks/trace).",
|
|
93
|
+
"inputSchema": {
|
|
94
|
+
"type": "object",
|
|
95
|
+
"properties": {"model": {"type": "string", "description": "model name"}},
|
|
96
|
+
"required": ["model"],
|
|
97
|
+
},
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
"name": "unresolved_tables",
|
|
101
|
+
"description": (
|
|
102
|
+
"External tables this project reads but does not define, sorted by how much "
|
|
103
|
+
"coverage they block: each entry has the table name, how many models reference "
|
|
104
|
+
"it, and how many of those are stuck (star_only or failed) because its columns "
|
|
105
|
+
"are unknown. Follow-up: Ripple never connects to a warehouse, but you likely "
|
|
106
|
+
"can. Fetch these tables' columns yourself and hand them back via ingest_schema. "
|
|
107
|
+
"Canonical query: SELECT table_name, column_name FROM information_schema.columns "
|
|
108
|
+
"WHERE table_name IN ('t1', 't2') ORDER BY table_name, ordinal_position. "
|
|
109
|
+
"Dialect notes: Snowflake stores unquoted identifiers uppercase, so match with "
|
|
110
|
+
"UPPER(table_name) or uppercase literals; BigQuery scopes it per dataset as "
|
|
111
|
+
"`project.dataset.INFORMATION_SCHEMA.COLUMNS`; Postgres and Redshift use "
|
|
112
|
+
"information_schema.columns as written (add table_schema = '...' if names "
|
|
113
|
+
"repeat across schemas). Then call ingest_schema with the results."
|
|
114
|
+
),
|
|
115
|
+
"inputSchema": {"type": "object", "properties": {}},
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"name": "ingest_schema",
|
|
119
|
+
"description": (
|
|
120
|
+
"Save warehouse column lists for external tables so Ripple can resolve them. "
|
|
121
|
+
'Accepts {"tables": {"name": {"columns": ["a", "b"]}}} or, straight '
|
|
122
|
+
'from a tabular query result, {"rows": [{"table": "name", "column": '
|
|
123
|
+
'"a"}]}. Table names may be bare or schema qualified; case does not matter. '
|
|
124
|
+
"Merges into .ripple/schemas.json at the project root (committable, share it "
|
|
125
|
+
"with your team) and returns what changed. Rebuild happens automatically on "
|
|
126
|
+
"next query."
|
|
127
|
+
),
|
|
128
|
+
"inputSchema": {
|
|
129
|
+
"type": "object",
|
|
130
|
+
"properties": {
|
|
131
|
+
"tables": {
|
|
132
|
+
"type": "object",
|
|
133
|
+
"description": 'table name -> {"columns": [...]} or a bare column list',
|
|
134
|
+
},
|
|
135
|
+
"rows": {
|
|
136
|
+
"type": "array",
|
|
137
|
+
"description": 'alternative tabular form: [{"table": ..., "column": ...}]',
|
|
138
|
+
"items": {
|
|
139
|
+
"type": "object",
|
|
140
|
+
"properties": {
|
|
141
|
+
"table": {"type": "string"},
|
|
142
|
+
"column": {"type": "string"},
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
},
|
|
148
|
+
},
|
|
149
|
+
{
|
|
150
|
+
"name": "ingest_usage",
|
|
151
|
+
"description": (
|
|
152
|
+
"Ingest a query-history export: what actually ran in the warehouse. "
|
|
153
|
+
"Takes a FILE PATH only, by design: never fetch or paste query rows "
|
|
154
|
+
"yourself, query text must not pass through a model. Have the user run "
|
|
155
|
+
'the export with a local command, for example: snow sql -q "SELECT '
|
|
156
|
+
"query_id, query_text, database_name, schema_name, user_name, "
|
|
157
|
+
"start_time FROM TABLE(INFORMATION_SCHEMA.QUERY_HISTORY("
|
|
158
|
+
'RESULT_LIMIT => 10000))" --format json > history.json, then pass '
|
|
159
|
+
"that path here. JSONL or a JSON array with a query_text field works; "
|
|
160
|
+
"Snowflake QUERY_HISTORY column names are understood as-is. Ripple "
|
|
161
|
+
"stores per-table counts only, never the query text."
|
|
162
|
+
),
|
|
163
|
+
"inputSchema": {
|
|
164
|
+
"type": "object",
|
|
165
|
+
"properties": {
|
|
166
|
+
"path": {
|
|
167
|
+
"type": "string",
|
|
168
|
+
"description": "path to the export file on this machine",
|
|
169
|
+
}
|
|
170
|
+
},
|
|
171
|
+
"required": ["path"],
|
|
172
|
+
},
|
|
173
|
+
},
|
|
174
|
+
{
|
|
175
|
+
"name": "collect_usage",
|
|
176
|
+
"description": (
|
|
177
|
+
"Collect query history by running the user's already-authenticated "
|
|
178
|
+
"warehouse CLI (snow, bq, or databricks) locally. Ripple's process "
|
|
179
|
+
"spawns the CLI, spools the rows to a temp file, aggregates, and "
|
|
180
|
+
"deletes the file inside one call: the query text never reaches you "
|
|
181
|
+
"and is never stored, and no credential is read or transmitted. "
|
|
182
|
+
"Call with no arguments first: you get back which CLIs are installed "
|
|
183
|
+
"and which connections exist. When several connections are "
|
|
184
|
+
"configured, NEVER pick one for the user; show the options and ask. "
|
|
185
|
+
"scope 'mine' (default) reads only the user's own queries and needs "
|
|
186
|
+
"no permission. scope 'account' covers everyone's queries and may "
|
|
187
|
+
"need a grant; a permission error includes the exact statement to "
|
|
188
|
+
"hand an admin. Databricks has no mine/account split: its history "
|
|
189
|
+
"API returns whatever this login can see, and the result says "
|
|
190
|
+
"'visible-to-this-login'. If the call times out because a browser sign-in "
|
|
191
|
+
"window opened, tell the user to finish signing in and call again."
|
|
192
|
+
),
|
|
193
|
+
"inputSchema": {
|
|
194
|
+
"type": "object",
|
|
195
|
+
"properties": {
|
|
196
|
+
"platform": {
|
|
197
|
+
"type": "string",
|
|
198
|
+
"enum": ["snowflake", "bigquery", "databricks"],
|
|
199
|
+
"description": "omit to get the capability table",
|
|
200
|
+
},
|
|
201
|
+
"connection": {
|
|
202
|
+
"type": "string",
|
|
203
|
+
"description": "connection or profile name (bigquery: project id)",
|
|
204
|
+
},
|
|
205
|
+
"days": {"type": "integer", "description": "window size, default 7"},
|
|
206
|
+
"scope": {"type": "string", "enum": ["mine", "account"]},
|
|
207
|
+
"region": {
|
|
208
|
+
"type": "string",
|
|
209
|
+
"description": "BigQuery INFORMATION_SCHEMA region, default us",
|
|
210
|
+
},
|
|
211
|
+
},
|
|
212
|
+
},
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
"name": "usage_summary",
|
|
216
|
+
"description": (
|
|
217
|
+
"Aggregated usage from the ingested query history: which models were "
|
|
218
|
+
"seen running in the window, which were not, attribution counts, and "
|
|
219
|
+
"the tables queries read outside this project. Aggregates only, no "
|
|
220
|
+
"query text. Never report a model as 'unused' from this; the honest "
|
|
221
|
+
"claim is 'not seen in this window', and the window is in the manifest."
|
|
222
|
+
),
|
|
223
|
+
"inputSchema": {"type": "object", "properties": {}},
|
|
224
|
+
},
|
|
225
|
+
]
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
INGEST_NEXT = (
|
|
229
|
+
"For each table, get its columns from the warehouse (DESCRIBE TABLE or "
|
|
230
|
+
"INFORMATION_SCHEMA.COLUMNS through your warehouse MCP) and call ingest_schema "
|
|
231
|
+
'with {"tables": {"<table>": {"columns": ["a", "b"]}}}. Tables with '
|
|
232
|
+
"blocked_models > 0 unlock the most."
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _summary_next(graph, summary: dict) -> str:
|
|
237
|
+
"""One sentence an agent can act on, the way the CLI ends on 'try:'.
|
|
238
|
+
|
|
239
|
+
Counted before it blames: on balboa only 9 of 17 review links pointed at
|
|
240
|
+
external tables; the rest came from ambiguous names and an unparsed CTE,
|
|
241
|
+
which no ingest fixes."""
|
|
242
|
+
review = summary.get("review_required_edges", 0)
|
|
243
|
+
if not review:
|
|
244
|
+
return "Ask breaks(model.column) about any column; model_columns lists a model's columns."
|
|
245
|
+
review_edges = [e for e in graph.edges if e.trust == "review_required"]
|
|
246
|
+
external = sum(1 for e in review_edges if "not found" in e.reason)
|
|
247
|
+
incomplete = [e for e in review_edges if "declared columns of" in e.reason]
|
|
248
|
+
other = review - external - len(incomplete)
|
|
249
|
+
parts = []
|
|
250
|
+
if incomplete:
|
|
251
|
+
tables = sorted({e.src_model for e in incomplete})[:5]
|
|
252
|
+
parts.append(
|
|
253
|
+
f"{len(incomplete)} of the {review} review links read columns that the "
|
|
254
|
+
f"declared schema lacks ({', '.join(tables)}): re-ingest complete column "
|
|
255
|
+
"lists for those tables."
|
|
256
|
+
)
|
|
257
|
+
if external:
|
|
258
|
+
parts.append(
|
|
259
|
+
f"{external} of the {review} review links point at tables outside the "
|
|
260
|
+
"project: call unresolved_tables, then ingest_schema with their column "
|
|
261
|
+
"lists, then ask breaks(model.column)."
|
|
262
|
+
)
|
|
263
|
+
if other > 0:
|
|
264
|
+
parts.append(
|
|
265
|
+
f"{other}{' other' if external or incomplete else ''} review links come from ambiguous "
|
|
266
|
+
"names, disabled models, or statements that did not fully parse; no ingest "
|
|
267
|
+
"fixes those, and each breaks hit carries its reason."
|
|
268
|
+
)
|
|
269
|
+
return " ".join(parts)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _hits_next(result: dict) -> str:
|
|
273
|
+
"""Counted like the summary: only hits whose reason is an external table
|
|
274
|
+
are sent to unresolved_tables; the rest are named for what they are."""
|
|
275
|
+
hits = [h for hits in result.get("by_model", {}).values() for h in hits]
|
|
276
|
+
review = [h for h in hits if h.get("trust") == "review_required"]
|
|
277
|
+
external = sum(1 for h in review if "not found" in (h.get("reason") or ""))
|
|
278
|
+
incomplete = sum(1 for h in review if "declared columns of" in (h.get("reason") or ""))
|
|
279
|
+
other = len(review) - external - incomplete
|
|
280
|
+
parts = []
|
|
281
|
+
if external:
|
|
282
|
+
parts.append(
|
|
283
|
+
f"{external} of the {len(review)} review hits trace through tables outside "
|
|
284
|
+
"the project: unresolved_tables lists them; ingest_schema their columns, "
|
|
285
|
+
"then ask again."
|
|
286
|
+
)
|
|
287
|
+
if incomplete:
|
|
288
|
+
parts.append(
|
|
289
|
+
f"{incomplete} read columns the declared schema lacks: re-ingest complete column lists."
|
|
290
|
+
)
|
|
291
|
+
if other > 0:
|
|
292
|
+
parts.append(
|
|
293
|
+
f"{other}{' other' if parts else ''} review hits come from ambiguous names, "
|
|
294
|
+
"disabled models, or statements that did not fully parse; each hit carries "
|
|
295
|
+
"its reason."
|
|
296
|
+
)
|
|
297
|
+
return " ".join(parts) or f"{result['review_required']} hits need review."
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _depth(arguments: dict) -> int:
|
|
301
|
+
"""The caller's depth if it is a sane integer; the default otherwise.
|
|
302
|
+
|
|
303
|
+
Agents send strings, floats and worse; a bad value must degrade to the
|
|
304
|
+
default, never crash the tool call."""
|
|
305
|
+
depth = arguments.get("depth")
|
|
306
|
+
if isinstance(depth, bool) or not isinstance(depth, int) or not 1 <= depth <= 10000:
|
|
307
|
+
return 25
|
|
308
|
+
return depth
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
class RippleService:
|
|
312
|
+
"""Builds and caches the lineage graph, rebuilding when the SQL, YAML, or
|
|
313
|
+
LookML files change.
|
|
314
|
+
|
|
315
|
+
The agent's loop is edit SQL then ask breaks; answering from a stale
|
|
316
|
+
graph would be a quiet lie, so staleness is checked on every call.
|
|
317
|
+
"""
|
|
318
|
+
|
|
319
|
+
def __init__(self, path: str = ".", dialect: str | None = None):
|
|
320
|
+
self.path = path
|
|
321
|
+
self.dialect = dialect
|
|
322
|
+
self._graph = None
|
|
323
|
+
self._fingerprint = None
|
|
324
|
+
self._lock = threading.RLock()
|
|
325
|
+
|
|
326
|
+
def _root(self) -> Path:
|
|
327
|
+
from ripple.project import find_project_root
|
|
328
|
+
|
|
329
|
+
return find_project_root(Path(self.path))
|
|
330
|
+
|
|
331
|
+
def _current_fingerprint(self):
|
|
332
|
+
from ripple.sourcefiles import stamp
|
|
333
|
+
|
|
334
|
+
return stamp(self._root())
|
|
335
|
+
|
|
336
|
+
def graph(self, refresh: bool = False):
|
|
337
|
+
# the walk stays outside the lock so requests do not queue behind each other's stats
|
|
338
|
+
fingerprint = self._current_fingerprint()
|
|
339
|
+
# serve() calls this from many threads at once; one build, not one per caller
|
|
340
|
+
with self._lock:
|
|
341
|
+
if self._graph is None or refresh or fingerprint != self._fingerprint:
|
|
342
|
+
from ripple import cache
|
|
343
|
+
from ripple.graph import LineageGraph
|
|
344
|
+
from ripple.project import load_project
|
|
345
|
+
|
|
346
|
+
# Same disk cache as the CLI: an MCP session dies with the client,
|
|
347
|
+
# and without this every client restart re-pays the full build,
|
|
348
|
+
# which on a big repo is exactly the tool-call the client times out.
|
|
349
|
+
root = self._root()
|
|
350
|
+
graph = None if refresh else cache.load(root, self.dialect)
|
|
351
|
+
if graph is None:
|
|
352
|
+
project = load_project(self.path, dialect=self.dialect)
|
|
353
|
+
graph = LineageGraph.build(project)
|
|
354
|
+
cache.store(graph, root, self.dialect)
|
|
355
|
+
self._graph = graph
|
|
356
|
+
self._fingerprint = fingerprint
|
|
357
|
+
return self._graph
|
|
358
|
+
|
|
359
|
+
def call(self, name: str, arguments: dict[str, Any]) -> dict:
|
|
360
|
+
if name == "lineage_summary":
|
|
361
|
+
graph = self.graph(refresh=bool(arguments.get("refresh")))
|
|
362
|
+
summary = graph.stats()
|
|
363
|
+
summary["project_warnings"] = graph.project.warnings
|
|
364
|
+
summary["next"] = _summary_next(graph, summary)
|
|
365
|
+
return summary
|
|
366
|
+
if name == "unresolved_tables":
|
|
367
|
+
unresolved = self.graph().unresolved_tables()
|
|
368
|
+
return {
|
|
369
|
+
"unresolved": unresolved,
|
|
370
|
+
"total": len(unresolved),
|
|
371
|
+
**(
|
|
372
|
+
{"next": INGEST_NEXT}
|
|
373
|
+
if unresolved
|
|
374
|
+
else {"note": "every referenced table resolves; nothing to ingest"}
|
|
375
|
+
),
|
|
376
|
+
}
|
|
377
|
+
if name == "ingest_schema":
|
|
378
|
+
from ripple.schemas import merge_schemas, normalize_tables
|
|
379
|
+
|
|
380
|
+
tables = normalize_tables(arguments)
|
|
381
|
+
before = self.graph().stats()
|
|
382
|
+
delta = merge_schemas(self._root(), tables)
|
|
383
|
+
after = self.graph(refresh=True).stats()
|
|
384
|
+
# the promise of this work is a smaller review count; show whether
|
|
385
|
+
# it moved, including an honest "still N" (the CLI already does)
|
|
386
|
+
delta["review_required_before"] = before["review_required_edges"]
|
|
387
|
+
delta["review_required_after"] = after["review_required_edges"]
|
|
388
|
+
delta["star_only_before"] = before["star_only"]
|
|
389
|
+
delta["star_only_after"] = after["star_only"]
|
|
390
|
+
delta["edges_before"] = before["edges"]
|
|
391
|
+
delta["edges_after"] = after["edges"]
|
|
392
|
+
delta["note"] = (
|
|
393
|
+
"graph rebuilt"
|
|
394
|
+
if (before["review_required_edges"], before["star_only"])
|
|
395
|
+
!= (after["review_required_edges"], after["star_only"])
|
|
396
|
+
else "graph rebuilt; nothing moved, so these tables were not what blocked the review links"
|
|
397
|
+
)
|
|
398
|
+
delta["next"] = (
|
|
399
|
+
"Re-ask breaks(model.column), or call lineage_summary to see how many "
|
|
400
|
+
"links still need review."
|
|
401
|
+
)
|
|
402
|
+
return delta
|
|
403
|
+
if name == "ingest_usage":
|
|
404
|
+
from ripple.usage import ingest_file
|
|
405
|
+
from ripple.usage.ingest import write_store
|
|
406
|
+
|
|
407
|
+
project = self.graph().project
|
|
408
|
+
store = ingest_file(
|
|
409
|
+
arguments["path"],
|
|
410
|
+
[m.name for m in project.models],
|
|
411
|
+
project.dialect,
|
|
412
|
+
model_aliases={m.name: set(m.aliases) for m in project.models},
|
|
413
|
+
)
|
|
414
|
+
write_store(self._root(), store)
|
|
415
|
+
return store
|
|
416
|
+
if name == "collect_usage":
|
|
417
|
+
from ripple.usage import collect as collect_mod
|
|
418
|
+
from ripple.usage.discover import capabilities
|
|
419
|
+
from ripple.usage.ingest import write_store
|
|
420
|
+
|
|
421
|
+
platform = arguments.get("platform")
|
|
422
|
+
if not platform:
|
|
423
|
+
return capabilities()
|
|
424
|
+
project = self.graph().project
|
|
425
|
+
# 55s: finish (or fail loudly) inside a typical 60s client tool budget
|
|
426
|
+
store = collect_mod.collect(
|
|
427
|
+
platform,
|
|
428
|
+
[m.name for m in project.models],
|
|
429
|
+
connection=arguments.get("connection"),
|
|
430
|
+
days=int(arguments.get("days") or 7),
|
|
431
|
+
scope=arguments.get("scope") or "mine",
|
|
432
|
+
region=arguments.get("region") or "us",
|
|
433
|
+
timeout=55.0,
|
|
434
|
+
model_aliases={m.name: set(m.aliases) for m in project.models},
|
|
435
|
+
)
|
|
436
|
+
write_store(self._root(), store)
|
|
437
|
+
return store
|
|
438
|
+
if name == "usage_summary":
|
|
439
|
+
from ripple.usage.ingest import read_store
|
|
440
|
+
|
|
441
|
+
store = read_store(self._root())
|
|
442
|
+
if store is None:
|
|
443
|
+
return {
|
|
444
|
+
"note": (
|
|
445
|
+
"no usage ingested yet: call collect_usage to run the "
|
|
446
|
+
"user's own warehouse CLI locally, or have the user "
|
|
447
|
+
"export query history and call ingest_usage with the path"
|
|
448
|
+
)
|
|
449
|
+
}
|
|
450
|
+
recorded = store.get("models_fingerprint")
|
|
451
|
+
if recorded:
|
|
452
|
+
from ripple.usage.ingest import models_fingerprint
|
|
453
|
+
|
|
454
|
+
current = models_fingerprint([m.name for m in self.graph().project.models])
|
|
455
|
+
if current != recorded:
|
|
456
|
+
store = {
|
|
457
|
+
**store,
|
|
458
|
+
"warning": (
|
|
459
|
+
"stale: the project's models changed since this usage "
|
|
460
|
+
"was ingested; per-model counts reflect the old model "
|
|
461
|
+
"set. Re-run collect_usage or ingest_usage."
|
|
462
|
+
),
|
|
463
|
+
}
|
|
464
|
+
return store
|
|
465
|
+
from ripple.graph import UnknownTarget
|
|
466
|
+
|
|
467
|
+
# the page path is for a person with a screen; a remote agent keeps the text
|
|
468
|
+
if name == "breaks":
|
|
469
|
+
graph = self.graph()
|
|
470
|
+
model, column = self._split(arguments["target"])
|
|
471
|
+
try:
|
|
472
|
+
result = graph.breaks(model, column, max_depth=_depth(arguments))
|
|
473
|
+
except UnknownTarget as e:
|
|
474
|
+
return {"error": str(e), "did_you_mean": e.suggestions}
|
|
475
|
+
if result.get("review_required"):
|
|
476
|
+
result["next"] = _hits_next(result)
|
|
477
|
+
result = attach(result, "breaks", noun_for(graph.project.mode))
|
|
478
|
+
if arguments.get("page"):
|
|
479
|
+
full = plain_text(breaks_lines(result["answer"], full=True))
|
|
480
|
+
page = answer_page.write_for(graph, self._root(), result["answer"], full)
|
|
481
|
+
result["html_path"] = str(page)
|
|
482
|
+
return result
|
|
483
|
+
if name == "trace":
|
|
484
|
+
graph = self.graph()
|
|
485
|
+
model, column = self._split(arguments["target"])
|
|
486
|
+
try:
|
|
487
|
+
result = graph.trace(model, column, max_depth=_depth(arguments))
|
|
488
|
+
except UnknownTarget as e:
|
|
489
|
+
return {"error": str(e), "did_you_mean": e.suggestions}
|
|
490
|
+
if any(h.get("trust") == "review_required" for h in result.get("upstream", [])):
|
|
491
|
+
result["next"] = (
|
|
492
|
+
"Some hops need review. unresolved_tables lists the external tables "
|
|
493
|
+
"whose columns would settle them; ingest_schema them, then ask again."
|
|
494
|
+
)
|
|
495
|
+
result = attach(result, "trace", noun_for(graph.project.mode))
|
|
496
|
+
if arguments.get("page"):
|
|
497
|
+
page = answer_page.write_for(graph, self._root(), result["answer"], result["text"])
|
|
498
|
+
result["html_path"] = str(page)
|
|
499
|
+
return result
|
|
500
|
+
if name == "model_columns":
|
|
501
|
+
graph = self.graph()
|
|
502
|
+
wanted = arguments["model"].lower()
|
|
503
|
+
# a dbt-disabled model is analyzed for its own schema but never
|
|
504
|
+
# binds references; the summary counts it, so it must be nameable
|
|
505
|
+
# (dbt-ga4 keeps 30 such models behind vars)
|
|
506
|
+
candidates = graph._candidates.get(wanted) or graph._all_candidates.get(wanted, [])
|
|
507
|
+
# a name shared by a model and a source is the model's here;
|
|
508
|
+
# two models sharing it is a real ambiguity, refused with the
|
|
509
|
+
# spellings that would settle it
|
|
510
|
+
analyzed = [c for c in candidates if c in graph.reports]
|
|
511
|
+
exact = [c for c in analyzed if c.lower() == wanted]
|
|
512
|
+
if len(exact) == 1:
|
|
513
|
+
analyzed = exact # the spelling is this model's own name
|
|
514
|
+
if len(analyzed) > 1:
|
|
515
|
+
return {
|
|
516
|
+
"error": f"'{arguments['model']}' names {len(analyzed)} models; ask by full name",
|
|
517
|
+
"did_you_mean": sorted(graph._askable_name(c) for c in analyzed),
|
|
518
|
+
}
|
|
519
|
+
report = graph.reports.get(analyzed[0]) if analyzed else None
|
|
520
|
+
if report is None:
|
|
521
|
+
known = sorted(graph.reports.keys())
|
|
522
|
+
close = difflib.get_close_matches(arguments["model"], known, n=5)
|
|
523
|
+
return {"error": f"unknown model '{arguments['model']}'", "did_you_mean": close}
|
|
524
|
+
model = next((m for m in graph.project.models if m.name == report.name), None)
|
|
525
|
+
return {
|
|
526
|
+
"model": report.name,
|
|
527
|
+
"columns": report.columns,
|
|
528
|
+
"status": report.status,
|
|
529
|
+
"analyzed": report.parsed,
|
|
530
|
+
**({"enabled": False} if model is not None and not model.enabled else {}),
|
|
531
|
+
}
|
|
532
|
+
raise ValueError(f"unknown tool: {name}")
|
|
533
|
+
|
|
534
|
+
@staticmethod
|
|
535
|
+
def _split(target: str) -> tuple[str, str]:
|
|
536
|
+
model, _, column = target.rpartition(".")
|
|
537
|
+
if not model:
|
|
538
|
+
raise ValueError(f"expected model.column, got '{target}'")
|
|
539
|
+
return model, column
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
def run_stdio(path: str = ".", dialect: str | None = None) -> None:
|
|
543
|
+
service = RippleService(path=path, dialect=dialect)
|
|
544
|
+
for line in sys.stdin:
|
|
545
|
+
line = line.strip()
|
|
546
|
+
if not line:
|
|
547
|
+
continue
|
|
548
|
+
try:
|
|
549
|
+
message = json.loads(line)
|
|
550
|
+
except json.JSONDecodeError:
|
|
551
|
+
continue
|
|
552
|
+
response = _handle(message, service)
|
|
553
|
+
if response is not None:
|
|
554
|
+
sys.stdout.write(json.dumps(response) + "\n")
|
|
555
|
+
sys.stdout.flush()
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _handle(message: dict, service: RippleService) -> dict | None:
|
|
559
|
+
method = message.get("method")
|
|
560
|
+
msg_id = message.get("id")
|
|
561
|
+
|
|
562
|
+
if method == "initialize":
|
|
563
|
+
requested = message.get("params", {}).get("protocolVersion")
|
|
564
|
+
version = requested if requested in SUPPORTED_PROTOCOL_VERSIONS else PROTOCOL_VERSION
|
|
565
|
+
info = {"name": "ripple", "version": _version()}
|
|
566
|
+
payload = {"protocolVersion": version, "capabilities": {"tools": {}}, "serverInfo": info}
|
|
567
|
+
return _result(msg_id, payload)
|
|
568
|
+
if msg_id is None: # notification
|
|
569
|
+
return None
|
|
570
|
+
if method == "tools/list":
|
|
571
|
+
return _result(msg_id, {"tools": TOOLS})
|
|
572
|
+
if method == "tools/call":
|
|
573
|
+
params = message.get("params", {})
|
|
574
|
+
try:
|
|
575
|
+
payload = service.call(params.get("name", ""), params.get("arguments", {}) or {})
|
|
576
|
+
text = json.dumps(payload, indent=2, default=str)
|
|
577
|
+
return _result(msg_id, {"content": [{"type": "text", "text": text}]})
|
|
578
|
+
except Exception as e:
|
|
579
|
+
content = [{"type": "text", "text": f"error: {e}"}]
|
|
580
|
+
return _result(msg_id, {"content": content, "isError": True})
|
|
581
|
+
if method == "ping":
|
|
582
|
+
return _result(msg_id, {})
|
|
583
|
+
return {
|
|
584
|
+
"jsonrpc": "2.0",
|
|
585
|
+
"id": msg_id,
|
|
586
|
+
"error": {"code": -32601, "message": f"method not found: {method}"},
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def _result(msg_id, payload: dict) -> dict:
|
|
591
|
+
return {"jsonrpc": "2.0", "id": msg_id, "result": payload}
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _version() -> str:
|
|
595
|
+
try:
|
|
596
|
+
from importlib.metadata import version
|
|
597
|
+
|
|
598
|
+
return version("ripple-sql")
|
|
599
|
+
except Exception:
|
|
600
|
+
return "0.1.0"
|
ripple/names.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Which spellings name which model.
|
|
2
|
+
|
|
3
|
+
`LineageGraph._require_target` accepts more than a model's canonical name:
|
|
4
|
+
an alias, a different case, a shorter dotted path. A surface with no graph
|
|
5
|
+
behind it (the answer page, answering a follow-up from an export) has to
|
|
6
|
+
resolve a typed name the same way, so the map it needs is built here rather
|
|
7
|
+
than growing graph.py.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def askable_names(graph) -> dict[str, list[str]]:
|
|
14
|
+
"""Every spelling the graph accepts for a model, mapped to the canonical
|
|
15
|
+
name or names it means.
|
|
16
|
+
|
|
17
|
+
Only the spellings that carry information: one that already is the
|
|
18
|
+
canonical name it resolves to is left out, which keeps the map small
|
|
19
|
+
enough to embed in a page."""
|
|
20
|
+
merged = {**graph._all_candidates, **graph._candidates}
|
|
21
|
+
return {
|
|
22
|
+
spelling: list(canonical)
|
|
23
|
+
for spelling, canonical in merged.items()
|
|
24
|
+
if len(canonical) > 1 or canonical[0] != spelling
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def unlinked_columns(graph) -> list[str]:
|
|
29
|
+
"""Columns no link touches. `select 1 as x` is a real answer (nothing
|
|
30
|
+
downstream reads it), so the page has to know the column exists."""
|
|
31
|
+
linked = set()
|
|
32
|
+
for edge in graph.edges:
|
|
33
|
+
linked.add(f"{edge.src_model}.{edge.src_column}")
|
|
34
|
+
linked.add(f"{edge.dst_model}.{edge.dst_column}")
|
|
35
|
+
return sorted(
|
|
36
|
+
column_id
|
|
37
|
+
for name, report in graph.reports.items()
|
|
38
|
+
for column in report.columns
|
|
39
|
+
if column != "*" and (column_id := f"{name}.{column}") not in linked
|
|
40
|
+
)
|