ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/usage/ingest.py ADDED
@@ -0,0 +1,414 @@
1
+ """Turn a query-history export into per-model counts, refusing what it can't know.
2
+
3
+ Accepts JSONL (one object per line) or a JSON array, so the native output of
4
+ `snow sql --format json` and `bq query --format=json` ingests as-is. Field
5
+ names follow Snowflake's QUERY_HISTORY columns with common aliases, matched
6
+ case-insensitively.
7
+
8
+ Attribution refuses instead of guessing: an unqualified table name in a
9
+ statement with no database/schema context could live anywhere, so it lands in
10
+ a visible "no context" bucket even when it happens to match a model's name.
11
+ That collision (same table name, different database) is exactly how usage
12
+ tools end up confidently wrong.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import json
19
+ from collections import Counter
20
+ from datetime import datetime, timezone
21
+ from pathlib import Path
22
+
23
+ TEXT_KEYS = ("query_text", "text", "sql", "statement")
24
+ ID_KEYS = ("query_id", "id", "statement_id", "job_id")
25
+ START_KEYS = ("start_time", "started_at", "start", "creation_time")
26
+ DATABASE_KEYS = ("database_name", "database", "catalog")
27
+ SCHEMA_KEYS = ("schema_name", "schema")
28
+ PRINCIPAL_KEYS = ("user_name", "user", "principal")
29
+
30
+ # Snowflake's INFORMATION_SCHEMA.QUERY_HISTORY hard cap. Exactly this many
31
+ # rows almost always means the export was cut short, not that the week ended.
32
+ SNOWFLAKE_ROW_CAP = 10_000
33
+
34
+ OUTSIDE_TABLE_LIMIT = 20
35
+
36
+
37
+ def store_path(root: str | Path) -> Path:
38
+ return Path(root) / ".ripple" / "usage.json"
39
+
40
+
41
+ def models_fingerprint(model_names: list[str]) -> str:
42
+ """Identity of the model set a store was computed against, so a summary
43
+ read after models were added or renamed can say 'stale' instead of
44
+ silently reporting the old project's inventory."""
45
+ return hashlib.sha256("|".join(sorted({n.lower() for n in model_names})).encode()).hexdigest()
46
+
47
+
48
+ def _get(record: dict, keys: tuple[str, ...]):
49
+ lowered = {str(k).lower(): v for k, v in record.items()}
50
+ for key in keys:
51
+ value = lowered.get(key)
52
+ if value not in (None, ""):
53
+ return value
54
+ return None
55
+
56
+
57
+ def _parse_time(value) -> datetime | None:
58
+ if value is None:
59
+ return None
60
+ if isinstance(value, (int, float)):
61
+ try:
62
+ seconds = value / 1000.0 if value > 1e12 else float(value)
63
+ return datetime.fromtimestamp(seconds, tz=timezone.utc)
64
+ except (OverflowError, OSError, ValueError):
65
+ return None
66
+ text = str(value).strip().replace("Z", "+00:00")
67
+ try:
68
+ parsed = datetime.fromisoformat(text)
69
+ except ValueError:
70
+ return None
71
+ if parsed.tzinfo is None:
72
+ # a valid export can mix offset-less and offset-bearing rows; naive
73
+ # values read as UTC, or min()/max() over the window raises TypeError
74
+ parsed = parsed.replace(tzinfo=timezone.utc)
75
+ return parsed
76
+
77
+
78
+ def _load_records(raw: bytes) -> tuple[list[dict], int]:
79
+ """Records plus the count of lines that were not usable."""
80
+ text = raw.decode("utf-8", errors="replace").strip()
81
+ if text.startswith("["):
82
+ try:
83
+ data = json.loads(text)
84
+ except json.JSONDecodeError:
85
+ return [], 1
86
+ records = [r for r in data if isinstance(r, dict)]
87
+ return records, len(data) - len(records)
88
+ records, invalid = [], 0
89
+ for line in text.splitlines():
90
+ line = line.strip()
91
+ if not line:
92
+ continue
93
+ try:
94
+ record = json.loads(line)
95
+ except json.JSONDecodeError:
96
+ invalid += 1
97
+ continue
98
+ if isinstance(record, dict):
99
+ records.append(record)
100
+ else:
101
+ invalid += 1
102
+ return records, invalid
103
+
104
+
105
+ def _statement_refs(statement) -> list[tuple[str, str, bool, bool]]:
106
+ """(name, display, qualified, is_write) per table reference, CTEs excluded.
107
+
108
+ Only statements that read or write data count. GRANT, SHOW, DESCRIBE,
109
+ ALTER and the rest are administrative: counting them made a model look
110
+ alive because someone touched its permissions, and SHOW TABLES IN a
111
+ schema fabricated the schema itself as an outside table.
112
+ """
113
+ from sqlglot import exp
114
+
115
+ activity_kinds = (
116
+ exp.Query,
117
+ exp.Insert,
118
+ exp.Create,
119
+ exp.Merge,
120
+ exp.Update,
121
+ exp.Delete,
122
+ exp.Drop,
123
+ exp.Copy,
124
+ )
125
+ if not isinstance(statement, activity_kinds):
126
+ return []
127
+ ctes = {c.alias_or_name.lower() for c in statement.find_all(exp.CTE) if c.alias_or_name}
128
+ write_ids: set[int] = set()
129
+ if isinstance(
130
+ statement, (exp.Insert, exp.Create, exp.Merge, exp.Update, exp.Delete, exp.Drop, exp.Copy)
131
+ ):
132
+ target = statement.this
133
+ if target is not None:
134
+ nodes = [target] if isinstance(target, exp.Table) else list(target.find_all(exp.Table))
135
+ write_ids.update(id(node) for node in nodes)
136
+
137
+ refs = []
138
+ for node in statement.find_all(exp.Table):
139
+ name = (node.name or "").lower()
140
+ if not name:
141
+ continue
142
+ if name in ctes and not (node.db or node.catalog):
143
+ # only an UNQUALIFIED reference can mean the CTE; a qualified
144
+ # analytics.orders inside a CTE named orders is the real table,
145
+ # and dropping it erased the statement's only genuine read
146
+ continue
147
+ parts = [p for p in (node.catalog, node.db, node.name) if p]
148
+ refs.append(
149
+ (name, ".".join(parts).lower(), bool(node.db or node.catalog), id(node) in write_ids)
150
+ )
151
+ return refs
152
+
153
+
154
+ def _identity_index(model_aliases: dict[str, set[str]] | None) -> tuple[dict, dict]:
155
+ """(dotted alias -> model, model -> dotted alias set), lowercased.
156
+
157
+ Only uniquely owned aliases resolve; a dotted spelling claimed by two
158
+ models resolves to neither. Bare aliases carry no identity and are
159
+ ignored here.
160
+ """
161
+ if not model_aliases:
162
+ return {}, {}
163
+ owners: dict[str, set[str]] = {}
164
+ dotted: dict[str, set[str]] = {}
165
+ for model, aliases in model_aliases.items():
166
+ lowered = model.lower()
167
+ for alias in aliases or ():
168
+ a = str(alias).lower()
169
+ if "." not in a:
170
+ continue
171
+ owners.setdefault(a, set()).add(lowered)
172
+ dotted.setdefault(lowered, set()).add(a)
173
+ index = {a: next(iter(ms)) for a, ms in owners.items() if len(ms) == 1}
174
+ return index, dotted
175
+
176
+
177
+ def _suffix_compatible(display: str, aliases: set[str]) -> bool:
178
+ return any(
179
+ display == a or a.endswith("." + display) or display.endswith("." + a) for a in aliases
180
+ )
181
+
182
+
183
+ def _ref_prefix(display: str, session_db: str | None) -> str | None:
184
+ """The database-ish evidence a leaf match rests on, for conflict
185
+ detection. A three-part reference names its database outright;
186
+ otherwise the session database is the evidence; a two-part reference
187
+ outside any session contributes its first token."""
188
+ parts = display.split(".")
189
+ if len(parts) >= 3:
190
+ return parts[0]
191
+ if session_db:
192
+ return session_db
193
+ if len(parts) == 2:
194
+ return parts[0]
195
+ return None
196
+
197
+
198
+ def ingest_file(
199
+ path: str | Path,
200
+ model_names: list[str],
201
+ dialect: str | None,
202
+ model_aliases: dict[str, set[str]] | None = None,
203
+ ) -> dict:
204
+ """Aggregate one export into the store dict. Raises ValueError when the
205
+ file holds no usable records, with the expected shape spelled out.
206
+
207
+ Attribution runs in two phases so identity conflicts are seen before
208
+ any counting: phase one parses every statement and gathers which
209
+ database each leaf-name match would rest on; phase two attributes.
210
+ A model with dbt-manifest identity (dotted aliases) rejects foreign
211
+ databases outright; a model without identity that is referenced from
212
+ several databases in one export is refused into the ambiguous bucket,
213
+ because merging them is exactly how usage counts end up confidently
214
+ wrong.
215
+ """
216
+ import sqlglot
217
+
218
+ path = Path(path)
219
+ raw = path.read_bytes()
220
+ records, invalid = _load_records(raw)
221
+ if not records:
222
+ raise ValueError(
223
+ f"no usable records in {path.name}. Expected JSONL (one object per line) "
224
+ "or a JSON array, each record with a query_text field "
225
+ "(Snowflake QUERY_HISTORY column names work as-is)."
226
+ )
227
+
228
+ occurrences = Counter(name.lower() for name in model_names)
229
+ unique_models = {n for n, c in occurrences.items() if c == 1}
230
+ ambiguous_models = {n for n, c in occurrences.items() if c > 1}
231
+ alias_index, dotted_aliases = _identity_index(model_aliases)
232
+
233
+ stats = dict.fromkeys(
234
+ (
235
+ "total",
236
+ "touched_tables",
237
+ "matched",
238
+ "no_context",
239
+ "outside_only",
240
+ "no_tables",
241
+ "unparsed",
242
+ ),
243
+ 0,
244
+ )
245
+ principals: set[str] = set()
246
+ seen_ids: set[str] = set()
247
+ duplicates = 0
248
+ times: list[datetime] = []
249
+ parsed_records: list[tuple] = []
250
+ leaf_prefixes: dict[str, set[str]] = {}
251
+
252
+ for record in records:
253
+ text = _get(record, TEXT_KEYS)
254
+ if not text or not isinstance(text, str):
255
+ invalid += 1
256
+ continue
257
+ qid = _get(record, ID_KEYS)
258
+ if qid is not None:
259
+ qid = str(qid)
260
+ if qid in seen_ids:
261
+ duplicates += 1
262
+ continue
263
+ seen_ids.add(qid)
264
+
265
+ stats["total"] += 1
266
+ principal = _get(record, PRINCIPAL_KEYS)
267
+ if principal:
268
+ principals.add(str(principal))
269
+ started = _parse_time(_get(record, START_KEYS))
270
+ if started:
271
+ times.append(started)
272
+ session_db = _get(record, DATABASE_KEYS)
273
+ session_db = str(session_db).lower() if session_db else None
274
+ has_context = bool(session_db or _get(record, SCHEMA_KEYS))
275
+
276
+ try:
277
+ statements = sqlglot.parse(text, read=dialect)
278
+ except Exception:
279
+ stats["unparsed"] += 1
280
+ continue
281
+
282
+ refs = {}
283
+ for statement in statements:
284
+ if statement is None:
285
+ continue
286
+ for name, display, qualified, is_write in _statement_refs(statement):
287
+ # keyed by the qualified spelling: a join of a.foo and b.foo
288
+ # is two tables, not one
289
+ key = (display, is_write)
290
+ prior = refs.get(key)
291
+ refs[key] = (name, (prior[1] if prior else False) or qualified)
292
+ if not refs:
293
+ stats["no_tables"] += 1
294
+ continue
295
+ stats["touched_tables"] += 1
296
+ parsed_records.append((started, has_context, refs))
297
+
298
+ for (display, _is_write), (name, qualified) in refs.items():
299
+ if (
300
+ name in unique_models
301
+ and name not in dotted_aliases
302
+ and display not in alias_index
303
+ and (qualified or has_context)
304
+ ):
305
+ prefix = _ref_prefix(display, session_db)
306
+ if prefix:
307
+ leaf_prefixes.setdefault(name, set()).add(prefix)
308
+
309
+ conflicted = {name for name, prefixes in leaf_prefixes.items() if len(prefixes) > 1}
310
+
311
+ models: dict[str, dict] = {}
312
+ outside: dict[str, int] = {}
313
+ ambiguous_hit: set[str] = set()
314
+
315
+ def count(name: str, is_write: bool, started) -> None:
316
+ entry = models.setdefault(name, {"reads": 0, "writes": 0, "last_seen": None})
317
+ entry["writes" if is_write else "reads"] += 1
318
+ if started and (entry["last_seen"] is None or started.isoformat() > entry["last_seen"]):
319
+ entry["last_seen"] = started.isoformat()
320
+
321
+ for started, has_context, refs in parsed_records:
322
+ matched = refused = external = 0
323
+ for (display, is_write), (name, qualified) in refs.items():
324
+ resolved = alias_index.get(display)
325
+ if resolved is not None:
326
+ matched += 1
327
+ count(resolved, is_write, started)
328
+ continue
329
+ if not qualified and not has_context:
330
+ # could be any database's table with this name; refusing here
331
+ # is the whole difference between a count and a guess
332
+ refused += 1
333
+ continue
334
+ if name in unique_models:
335
+ known = dotted_aliases.get(name)
336
+ if known and qualified and not _suffix_compatible(display, known):
337
+ # identity is known and this reference names somewhere
338
+ # else: another database's table, not this model
339
+ external += 1
340
+ outside[display] = outside.get(display, 0) + 1
341
+ continue
342
+ if name in conflicted:
343
+ ambiguous_hit.add(name)
344
+ continue
345
+ matched += 1
346
+ count(name, is_write, started)
347
+ elif name in ambiguous_models:
348
+ ambiguous_hit.add(name)
349
+ else:
350
+ external += 1
351
+ outside[display] = outside.get(display, 0) + 1
352
+
353
+ if matched:
354
+ stats["matched"] += 1
355
+ elif refused:
356
+ stats["no_context"] += 1
357
+ elif external:
358
+ stats["outside_only"] += 1
359
+
360
+ window_start = min(times).isoformat() if times else None
361
+ window_end = max(times).isoformat() if times else None
362
+ days = ((max(times) - min(times)).total_seconds() / 86400.0) if len(times) > 1 else None
363
+
364
+ truncated = "unknown (hand-carried export)"
365
+ if len(records) == SNOWFLAKE_ROW_CAP:
366
+ cap_name = (
367
+ "Snowflake's INFORMATION_SCHEMA cap"
368
+ if dialect == "snowflake"
369
+ else "the export's row limit"
370
+ )
371
+ truncated = (
372
+ f"likely: exactly {SNOWFLAKE_ROW_CAP:,} rows is {cap_name}, "
373
+ "so the window was probably cut short"
374
+ )
375
+
376
+ top_outside = dict(sorted(outside.items(), key=lambda kv: -kv[1])[:OUTSIDE_TABLE_LIMIT])
377
+ seen = set(models)
378
+ not_seen = sorted(n for n in unique_models if n not in seen)
379
+
380
+ return {
381
+ "version": 1,
382
+ "models_fingerprint": models_fingerprint(model_names),
383
+ "manifest": {
384
+ "source": path.name,
385
+ "sha256": hashlib.sha256(raw).hexdigest(),
386
+ "ingested_at": datetime.now(tz=timezone.utc).isoformat(),
387
+ "invalid_lines": invalid,
388
+ "duplicate_ids": duplicates,
389
+ "window": {"start": window_start, "end": window_end, "days": days},
390
+ "principals": len(principals),
391
+ "truncated": truncated,
392
+ },
393
+ "project_models": len(unique_models) + len(ambiguous_models),
394
+ "statements": stats,
395
+ "models": models,
396
+ "not_seen": not_seen,
397
+ "ambiguous_names": sorted(ambiguous_hit),
398
+ "outside": top_outside,
399
+ }
400
+
401
+
402
+ def write_store(root: str | Path, store: dict) -> Path:
403
+ path = store_path(root)
404
+ path.parent.mkdir(parents=True, exist_ok=True)
405
+ path.write_text(json.dumps(store, indent=2, sort_keys=True) + "\n", encoding="utf-8")
406
+ return path
407
+
408
+
409
+ def read_store(root: str | Path) -> dict | None:
410
+ path = store_path(root)
411
+ try:
412
+ return json.loads(path.read_text(encoding="utf-8"))
413
+ except (OSError, json.JSONDecodeError):
414
+ return None
ripple/usage/report.py ADDED
@@ -0,0 +1,131 @@
1
+ """Render the usage store for a human. Plain text, honesty built into the copy.
2
+
3
+ The banned word is banned here above all: nothing this module prints may call
4
+ a model "unused". The window is always printed next to any absence claim,
5
+ because the absence is a fact about the window, not about the model.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ SEEN_LIMIT = 10
11
+ NOT_SEEN_LIMIT = 8
12
+ OUTSIDE_LIMIT = 5
13
+
14
+
15
+ def _day(iso: str | None) -> str:
16
+ return iso[:10] if iso else "?"
17
+
18
+
19
+ def _window_phrase(manifest: dict) -> str:
20
+ window = manifest["window"]
21
+ if not window["start"]:
22
+ return "no timestamps in this export"
23
+ phrase = f"{_day(window['start'])} to {_day(window['end'])}"
24
+ if window["days"] is not None:
25
+ phrase += f" ({window['days']:.1f} days)"
26
+ return phrase
27
+
28
+
29
+ def render(store: dict) -> str:
30
+ manifest = store["manifest"]
31
+ stats = store["statements"]
32
+ users = f"{manifest['principals']} user" + ("s" if manifest["principals"] != 1 else "")
33
+ collected = manifest.get("collected")
34
+ # a collected store's "source" is a temp spool filename the user never
35
+ # made; say where the rows actually came from instead
36
+ source = (
37
+ f"{collected['platform']} ({collected['connection']})" if collected else manifest["source"]
38
+ )
39
+ lines = [f"{stats['total']:,} statements from {source} · {_window_phrase(manifest)} · {users}"]
40
+ if not str(manifest["truncated"]).startswith("unknown"):
41
+ lines.append(f" export truncated? {manifest['truncated']}")
42
+
43
+ touched = stats["touched_tables"]
44
+ if touched:
45
+ pct = 100.0 * stats["matched"] / touched
46
+ lines.append(
47
+ f"\n{stats['matched']:,} of {touched:,} table-touching statements "
48
+ f"matched this project's models ({pct:.0f}%)"
49
+ )
50
+ parts = []
51
+ if stats["no_context"]:
52
+ parts.append(f"{stats['no_context']:,} skipped rather than guessed (no database context)")
53
+ if stats["outside_only"]:
54
+ parts.append(f"{stats['outside_only']:,} only touch tables outside this project")
55
+ if stats["unparsed"]:
56
+ parts.append(f"{stats['unparsed']:,} would not parse")
57
+ if stats["no_tables"]:
58
+ parts.append(f"{stats['no_tables']:,} touch no tables (SHOW, USE, ...)")
59
+ for part in parts:
60
+ lines.append(f" {part}")
61
+
62
+ models = store["models"]
63
+ if models:
64
+ lines.append(f"\nSeen running: {len(models)} of {store['project_models']} models")
65
+ ranked = sorted(models.items(), key=lambda kv: -(kv[1]["reads"] + kv[1]["writes"]))
66
+ width = max(len(name) for name, _ in ranked[:SEEN_LIMIT])
67
+ for name, entry in ranked[:SEEN_LIMIT]:
68
+ lines.append(
69
+ f" {name.ljust(width)} {entry['reads']} reads · {entry['writes']} writes · "
70
+ f"last seen {_day(entry['last_seen'])}"
71
+ )
72
+ if len(ranked) > SEEN_LIMIT:
73
+ lines.append(f" ...and {len(ranked) - SEEN_LIMIT} more")
74
+
75
+ not_seen = store["not_seen"]
76
+ if not_seen:
77
+ preview = ", ".join(not_seen[:NOT_SEEN_LIMIT])
78
+ more = (
79
+ f", and {len(not_seen) - NOT_SEEN_LIMIT} more" if len(not_seen) > NOT_SEEN_LIMIT else ""
80
+ )
81
+ lines.append(f"\nNot seen in this window: {len(not_seen)} models")
82
+ lines.append(f" {preview}{more}")
83
+ lines.append(" This window can't call a model dead: a quarterly job looks identical to a")
84
+ lines.append(" dead one. Not seen here means exactly that, and nothing more.")
85
+
86
+ if store["ambiguous_names"]:
87
+ lines.append(
88
+ f"\n{len(store['ambiguous_names'])} name(s) claimed by several models or several "
89
+ f"databases were left uncounted rather than guessed: "
90
+ f"{', '.join(store['ambiguous_names'])}"
91
+ )
92
+
93
+ outside = store["outside"]
94
+ if outside:
95
+ lines.append("\nAlso read by these queries, but defined outside this project:")
96
+ ranked_outside = sorted(outside.items(), key=lambda kv: -kv[1])
97
+ for table, count in ranked_outside[:OUTSIDE_LIMIT]:
98
+ lines.append(f" {table} ({count})")
99
+ if len(ranked_outside) > OUTSIDE_LIMIT:
100
+ lines.append(f" ...and {len(ranked_outside) - OUTSIDE_LIMIT} more")
101
+ lines.append(" That's activity your SQL files alone can't see: dashboards, scripts,")
102
+ lines.append(" other teams. Their columns can be added with ripple ingest-schema.")
103
+
104
+ return "\n".join(lines)
105
+
106
+
107
+ def render_empty() -> str:
108
+ return """No usage data yet. The short way, if snow, bq, or databricks is set up here:
109
+
110
+ ripple collect-usage
111
+
112
+ runs your own warehouse CLI with the login you already have, ingests the
113
+ result, and deletes the raw rows. Or export by hand and hand the file over:
114
+
115
+ ripple ingest-usage history.json
116
+
117
+ Snowflake, your own queries, last 7 days, no permission needed:
118
+
119
+ snow sql -q "SELECT query_id, query_text, database_name, schema_name,
120
+ user_name, start_time
121
+ FROM TABLE(INFORMATION_SCHEMA.QUERY_HISTORY(RESULT_LIMIT => 10000))" \\
122
+ --format json > history.json
123
+
124
+ Getting exactly 10,000 rows back means you hit Snowflake's cap and the window
125
+ was cut short; Ripple will say so. For everyone's queries over 365 days, the
126
+ same columns come from SNOWFLAKE.ACCOUNT_USAGE.QUERY_HISTORY, which needs one
127
+ admin grant (IMPORTED PRIVILEGES on the SNOWFLAKE database).
128
+
129
+ Any JSONL or JSON-array file with a query_text field works, whichever
130
+ warehouse it came from. The file stays on this machine, and Ripple stores
131
+ per-table counts only, never the query text itself."""