ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/usage/ingest.py
ADDED
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
"""Turn a query-history export into per-model counts, refusing what it can't know.
|
|
2
|
+
|
|
3
|
+
Accepts JSONL (one object per line) or a JSON array, so the native output of
|
|
4
|
+
`snow sql --format json` and `bq query --format=json` ingests as-is. Field
|
|
5
|
+
names follow Snowflake's QUERY_HISTORY columns with common aliases, matched
|
|
6
|
+
case-insensitively.
|
|
7
|
+
|
|
8
|
+
Attribution refuses instead of guessing: an unqualified table name in a
|
|
9
|
+
statement with no database/schema context could live anywhere, so it lands in
|
|
10
|
+
a visible "no context" bucket even when it happens to match a model's name.
|
|
11
|
+
That collision (same table name, different database) is exactly how usage
|
|
12
|
+
tools end up confidently wrong.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import json
|
|
19
|
+
from collections import Counter
|
|
20
|
+
from datetime import datetime, timezone
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
TEXT_KEYS = ("query_text", "text", "sql", "statement")
|
|
24
|
+
ID_KEYS = ("query_id", "id", "statement_id", "job_id")
|
|
25
|
+
START_KEYS = ("start_time", "started_at", "start", "creation_time")
|
|
26
|
+
DATABASE_KEYS = ("database_name", "database", "catalog")
|
|
27
|
+
SCHEMA_KEYS = ("schema_name", "schema")
|
|
28
|
+
PRINCIPAL_KEYS = ("user_name", "user", "principal")
|
|
29
|
+
|
|
30
|
+
# Snowflake's INFORMATION_SCHEMA.QUERY_HISTORY hard cap. Exactly this many
|
|
31
|
+
# rows almost always means the export was cut short, not that the week ended.
|
|
32
|
+
SNOWFLAKE_ROW_CAP = 10_000
|
|
33
|
+
|
|
34
|
+
OUTSIDE_TABLE_LIMIT = 20
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def store_path(root: str | Path) -> Path:
|
|
38
|
+
return Path(root) / ".ripple" / "usage.json"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def models_fingerprint(model_names: list[str]) -> str:
|
|
42
|
+
"""Identity of the model set a store was computed against, so a summary
|
|
43
|
+
read after models were added or renamed can say 'stale' instead of
|
|
44
|
+
silently reporting the old project's inventory."""
|
|
45
|
+
return hashlib.sha256("|".join(sorted({n.lower() for n in model_names})).encode()).hexdigest()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _get(record: dict, keys: tuple[str, ...]):
|
|
49
|
+
lowered = {str(k).lower(): v for k, v in record.items()}
|
|
50
|
+
for key in keys:
|
|
51
|
+
value = lowered.get(key)
|
|
52
|
+
if value not in (None, ""):
|
|
53
|
+
return value
|
|
54
|
+
return None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _parse_time(value) -> datetime | None:
|
|
58
|
+
if value is None:
|
|
59
|
+
return None
|
|
60
|
+
if isinstance(value, (int, float)):
|
|
61
|
+
try:
|
|
62
|
+
seconds = value / 1000.0 if value > 1e12 else float(value)
|
|
63
|
+
return datetime.fromtimestamp(seconds, tz=timezone.utc)
|
|
64
|
+
except (OverflowError, OSError, ValueError):
|
|
65
|
+
return None
|
|
66
|
+
text = str(value).strip().replace("Z", "+00:00")
|
|
67
|
+
try:
|
|
68
|
+
parsed = datetime.fromisoformat(text)
|
|
69
|
+
except ValueError:
|
|
70
|
+
return None
|
|
71
|
+
if parsed.tzinfo is None:
|
|
72
|
+
# a valid export can mix offset-less and offset-bearing rows; naive
|
|
73
|
+
# values read as UTC, or min()/max() over the window raises TypeError
|
|
74
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
75
|
+
return parsed
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _load_records(raw: bytes) -> tuple[list[dict], int]:
|
|
79
|
+
"""Records plus the count of lines that were not usable."""
|
|
80
|
+
text = raw.decode("utf-8", errors="replace").strip()
|
|
81
|
+
if text.startswith("["):
|
|
82
|
+
try:
|
|
83
|
+
data = json.loads(text)
|
|
84
|
+
except json.JSONDecodeError:
|
|
85
|
+
return [], 1
|
|
86
|
+
records = [r for r in data if isinstance(r, dict)]
|
|
87
|
+
return records, len(data) - len(records)
|
|
88
|
+
records, invalid = [], 0
|
|
89
|
+
for line in text.splitlines():
|
|
90
|
+
line = line.strip()
|
|
91
|
+
if not line:
|
|
92
|
+
continue
|
|
93
|
+
try:
|
|
94
|
+
record = json.loads(line)
|
|
95
|
+
except json.JSONDecodeError:
|
|
96
|
+
invalid += 1
|
|
97
|
+
continue
|
|
98
|
+
if isinstance(record, dict):
|
|
99
|
+
records.append(record)
|
|
100
|
+
else:
|
|
101
|
+
invalid += 1
|
|
102
|
+
return records, invalid
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _statement_refs(statement) -> list[tuple[str, str, bool, bool]]:
|
|
106
|
+
"""(name, display, qualified, is_write) per table reference, CTEs excluded.
|
|
107
|
+
|
|
108
|
+
Only statements that read or write data count. GRANT, SHOW, DESCRIBE,
|
|
109
|
+
ALTER and the rest are administrative: counting them made a model look
|
|
110
|
+
alive because someone touched its permissions, and SHOW TABLES IN a
|
|
111
|
+
schema fabricated the schema itself as an outside table.
|
|
112
|
+
"""
|
|
113
|
+
from sqlglot import exp
|
|
114
|
+
|
|
115
|
+
activity_kinds = (
|
|
116
|
+
exp.Query,
|
|
117
|
+
exp.Insert,
|
|
118
|
+
exp.Create,
|
|
119
|
+
exp.Merge,
|
|
120
|
+
exp.Update,
|
|
121
|
+
exp.Delete,
|
|
122
|
+
exp.Drop,
|
|
123
|
+
exp.Copy,
|
|
124
|
+
)
|
|
125
|
+
if not isinstance(statement, activity_kinds):
|
|
126
|
+
return []
|
|
127
|
+
ctes = {c.alias_or_name.lower() for c in statement.find_all(exp.CTE) if c.alias_or_name}
|
|
128
|
+
write_ids: set[int] = set()
|
|
129
|
+
if isinstance(
|
|
130
|
+
statement, (exp.Insert, exp.Create, exp.Merge, exp.Update, exp.Delete, exp.Drop, exp.Copy)
|
|
131
|
+
):
|
|
132
|
+
target = statement.this
|
|
133
|
+
if target is not None:
|
|
134
|
+
nodes = [target] if isinstance(target, exp.Table) else list(target.find_all(exp.Table))
|
|
135
|
+
write_ids.update(id(node) for node in nodes)
|
|
136
|
+
|
|
137
|
+
refs = []
|
|
138
|
+
for node in statement.find_all(exp.Table):
|
|
139
|
+
name = (node.name or "").lower()
|
|
140
|
+
if not name:
|
|
141
|
+
continue
|
|
142
|
+
if name in ctes and not (node.db or node.catalog):
|
|
143
|
+
# only an UNQUALIFIED reference can mean the CTE; a qualified
|
|
144
|
+
# analytics.orders inside a CTE named orders is the real table,
|
|
145
|
+
# and dropping it erased the statement's only genuine read
|
|
146
|
+
continue
|
|
147
|
+
parts = [p for p in (node.catalog, node.db, node.name) if p]
|
|
148
|
+
refs.append(
|
|
149
|
+
(name, ".".join(parts).lower(), bool(node.db or node.catalog), id(node) in write_ids)
|
|
150
|
+
)
|
|
151
|
+
return refs
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _identity_index(model_aliases: dict[str, set[str]] | None) -> tuple[dict, dict]:
|
|
155
|
+
"""(dotted alias -> model, model -> dotted alias set), lowercased.
|
|
156
|
+
|
|
157
|
+
Only uniquely owned aliases resolve; a dotted spelling claimed by two
|
|
158
|
+
models resolves to neither. Bare aliases carry no identity and are
|
|
159
|
+
ignored here.
|
|
160
|
+
"""
|
|
161
|
+
if not model_aliases:
|
|
162
|
+
return {}, {}
|
|
163
|
+
owners: dict[str, set[str]] = {}
|
|
164
|
+
dotted: dict[str, set[str]] = {}
|
|
165
|
+
for model, aliases in model_aliases.items():
|
|
166
|
+
lowered = model.lower()
|
|
167
|
+
for alias in aliases or ():
|
|
168
|
+
a = str(alias).lower()
|
|
169
|
+
if "." not in a:
|
|
170
|
+
continue
|
|
171
|
+
owners.setdefault(a, set()).add(lowered)
|
|
172
|
+
dotted.setdefault(lowered, set()).add(a)
|
|
173
|
+
index = {a: next(iter(ms)) for a, ms in owners.items() if len(ms) == 1}
|
|
174
|
+
return index, dotted
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _suffix_compatible(display: str, aliases: set[str]) -> bool:
|
|
178
|
+
return any(
|
|
179
|
+
display == a or a.endswith("." + display) or display.endswith("." + a) for a in aliases
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _ref_prefix(display: str, session_db: str | None) -> str | None:
|
|
184
|
+
"""The database-ish evidence a leaf match rests on, for conflict
|
|
185
|
+
detection. A three-part reference names its database outright;
|
|
186
|
+
otherwise the session database is the evidence; a two-part reference
|
|
187
|
+
outside any session contributes its first token."""
|
|
188
|
+
parts = display.split(".")
|
|
189
|
+
if len(parts) >= 3:
|
|
190
|
+
return parts[0]
|
|
191
|
+
if session_db:
|
|
192
|
+
return session_db
|
|
193
|
+
if len(parts) == 2:
|
|
194
|
+
return parts[0]
|
|
195
|
+
return None
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def ingest_file(
|
|
199
|
+
path: str | Path,
|
|
200
|
+
model_names: list[str],
|
|
201
|
+
dialect: str | None,
|
|
202
|
+
model_aliases: dict[str, set[str]] | None = None,
|
|
203
|
+
) -> dict:
|
|
204
|
+
"""Aggregate one export into the store dict. Raises ValueError when the
|
|
205
|
+
file holds no usable records, with the expected shape spelled out.
|
|
206
|
+
|
|
207
|
+
Attribution runs in two phases so identity conflicts are seen before
|
|
208
|
+
any counting: phase one parses every statement and gathers which
|
|
209
|
+
database each leaf-name match would rest on; phase two attributes.
|
|
210
|
+
A model with dbt-manifest identity (dotted aliases) rejects foreign
|
|
211
|
+
databases outright; a model without identity that is referenced from
|
|
212
|
+
several databases in one export is refused into the ambiguous bucket,
|
|
213
|
+
because merging them is exactly how usage counts end up confidently
|
|
214
|
+
wrong.
|
|
215
|
+
"""
|
|
216
|
+
import sqlglot
|
|
217
|
+
|
|
218
|
+
path = Path(path)
|
|
219
|
+
raw = path.read_bytes()
|
|
220
|
+
records, invalid = _load_records(raw)
|
|
221
|
+
if not records:
|
|
222
|
+
raise ValueError(
|
|
223
|
+
f"no usable records in {path.name}. Expected JSONL (one object per line) "
|
|
224
|
+
"or a JSON array, each record with a query_text field "
|
|
225
|
+
"(Snowflake QUERY_HISTORY column names work as-is)."
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
occurrences = Counter(name.lower() for name in model_names)
|
|
229
|
+
unique_models = {n for n, c in occurrences.items() if c == 1}
|
|
230
|
+
ambiguous_models = {n for n, c in occurrences.items() if c > 1}
|
|
231
|
+
alias_index, dotted_aliases = _identity_index(model_aliases)
|
|
232
|
+
|
|
233
|
+
stats = dict.fromkeys(
|
|
234
|
+
(
|
|
235
|
+
"total",
|
|
236
|
+
"touched_tables",
|
|
237
|
+
"matched",
|
|
238
|
+
"no_context",
|
|
239
|
+
"outside_only",
|
|
240
|
+
"no_tables",
|
|
241
|
+
"unparsed",
|
|
242
|
+
),
|
|
243
|
+
0,
|
|
244
|
+
)
|
|
245
|
+
principals: set[str] = set()
|
|
246
|
+
seen_ids: set[str] = set()
|
|
247
|
+
duplicates = 0
|
|
248
|
+
times: list[datetime] = []
|
|
249
|
+
parsed_records: list[tuple] = []
|
|
250
|
+
leaf_prefixes: dict[str, set[str]] = {}
|
|
251
|
+
|
|
252
|
+
for record in records:
|
|
253
|
+
text = _get(record, TEXT_KEYS)
|
|
254
|
+
if not text or not isinstance(text, str):
|
|
255
|
+
invalid += 1
|
|
256
|
+
continue
|
|
257
|
+
qid = _get(record, ID_KEYS)
|
|
258
|
+
if qid is not None:
|
|
259
|
+
qid = str(qid)
|
|
260
|
+
if qid in seen_ids:
|
|
261
|
+
duplicates += 1
|
|
262
|
+
continue
|
|
263
|
+
seen_ids.add(qid)
|
|
264
|
+
|
|
265
|
+
stats["total"] += 1
|
|
266
|
+
principal = _get(record, PRINCIPAL_KEYS)
|
|
267
|
+
if principal:
|
|
268
|
+
principals.add(str(principal))
|
|
269
|
+
started = _parse_time(_get(record, START_KEYS))
|
|
270
|
+
if started:
|
|
271
|
+
times.append(started)
|
|
272
|
+
session_db = _get(record, DATABASE_KEYS)
|
|
273
|
+
session_db = str(session_db).lower() if session_db else None
|
|
274
|
+
has_context = bool(session_db or _get(record, SCHEMA_KEYS))
|
|
275
|
+
|
|
276
|
+
try:
|
|
277
|
+
statements = sqlglot.parse(text, read=dialect)
|
|
278
|
+
except Exception:
|
|
279
|
+
stats["unparsed"] += 1
|
|
280
|
+
continue
|
|
281
|
+
|
|
282
|
+
refs = {}
|
|
283
|
+
for statement in statements:
|
|
284
|
+
if statement is None:
|
|
285
|
+
continue
|
|
286
|
+
for name, display, qualified, is_write in _statement_refs(statement):
|
|
287
|
+
# keyed by the qualified spelling: a join of a.foo and b.foo
|
|
288
|
+
# is two tables, not one
|
|
289
|
+
key = (display, is_write)
|
|
290
|
+
prior = refs.get(key)
|
|
291
|
+
refs[key] = (name, (prior[1] if prior else False) or qualified)
|
|
292
|
+
if not refs:
|
|
293
|
+
stats["no_tables"] += 1
|
|
294
|
+
continue
|
|
295
|
+
stats["touched_tables"] += 1
|
|
296
|
+
parsed_records.append((started, has_context, refs))
|
|
297
|
+
|
|
298
|
+
for (display, _is_write), (name, qualified) in refs.items():
|
|
299
|
+
if (
|
|
300
|
+
name in unique_models
|
|
301
|
+
and name not in dotted_aliases
|
|
302
|
+
and display not in alias_index
|
|
303
|
+
and (qualified or has_context)
|
|
304
|
+
):
|
|
305
|
+
prefix = _ref_prefix(display, session_db)
|
|
306
|
+
if prefix:
|
|
307
|
+
leaf_prefixes.setdefault(name, set()).add(prefix)
|
|
308
|
+
|
|
309
|
+
conflicted = {name for name, prefixes in leaf_prefixes.items() if len(prefixes) > 1}
|
|
310
|
+
|
|
311
|
+
models: dict[str, dict] = {}
|
|
312
|
+
outside: dict[str, int] = {}
|
|
313
|
+
ambiguous_hit: set[str] = set()
|
|
314
|
+
|
|
315
|
+
def count(name: str, is_write: bool, started) -> None:
|
|
316
|
+
entry = models.setdefault(name, {"reads": 0, "writes": 0, "last_seen": None})
|
|
317
|
+
entry["writes" if is_write else "reads"] += 1
|
|
318
|
+
if started and (entry["last_seen"] is None or started.isoformat() > entry["last_seen"]):
|
|
319
|
+
entry["last_seen"] = started.isoformat()
|
|
320
|
+
|
|
321
|
+
for started, has_context, refs in parsed_records:
|
|
322
|
+
matched = refused = external = 0
|
|
323
|
+
for (display, is_write), (name, qualified) in refs.items():
|
|
324
|
+
resolved = alias_index.get(display)
|
|
325
|
+
if resolved is not None:
|
|
326
|
+
matched += 1
|
|
327
|
+
count(resolved, is_write, started)
|
|
328
|
+
continue
|
|
329
|
+
if not qualified and not has_context:
|
|
330
|
+
# could be any database's table with this name; refusing here
|
|
331
|
+
# is the whole difference between a count and a guess
|
|
332
|
+
refused += 1
|
|
333
|
+
continue
|
|
334
|
+
if name in unique_models:
|
|
335
|
+
known = dotted_aliases.get(name)
|
|
336
|
+
if known and qualified and not _suffix_compatible(display, known):
|
|
337
|
+
# identity is known and this reference names somewhere
|
|
338
|
+
# else: another database's table, not this model
|
|
339
|
+
external += 1
|
|
340
|
+
outside[display] = outside.get(display, 0) + 1
|
|
341
|
+
continue
|
|
342
|
+
if name in conflicted:
|
|
343
|
+
ambiguous_hit.add(name)
|
|
344
|
+
continue
|
|
345
|
+
matched += 1
|
|
346
|
+
count(name, is_write, started)
|
|
347
|
+
elif name in ambiguous_models:
|
|
348
|
+
ambiguous_hit.add(name)
|
|
349
|
+
else:
|
|
350
|
+
external += 1
|
|
351
|
+
outside[display] = outside.get(display, 0) + 1
|
|
352
|
+
|
|
353
|
+
if matched:
|
|
354
|
+
stats["matched"] += 1
|
|
355
|
+
elif refused:
|
|
356
|
+
stats["no_context"] += 1
|
|
357
|
+
elif external:
|
|
358
|
+
stats["outside_only"] += 1
|
|
359
|
+
|
|
360
|
+
window_start = min(times).isoformat() if times else None
|
|
361
|
+
window_end = max(times).isoformat() if times else None
|
|
362
|
+
days = ((max(times) - min(times)).total_seconds() / 86400.0) if len(times) > 1 else None
|
|
363
|
+
|
|
364
|
+
truncated = "unknown (hand-carried export)"
|
|
365
|
+
if len(records) == SNOWFLAKE_ROW_CAP:
|
|
366
|
+
cap_name = (
|
|
367
|
+
"Snowflake's INFORMATION_SCHEMA cap"
|
|
368
|
+
if dialect == "snowflake"
|
|
369
|
+
else "the export's row limit"
|
|
370
|
+
)
|
|
371
|
+
truncated = (
|
|
372
|
+
f"likely: exactly {SNOWFLAKE_ROW_CAP:,} rows is {cap_name}, "
|
|
373
|
+
"so the window was probably cut short"
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
top_outside = dict(sorted(outside.items(), key=lambda kv: -kv[1])[:OUTSIDE_TABLE_LIMIT])
|
|
377
|
+
seen = set(models)
|
|
378
|
+
not_seen = sorted(n for n in unique_models if n not in seen)
|
|
379
|
+
|
|
380
|
+
return {
|
|
381
|
+
"version": 1,
|
|
382
|
+
"models_fingerprint": models_fingerprint(model_names),
|
|
383
|
+
"manifest": {
|
|
384
|
+
"source": path.name,
|
|
385
|
+
"sha256": hashlib.sha256(raw).hexdigest(),
|
|
386
|
+
"ingested_at": datetime.now(tz=timezone.utc).isoformat(),
|
|
387
|
+
"invalid_lines": invalid,
|
|
388
|
+
"duplicate_ids": duplicates,
|
|
389
|
+
"window": {"start": window_start, "end": window_end, "days": days},
|
|
390
|
+
"principals": len(principals),
|
|
391
|
+
"truncated": truncated,
|
|
392
|
+
},
|
|
393
|
+
"project_models": len(unique_models) + len(ambiguous_models),
|
|
394
|
+
"statements": stats,
|
|
395
|
+
"models": models,
|
|
396
|
+
"not_seen": not_seen,
|
|
397
|
+
"ambiguous_names": sorted(ambiguous_hit),
|
|
398
|
+
"outside": top_outside,
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def write_store(root: str | Path, store: dict) -> Path:
|
|
403
|
+
path = store_path(root)
|
|
404
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
405
|
+
path.write_text(json.dumps(store, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
|
406
|
+
return path
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def read_store(root: str | Path) -> dict | None:
|
|
410
|
+
path = store_path(root)
|
|
411
|
+
try:
|
|
412
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
413
|
+
except (OSError, json.JSONDecodeError):
|
|
414
|
+
return None
|
ripple/usage/report.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Render the usage store for a human. Plain text, honesty built into the copy.
|
|
2
|
+
|
|
3
|
+
The banned word is banned here above all: nothing this module prints may call
|
|
4
|
+
a model "unused". The window is always printed next to any absence claim,
|
|
5
|
+
because the absence is a fact about the window, not about the model.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
SEEN_LIMIT = 10
|
|
11
|
+
NOT_SEEN_LIMIT = 8
|
|
12
|
+
OUTSIDE_LIMIT = 5
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _day(iso: str | None) -> str:
|
|
16
|
+
return iso[:10] if iso else "?"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _window_phrase(manifest: dict) -> str:
|
|
20
|
+
window = manifest["window"]
|
|
21
|
+
if not window["start"]:
|
|
22
|
+
return "no timestamps in this export"
|
|
23
|
+
phrase = f"{_day(window['start'])} to {_day(window['end'])}"
|
|
24
|
+
if window["days"] is not None:
|
|
25
|
+
phrase += f" ({window['days']:.1f} days)"
|
|
26
|
+
return phrase
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def render(store: dict) -> str:
|
|
30
|
+
manifest = store["manifest"]
|
|
31
|
+
stats = store["statements"]
|
|
32
|
+
users = f"{manifest['principals']} user" + ("s" if manifest["principals"] != 1 else "")
|
|
33
|
+
collected = manifest.get("collected")
|
|
34
|
+
# a collected store's "source" is a temp spool filename the user never
|
|
35
|
+
# made; say where the rows actually came from instead
|
|
36
|
+
source = (
|
|
37
|
+
f"{collected['platform']} ({collected['connection']})" if collected else manifest["source"]
|
|
38
|
+
)
|
|
39
|
+
lines = [f"{stats['total']:,} statements from {source} · {_window_phrase(manifest)} · {users}"]
|
|
40
|
+
if not str(manifest["truncated"]).startswith("unknown"):
|
|
41
|
+
lines.append(f" export truncated? {manifest['truncated']}")
|
|
42
|
+
|
|
43
|
+
touched = stats["touched_tables"]
|
|
44
|
+
if touched:
|
|
45
|
+
pct = 100.0 * stats["matched"] / touched
|
|
46
|
+
lines.append(
|
|
47
|
+
f"\n{stats['matched']:,} of {touched:,} table-touching statements "
|
|
48
|
+
f"matched this project's models ({pct:.0f}%)"
|
|
49
|
+
)
|
|
50
|
+
parts = []
|
|
51
|
+
if stats["no_context"]:
|
|
52
|
+
parts.append(f"{stats['no_context']:,} skipped rather than guessed (no database context)")
|
|
53
|
+
if stats["outside_only"]:
|
|
54
|
+
parts.append(f"{stats['outside_only']:,} only touch tables outside this project")
|
|
55
|
+
if stats["unparsed"]:
|
|
56
|
+
parts.append(f"{stats['unparsed']:,} would not parse")
|
|
57
|
+
if stats["no_tables"]:
|
|
58
|
+
parts.append(f"{stats['no_tables']:,} touch no tables (SHOW, USE, ...)")
|
|
59
|
+
for part in parts:
|
|
60
|
+
lines.append(f" {part}")
|
|
61
|
+
|
|
62
|
+
models = store["models"]
|
|
63
|
+
if models:
|
|
64
|
+
lines.append(f"\nSeen running: {len(models)} of {store['project_models']} models")
|
|
65
|
+
ranked = sorted(models.items(), key=lambda kv: -(kv[1]["reads"] + kv[1]["writes"]))
|
|
66
|
+
width = max(len(name) for name, _ in ranked[:SEEN_LIMIT])
|
|
67
|
+
for name, entry in ranked[:SEEN_LIMIT]:
|
|
68
|
+
lines.append(
|
|
69
|
+
f" {name.ljust(width)} {entry['reads']} reads · {entry['writes']} writes · "
|
|
70
|
+
f"last seen {_day(entry['last_seen'])}"
|
|
71
|
+
)
|
|
72
|
+
if len(ranked) > SEEN_LIMIT:
|
|
73
|
+
lines.append(f" ...and {len(ranked) - SEEN_LIMIT} more")
|
|
74
|
+
|
|
75
|
+
not_seen = store["not_seen"]
|
|
76
|
+
if not_seen:
|
|
77
|
+
preview = ", ".join(not_seen[:NOT_SEEN_LIMIT])
|
|
78
|
+
more = (
|
|
79
|
+
f", and {len(not_seen) - NOT_SEEN_LIMIT} more" if len(not_seen) > NOT_SEEN_LIMIT else ""
|
|
80
|
+
)
|
|
81
|
+
lines.append(f"\nNot seen in this window: {len(not_seen)} models")
|
|
82
|
+
lines.append(f" {preview}{more}")
|
|
83
|
+
lines.append(" This window can't call a model dead: a quarterly job looks identical to a")
|
|
84
|
+
lines.append(" dead one. Not seen here means exactly that, and nothing more.")
|
|
85
|
+
|
|
86
|
+
if store["ambiguous_names"]:
|
|
87
|
+
lines.append(
|
|
88
|
+
f"\n{len(store['ambiguous_names'])} name(s) claimed by several models or several "
|
|
89
|
+
f"databases were left uncounted rather than guessed: "
|
|
90
|
+
f"{', '.join(store['ambiguous_names'])}"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
outside = store["outside"]
|
|
94
|
+
if outside:
|
|
95
|
+
lines.append("\nAlso read by these queries, but defined outside this project:")
|
|
96
|
+
ranked_outside = sorted(outside.items(), key=lambda kv: -kv[1])
|
|
97
|
+
for table, count in ranked_outside[:OUTSIDE_LIMIT]:
|
|
98
|
+
lines.append(f" {table} ({count})")
|
|
99
|
+
if len(ranked_outside) > OUTSIDE_LIMIT:
|
|
100
|
+
lines.append(f" ...and {len(ranked_outside) - OUTSIDE_LIMIT} more")
|
|
101
|
+
lines.append(" That's activity your SQL files alone can't see: dashboards, scripts,")
|
|
102
|
+
lines.append(" other teams. Their columns can be added with ripple ingest-schema.")
|
|
103
|
+
|
|
104
|
+
return "\n".join(lines)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def render_empty() -> str:
|
|
108
|
+
return """No usage data yet. The short way, if snow, bq, or databricks is set up here:
|
|
109
|
+
|
|
110
|
+
ripple collect-usage
|
|
111
|
+
|
|
112
|
+
runs your own warehouse CLI with the login you already have, ingests the
|
|
113
|
+
result, and deletes the raw rows. Or export by hand and hand the file over:
|
|
114
|
+
|
|
115
|
+
ripple ingest-usage history.json
|
|
116
|
+
|
|
117
|
+
Snowflake, your own queries, last 7 days, no permission needed:
|
|
118
|
+
|
|
119
|
+
snow sql -q "SELECT query_id, query_text, database_name, schema_name,
|
|
120
|
+
user_name, start_time
|
|
121
|
+
FROM TABLE(INFORMATION_SCHEMA.QUERY_HISTORY(RESULT_LIMIT => 10000))" \\
|
|
122
|
+
--format json > history.json
|
|
123
|
+
|
|
124
|
+
Getting exactly 10,000 rows back means you hit Snowflake's cap and the window
|
|
125
|
+
was cut short; Ripple will say so. For everyone's queries over 365 days, the
|
|
126
|
+
same columns come from SNOWFLAKE.ACCOUNT_USAGE.QUERY_HISTORY, which needs one
|
|
127
|
+
admin grant (IMPORTED PRIVILEGES on the SNOWFLAKE database).
|
|
128
|
+
|
|
129
|
+
Any JSONL or JSON-array file with a query_text field works, whichever
|
|
130
|
+
warehouse it came from. The file stays on this machine, and Ripple stores
|
|
131
|
+
per-table counts only, never the query text itself."""
|