ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Base tables declared by a schema file beside no SQL.
|
|
2
|
+
|
|
3
|
+
Mozilla's bigquery-etl loads many tables from outside the repo and describes
|
|
4
|
+
each one only by sql/<project>/<dataset>/<table>/schema.yaml (BigQuery's
|
|
5
|
+
`fields:` format). Without those columns a view's SELECT * over such a table
|
|
6
|
+
cannot expand, and every view downstream of it degrades to review_required.
|
|
7
|
+
A directory holding a schema file and no .sql becomes a base-table model
|
|
8
|
+
carrying the declared columns, the same way CREATE TABLE (col defs) does.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
import yaml
|
|
17
|
+
|
|
18
|
+
from ripple.loaders.identity import directory_identity
|
|
19
|
+
from ripple.loaders.types import IGNORE_DIRS, Model, relative_posix
|
|
20
|
+
|
|
21
|
+
SIDECAR_NAMES = ("schema.yaml", "schema.yml", "schema.json")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _declared_fields(path: Path) -> list[str]:
|
|
25
|
+
try:
|
|
26
|
+
text = path.read_text(errors="replace", encoding="utf-8")
|
|
27
|
+
doc = json.loads(text) if path.suffix == ".json" else yaml.safe_load(text)
|
|
28
|
+
except (OSError, ValueError, yaml.YAMLError):
|
|
29
|
+
return []
|
|
30
|
+
fields = doc.get("fields") if isinstance(doc, dict) else doc
|
|
31
|
+
if not isinstance(fields, list):
|
|
32
|
+
return []
|
|
33
|
+
return [f["name"] for f in fields if isinstance(f, dict) and isinstance(f.get("name"), str)]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def sidecar_base_tables(root: Path, sql_dirs: set[Path], dialect: str) -> list[Model]:
|
|
37
|
+
models: list[Model] = []
|
|
38
|
+
for name in SIDECAR_NAMES:
|
|
39
|
+
for path in sorted(root.rglob(name)):
|
|
40
|
+
parts = path.relative_to(root).parts[:-1]
|
|
41
|
+
if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
|
|
42
|
+
continue
|
|
43
|
+
if path.parent in sql_dirs or len(parts) < 1:
|
|
44
|
+
continue
|
|
45
|
+
columns = _declared_fields(path)
|
|
46
|
+
if not columns:
|
|
47
|
+
continue
|
|
48
|
+
rel = relative_posix(path, root)
|
|
49
|
+
identity = directory_identity(str(Path(rel).with_name("query.sql")))
|
|
50
|
+
if identity is None:
|
|
51
|
+
continue
|
|
52
|
+
table, dotted = identity
|
|
53
|
+
model = Model(
|
|
54
|
+
name=table,
|
|
55
|
+
sql="",
|
|
56
|
+
path=rel,
|
|
57
|
+
declared_columns=columns,
|
|
58
|
+
evidence="plain_sql",
|
|
59
|
+
dialect=dialect,
|
|
60
|
+
uid=f"{rel}::{table}",
|
|
61
|
+
)
|
|
62
|
+
model.aliases |= {a.lower() for a in dotted}
|
|
63
|
+
models.append(model)
|
|
64
|
+
sql_dirs.add(path.parent) # one sidecar per directory wins
|
|
65
|
+
return models
|
ripple/loaders/sqldir.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""Plain SQL directory loading, with the per-file parse budget."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from ripple.engine.budget import TIMED_OUT, budget_seconds, max_sql_bytes, run_within_budget
|
|
8
|
+
from ripple.loaders.identity import directory_identity
|
|
9
|
+
from ripple.loaders.types import IGNORE_DIRS, Model, Project, relative_posix
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _iter_sql_files(root: Path, exclude: list[Path] | None = None) -> tuple[list[Path], int]:
|
|
13
|
+
"""Every .sql under root, minus vendored/build/hidden dirs.
|
|
14
|
+
|
|
15
|
+
Returns (kept files, count skipped) so the loader can report what it
|
|
16
|
+
passed over instead of silently dropping it. `exclude` drops whole subtrees,
|
|
17
|
+
used to keep a dbt project's macros out of the beside-dbt fallback.
|
|
18
|
+
"""
|
|
19
|
+
kept: list[Path] = []
|
|
20
|
+
skipped = 0
|
|
21
|
+
excluded = [d.resolve() for d in (exclude or [])]
|
|
22
|
+
for path in sorted(root.rglob("*.sql")):
|
|
23
|
+
dirs = path.relative_to(root).parts[:-1]
|
|
24
|
+
if any(part in IGNORE_DIRS or part.startswith(".") for part in dirs):
|
|
25
|
+
skipped += 1
|
|
26
|
+
continue
|
|
27
|
+
if any(path.resolve().is_relative_to(d) for d in excluded):
|
|
28
|
+
skipped += 1
|
|
29
|
+
continue
|
|
30
|
+
kept.append(path)
|
|
31
|
+
return kept, skipped
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _sniff_sql_dir_dialect(root: Path, sample: str) -> tuple[str, bool]:
|
|
35
|
+
"""(dialect, assumed). Read the SQL itself for a dialect signal (backticks,
|
|
36
|
+
psql directives, ENGINE=, INFORMATION_SCHEMA); postgres is the honest default
|
|
37
|
+
for a loose pile of .sql, and `assumed` tells the caller to say so."""
|
|
38
|
+
if (root / "supabase" / "config.toml").is_file() or (root / "supabase").is_dir():
|
|
39
|
+
return "postgres", False
|
|
40
|
+
from ripple.engine.sql_script import sniff_dialect
|
|
41
|
+
|
|
42
|
+
sniffed = sniff_dialect(sample)
|
|
43
|
+
if sniffed:
|
|
44
|
+
return sniffed, False
|
|
45
|
+
return "postgres", True
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _expand_within_budget(
|
|
49
|
+
stem: str,
|
|
50
|
+
text: str,
|
|
51
|
+
dialect: str | None,
|
|
52
|
+
budget: float,
|
|
53
|
+
warnings: list[str] | None = None,
|
|
54
|
+
):
|
|
55
|
+
"""expand_script under the shared budget; None means the budget was blown."""
|
|
56
|
+
from ripple.engine import sql_script
|
|
57
|
+
|
|
58
|
+
result = run_within_budget(
|
|
59
|
+
stem, lambda: sql_script.expand_script(stem, text, dialect, warnings=warnings), budget
|
|
60
|
+
)
|
|
61
|
+
return None if result is TIMED_OUT else result
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _load_sql_dir(root: Path, dialect: str | None, exclude: list[Path] | None = None) -> Project:
|
|
65
|
+
files, skipped = _iter_sql_files(root, exclude)
|
|
66
|
+
texts = [(f, f.read_text(errors="replace", encoding="utf-8")) for f in files]
|
|
67
|
+
|
|
68
|
+
from ripple.engine.sql_script import sniff_dialect
|
|
69
|
+
|
|
70
|
+
warnings: list[str] = []
|
|
71
|
+
assumed = False
|
|
72
|
+
if dialect:
|
|
73
|
+
resolved = dialect
|
|
74
|
+
else:
|
|
75
|
+
# sample a slice of EVERY file, not just the first few: a bigquery signal
|
|
76
|
+
# (INFORMATION_SCHEMA) or mysql signal may live deep in the tree.
|
|
77
|
+
sample = "\n".join(t[:4000] for _, t in texts)[:200000]
|
|
78
|
+
resolved, assumed = _sniff_sql_dir_dialect(root, sample)
|
|
79
|
+
if assumed and texts:
|
|
80
|
+
warnings.append(
|
|
81
|
+
f"Assuming {resolved} dialect (no dbt profile or warehouse to read). "
|
|
82
|
+
"Pass --dialect to override."
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
# a .sql file can hold one query or a 15,000-statement dump. Expand each into
|
|
86
|
+
# the models it actually defines, so a dump becomes its views and tables, not
|
|
87
|
+
# one unparseable blob.
|
|
88
|
+
models: list[Model] = []
|
|
89
|
+
stem_claims: list[tuple[Model, str]] = []
|
|
90
|
+
budget = budget_seconds()
|
|
91
|
+
for sql_file, text in texts:
|
|
92
|
+
rel = relative_posix(sql_file, root)
|
|
93
|
+
# per-file dialect: one repo can hold the same schema in several dialects
|
|
94
|
+
# (Chinook ships Postgres, MySQL, Oracle, SQL Server variants side by side).
|
|
95
|
+
# An explicit --dialect wins; otherwise sniff this file, then fall back to
|
|
96
|
+
# the repo-level guess.
|
|
97
|
+
file_dialect = dialect or sniff_dialect(text) or resolved
|
|
98
|
+
limit = max_sql_bytes()
|
|
99
|
+
if limit and len(text) > limit and ";" not in text:
|
|
100
|
+
# a multi-MB file with no statement separator is one generated
|
|
101
|
+
# statement (a 3.7MB one-liner cost 13s to parse and 139s to
|
|
102
|
+
# analyze); a dump of many small statements has semicolons and
|
|
103
|
+
# still expands normally
|
|
104
|
+
warnings.append(
|
|
105
|
+
f"{rel}: a single {len(text) / 1e6:.1f}MB statement, skipped as "
|
|
106
|
+
"generated (RIPPLE_MAX_SQL_BYTES to raise)."
|
|
107
|
+
)
|
|
108
|
+
models.append(
|
|
109
|
+
Model(
|
|
110
|
+
name=sql_file.stem,
|
|
111
|
+
sql="",
|
|
112
|
+
path=rel,
|
|
113
|
+
evidence="plain_sql",
|
|
114
|
+
dialect=file_dialect,
|
|
115
|
+
uid=f"{rel}::{sql_file.stem}",
|
|
116
|
+
load_error=(
|
|
117
|
+
f"single statement is {len(text) / 1e6:.1f}MB, over the "
|
|
118
|
+
f"{limit / 1e6:g}MB limit: {rel}"
|
|
119
|
+
),
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
continue
|
|
123
|
+
script_warnings: list[str] = []
|
|
124
|
+
expanded = _expand_within_budget(
|
|
125
|
+
sql_file.stem, text, file_dialect, budget, warnings=script_warnings
|
|
126
|
+
)
|
|
127
|
+
warnings.extend(f"{rel}: {w}" for w in script_warnings)
|
|
128
|
+
if expanded is None:
|
|
129
|
+
warnings.append(
|
|
130
|
+
f"{rel}: parse exceeded the {budget:g}s per-file budget; reported as "
|
|
131
|
+
"timed_out (RIPPLE_PARSE_BUDGET_S to raise)."
|
|
132
|
+
)
|
|
133
|
+
models.append(
|
|
134
|
+
Model(
|
|
135
|
+
name=sql_file.stem,
|
|
136
|
+
sql="",
|
|
137
|
+
path=rel,
|
|
138
|
+
evidence="plain_sql",
|
|
139
|
+
dialect=file_dialect,
|
|
140
|
+
uid=f"{rel}::{sql_file.stem}",
|
|
141
|
+
load_error=f"parse timed out after {budget:g}s: {rel}",
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
continue
|
|
145
|
+
# dedup only WITHIN a file (a base table and a later INSERT/CTAS that fills
|
|
146
|
+
# it). Same name across different files is a real collision that
|
|
147
|
+
# _qualify_collisions path-qualifies, so it must NOT be dropped here.
|
|
148
|
+
# dedup is keyed by the QUALIFIED identity, not the bare name:
|
|
149
|
+
# sales.summary and hr.summary in one migration are different
|
|
150
|
+
# relations, and merging them would bind hr references to the sales
|
|
151
|
+
# derivation (the review of the round-3 fixes)
|
|
152
|
+
here: dict[str, Model] = {}
|
|
153
|
+
clone_here: set[str] = set()
|
|
154
|
+
for sm in expanded:
|
|
155
|
+
identity = (sm.qualified_name or sm.name).lower()
|
|
156
|
+
existing = here.get(identity)
|
|
157
|
+
if existing is not None:
|
|
158
|
+
# an empty base table upgrades to any derivation (old rule);
|
|
159
|
+
# a schema clone (select * ... where 0=1) upgrades to the
|
|
160
|
+
# later real derivation, which it exists to receive
|
|
161
|
+
# (usaspending's subaward loader, holdout round 2). A real
|
|
162
|
+
# derivation is never replaced. Parents always merge: a
|
|
163
|
+
# second statement writing the same target is a real
|
|
164
|
+
# dependency even when its SQL is not the one analyzed.
|
|
165
|
+
replaceable = not existing.sql.strip() or (
|
|
166
|
+
identity in clone_here and not sm.schema_clone
|
|
167
|
+
)
|
|
168
|
+
if replaceable and sm.sql.strip():
|
|
169
|
+
existing.sql = sm.sql
|
|
170
|
+
# the analyzed SQL is now the UPDATE's; without its
|
|
171
|
+
# self_read the graph drops the prior-state edges as
|
|
172
|
+
# CTE loops (the review of PR #28)
|
|
173
|
+
existing.self_read = sm.self_read
|
|
174
|
+
if sm.schema_clone:
|
|
175
|
+
clone_here.add(identity)
|
|
176
|
+
else:
|
|
177
|
+
clone_here.discard(identity)
|
|
178
|
+
elif sm.self_read and sm.sql.strip():
|
|
179
|
+
# a later UPDATE of a target that already has a real
|
|
180
|
+
# derivation adds lineage, it does not replace any: the
|
|
181
|
+
# statement joins the model's analyzed set (nycdb,
|
|
182
|
+
# PR #28 gap 3)
|
|
183
|
+
existing.extra_sqls.append(sm.sql)
|
|
184
|
+
existing.self_read = True
|
|
185
|
+
existing.declared_parents |= set(sm.parents)
|
|
186
|
+
continue
|
|
187
|
+
model = Model(
|
|
188
|
+
name=sm.name,
|
|
189
|
+
sql=sm.sql,
|
|
190
|
+
path=rel,
|
|
191
|
+
declared_parents=set(sm.parents),
|
|
192
|
+
declared_columns=list(sm.columns),
|
|
193
|
+
evidence="plain_sql",
|
|
194
|
+
dialect=file_dialect,
|
|
195
|
+
uid=f"{rel}::{identity}",
|
|
196
|
+
self_read=sm.self_read,
|
|
197
|
+
is_function=sm.is_function,
|
|
198
|
+
)
|
|
199
|
+
# a bare SELECT is named after its file; when the file is named
|
|
200
|
+
# for its role (query.sql), the directory is the relation
|
|
201
|
+
by_directory = (
|
|
202
|
+
directory_identity(rel) if sm.name.lower() == sql_file.stem.lower() else None
|
|
203
|
+
)
|
|
204
|
+
if by_directory:
|
|
205
|
+
model.name, dotted = by_directory
|
|
206
|
+
model.aliases |= {a.lower() for a in dotted}
|
|
207
|
+
if sm.schema_clone:
|
|
208
|
+
clone_here.add(identity)
|
|
209
|
+
# the qualified spelling the SQL wrote (disclosure.v_sum) answers
|
|
210
|
+
# queries and folds scoring; the display name stays bare
|
|
211
|
+
if sm.qualified_name and sm.qualified_name.lower() != sm.name.lower():
|
|
212
|
+
model.aliases.add(sm.qualified_name.lower())
|
|
213
|
+
here[identity] = model
|
|
214
|
+
models.append(model)
|
|
215
|
+
# a file is known to its team by its filename: claim the stem for the
|
|
216
|
+
# file's relations, granted after the sweep only if no other file and
|
|
217
|
+
# no real model owns the name. A multi-relation file's stem becomes a
|
|
218
|
+
# deliberately ambiguous alias; the graph disambiguates by the asked
|
|
219
|
+
# column (patentsview's webtool_tables, holdout round 5) and raises
|
|
220
|
+
# when several claimants carry it.
|
|
221
|
+
stem = sql_file.stem.lower()
|
|
222
|
+
for defined in here.values():
|
|
223
|
+
if defined.name.lower() != stem:
|
|
224
|
+
stem_claims.append((rel, defined, stem))
|
|
225
|
+
|
|
226
|
+
from ripple.loaders.sidecar import sidecar_base_tables
|
|
227
|
+
|
|
228
|
+
models.extend(sidecar_base_tables(root, {f.parent for f in files}, resolved))
|
|
229
|
+
|
|
230
|
+
# grant stem aliases only where the name is unowned and claimed by ONE
|
|
231
|
+
# file: a stem matching a real model, or shared by two files, would
|
|
232
|
+
# silently redirect breaks/trace to the wrong relation
|
|
233
|
+
taken = {m.name.lower() for m in models}
|
|
234
|
+
for m in models:
|
|
235
|
+
taken |= {a.lower() for a in m.aliases}
|
|
236
|
+
files_per_stem: dict[str, set[str]] = {}
|
|
237
|
+
for claim_rel, _, stem in stem_claims:
|
|
238
|
+
files_per_stem.setdefault(stem, set()).add(claim_rel)
|
|
239
|
+
for _, claimant, stem in stem_claims:
|
|
240
|
+
if len(files_per_stem[stem]) == 1 and stem not in taken:
|
|
241
|
+
claimant.stem_aliases.add(stem)
|
|
242
|
+
|
|
243
|
+
if skipped:
|
|
244
|
+
warnings.append(
|
|
245
|
+
f"Skipped {skipped} .sql file{'s' if skipped != 1 else ''} in "
|
|
246
|
+
"vendored or build directories."
|
|
247
|
+
)
|
|
248
|
+
if not models:
|
|
249
|
+
warnings.append(
|
|
250
|
+
f"No .sql files found under {root}. Run ripple inside a dbt project "
|
|
251
|
+
"or a folder that contains .sql files."
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
return Project(
|
|
255
|
+
root=root,
|
|
256
|
+
dialect=resolved,
|
|
257
|
+
mode="sql-dir",
|
|
258
|
+
models=models,
|
|
259
|
+
sources=[],
|
|
260
|
+
warnings=warnings,
|
|
261
|
+
dialect_assumed=assumed,
|
|
262
|
+
)
|
ripple/loaders/types.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
"""Shared model/project types and discovery constants."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
ADAPTER_TO_DIALECT = {
|
|
10
|
+
"snowflake": "snowflake",
|
|
11
|
+
"bigquery": "bigquery",
|
|
12
|
+
"databricks": "databricks",
|
|
13
|
+
"spark": "spark",
|
|
14
|
+
"redshift": "redshift",
|
|
15
|
+
"postgres": "postgres",
|
|
16
|
+
"duckdb": "duckdb",
|
|
17
|
+
"trino": "trino",
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
# ref()/source() anywhere in the Jinja, including nested inside macro args
|
|
21
|
+
# like {{ dbt_audit(ref('base'), ...) }}. Two-arg ref('package', 'model')
|
|
22
|
+
# takes the last argument as the model name.
|
|
23
|
+
_REF_RE = re.compile(r"\bref\(\s*['\"]([^'\"]+)['\"]\s*(?:,\s*['\"]([^'\"]+)['\"]\s*)?\)")
|
|
24
|
+
_SOURCE_RE = re.compile(r"\bsource\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)")
|
|
25
|
+
_VAR_NAME_RE = re.compile(r"\bvar\(\s*['\"]([^'\"]+)['\"]")
|
|
26
|
+
|
|
27
|
+
# Directories that never hold the project's own SQL. Skipped when scanning a
|
|
28
|
+
# plain repo so the map is the code you wrote, not vendored or generated junk.
|
|
29
|
+
# Hidden dirs (dotfiles) are skipped separately.
|
|
30
|
+
IGNORE_DIRS = frozenset(
|
|
31
|
+
{
|
|
32
|
+
"node_modules",
|
|
33
|
+
"bower_components",
|
|
34
|
+
"venv",
|
|
35
|
+
"env",
|
|
36
|
+
"virtualenv",
|
|
37
|
+
"site-packages",
|
|
38
|
+
"__pycache__",
|
|
39
|
+
"target",
|
|
40
|
+
"dist",
|
|
41
|
+
"build",
|
|
42
|
+
"out",
|
|
43
|
+
"dbt_packages",
|
|
44
|
+
"dbt_modules",
|
|
45
|
+
"vendor",
|
|
46
|
+
"vendored",
|
|
47
|
+
"third_party",
|
|
48
|
+
}
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class Model:
|
|
54
|
+
name: str
|
|
55
|
+
sql: str
|
|
56
|
+
path: str
|
|
57
|
+
# names this model can be referenced by in SQL (relation name, alias, bare name)
|
|
58
|
+
aliases: set[str] = field(default_factory=set)
|
|
59
|
+
# parent model/source names declared by dbt (manifest mode) or found via ref()/source()
|
|
60
|
+
declared_parents: set[str] = field(default_factory=set)
|
|
61
|
+
is_source: bool = False
|
|
62
|
+
# "manifest" | "raw_jinja" | "plain_sql"
|
|
63
|
+
evidence: str = "plain_sql"
|
|
64
|
+
# per-model SQL dialect. In a monorepo each dbt project is its own island
|
|
65
|
+
# with its own warehouse; a model parses in its island's dialect, not one
|
|
66
|
+
# dialect forced on the whole tree. None means "use the project default".
|
|
67
|
+
dialect: str | None = None
|
|
68
|
+
# columns declared by a CREATE TABLE (the schema). A base table has these but
|
|
69
|
+
# no SQL to analyze; the graph uses them so a downstream SELECT * can resolve.
|
|
70
|
+
declared_columns: list[str] = field(default_factory=list)
|
|
71
|
+
# stable identity: dbt unique_id in manifest mode, project-relative path
|
|
72
|
+
# elsewhere. Display names can collide; this must not.
|
|
73
|
+
uid: str = ""
|
|
74
|
+
# vars declared by THIS model's own dbt_project.yml. Like dialect: set per
|
|
75
|
+
# model in a monorepo so islands keep their own values; None means "use the
|
|
76
|
+
# project default".
|
|
77
|
+
jinja_vars: dict | None = None
|
|
78
|
+
# dbt enabled config (dbt_project.yml +enabled or {{ config(enabled=...) }}).
|
|
79
|
+
# A disabled model stays a visible node but never binds refs or feeds
|
|
80
|
+
# parent schemas: dbt would not build it.
|
|
81
|
+
enabled: bool = True
|
|
82
|
+
# set when the file's parse blew the per-file budget; the graph reports the
|
|
83
|
+
# model as timed_out instead of pretending it was analyzed
|
|
84
|
+
load_error: str | None = None
|
|
85
|
+
# an UPDATE-derived model reads its own prior state; the graph keeps
|
|
86
|
+
# its self edges instead of treating them as CTE loops
|
|
87
|
+
self_read: bool = False
|
|
88
|
+
# further analyzed statements for the same relation: a table built by
|
|
89
|
+
# CREATE plus several UPDATEs (in one file or across files) is ONE
|
|
90
|
+
# model whose lineage is the union of its statements (nycdb, the
|
|
91
|
+
# cross-file gap named in PR #28)
|
|
92
|
+
extra_sqls: list[str] = field(default_factory=list)
|
|
93
|
+
# spellings granted from a multi-relation file's stem. Kept apart from
|
|
94
|
+
# aliases because they are DELIBERATELY ambiguous: the graph may pick
|
|
95
|
+
# among stem claimants by the asked column, but never among real
|
|
96
|
+
# collisions (the review of PR #28)
|
|
97
|
+
stem_aliases: set[str] = field(default_factory=set)
|
|
98
|
+
# minted from CREATE FUNCTION (a traced TVF): a table-function call by
|
|
99
|
+
# this written name binds here instead of degrading as a
|
|
100
|
+
# function/relation collision (cycle-13 review, F11)
|
|
101
|
+
is_function: bool = False
|
|
102
|
+
|
|
103
|
+
def __post_init__(self):
|
|
104
|
+
if not self.uid:
|
|
105
|
+
self.uid = self.path or self.name
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass
|
|
109
|
+
class Project:
|
|
110
|
+
root: Path
|
|
111
|
+
dialect: str
|
|
112
|
+
mode: str # "dbt-manifest" | "dbt-raw" | "sql-dir"
|
|
113
|
+
models: list[Model]
|
|
114
|
+
sources: list[Model]
|
|
115
|
+
warnings: list[str] = field(default_factory=list)
|
|
116
|
+
# True when nothing declared the dialect and it was guessed or sniffed
|
|
117
|
+
dialect_assumed: bool = False
|
|
118
|
+
# (package_name | None, macro_source_text) pairs
|
|
119
|
+
macro_sources: list = field(default_factory=list)
|
|
120
|
+
package_name: str | None = None
|
|
121
|
+
# BI edges declared as code (dbt metrics, exposures): model column -> BI node
|
|
122
|
+
downstream: list = field(default_factory=list)
|
|
123
|
+
# vars declared in dbt_project.yml, resolved by var() during tier-b renders
|
|
124
|
+
jinja_vars: dict = field(default_factory=dict)
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def name_candidates(self) -> dict[str, list[str]]:
|
|
128
|
+
"""Every alias (lowercased) -> all ENABLED model/source names it could
|
|
129
|
+
mean. Disabled models never bind refs: dbt would not build them.
|
|
130
|
+
|
|
131
|
+
More than one candidate means the name is ambiguous in this project;
|
|
132
|
+
the graph layer downgrades trust instead of guessing silently.
|
|
133
|
+
"""
|
|
134
|
+
return self._name_index(include_disabled=False)
|
|
135
|
+
|
|
136
|
+
@property
|
|
137
|
+
def all_name_candidates(self) -> dict[str, list[str]]:
|
|
138
|
+
"""name_candidates including disabled models, for lookups that are
|
|
139
|
+
about a model's own analysis (parent schemas, ordering), not binding."""
|
|
140
|
+
return self._name_index(include_disabled=True)
|
|
141
|
+
|
|
142
|
+
def _name_index(self, include_disabled: bool) -> dict[str, list[str]]:
|
|
143
|
+
index: dict[str, list[str]] = {}
|
|
144
|
+
for m in [*self.sources, *self.models]:
|
|
145
|
+
if not include_disabled and not m.enabled:
|
|
146
|
+
continue
|
|
147
|
+
for alias in {m.name, *m.aliases, *m.stem_aliases}:
|
|
148
|
+
bucket = index.setdefault(alias.lower(), [])
|
|
149
|
+
if m.name not in bucket:
|
|
150
|
+
bucket.append(m.name)
|
|
151
|
+
return index
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def stem_alias_names(self) -> set[str]:
|
|
155
|
+
"""Lowercased spellings granted from file stems; the only names the
|
|
156
|
+
graph may disambiguate by the asked column."""
|
|
157
|
+
return {a.lower() for m in self.models for a in m.stem_aliases}
|
|
158
|
+
|
|
159
|
+
@property
|
|
160
|
+
def name_index(self) -> dict[str, str]:
|
|
161
|
+
"""Every alias (lowercased) -> first canonical name. Prefer
|
|
162
|
+
name_candidates anywhere ambiguity matters."""
|
|
163
|
+
return {alias: names[0] for alias, names in self.name_candidates.items()}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def find_project_root(start: Path) -> Path:
|
|
167
|
+
"""Walk up from start looking for dbt_project.yml; fall back to start."""
|
|
168
|
+
current = start.resolve()
|
|
169
|
+
for candidate in [current, *current.parents]:
|
|
170
|
+
if (candidate / "dbt_project.yml").exists():
|
|
171
|
+
return candidate
|
|
172
|
+
return current
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _find_dbt_projects(root: Path) -> list[Path]:
|
|
176
|
+
"""Every real dbt project dir under root, deepest-first-stable.
|
|
177
|
+
|
|
178
|
+
Excludes vendored packages: dbt_packages/ and dbt_modules/ ship their own
|
|
179
|
+
dbt_project.yml, and those are dependencies, not the repo's own projects.
|
|
180
|
+
"""
|
|
181
|
+
found: list[Path] = []
|
|
182
|
+
for yml in sorted(root.rglob("dbt_project.yml")):
|
|
183
|
+
parts = yml.relative_to(root).parts[:-1]
|
|
184
|
+
if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
|
|
185
|
+
continue
|
|
186
|
+
found.append(yml.parent)
|
|
187
|
+
return found
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
DEFAULT_DIALECT = "snowflake"
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def relative_posix(path, root) -> str:
|
|
194
|
+
"""A path relative to the project root, with forward slashes on every
|
|
195
|
+
OS, so a model's path is the same string in a PR comment written on
|
|
196
|
+
Linux and on the laptop that reads it."""
|
|
197
|
+
return path.relative_to(root).as_posix()
|
ripple/lookml.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Downstream reach into Looker by parsing LookML files.
|
|
2
|
+
|
|
3
|
+
LookML is declarative code in git: a view binds to a warehouse table, and every
|
|
4
|
+
field's `sql:` names the columns it reads. Resolving `${TABLE}.col` and `${field}`
|
|
5
|
+
references gives `view.field -> table.column` with no connection to Looker. Where
|
|
6
|
+
the mapping only exists at query time (Liquid branching, PDT-compiled SQL), the
|
|
7
|
+
edge is marked review_required rather than guessed.
|
|
8
|
+
|
|
9
|
+
Needs the optional `looker` extra (`lkml`); absent it, this yields nothing.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import yaml
|
|
18
|
+
|
|
19
|
+
from ripple.project import IGNORE_DIRS
|
|
20
|
+
from ripple.semantic import DownstreamEdge
|
|
21
|
+
|
|
22
|
+
_TABLE_COL = re.compile(r"\$\{TABLE\}\.(\w+)")
|
|
23
|
+
_FIELD_REF = re.compile(r"\$\{(\w+)\}") # ${other_field}, same view
|
|
24
|
+
_LIQUID = re.compile(r"\{%|\{\{") # Liquid: only resolves at query time
|
|
25
|
+
_FIELD_KINDS = ("dimensions", "measures", "dimension_groups", "filters")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _table(view: dict) -> str:
|
|
29
|
+
"""The physical table a view reads. `sql_table_name` wins, else the view name;
|
|
30
|
+
schema/database qualifiers are dropped so it matches a model node."""
|
|
31
|
+
raw = view.get("sql_table_name") or view.get("name", "")
|
|
32
|
+
return raw.split(".")[-1].strip().strip('`"') or view.get("name", "")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _resolve(sql: str, by_name: dict, seen: set | None = None) -> tuple[str, set[str], str]:
|
|
36
|
+
"""(trust, columns, reason) for one field's SQL."""
|
|
37
|
+
seen = seen or set()
|
|
38
|
+
if _LIQUID.search(sql):
|
|
39
|
+
return (
|
|
40
|
+
"review_required",
|
|
41
|
+
set(_TABLE_COL.findall(sql)),
|
|
42
|
+
"Liquid branch, resolved at query time",
|
|
43
|
+
)
|
|
44
|
+
cols = set(_TABLE_COL.findall(sql))
|
|
45
|
+
trust, reason = "verified", ""
|
|
46
|
+
for ref in _FIELD_REF.findall(sql):
|
|
47
|
+
if ref == "TABLE" or ref in seen:
|
|
48
|
+
continue
|
|
49
|
+
seen.add(ref)
|
|
50
|
+
target = by_name.get(ref)
|
|
51
|
+
if target is None:
|
|
52
|
+
trust, reason = "review_required", "references a field outside this view"
|
|
53
|
+
continue
|
|
54
|
+
t2, sub, r2 = _resolve(target, by_name, seen)
|
|
55
|
+
cols |= sub
|
|
56
|
+
if t2 == "review_required":
|
|
57
|
+
trust, reason = "review_required", r2
|
|
58
|
+
return trust, cols, reason
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _view_edges(view: dict) -> list[DownstreamEdge]:
|
|
62
|
+
name = view.get("name")
|
|
63
|
+
if not name:
|
|
64
|
+
return []
|
|
65
|
+
table = _table(view)
|
|
66
|
+
derived = "derived_table" in view
|
|
67
|
+
fields: dict[str, str] = {}
|
|
68
|
+
for kind in _FIELD_KINDS:
|
|
69
|
+
for f in view.get(kind) or []:
|
|
70
|
+
if isinstance(f, dict) and f.get("name"):
|
|
71
|
+
# a field with no sql defaults to the same-named column
|
|
72
|
+
fields[f["name"]] = (f.get("sql") or f"${{TABLE}}.{f['name']}").strip().rstrip(";")
|
|
73
|
+
|
|
74
|
+
edges: list[DownstreamEdge] = []
|
|
75
|
+
for field, sql in fields.items():
|
|
76
|
+
if derived:
|
|
77
|
+
# a derived table's columns come from its own SQL, not the base table
|
|
78
|
+
edges.append(
|
|
79
|
+
DownstreamEdge(
|
|
80
|
+
table,
|
|
81
|
+
field,
|
|
82
|
+
f"looker:{name}.{field}",
|
|
83
|
+
field,
|
|
84
|
+
"review_required",
|
|
85
|
+
"column of a derived table, verify against its SQL",
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
continue
|
|
89
|
+
trust, cols, reason = _resolve(sql, fields)
|
|
90
|
+
for col in sorted(cols):
|
|
91
|
+
edges.append(DownstreamEdge(table, col, f"looker:{name}.{field}", field, trust, reason))
|
|
92
|
+
return edges
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _dashboard_edges(doc) -> list[DownstreamEdge]:
|
|
96
|
+
"""A LookML dashboard (YAML) tile references `view.field`; that field feeds the
|
|
97
|
+
dashboard, so the field node hangs a node-level edge to the dashboard."""
|
|
98
|
+
edges: list[DownstreamEdge] = []
|
|
99
|
+
dashboards = doc if isinstance(doc, list) else [doc]
|
|
100
|
+
for dash in dashboards:
|
|
101
|
+
if not isinstance(dash, dict) or "dashboard" not in dash:
|
|
102
|
+
continue
|
|
103
|
+
dname = dash["dashboard"]
|
|
104
|
+
refs: set[str] = set()
|
|
105
|
+
for el in dash.get("elements") or []:
|
|
106
|
+
if not isinstance(el, dict):
|
|
107
|
+
continue
|
|
108
|
+
for key in ("fields", "dimensions", "measures"):
|
|
109
|
+
for ref in el.get(key) or []:
|
|
110
|
+
if isinstance(ref, str) and "." in ref:
|
|
111
|
+
refs.add(ref)
|
|
112
|
+
for ref in sorted(refs):
|
|
113
|
+
view, _, field = ref.rpartition(".")
|
|
114
|
+
view = view.split(".")[-1] # drop model.explore. prefix if present
|
|
115
|
+
edges.append(
|
|
116
|
+
DownstreamEdge(
|
|
117
|
+
f"looker:{view}.{field}",
|
|
118
|
+
field,
|
|
119
|
+
f"dashboard:{dname}",
|
|
120
|
+
"*",
|
|
121
|
+
"review_required",
|
|
122
|
+
"dashboard tile, node-level",
|
|
123
|
+
)
|
|
124
|
+
)
|
|
125
|
+
return edges
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def load_lookml_downstream(root: Path) -> list[DownstreamEdge]:
|
|
129
|
+
"""Every field-to-column edge (and dashboard leaf) resolvable from LookML files."""
|
|
130
|
+
try:
|
|
131
|
+
import lkml
|
|
132
|
+
except ImportError:
|
|
133
|
+
return []
|
|
134
|
+
|
|
135
|
+
edges: list[DownstreamEdge] = []
|
|
136
|
+
for path in sorted(root.rglob("*.lkml")):
|
|
137
|
+
parts = path.relative_to(root).parts[:-1]
|
|
138
|
+
if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
|
|
139
|
+
continue
|
|
140
|
+
try:
|
|
141
|
+
tree = lkml.load(path.read_text(errors="replace", encoding="utf-8"))
|
|
142
|
+
except Exception:
|
|
143
|
+
continue
|
|
144
|
+
for view in tree.get("views") or []:
|
|
145
|
+
edges.extend(_view_edges(view))
|
|
146
|
+
|
|
147
|
+
for path in sorted(root.rglob("*.dashboard.lookml")):
|
|
148
|
+
parts = path.relative_to(root).parts[:-1]
|
|
149
|
+
if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
|
|
150
|
+
continue
|
|
151
|
+
try:
|
|
152
|
+
doc = yaml.safe_load(path.read_text(errors="replace", encoding="utf-8"))
|
|
153
|
+
except yaml.YAMLError:
|
|
154
|
+
continue
|
|
155
|
+
edges.extend(_dashboard_edges(doc))
|
|
156
|
+
|
|
157
|
+
# dedupe
|
|
158
|
+
seen, out = set(), []
|
|
159
|
+
for e in edges:
|
|
160
|
+
if e not in seen:
|
|
161
|
+
seen.add(e)
|
|
162
|
+
out.append(e)
|
|
163
|
+
return out
|