ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,65 @@
1
+ """Base tables declared by a schema file beside no SQL.
2
+
3
+ Mozilla's bigquery-etl loads many tables from outside the repo and describes
4
+ each one only by sql/<project>/<dataset>/<table>/schema.yaml (BigQuery's
5
+ `fields:` format). Without those columns a view's SELECT * over such a table
6
+ cannot expand, and every view downstream of it degrades to review_required.
7
+ A directory holding a schema file and no .sql becomes a base-table model
8
+ carrying the declared columns, the same way CREATE TABLE (col defs) does.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ from pathlib import Path
15
+
16
+ import yaml
17
+
18
+ from ripple.loaders.identity import directory_identity
19
+ from ripple.loaders.types import IGNORE_DIRS, Model, relative_posix
20
+
21
+ SIDECAR_NAMES = ("schema.yaml", "schema.yml", "schema.json")
22
+
23
+
24
+ def _declared_fields(path: Path) -> list[str]:
25
+ try:
26
+ text = path.read_text(errors="replace", encoding="utf-8")
27
+ doc = json.loads(text) if path.suffix == ".json" else yaml.safe_load(text)
28
+ except (OSError, ValueError, yaml.YAMLError):
29
+ return []
30
+ fields = doc.get("fields") if isinstance(doc, dict) else doc
31
+ if not isinstance(fields, list):
32
+ return []
33
+ return [f["name"] for f in fields if isinstance(f, dict) and isinstance(f.get("name"), str)]
34
+
35
+
36
+ def sidecar_base_tables(root: Path, sql_dirs: set[Path], dialect: str) -> list[Model]:
37
+ models: list[Model] = []
38
+ for name in SIDECAR_NAMES:
39
+ for path in sorted(root.rglob(name)):
40
+ parts = path.relative_to(root).parts[:-1]
41
+ if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
42
+ continue
43
+ if path.parent in sql_dirs or len(parts) < 1:
44
+ continue
45
+ columns = _declared_fields(path)
46
+ if not columns:
47
+ continue
48
+ rel = relative_posix(path, root)
49
+ identity = directory_identity(str(Path(rel).with_name("query.sql")))
50
+ if identity is None:
51
+ continue
52
+ table, dotted = identity
53
+ model = Model(
54
+ name=table,
55
+ sql="",
56
+ path=rel,
57
+ declared_columns=columns,
58
+ evidence="plain_sql",
59
+ dialect=dialect,
60
+ uid=f"{rel}::{table}",
61
+ )
62
+ model.aliases |= {a.lower() for a in dotted}
63
+ models.append(model)
64
+ sql_dirs.add(path.parent) # one sidecar per directory wins
65
+ return models
@@ -0,0 +1,262 @@
1
+ """Plain SQL directory loading, with the per-file parse budget."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+
7
+ from ripple.engine.budget import TIMED_OUT, budget_seconds, max_sql_bytes, run_within_budget
8
+ from ripple.loaders.identity import directory_identity
9
+ from ripple.loaders.types import IGNORE_DIRS, Model, Project, relative_posix
10
+
11
+
12
+ def _iter_sql_files(root: Path, exclude: list[Path] | None = None) -> tuple[list[Path], int]:
13
+ """Every .sql under root, minus vendored/build/hidden dirs.
14
+
15
+ Returns (kept files, count skipped) so the loader can report what it
16
+ passed over instead of silently dropping it. `exclude` drops whole subtrees,
17
+ used to keep a dbt project's macros out of the beside-dbt fallback.
18
+ """
19
+ kept: list[Path] = []
20
+ skipped = 0
21
+ excluded = [d.resolve() for d in (exclude or [])]
22
+ for path in sorted(root.rglob("*.sql")):
23
+ dirs = path.relative_to(root).parts[:-1]
24
+ if any(part in IGNORE_DIRS or part.startswith(".") for part in dirs):
25
+ skipped += 1
26
+ continue
27
+ if any(path.resolve().is_relative_to(d) for d in excluded):
28
+ skipped += 1
29
+ continue
30
+ kept.append(path)
31
+ return kept, skipped
32
+
33
+
34
+ def _sniff_sql_dir_dialect(root: Path, sample: str) -> tuple[str, bool]:
35
+ """(dialect, assumed). Read the SQL itself for a dialect signal (backticks,
36
+ psql directives, ENGINE=, INFORMATION_SCHEMA); postgres is the honest default
37
+ for a loose pile of .sql, and `assumed` tells the caller to say so."""
38
+ if (root / "supabase" / "config.toml").is_file() or (root / "supabase").is_dir():
39
+ return "postgres", False
40
+ from ripple.engine.sql_script import sniff_dialect
41
+
42
+ sniffed = sniff_dialect(sample)
43
+ if sniffed:
44
+ return sniffed, False
45
+ return "postgres", True
46
+
47
+
48
+ def _expand_within_budget(
49
+ stem: str,
50
+ text: str,
51
+ dialect: str | None,
52
+ budget: float,
53
+ warnings: list[str] | None = None,
54
+ ):
55
+ """expand_script under the shared budget; None means the budget was blown."""
56
+ from ripple.engine import sql_script
57
+
58
+ result = run_within_budget(
59
+ stem, lambda: sql_script.expand_script(stem, text, dialect, warnings=warnings), budget
60
+ )
61
+ return None if result is TIMED_OUT else result
62
+
63
+
64
+ def _load_sql_dir(root: Path, dialect: str | None, exclude: list[Path] | None = None) -> Project:
65
+ files, skipped = _iter_sql_files(root, exclude)
66
+ texts = [(f, f.read_text(errors="replace", encoding="utf-8")) for f in files]
67
+
68
+ from ripple.engine.sql_script import sniff_dialect
69
+
70
+ warnings: list[str] = []
71
+ assumed = False
72
+ if dialect:
73
+ resolved = dialect
74
+ else:
75
+ # sample a slice of EVERY file, not just the first few: a bigquery signal
76
+ # (INFORMATION_SCHEMA) or mysql signal may live deep in the tree.
77
+ sample = "\n".join(t[:4000] for _, t in texts)[:200000]
78
+ resolved, assumed = _sniff_sql_dir_dialect(root, sample)
79
+ if assumed and texts:
80
+ warnings.append(
81
+ f"Assuming {resolved} dialect (no dbt profile or warehouse to read). "
82
+ "Pass --dialect to override."
83
+ )
84
+
85
+ # a .sql file can hold one query or a 15,000-statement dump. Expand each into
86
+ # the models it actually defines, so a dump becomes its views and tables, not
87
+ # one unparseable blob.
88
+ models: list[Model] = []
89
+ stem_claims: list[tuple[Model, str]] = []
90
+ budget = budget_seconds()
91
+ for sql_file, text in texts:
92
+ rel = relative_posix(sql_file, root)
93
+ # per-file dialect: one repo can hold the same schema in several dialects
94
+ # (Chinook ships Postgres, MySQL, Oracle, SQL Server variants side by side).
95
+ # An explicit --dialect wins; otherwise sniff this file, then fall back to
96
+ # the repo-level guess.
97
+ file_dialect = dialect or sniff_dialect(text) or resolved
98
+ limit = max_sql_bytes()
99
+ if limit and len(text) > limit and ";" not in text:
100
+ # a multi-MB file with no statement separator is one generated
101
+ # statement (a 3.7MB one-liner cost 13s to parse and 139s to
102
+ # analyze); a dump of many small statements has semicolons and
103
+ # still expands normally
104
+ warnings.append(
105
+ f"{rel}: a single {len(text) / 1e6:.1f}MB statement, skipped as "
106
+ "generated (RIPPLE_MAX_SQL_BYTES to raise)."
107
+ )
108
+ models.append(
109
+ Model(
110
+ name=sql_file.stem,
111
+ sql="",
112
+ path=rel,
113
+ evidence="plain_sql",
114
+ dialect=file_dialect,
115
+ uid=f"{rel}::{sql_file.stem}",
116
+ load_error=(
117
+ f"single statement is {len(text) / 1e6:.1f}MB, over the "
118
+ f"{limit / 1e6:g}MB limit: {rel}"
119
+ ),
120
+ )
121
+ )
122
+ continue
123
+ script_warnings: list[str] = []
124
+ expanded = _expand_within_budget(
125
+ sql_file.stem, text, file_dialect, budget, warnings=script_warnings
126
+ )
127
+ warnings.extend(f"{rel}: {w}" for w in script_warnings)
128
+ if expanded is None:
129
+ warnings.append(
130
+ f"{rel}: parse exceeded the {budget:g}s per-file budget; reported as "
131
+ "timed_out (RIPPLE_PARSE_BUDGET_S to raise)."
132
+ )
133
+ models.append(
134
+ Model(
135
+ name=sql_file.stem,
136
+ sql="",
137
+ path=rel,
138
+ evidence="plain_sql",
139
+ dialect=file_dialect,
140
+ uid=f"{rel}::{sql_file.stem}",
141
+ load_error=f"parse timed out after {budget:g}s: {rel}",
142
+ )
143
+ )
144
+ continue
145
+ # dedup only WITHIN a file (a base table and a later INSERT/CTAS that fills
146
+ # it). Same name across different files is a real collision that
147
+ # _qualify_collisions path-qualifies, so it must NOT be dropped here.
148
+ # dedup is keyed by the QUALIFIED identity, not the bare name:
149
+ # sales.summary and hr.summary in one migration are different
150
+ # relations, and merging them would bind hr references to the sales
151
+ # derivation (the review of the round-3 fixes)
152
+ here: dict[str, Model] = {}
153
+ clone_here: set[str] = set()
154
+ for sm in expanded:
155
+ identity = (sm.qualified_name or sm.name).lower()
156
+ existing = here.get(identity)
157
+ if existing is not None:
158
+ # an empty base table upgrades to any derivation (old rule);
159
+ # a schema clone (select * ... where 0=1) upgrades to the
160
+ # later real derivation, which it exists to receive
161
+ # (usaspending's subaward loader, holdout round 2). A real
162
+ # derivation is never replaced. Parents always merge: a
163
+ # second statement writing the same target is a real
164
+ # dependency even when its SQL is not the one analyzed.
165
+ replaceable = not existing.sql.strip() or (
166
+ identity in clone_here and not sm.schema_clone
167
+ )
168
+ if replaceable and sm.sql.strip():
169
+ existing.sql = sm.sql
170
+ # the analyzed SQL is now the UPDATE's; without its
171
+ # self_read the graph drops the prior-state edges as
172
+ # CTE loops (the review of PR #28)
173
+ existing.self_read = sm.self_read
174
+ if sm.schema_clone:
175
+ clone_here.add(identity)
176
+ else:
177
+ clone_here.discard(identity)
178
+ elif sm.self_read and sm.sql.strip():
179
+ # a later UPDATE of a target that already has a real
180
+ # derivation adds lineage, it does not replace any: the
181
+ # statement joins the model's analyzed set (nycdb,
182
+ # PR #28 gap 3)
183
+ existing.extra_sqls.append(sm.sql)
184
+ existing.self_read = True
185
+ existing.declared_parents |= set(sm.parents)
186
+ continue
187
+ model = Model(
188
+ name=sm.name,
189
+ sql=sm.sql,
190
+ path=rel,
191
+ declared_parents=set(sm.parents),
192
+ declared_columns=list(sm.columns),
193
+ evidence="plain_sql",
194
+ dialect=file_dialect,
195
+ uid=f"{rel}::{identity}",
196
+ self_read=sm.self_read,
197
+ is_function=sm.is_function,
198
+ )
199
+ # a bare SELECT is named after its file; when the file is named
200
+ # for its role (query.sql), the directory is the relation
201
+ by_directory = (
202
+ directory_identity(rel) if sm.name.lower() == sql_file.stem.lower() else None
203
+ )
204
+ if by_directory:
205
+ model.name, dotted = by_directory
206
+ model.aliases |= {a.lower() for a in dotted}
207
+ if sm.schema_clone:
208
+ clone_here.add(identity)
209
+ # the qualified spelling the SQL wrote (disclosure.v_sum) answers
210
+ # queries and folds scoring; the display name stays bare
211
+ if sm.qualified_name and sm.qualified_name.lower() != sm.name.lower():
212
+ model.aliases.add(sm.qualified_name.lower())
213
+ here[identity] = model
214
+ models.append(model)
215
+ # a file is known to its team by its filename: claim the stem for the
216
+ # file's relations, granted after the sweep only if no other file and
217
+ # no real model owns the name. A multi-relation file's stem becomes a
218
+ # deliberately ambiguous alias; the graph disambiguates by the asked
219
+ # column (patentsview's webtool_tables, holdout round 5) and raises
220
+ # when several claimants carry it.
221
+ stem = sql_file.stem.lower()
222
+ for defined in here.values():
223
+ if defined.name.lower() != stem:
224
+ stem_claims.append((rel, defined, stem))
225
+
226
+ from ripple.loaders.sidecar import sidecar_base_tables
227
+
228
+ models.extend(sidecar_base_tables(root, {f.parent for f in files}, resolved))
229
+
230
+ # grant stem aliases only where the name is unowned and claimed by ONE
231
+ # file: a stem matching a real model, or shared by two files, would
232
+ # silently redirect breaks/trace to the wrong relation
233
+ taken = {m.name.lower() for m in models}
234
+ for m in models:
235
+ taken |= {a.lower() for a in m.aliases}
236
+ files_per_stem: dict[str, set[str]] = {}
237
+ for claim_rel, _, stem in stem_claims:
238
+ files_per_stem.setdefault(stem, set()).add(claim_rel)
239
+ for _, claimant, stem in stem_claims:
240
+ if len(files_per_stem[stem]) == 1 and stem not in taken:
241
+ claimant.stem_aliases.add(stem)
242
+
243
+ if skipped:
244
+ warnings.append(
245
+ f"Skipped {skipped} .sql file{'s' if skipped != 1 else ''} in "
246
+ "vendored or build directories."
247
+ )
248
+ if not models:
249
+ warnings.append(
250
+ f"No .sql files found under {root}. Run ripple inside a dbt project "
251
+ "or a folder that contains .sql files."
252
+ )
253
+
254
+ return Project(
255
+ root=root,
256
+ dialect=resolved,
257
+ mode="sql-dir",
258
+ models=models,
259
+ sources=[],
260
+ warnings=warnings,
261
+ dialect_assumed=assumed,
262
+ )
@@ -0,0 +1,197 @@
1
+ """Shared model/project types and discovery constants."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass, field
7
+ from pathlib import Path
8
+
9
+ ADAPTER_TO_DIALECT = {
10
+ "snowflake": "snowflake",
11
+ "bigquery": "bigquery",
12
+ "databricks": "databricks",
13
+ "spark": "spark",
14
+ "redshift": "redshift",
15
+ "postgres": "postgres",
16
+ "duckdb": "duckdb",
17
+ "trino": "trino",
18
+ }
19
+
20
+ # ref()/source() anywhere in the Jinja, including nested inside macro args
21
+ # like {{ dbt_audit(ref('base'), ...) }}. Two-arg ref('package', 'model')
22
+ # takes the last argument as the model name.
23
+ _REF_RE = re.compile(r"\bref\(\s*['\"]([^'\"]+)['\"]\s*(?:,\s*['\"]([^'\"]+)['\"]\s*)?\)")
24
+ _SOURCE_RE = re.compile(r"\bsource\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)")
25
+ _VAR_NAME_RE = re.compile(r"\bvar\(\s*['\"]([^'\"]+)['\"]")
26
+
27
+ # Directories that never hold the project's own SQL. Skipped when scanning a
28
+ # plain repo so the map is the code you wrote, not vendored or generated junk.
29
+ # Hidden dirs (dotfiles) are skipped separately.
30
+ IGNORE_DIRS = frozenset(
31
+ {
32
+ "node_modules",
33
+ "bower_components",
34
+ "venv",
35
+ "env",
36
+ "virtualenv",
37
+ "site-packages",
38
+ "__pycache__",
39
+ "target",
40
+ "dist",
41
+ "build",
42
+ "out",
43
+ "dbt_packages",
44
+ "dbt_modules",
45
+ "vendor",
46
+ "vendored",
47
+ "third_party",
48
+ }
49
+ )
50
+
51
+
52
+ @dataclass
53
+ class Model:
54
+ name: str
55
+ sql: str
56
+ path: str
57
+ # names this model can be referenced by in SQL (relation name, alias, bare name)
58
+ aliases: set[str] = field(default_factory=set)
59
+ # parent model/source names declared by dbt (manifest mode) or found via ref()/source()
60
+ declared_parents: set[str] = field(default_factory=set)
61
+ is_source: bool = False
62
+ # "manifest" | "raw_jinja" | "plain_sql"
63
+ evidence: str = "plain_sql"
64
+ # per-model SQL dialect. In a monorepo each dbt project is its own island
65
+ # with its own warehouse; a model parses in its island's dialect, not one
66
+ # dialect forced on the whole tree. None means "use the project default".
67
+ dialect: str | None = None
68
+ # columns declared by a CREATE TABLE (the schema). A base table has these but
69
+ # no SQL to analyze; the graph uses them so a downstream SELECT * can resolve.
70
+ declared_columns: list[str] = field(default_factory=list)
71
+ # stable identity: dbt unique_id in manifest mode, project-relative path
72
+ # elsewhere. Display names can collide; this must not.
73
+ uid: str = ""
74
+ # vars declared by THIS model's own dbt_project.yml. Like dialect: set per
75
+ # model in a monorepo so islands keep their own values; None means "use the
76
+ # project default".
77
+ jinja_vars: dict | None = None
78
+ # dbt enabled config (dbt_project.yml +enabled or {{ config(enabled=...) }}).
79
+ # A disabled model stays a visible node but never binds refs or feeds
80
+ # parent schemas: dbt would not build it.
81
+ enabled: bool = True
82
+ # set when the file's parse blew the per-file budget; the graph reports the
83
+ # model as timed_out instead of pretending it was analyzed
84
+ load_error: str | None = None
85
+ # an UPDATE-derived model reads its own prior state; the graph keeps
86
+ # its self edges instead of treating them as CTE loops
87
+ self_read: bool = False
88
+ # further analyzed statements for the same relation: a table built by
89
+ # CREATE plus several UPDATEs (in one file or across files) is ONE
90
+ # model whose lineage is the union of its statements (nycdb, the
91
+ # cross-file gap named in PR #28)
92
+ extra_sqls: list[str] = field(default_factory=list)
93
+ # spellings granted from a multi-relation file's stem. Kept apart from
94
+ # aliases because they are DELIBERATELY ambiguous: the graph may pick
95
+ # among stem claimants by the asked column, but never among real
96
+ # collisions (the review of PR #28)
97
+ stem_aliases: set[str] = field(default_factory=set)
98
+ # minted from CREATE FUNCTION (a traced TVF): a table-function call by
99
+ # this written name binds here instead of degrading as a
100
+ # function/relation collision (cycle-13 review, F11)
101
+ is_function: bool = False
102
+
103
+ def __post_init__(self):
104
+ if not self.uid:
105
+ self.uid = self.path or self.name
106
+
107
+
108
+ @dataclass
109
+ class Project:
110
+ root: Path
111
+ dialect: str
112
+ mode: str # "dbt-manifest" | "dbt-raw" | "sql-dir"
113
+ models: list[Model]
114
+ sources: list[Model]
115
+ warnings: list[str] = field(default_factory=list)
116
+ # True when nothing declared the dialect and it was guessed or sniffed
117
+ dialect_assumed: bool = False
118
+ # (package_name | None, macro_source_text) pairs
119
+ macro_sources: list = field(default_factory=list)
120
+ package_name: str | None = None
121
+ # BI edges declared as code (dbt metrics, exposures): model column -> BI node
122
+ downstream: list = field(default_factory=list)
123
+ # vars declared in dbt_project.yml, resolved by var() during tier-b renders
124
+ jinja_vars: dict = field(default_factory=dict)
125
+
126
+ @property
127
+ def name_candidates(self) -> dict[str, list[str]]:
128
+ """Every alias (lowercased) -> all ENABLED model/source names it could
129
+ mean. Disabled models never bind refs: dbt would not build them.
130
+
131
+ More than one candidate means the name is ambiguous in this project;
132
+ the graph layer downgrades trust instead of guessing silently.
133
+ """
134
+ return self._name_index(include_disabled=False)
135
+
136
+ @property
137
+ def all_name_candidates(self) -> dict[str, list[str]]:
138
+ """name_candidates including disabled models, for lookups that are
139
+ about a model's own analysis (parent schemas, ordering), not binding."""
140
+ return self._name_index(include_disabled=True)
141
+
142
+ def _name_index(self, include_disabled: bool) -> dict[str, list[str]]:
143
+ index: dict[str, list[str]] = {}
144
+ for m in [*self.sources, *self.models]:
145
+ if not include_disabled and not m.enabled:
146
+ continue
147
+ for alias in {m.name, *m.aliases, *m.stem_aliases}:
148
+ bucket = index.setdefault(alias.lower(), [])
149
+ if m.name not in bucket:
150
+ bucket.append(m.name)
151
+ return index
152
+
153
+ @property
154
+ def stem_alias_names(self) -> set[str]:
155
+ """Lowercased spellings granted from file stems; the only names the
156
+ graph may disambiguate by the asked column."""
157
+ return {a.lower() for m in self.models for a in m.stem_aliases}
158
+
159
+ @property
160
+ def name_index(self) -> dict[str, str]:
161
+ """Every alias (lowercased) -> first canonical name. Prefer
162
+ name_candidates anywhere ambiguity matters."""
163
+ return {alias: names[0] for alias, names in self.name_candidates.items()}
164
+
165
+
166
+ def find_project_root(start: Path) -> Path:
167
+ """Walk up from start looking for dbt_project.yml; fall back to start."""
168
+ current = start.resolve()
169
+ for candidate in [current, *current.parents]:
170
+ if (candidate / "dbt_project.yml").exists():
171
+ return candidate
172
+ return current
173
+
174
+
175
+ def _find_dbt_projects(root: Path) -> list[Path]:
176
+ """Every real dbt project dir under root, deepest-first-stable.
177
+
178
+ Excludes vendored packages: dbt_packages/ and dbt_modules/ ship their own
179
+ dbt_project.yml, and those are dependencies, not the repo's own projects.
180
+ """
181
+ found: list[Path] = []
182
+ for yml in sorted(root.rglob("dbt_project.yml")):
183
+ parts = yml.relative_to(root).parts[:-1]
184
+ if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
185
+ continue
186
+ found.append(yml.parent)
187
+ return found
188
+
189
+
190
+ DEFAULT_DIALECT = "snowflake"
191
+
192
+
193
+ def relative_posix(path, root) -> str:
194
+ """A path relative to the project root, with forward slashes on every
195
+ OS, so a model's path is the same string in a PR comment written on
196
+ Linux and on the laptop that reads it."""
197
+ return path.relative_to(root).as_posix()
ripple/lookml.py ADDED
@@ -0,0 +1,163 @@
1
+ """Downstream reach into Looker by parsing LookML files.
2
+
3
+ LookML is declarative code in git: a view binds to a warehouse table, and every
4
+ field's `sql:` names the columns it reads. Resolving `${TABLE}.col` and `${field}`
5
+ references gives `view.field -> table.column` with no connection to Looker. Where
6
+ the mapping only exists at query time (Liquid branching, PDT-compiled SQL), the
7
+ edge is marked review_required rather than guessed.
8
+
9
+ Needs the optional `looker` extra (`lkml`); absent it, this yields nothing.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+ from pathlib import Path
16
+
17
+ import yaml
18
+
19
+ from ripple.project import IGNORE_DIRS
20
+ from ripple.semantic import DownstreamEdge
21
+
22
+ _TABLE_COL = re.compile(r"\$\{TABLE\}\.(\w+)")
23
+ _FIELD_REF = re.compile(r"\$\{(\w+)\}") # ${other_field}, same view
24
+ _LIQUID = re.compile(r"\{%|\{\{") # Liquid: only resolves at query time
25
+ _FIELD_KINDS = ("dimensions", "measures", "dimension_groups", "filters")
26
+
27
+
28
+ def _table(view: dict) -> str:
29
+ """The physical table a view reads. `sql_table_name` wins, else the view name;
30
+ schema/database qualifiers are dropped so it matches a model node."""
31
+ raw = view.get("sql_table_name") or view.get("name", "")
32
+ return raw.split(".")[-1].strip().strip('`"') or view.get("name", "")
33
+
34
+
35
+ def _resolve(sql: str, by_name: dict, seen: set | None = None) -> tuple[str, set[str], str]:
36
+ """(trust, columns, reason) for one field's SQL."""
37
+ seen = seen or set()
38
+ if _LIQUID.search(sql):
39
+ return (
40
+ "review_required",
41
+ set(_TABLE_COL.findall(sql)),
42
+ "Liquid branch, resolved at query time",
43
+ )
44
+ cols = set(_TABLE_COL.findall(sql))
45
+ trust, reason = "verified", ""
46
+ for ref in _FIELD_REF.findall(sql):
47
+ if ref == "TABLE" or ref in seen:
48
+ continue
49
+ seen.add(ref)
50
+ target = by_name.get(ref)
51
+ if target is None:
52
+ trust, reason = "review_required", "references a field outside this view"
53
+ continue
54
+ t2, sub, r2 = _resolve(target, by_name, seen)
55
+ cols |= sub
56
+ if t2 == "review_required":
57
+ trust, reason = "review_required", r2
58
+ return trust, cols, reason
59
+
60
+
61
+ def _view_edges(view: dict) -> list[DownstreamEdge]:
62
+ name = view.get("name")
63
+ if not name:
64
+ return []
65
+ table = _table(view)
66
+ derived = "derived_table" in view
67
+ fields: dict[str, str] = {}
68
+ for kind in _FIELD_KINDS:
69
+ for f in view.get(kind) or []:
70
+ if isinstance(f, dict) and f.get("name"):
71
+ # a field with no sql defaults to the same-named column
72
+ fields[f["name"]] = (f.get("sql") or f"${{TABLE}}.{f['name']}").strip().rstrip(";")
73
+
74
+ edges: list[DownstreamEdge] = []
75
+ for field, sql in fields.items():
76
+ if derived:
77
+ # a derived table's columns come from its own SQL, not the base table
78
+ edges.append(
79
+ DownstreamEdge(
80
+ table,
81
+ field,
82
+ f"looker:{name}.{field}",
83
+ field,
84
+ "review_required",
85
+ "column of a derived table, verify against its SQL",
86
+ )
87
+ )
88
+ continue
89
+ trust, cols, reason = _resolve(sql, fields)
90
+ for col in sorted(cols):
91
+ edges.append(DownstreamEdge(table, col, f"looker:{name}.{field}", field, trust, reason))
92
+ return edges
93
+
94
+
95
+ def _dashboard_edges(doc) -> list[DownstreamEdge]:
96
+ """A LookML dashboard (YAML) tile references `view.field`; that field feeds the
97
+ dashboard, so the field node hangs a node-level edge to the dashboard."""
98
+ edges: list[DownstreamEdge] = []
99
+ dashboards = doc if isinstance(doc, list) else [doc]
100
+ for dash in dashboards:
101
+ if not isinstance(dash, dict) or "dashboard" not in dash:
102
+ continue
103
+ dname = dash["dashboard"]
104
+ refs: set[str] = set()
105
+ for el in dash.get("elements") or []:
106
+ if not isinstance(el, dict):
107
+ continue
108
+ for key in ("fields", "dimensions", "measures"):
109
+ for ref in el.get(key) or []:
110
+ if isinstance(ref, str) and "." in ref:
111
+ refs.add(ref)
112
+ for ref in sorted(refs):
113
+ view, _, field = ref.rpartition(".")
114
+ view = view.split(".")[-1] # drop model.explore. prefix if present
115
+ edges.append(
116
+ DownstreamEdge(
117
+ f"looker:{view}.{field}",
118
+ field,
119
+ f"dashboard:{dname}",
120
+ "*",
121
+ "review_required",
122
+ "dashboard tile, node-level",
123
+ )
124
+ )
125
+ return edges
126
+
127
+
128
+ def load_lookml_downstream(root: Path) -> list[DownstreamEdge]:
129
+ """Every field-to-column edge (and dashboard leaf) resolvable from LookML files."""
130
+ try:
131
+ import lkml
132
+ except ImportError:
133
+ return []
134
+
135
+ edges: list[DownstreamEdge] = []
136
+ for path in sorted(root.rglob("*.lkml")):
137
+ parts = path.relative_to(root).parts[:-1]
138
+ if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
139
+ continue
140
+ try:
141
+ tree = lkml.load(path.read_text(errors="replace", encoding="utf-8"))
142
+ except Exception:
143
+ continue
144
+ for view in tree.get("views") or []:
145
+ edges.extend(_view_edges(view))
146
+
147
+ for path in sorted(root.rglob("*.dashboard.lookml")):
148
+ parts = path.relative_to(root).parts[:-1]
149
+ if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
150
+ continue
151
+ try:
152
+ doc = yaml.safe_load(path.read_text(errors="replace", encoding="utf-8"))
153
+ except yaml.YAMLError:
154
+ continue
155
+ edges.extend(_dashboard_edges(doc))
156
+
157
+ # dedupe
158
+ seen, out = set(), []
159
+ for e in edges:
160
+ if e not in seen:
161
+ seen.add(e)
162
+ out.append(e)
163
+ return out