ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
"""dbt_project.yml knobs: vars, enabled rules, macro and model paths."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _project_vars(root: Path) -> dict:
|
|
10
|
+
"""Declared vars from dbt_project.yml. A sub-dict scoped under the
|
|
11
|
+
project's own name is flattened in (dbt per-package scoping); scopes for
|
|
12
|
+
other packages are left as-is since their names aren't known here."""
|
|
13
|
+
project_yml = root / "dbt_project.yml"
|
|
14
|
+
if not project_yml.is_file():
|
|
15
|
+
return {}
|
|
16
|
+
import yaml
|
|
17
|
+
|
|
18
|
+
text = project_yml.read_text(errors="replace", encoding="utf-8")
|
|
19
|
+
try:
|
|
20
|
+
parsed = yaml.safe_load(text)
|
|
21
|
+
except yaml.YAMLError:
|
|
22
|
+
# hand-edited ymls with inconsistent indentation still deserve their
|
|
23
|
+
# flat vars; salvage them line by line
|
|
24
|
+
return _flat_vars_fallback(text)
|
|
25
|
+
if not isinstance(parsed, dict) or not isinstance(parsed.get("vars"), dict):
|
|
26
|
+
return {}
|
|
27
|
+
declared = dict(parsed["vars"])
|
|
28
|
+
scoped = declared.get(parsed.get("name"))
|
|
29
|
+
if isinstance(scoped, dict):
|
|
30
|
+
declared.pop(parsed.get("name"))
|
|
31
|
+
declared.update(scoped)
|
|
32
|
+
return declared
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _flat_vars_fallback(text: str) -> dict:
|
|
36
|
+
import yaml
|
|
37
|
+
|
|
38
|
+
lines = text.splitlines()
|
|
39
|
+
start = next(
|
|
40
|
+
(i for i, line in enumerate(lines) if re.match(r"^\s*vars:\s*(#.*)?$", line)), None
|
|
41
|
+
)
|
|
42
|
+
if start is None:
|
|
43
|
+
return {}
|
|
44
|
+
block: list[str] = []
|
|
45
|
+
for line in lines[start + 1 :]:
|
|
46
|
+
if re.match(r"^\s*[A-Za-z0-9_.-]+:\s*\S", line):
|
|
47
|
+
block.append(line.strip())
|
|
48
|
+
elif line.strip():
|
|
49
|
+
break
|
|
50
|
+
try:
|
|
51
|
+
parsed = yaml.safe_load("\n".join(block))
|
|
52
|
+
except yaml.YAMLError:
|
|
53
|
+
return {}
|
|
54
|
+
return parsed if isinstance(parsed, dict) else {}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _enabled_overrides(
|
|
58
|
+
root: Path, dialect: str, project_vars: dict
|
|
59
|
+
) -> list[tuple[tuple[str, ...], bool]]:
|
|
60
|
+
"""(path prefix under the model dir, enabled) rules from the project's own
|
|
61
|
+
models: block in dbt_project.yml. dbt packages ship every dialect variant
|
|
62
|
+
of a model and gate them with enabled configs; ignoring those poisons ref
|
|
63
|
+
binding with variants dbt would never build."""
|
|
64
|
+
project_yml = root / "dbt_project.yml"
|
|
65
|
+
if not project_yml.is_file():
|
|
66
|
+
return []
|
|
67
|
+
import yaml
|
|
68
|
+
|
|
69
|
+
try:
|
|
70
|
+
parsed = yaml.safe_load(project_yml.read_text(errors="replace", encoding="utf-8"))
|
|
71
|
+
except yaml.YAMLError:
|
|
72
|
+
return []
|
|
73
|
+
if not isinstance(parsed, dict) or not isinstance(parsed.get("models"), dict):
|
|
74
|
+
return []
|
|
75
|
+
tree = parsed["models"].get(parsed.get("name"))
|
|
76
|
+
if not isinstance(tree, dict):
|
|
77
|
+
return []
|
|
78
|
+
rules: list[tuple[tuple[str, ...], bool]] = []
|
|
79
|
+
|
|
80
|
+
def walk(node: dict, prefix: tuple[str, ...]) -> None:
|
|
81
|
+
raw = node.get("+enabled", node.get("enabled"))
|
|
82
|
+
if raw is not None and not isinstance(raw, dict):
|
|
83
|
+
decided = _eval_enabled(raw, dialect, project_vars)
|
|
84
|
+
if decided is not None:
|
|
85
|
+
rules.append((prefix, decided))
|
|
86
|
+
for key, child in node.items():
|
|
87
|
+
if isinstance(child, dict) and not key.startswith("+"):
|
|
88
|
+
walk(child, prefix + (key,))
|
|
89
|
+
|
|
90
|
+
walk(tree, ())
|
|
91
|
+
return rules
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _enabled_for(parts: tuple[str, ...], rules: list[tuple[tuple[str, ...], bool]]) -> bool | None:
|
|
95
|
+
"""Deepest matching rule wins, dbt-style. None when no rule applies."""
|
|
96
|
+
decided, best = None, -1
|
|
97
|
+
for prefix, value in rules:
|
|
98
|
+
if len(prefix) > best and parts[: len(prefix)] == prefix:
|
|
99
|
+
decided, best = value, len(prefix)
|
|
100
|
+
return decided
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
_CONFIG_CALL_RE = re.compile(r"{{-?\s*config\s*\(", re.I)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _file_enabled(raw: str, dialect: str, project_vars: dict) -> bool | None:
|
|
107
|
+
"""enabled= from a model file's own {{ config(...) }}, if it decides."""
|
|
108
|
+
match = _CONFIG_CALL_RE.search(raw)
|
|
109
|
+
if match is None:
|
|
110
|
+
return None
|
|
111
|
+
depth, i = 1, match.end()
|
|
112
|
+
while i < len(raw) and depth:
|
|
113
|
+
if raw[i] == "(":
|
|
114
|
+
depth += 1
|
|
115
|
+
elif raw[i] == ")":
|
|
116
|
+
depth -= 1
|
|
117
|
+
i += 1
|
|
118
|
+
body = raw[match.end() : i - 1]
|
|
119
|
+
arg = re.search(r"\benabled\s*=", body)
|
|
120
|
+
if arg is None:
|
|
121
|
+
return None
|
|
122
|
+
expr: list[str] = []
|
|
123
|
+
depth = 0
|
|
124
|
+
for ch in body[arg.end() :]:
|
|
125
|
+
if ch in "([{":
|
|
126
|
+
depth += 1
|
|
127
|
+
elif ch in ")]}":
|
|
128
|
+
depth -= 1
|
|
129
|
+
elif ch == "," and depth == 0:
|
|
130
|
+
break
|
|
131
|
+
expr.append(ch)
|
|
132
|
+
return _render_enabled("{{ (" + "".join(expr).strip() + ") }}", dialect, project_vars)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _eval_enabled(value, dialect: str, project_vars: dict) -> bool | None:
|
|
136
|
+
"""True/False when a literal or a target/var-decided expression settles
|
|
137
|
+
it; None (treated as enabled) when it cannot be evaluated honestly."""
|
|
138
|
+
if isinstance(value, bool):
|
|
139
|
+
return value
|
|
140
|
+
if not isinstance(value, str):
|
|
141
|
+
return None
|
|
142
|
+
text = value.strip()
|
|
143
|
+
if text.lower() in ("true", "false"):
|
|
144
|
+
return text.lower() == "true"
|
|
145
|
+
if "{{" in text or "{%" in text:
|
|
146
|
+
return _render_enabled(text, dialect, project_vars)
|
|
147
|
+
return None
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _render_enabled(template: str, dialect: str, project_vars: dict) -> bool | None:
|
|
151
|
+
"""Evaluate an enabled jinja expression with target.type = the project
|
|
152
|
+
dialect and declared vars. Anything unresolvable renders as None: an
|
|
153
|
+
un-evaluatable gate must not disable a model."""
|
|
154
|
+
from jinja2 import Environment, StrictUndefined
|
|
155
|
+
|
|
156
|
+
env = Environment(undefined=StrictUndefined)
|
|
157
|
+
# dbt's as_bool coerces after render; the comparison already yields a bool
|
|
158
|
+
env.filters["as_bool"] = lambda value: value
|
|
159
|
+
missing = object()
|
|
160
|
+
|
|
161
|
+
def _var(name, default=missing):
|
|
162
|
+
if name in project_vars:
|
|
163
|
+
return project_vars[name]
|
|
164
|
+
if default is missing:
|
|
165
|
+
raise KeyError(name)
|
|
166
|
+
return default
|
|
167
|
+
|
|
168
|
+
try:
|
|
169
|
+
rendered = (
|
|
170
|
+
env.from_string(template)
|
|
171
|
+
.render(var=_var, target={"type": dialect, "name": "prod"})
|
|
172
|
+
.strip()
|
|
173
|
+
)
|
|
174
|
+
except Exception:
|
|
175
|
+
return None
|
|
176
|
+
if rendered == "True":
|
|
177
|
+
return True
|
|
178
|
+
if rendered == "False":
|
|
179
|
+
return False
|
|
180
|
+
return None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _macro_dir_names(project_yml: Path) -> list[str]:
|
|
184
|
+
"""macro-paths from a dbt_project.yml, defaulting to dbt's ["macros"].
|
|
185
|
+
|
|
186
|
+
Real YAML parsing, because block-style lists are valid dbt config and a
|
|
187
|
+
flow-only regex silently falls back to macros/ on them, which is the same
|
|
188
|
+
zero-macros failure this exists to fix (make-open-data, holdout round 2:
|
|
189
|
+
macro-paths: ["5_macros"], every macro call rendered to nothing)."""
|
|
190
|
+
if not project_yml.is_file():
|
|
191
|
+
return ["macros"]
|
|
192
|
+
text = project_yml.read_text(errors="replace", encoding="utf-8")
|
|
193
|
+
try:
|
|
194
|
+
import yaml
|
|
195
|
+
|
|
196
|
+
data = yaml.safe_load(text) or {}
|
|
197
|
+
names = data.get("macro-paths")
|
|
198
|
+
if isinstance(names, list):
|
|
199
|
+
cleaned = [n for n in names if isinstance(n, str) and n.strip()]
|
|
200
|
+
if cleaned:
|
|
201
|
+
return cleaned
|
|
202
|
+
except Exception:
|
|
203
|
+
match = re.search(r"macro-paths:\s*\[([^\]]*)\]", text)
|
|
204
|
+
if match:
|
|
205
|
+
names = [
|
|
206
|
+
part.strip().strip("'\"") for part in match.group(1).split(",") if part.strip()
|
|
207
|
+
]
|
|
208
|
+
if names:
|
|
209
|
+
return names
|
|
210
|
+
return ["macros"]
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _macro_sources(root: Path) -> list[tuple[str | None, str]]:
|
|
214
|
+
"""(package, source) for the project's own macros and vendored packages.
|
|
215
|
+
|
|
216
|
+
The project's own macros carry the dbt_project.yml name as their package
|
|
217
|
+
so both bare and self-namespaced calls ({{ ga4.unnest_key(...) }} inside
|
|
218
|
+
the ga4 package itself) resolve. macro-paths is honored for the root
|
|
219
|
+
project and every vendored package alike.
|
|
220
|
+
"""
|
|
221
|
+
root_package = None
|
|
222
|
+
project_yml = root / "dbt_project.yml"
|
|
223
|
+
if project_yml.is_file():
|
|
224
|
+
match = re.search(
|
|
225
|
+
r"^name:\s*['\"]?([A-Za-z0-9_]+)",
|
|
226
|
+
project_yml.read_text(errors="replace", encoding="utf-8"),
|
|
227
|
+
re.M,
|
|
228
|
+
)
|
|
229
|
+
if match:
|
|
230
|
+
root_package = match.group(1)
|
|
231
|
+
pairs: list[tuple[str | None, str]] = []
|
|
232
|
+
seen_files: set[Path] = set()
|
|
233
|
+
|
|
234
|
+
def _load_dirs(base: Path, dir_names: list[str], package: str | None) -> None:
|
|
235
|
+
for dir_name in dir_names:
|
|
236
|
+
macros_dir = base / dir_name
|
|
237
|
+
if not macros_dir.is_dir():
|
|
238
|
+
continue
|
|
239
|
+
for macro_file in sorted(macros_dir.rglob("*.sql")):
|
|
240
|
+
resolved = macro_file.resolve()
|
|
241
|
+
if resolved in seen_files:
|
|
242
|
+
continue
|
|
243
|
+
seen_files.add(resolved)
|
|
244
|
+
try:
|
|
245
|
+
pairs.append(
|
|
246
|
+
(package, macro_file.read_text(errors="replace", encoding="utf-8"))
|
|
247
|
+
)
|
|
248
|
+
except OSError:
|
|
249
|
+
continue
|
|
250
|
+
|
|
251
|
+
_load_dirs(root, _macro_dir_names(project_yml), root_package)
|
|
252
|
+
for packages_dir_name in ("dbt_packages", "dbt_modules"):
|
|
253
|
+
packages_dir = root / packages_dir_name
|
|
254
|
+
if not packages_dir.is_dir():
|
|
255
|
+
continue
|
|
256
|
+
for package_dir in sorted(p for p in packages_dir.iterdir() if p.is_dir()):
|
|
257
|
+
_load_dirs(
|
|
258
|
+
package_dir,
|
|
259
|
+
_macro_dir_names(package_dir / "dbt_project.yml"),
|
|
260
|
+
package_dir.name,
|
|
261
|
+
)
|
|
262
|
+
return pairs
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _dbt_model_dirs(root: Path) -> list[Path]:
|
|
266
|
+
dirs = []
|
|
267
|
+
project_yml = (root / "dbt_project.yml").read_text(errors="replace", encoding="utf-8")
|
|
268
|
+
# cheap YAML-free extraction of model-paths; dbt defaults to ["models"]
|
|
269
|
+
match = re.search(r"model-paths:\s*\[([^\]]*)\]", project_yml)
|
|
270
|
+
if match:
|
|
271
|
+
names = [part.strip().strip("'\"") for part in match.group(1).split(",") if part.strip()]
|
|
272
|
+
else:
|
|
273
|
+
names = re.findall(r"model-paths:\s*\n((?:\s*-\s*.+\n?)+)", project_yml)
|
|
274
|
+
names = re.findall(r"-\s*['\"]?([^'\"\n]+)", names[0]) if names else ["models"]
|
|
275
|
+
for name in names or ["models"]:
|
|
276
|
+
candidate = root / name.strip()
|
|
277
|
+
if candidate.is_dir():
|
|
278
|
+
dirs.append(candidate)
|
|
279
|
+
return dirs or ([root / "models"] if (root / "models").is_dir() else [])
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _seed_dirs(project_dir: Path) -> list[Path]:
|
|
283
|
+
"""The configured seed-paths (default seeds/ and data/) that exist."""
|
|
284
|
+
names = ["seeds", "data"]
|
|
285
|
+
yml = project_dir / "dbt_project.yml"
|
|
286
|
+
if yml.is_file():
|
|
287
|
+
try:
|
|
288
|
+
import yaml
|
|
289
|
+
|
|
290
|
+
loaded = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
|
|
291
|
+
value = (loaded or {}).get("seed-paths") if isinstance(loaded, dict) else None
|
|
292
|
+
if isinstance(value, str):
|
|
293
|
+
value = [value]
|
|
294
|
+
if isinstance(value, list) and value:
|
|
295
|
+
names = [v for v in value if isinstance(v, str) and v.strip()]
|
|
296
|
+
except Exception:
|
|
297
|
+
pass
|
|
298
|
+
return [project_dir / n for n in names if (project_dir / n).is_dir()]
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
|
|
302
|
+
"""The directories a dbt project owns: its configured model/macro/test/
|
|
303
|
+
seed/snapshot/analysis paths plus dbt's machinery. Anything else under
|
|
304
|
+
the project dir is the team's own SQL, not dbt's. Configured spellings
|
|
305
|
+
are read from dbt_project.yml (review: analysis-paths: ["queries"]
|
|
306
|
+
must not be swept into raw models); the defaults stay excluded too,
|
|
307
|
+
since stale default dirs linger on disk."""
|
|
308
|
+
defaults = {
|
|
309
|
+
"analysis-paths": ["analyses", "analysis"],
|
|
310
|
+
"test-paths": ["tests"],
|
|
311
|
+
"seed-paths": ["seeds", "data"],
|
|
312
|
+
"snapshot-paths": ["snapshots"],
|
|
313
|
+
"docs-paths": [],
|
|
314
|
+
"asset-paths": [],
|
|
315
|
+
"target-path": ["target"],
|
|
316
|
+
"packages-install-path": ["dbt_packages", "dbt_modules"],
|
|
317
|
+
"log-path": ["logs"],
|
|
318
|
+
}
|
|
319
|
+
config: dict = {}
|
|
320
|
+
yml = project_dir / "dbt_project.yml"
|
|
321
|
+
if yml.is_file():
|
|
322
|
+
try:
|
|
323
|
+
import yaml
|
|
324
|
+
|
|
325
|
+
loaded = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
|
|
326
|
+
config = loaded if isinstance(loaded, dict) else {}
|
|
327
|
+
except Exception:
|
|
328
|
+
config = {}
|
|
329
|
+
names: set[str] = set()
|
|
330
|
+
for key, fallback in defaults.items():
|
|
331
|
+
value = config.get(key, [])
|
|
332
|
+
if isinstance(value, str):
|
|
333
|
+
value = [value]
|
|
334
|
+
names.update(v for v in (value or []) if isinstance(v, str) and v.strip())
|
|
335
|
+
names.update(fallback)
|
|
336
|
+
owned = [project_dir / name for name in sorted(names)]
|
|
337
|
+
owned += _dbt_model_dirs(project_dir)
|
|
338
|
+
owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
|
|
339
|
+
return [d for d in owned if d.is_dir()]
|
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
"""Cross-model identity: collisions, suffix lookups, ingested schemas."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from ripple.loaders.types import Model, Project
|
|
8
|
+
|
|
9
|
+
_WHOLE_PLACEHOLDER_RE = re.compile(r"^(\{\{[^}]*\}\}|\{[^{}]*\})$")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _split_outside_braces(name: str) -> list[str]:
|
|
13
|
+
parts, depth, cur = [], 0, []
|
|
14
|
+
for ch in name:
|
|
15
|
+
if ch == "{":
|
|
16
|
+
depth += 1
|
|
17
|
+
elif ch == "}":
|
|
18
|
+
depth = max(depth - 1, 0)
|
|
19
|
+
if ch == "." and depth == 0:
|
|
20
|
+
parts.append("".join(cur))
|
|
21
|
+
cur = []
|
|
22
|
+
else:
|
|
23
|
+
cur.append(ch)
|
|
24
|
+
parts.append("".join(cur))
|
|
25
|
+
return parts
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def strip_placeholders(name: str) -> str:
|
|
29
|
+
"""{hscic}.ccgs -> ccgs, {project}.hscic.bnf -> hscic.bnf,
|
|
30
|
+
{{params.dataset_name}}.blocks -> blocks. A qualifier that is wholly a
|
|
31
|
+
placeholder is deploy config, not identity (openprescribing,
|
|
32
|
+
bitcoin_etl); the relation a person ingests is what is left. A segment
|
|
33
|
+
that mixes text and a placeholder (blocks{{params.ds_postfix}}) is a
|
|
34
|
+
templated NAME, a different table each deploy, and is returned unchanged
|
|
35
|
+
so it stays cited as written (round 11)."""
|
|
36
|
+
parts = _split_outside_braces(name)
|
|
37
|
+
kept = [p for p in parts if p and not _WHOLE_PLACEHOLDER_RE.match(p)]
|
|
38
|
+
if any("{" in p for p in kept):
|
|
39
|
+
return name
|
|
40
|
+
return ".".join(kept)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _name_suffixes(name: str) -> set[str]:
|
|
44
|
+
"""Every suffix a qualified name answers to: db.schema.t -> {db.schema.t,
|
|
45
|
+
schema.t, t}."""
|
|
46
|
+
parts = name.split(".")
|
|
47
|
+
return {".".join(parts[i:]) for i in range(len(parts))}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _qualifiers_by_bare_name(project: Project) -> dict[str, set[str]]:
|
|
51
|
+
"""bare relation name -> every qualified form the project's SQL writes.
|
|
52
|
+
|
|
53
|
+
Only the ingest path needs this, so it parses lazily and only when a
|
|
54
|
+
schemas.json exists. Two entries for one bare name means the name does not
|
|
55
|
+
identify a relation in this repo, whatever a warehouse returned for it.
|
|
56
|
+
"""
|
|
57
|
+
import sqlglot
|
|
58
|
+
from sqlglot import exp
|
|
59
|
+
|
|
60
|
+
from ripple.engine.column_ref import qualified_table_name
|
|
61
|
+
|
|
62
|
+
seen: dict[str, set[str]] = {}
|
|
63
|
+
for model in project.models:
|
|
64
|
+
if not model.sql.strip():
|
|
65
|
+
continue
|
|
66
|
+
try:
|
|
67
|
+
parsed = sqlglot.parse(
|
|
68
|
+
model.sql, dialect=model.dialect or project.dialect, error_level=None
|
|
69
|
+
)
|
|
70
|
+
except Exception:
|
|
71
|
+
continue
|
|
72
|
+
for statement in parsed:
|
|
73
|
+
if statement is None:
|
|
74
|
+
continue
|
|
75
|
+
for table in statement.find_all(exp.Table):
|
|
76
|
+
if not table.name or not (table.db or table.catalog):
|
|
77
|
+
continue
|
|
78
|
+
# placeholder qualifiers are deploy config, not identity;
|
|
79
|
+
# counting {{ params.db }}.s.t and s.t as two relations
|
|
80
|
+
# refused bare ingested schemas as falsely ambiguous
|
|
81
|
+
qualified = strip_placeholders(qualified_table_name(table).lower())
|
|
82
|
+
if "." not in qualified:
|
|
83
|
+
continue
|
|
84
|
+
seen.setdefault(table.name.lower(), set()).add(qualified)
|
|
85
|
+
return seen
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
|
|
89
|
+
"""Expose .ripple/schemas.json tables as known external relations.
|
|
90
|
+
|
|
91
|
+
A name matching an existing schema-less relation (a dbt source, a base
|
|
92
|
+
table with no columns) attaches its columns there; anything else becomes
|
|
93
|
+
a new source with declared columns. A model the project actually derives
|
|
94
|
+
keeps its own analysis: ingested schemas fill gaps, they never override.
|
|
95
|
+
"""
|
|
96
|
+
from ripple.schemas import load_schemas
|
|
97
|
+
|
|
98
|
+
# the CLI and MCP write .ripple/schemas.json at the repo root they were
|
|
99
|
+
# run from; a dbt project in a subfolder (balboa's transform/) has its
|
|
100
|
+
# own root, and reading only there made every ingest a silent no-op
|
|
101
|
+
tables: dict = {}
|
|
102
|
+
for root in dict.fromkeys([project.root, *extra_roots]):
|
|
103
|
+
tables.update(load_schemas(root) or {})
|
|
104
|
+
if not tables:
|
|
105
|
+
return
|
|
106
|
+
# every relation a spelling names, not the first: a monorepo declares the
|
|
107
|
+
# same source in several projects (mattermost-dwh), and columns attached
|
|
108
|
+
# to one left the readers of the other star_only
|
|
109
|
+
index: dict[str, list[Model]] = {}
|
|
110
|
+
for m in [*project.models, *project.sources]:
|
|
111
|
+
for alias in {m.name, *m.aliases}:
|
|
112
|
+
bucket = index.setdefault(alias.lower(), [])
|
|
113
|
+
# identity, not equality: two projects' declarations of one
|
|
114
|
+
# source compare equal field by field and are distinct objects
|
|
115
|
+
if not any(existing is m for existing in bucket):
|
|
116
|
+
bucket.append(m)
|
|
117
|
+
qualifiers = _qualifiers_by_bare_name(project)
|
|
118
|
+
# a bare spelling a model owns is that model's: an ingested raw.orders
|
|
119
|
+
# granting itself "orders" made model_columns("orders") answer for the
|
|
120
|
+
# source (same rule dbt sources already follow)
|
|
121
|
+
owned = {m.name.lower() for m in project.models if m.sql.strip()}
|
|
122
|
+
for name, columns in tables.items():
|
|
123
|
+
if "." not in name:
|
|
124
|
+
# A bare ingested name cannot answer for two different relations.
|
|
125
|
+
# raw.orders and staging.orders both matched one `orders` entry and
|
|
126
|
+
# every edge came out high_confidence, when the same edges are
|
|
127
|
+
# review_required without the ingest. Refuse rather than pick one:
|
|
128
|
+
# the fix is to re-ingest qualified, and only the user knows which
|
|
129
|
+
# relation the columns came from.
|
|
130
|
+
seen = qualifiers.get(name.lower(), set())
|
|
131
|
+
if len(seen) > 1:
|
|
132
|
+
project.warnings.append(
|
|
133
|
+
f"Ingested schema '{name}' matches {len(seen)} relations "
|
|
134
|
+
f"({', '.join(sorted(seen))}); it was not attached. "
|
|
135
|
+
"Re-ingest using qualified names."
|
|
136
|
+
)
|
|
137
|
+
continue
|
|
138
|
+
candidates = sorted(_name_suffixes(name), key=len, reverse=True)
|
|
139
|
+
parts = name.split(".")
|
|
140
|
+
if len(parts) >= 2:
|
|
141
|
+
candidates.append(f"{parts[-2]}__{parts[-1]}")
|
|
142
|
+
# only a schema-less relation can take these columns; a bare suffix
|
|
143
|
+
# that names a model with SQL (Mozilla's events_stream beside an
|
|
144
|
+
# ingest of mdn_fred.events_stream) must not swallow the ingest
|
|
145
|
+
targets = next(
|
|
146
|
+
(
|
|
147
|
+
[t for t in index[k] if not t.sql.strip()]
|
|
148
|
+
for k in candidates
|
|
149
|
+
if k in index and any(not t.sql.strip() for t in index[k])
|
|
150
|
+
),
|
|
151
|
+
None,
|
|
152
|
+
)
|
|
153
|
+
if targets:
|
|
154
|
+
for target in targets:
|
|
155
|
+
known = {c.lower() for c in target.declared_columns}
|
|
156
|
+
target.declared_columns.extend(c for c in columns if c not in known)
|
|
157
|
+
target.aliases |= {a for a in _name_suffixes(name) if a.lower() not in owned}
|
|
158
|
+
else:
|
|
159
|
+
source = Model(
|
|
160
|
+
name=name,
|
|
161
|
+
sql="",
|
|
162
|
+
path="",
|
|
163
|
+
aliases={a for a in _name_suffixes(name) if a.lower() not in owned},
|
|
164
|
+
declared_columns=list(columns),
|
|
165
|
+
is_source=True,
|
|
166
|
+
evidence="ingested",
|
|
167
|
+
)
|
|
168
|
+
project.sources.append(source)
|
|
169
|
+
for alias in {source.name, *source.aliases}:
|
|
170
|
+
index.setdefault(alias.lower(), []).append(source)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _qualify_collisions(project: Project) -> None:
|
|
174
|
+
"""Two models with the same display name must never merge into one node.
|
|
175
|
+
|
|
176
|
+
Colliding models get path-qualified canonical names; the bare name stays
|
|
177
|
+
as an alias on both, so a reference to it resolves ambiguously (which the
|
|
178
|
+
graph layer flags review_required) instead of silently binding.
|
|
179
|
+
"""
|
|
180
|
+
by_name: dict[str, list[Model]] = {}
|
|
181
|
+
for model in project.models:
|
|
182
|
+
# resolution is case-insensitive (name_candidates lowercases), so
|
|
183
|
+
# collision detection in a sql dir must fold case the same way
|
|
184
|
+
key = model.name.lower() if project.mode == "sql-dir" else model.name
|
|
185
|
+
by_name.setdefault(key, []).append(model)
|
|
186
|
+
taken = {(m.name.lower() if project.mode == "sql-dir" else m.name) for m in project.models}
|
|
187
|
+
for name, group in by_name.items():
|
|
188
|
+
if len(group) < 2:
|
|
189
|
+
continue
|
|
190
|
+
if project.mode == "sql-dir":
|
|
191
|
+
group = _collapse_sql_dir_tables(project, group)
|
|
192
|
+
if len(group) < 2:
|
|
193
|
+
continue
|
|
194
|
+
project.warnings.append(
|
|
195
|
+
f"{len(group)} models share the name '{name}'; qualified by path. "
|
|
196
|
+
"References to the bare name are flagged for review."
|
|
197
|
+
)
|
|
198
|
+
# dbt compiles exactly one model per stem; a swept twin outside the
|
|
199
|
+
# dbt-owned paths (pudl's schema_inputs/ fixtures, holdout round 6)
|
|
200
|
+
# is not deployed under that name. When one member is the dbt model,
|
|
201
|
+
# the bare identity is provably its and only the twins rename.
|
|
202
|
+
dbt_owned = [m for m in group if m.evidence in ("manifest", "raw_jinja")]
|
|
203
|
+
keeper = dbt_owned[0] if len(dbt_owned) == 1 else None
|
|
204
|
+
for model in group:
|
|
205
|
+
if model is keeper:
|
|
206
|
+
continue
|
|
207
|
+
if project.mode == "sql-dir" and not model.sql.strip():
|
|
208
|
+
continue # a base table keeps its table-level name
|
|
209
|
+
base = re.sub(r"\.sql$", "", model.path).replace("/", "__") or model.name
|
|
210
|
+
suffix = model.name.lower()
|
|
211
|
+
if base.lower() == suffix or base.lower().endswith("__" + suffix):
|
|
212
|
+
qualified = base
|
|
213
|
+
else:
|
|
214
|
+
# a multi-relation file: the path alone names the FILE, and
|
|
215
|
+
# holdout round 4 collapsed eleven FAC views into one node
|
|
216
|
+
# this way; the relation keeps its identity inside the path
|
|
217
|
+
qualified = f"{base}__{model.name}"
|
|
218
|
+
# the rename must never assign one canonical name to two
|
|
219
|
+
# relations (review: foo__bar.sql defining bar AND
|
|
220
|
+
# foo__bar); fall through to ever-longer spellings until unique
|
|
221
|
+
key = qualified.lower() if project.mode == "sql-dir" else qualified
|
|
222
|
+
if key in taken and qualified != model.name:
|
|
223
|
+
qualified = f"{base}__{model.name}"
|
|
224
|
+
key = qualified.lower() if project.mode == "sql-dir" else qualified
|
|
225
|
+
while key in taken and key != (
|
|
226
|
+
model.name.lower() if project.mode == "sql-dir" else model.name
|
|
227
|
+
):
|
|
228
|
+
qualified = f"{qualified}__{model.name}"
|
|
229
|
+
key = qualified.lower() if project.mode == "sql-dir" else qualified
|
|
230
|
+
model.aliases.add(name)
|
|
231
|
+
model.name = qualified
|
|
232
|
+
taken.add(key)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _collapse_sql_dir_tables(project: Project, group: list[Model]) -> list[Model]:
|
|
236
|
+
"""In a plain sql dir the same CREATE TABLE name in several dump files is
|
|
237
|
+
one table-level relation (one dump = one database), not a collision worth
|
|
238
|
+
a path-qualified name. Schema-only definitions merge; models carrying a
|
|
239
|
+
real derivation stay distinct, EXCEPT prior-state derivations: a table
|
|
240
|
+
built by CREATE plus UPDATEs across files (nycdb's add_columns.sql and
|
|
241
|
+
full_text.sql both writing dobjobs, the gap named in PR #28) is one
|
|
242
|
+
relation whose lineage is the union of its statements."""
|
|
243
|
+
tables = [m for m in group if not m.sql.strip()]
|
|
244
|
+
if len(tables) >= 2:
|
|
245
|
+
# prefer the dialect-folded (lowercase) spelling as the surviving name
|
|
246
|
+
keep = next((m for m in tables if m.name == m.name.lower()), tables[0])
|
|
247
|
+
for other in tables:
|
|
248
|
+
if other is keep:
|
|
249
|
+
continue
|
|
250
|
+
for column in other.declared_columns:
|
|
251
|
+
if column not in keep.declared_columns:
|
|
252
|
+
keep.declared_columns.append(column)
|
|
253
|
+
keep.aliases |= other.aliases
|
|
254
|
+
project.models.remove(other)
|
|
255
|
+
group = [keep, *(m for m in group if m.sql.strip())]
|
|
256
|
+
return _merge_self_read_derivations(project, group)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _qualified_spellings(model: Model) -> set[str]:
|
|
260
|
+
return {a for a in model.aliases if "." in a}
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _merge_self_read_derivations(project: Project, group: list[Model]) -> list[Model]:
|
|
264
|
+
"""Fold UPDATE-derived models into the one model that owns the table.
|
|
265
|
+
|
|
266
|
+
Only prior-state (self_read) derivations merge, only when at most one
|
|
267
|
+
creator exists (two creators is a real collision), and only when the
|
|
268
|
+
qualified spellings agree: sales.summary and hr.summary stay two
|
|
269
|
+
relations whatever the bare name says (the round-3 identity rule)."""
|
|
270
|
+
creators = [m for m in group if m.sql.strip() and not m.self_read]
|
|
271
|
+
updates = [m for m in group if m.sql.strip() and m.self_read]
|
|
272
|
+
if not updates or len(creators) > 1:
|
|
273
|
+
return group
|
|
274
|
+
bases = [m for m in group if not m.sql.strip()]
|
|
275
|
+
anchor = creators[0] if creators else (bases[0] if bases else updates[0])
|
|
276
|
+
kept = [anchor, *(m for m in bases if m is not anchor)]
|
|
277
|
+
for update in updates:
|
|
278
|
+
if update is anchor:
|
|
279
|
+
continue
|
|
280
|
+
anchor_q = _qualified_spellings(anchor)
|
|
281
|
+
update_q = _qualified_spellings(update)
|
|
282
|
+
# a bare updater against a schema-qualified creator proves nothing:
|
|
283
|
+
# the update could run under any search_path (review of
|
|
284
|
+
# cycle 6); merging needs both bare or an agreeing qualification
|
|
285
|
+
if (anchor_q or update_q) and not (anchor_q & update_q):
|
|
286
|
+
kept.append(update)
|
|
287
|
+
continue
|
|
288
|
+
if not anchor.sql.strip():
|
|
289
|
+
anchor.sql = update.sql
|
|
290
|
+
else:
|
|
291
|
+
anchor.extra_sqls.append(update.sql)
|
|
292
|
+
anchor.extra_sqls.extend(update.extra_sqls)
|
|
293
|
+
anchor.self_read = True
|
|
294
|
+
anchor.declared_parents |= update.declared_parents
|
|
295
|
+
anchor.aliases |= update.aliases
|
|
296
|
+
anchor.stem_aliases |= update.stem_aliases
|
|
297
|
+
for column in update.declared_columns:
|
|
298
|
+
if column not in anchor.declared_columns:
|
|
299
|
+
anchor.declared_columns.append(column)
|
|
300
|
+
project.models.remove(update)
|
|
301
|
+
return kept
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
ROLE_STEMS = frozenset({"query", "view", "model", "table", "select", "main", "definition"})
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def directory_identity(rel: str) -> tuple[str, set[str]] | None:
|
|
308
|
+
"""(name, dotted aliases) for a file whose stem names its role, not the relation.
|
|
309
|
+
|
|
310
|
+
Mozilla's bigquery-etl keeps 887 query.sql and 803 view.sql files at
|
|
311
|
+
sql/<project>/<dataset>/<table>/ and references each relation as
|
|
312
|
+
`project.dataset.table`. Named by stem, every model was "query", the
|
|
313
|
+
collision handler mangled them into path names no reference could match,
|
|
314
|
+
and a 2,096-model repo came out with 0 verified edges. The directory is
|
|
315
|
+
the relation; its two nearest ancestors are the qualified spellings.
|
|
316
|
+
Returns None when the stem carries identity as usual. checks.sql and
|
|
317
|
+
script.sql are deliberately not roles: a check reads the relation, it
|
|
318
|
+
does not define it, and claiming the directory would collide with the
|
|
319
|
+
real definition beside it.
|
|
320
|
+
"""
|
|
321
|
+
from pathlib import Path
|
|
322
|
+
|
|
323
|
+
parts = Path(rel).parts
|
|
324
|
+
if len(parts) < 2 or Path(parts[-1]).stem.lower() not in ROLE_STEMS:
|
|
325
|
+
return None
|
|
326
|
+
dirs = parts[:-1]
|
|
327
|
+
aliases = {".".join(dirs[-n:]) for n in (2, 3) if len(dirs) >= n}
|
|
328
|
+
return dirs[-1], aliases
|