ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,339 @@
1
+ """dbt_project.yml knobs: vars, enabled rules, macro and model paths."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from pathlib import Path
7
+
8
+
9
+ def _project_vars(root: Path) -> dict:
10
+ """Declared vars from dbt_project.yml. A sub-dict scoped under the
11
+ project's own name is flattened in (dbt per-package scoping); scopes for
12
+ other packages are left as-is since their names aren't known here."""
13
+ project_yml = root / "dbt_project.yml"
14
+ if not project_yml.is_file():
15
+ return {}
16
+ import yaml
17
+
18
+ text = project_yml.read_text(errors="replace", encoding="utf-8")
19
+ try:
20
+ parsed = yaml.safe_load(text)
21
+ except yaml.YAMLError:
22
+ # hand-edited ymls with inconsistent indentation still deserve their
23
+ # flat vars; salvage them line by line
24
+ return _flat_vars_fallback(text)
25
+ if not isinstance(parsed, dict) or not isinstance(parsed.get("vars"), dict):
26
+ return {}
27
+ declared = dict(parsed["vars"])
28
+ scoped = declared.get(parsed.get("name"))
29
+ if isinstance(scoped, dict):
30
+ declared.pop(parsed.get("name"))
31
+ declared.update(scoped)
32
+ return declared
33
+
34
+
35
+ def _flat_vars_fallback(text: str) -> dict:
36
+ import yaml
37
+
38
+ lines = text.splitlines()
39
+ start = next(
40
+ (i for i, line in enumerate(lines) if re.match(r"^\s*vars:\s*(#.*)?$", line)), None
41
+ )
42
+ if start is None:
43
+ return {}
44
+ block: list[str] = []
45
+ for line in lines[start + 1 :]:
46
+ if re.match(r"^\s*[A-Za-z0-9_.-]+:\s*\S", line):
47
+ block.append(line.strip())
48
+ elif line.strip():
49
+ break
50
+ try:
51
+ parsed = yaml.safe_load("\n".join(block))
52
+ except yaml.YAMLError:
53
+ return {}
54
+ return parsed if isinstance(parsed, dict) else {}
55
+
56
+
57
+ def _enabled_overrides(
58
+ root: Path, dialect: str, project_vars: dict
59
+ ) -> list[tuple[tuple[str, ...], bool]]:
60
+ """(path prefix under the model dir, enabled) rules from the project's own
61
+ models: block in dbt_project.yml. dbt packages ship every dialect variant
62
+ of a model and gate them with enabled configs; ignoring those poisons ref
63
+ binding with variants dbt would never build."""
64
+ project_yml = root / "dbt_project.yml"
65
+ if not project_yml.is_file():
66
+ return []
67
+ import yaml
68
+
69
+ try:
70
+ parsed = yaml.safe_load(project_yml.read_text(errors="replace", encoding="utf-8"))
71
+ except yaml.YAMLError:
72
+ return []
73
+ if not isinstance(parsed, dict) or not isinstance(parsed.get("models"), dict):
74
+ return []
75
+ tree = parsed["models"].get(parsed.get("name"))
76
+ if not isinstance(tree, dict):
77
+ return []
78
+ rules: list[tuple[tuple[str, ...], bool]] = []
79
+
80
+ def walk(node: dict, prefix: tuple[str, ...]) -> None:
81
+ raw = node.get("+enabled", node.get("enabled"))
82
+ if raw is not None and not isinstance(raw, dict):
83
+ decided = _eval_enabled(raw, dialect, project_vars)
84
+ if decided is not None:
85
+ rules.append((prefix, decided))
86
+ for key, child in node.items():
87
+ if isinstance(child, dict) and not key.startswith("+"):
88
+ walk(child, prefix + (key,))
89
+
90
+ walk(tree, ())
91
+ return rules
92
+
93
+
94
+ def _enabled_for(parts: tuple[str, ...], rules: list[tuple[tuple[str, ...], bool]]) -> bool | None:
95
+ """Deepest matching rule wins, dbt-style. None when no rule applies."""
96
+ decided, best = None, -1
97
+ for prefix, value in rules:
98
+ if len(prefix) > best and parts[: len(prefix)] == prefix:
99
+ decided, best = value, len(prefix)
100
+ return decided
101
+
102
+
103
+ _CONFIG_CALL_RE = re.compile(r"{{-?\s*config\s*\(", re.I)
104
+
105
+
106
+ def _file_enabled(raw: str, dialect: str, project_vars: dict) -> bool | None:
107
+ """enabled= from a model file's own {{ config(...) }}, if it decides."""
108
+ match = _CONFIG_CALL_RE.search(raw)
109
+ if match is None:
110
+ return None
111
+ depth, i = 1, match.end()
112
+ while i < len(raw) and depth:
113
+ if raw[i] == "(":
114
+ depth += 1
115
+ elif raw[i] == ")":
116
+ depth -= 1
117
+ i += 1
118
+ body = raw[match.end() : i - 1]
119
+ arg = re.search(r"\benabled\s*=", body)
120
+ if arg is None:
121
+ return None
122
+ expr: list[str] = []
123
+ depth = 0
124
+ for ch in body[arg.end() :]:
125
+ if ch in "([{":
126
+ depth += 1
127
+ elif ch in ")]}":
128
+ depth -= 1
129
+ elif ch == "," and depth == 0:
130
+ break
131
+ expr.append(ch)
132
+ return _render_enabled("{{ (" + "".join(expr).strip() + ") }}", dialect, project_vars)
133
+
134
+
135
+ def _eval_enabled(value, dialect: str, project_vars: dict) -> bool | None:
136
+ """True/False when a literal or a target/var-decided expression settles
137
+ it; None (treated as enabled) when it cannot be evaluated honestly."""
138
+ if isinstance(value, bool):
139
+ return value
140
+ if not isinstance(value, str):
141
+ return None
142
+ text = value.strip()
143
+ if text.lower() in ("true", "false"):
144
+ return text.lower() == "true"
145
+ if "{{" in text or "{%" in text:
146
+ return _render_enabled(text, dialect, project_vars)
147
+ return None
148
+
149
+
150
+ def _render_enabled(template: str, dialect: str, project_vars: dict) -> bool | None:
151
+ """Evaluate an enabled jinja expression with target.type = the project
152
+ dialect and declared vars. Anything unresolvable renders as None: an
153
+ un-evaluatable gate must not disable a model."""
154
+ from jinja2 import Environment, StrictUndefined
155
+
156
+ env = Environment(undefined=StrictUndefined)
157
+ # dbt's as_bool coerces after render; the comparison already yields a bool
158
+ env.filters["as_bool"] = lambda value: value
159
+ missing = object()
160
+
161
+ def _var(name, default=missing):
162
+ if name in project_vars:
163
+ return project_vars[name]
164
+ if default is missing:
165
+ raise KeyError(name)
166
+ return default
167
+
168
+ try:
169
+ rendered = (
170
+ env.from_string(template)
171
+ .render(var=_var, target={"type": dialect, "name": "prod"})
172
+ .strip()
173
+ )
174
+ except Exception:
175
+ return None
176
+ if rendered == "True":
177
+ return True
178
+ if rendered == "False":
179
+ return False
180
+ return None
181
+
182
+
183
+ def _macro_dir_names(project_yml: Path) -> list[str]:
184
+ """macro-paths from a dbt_project.yml, defaulting to dbt's ["macros"].
185
+
186
+ Real YAML parsing, because block-style lists are valid dbt config and a
187
+ flow-only regex silently falls back to macros/ on them, which is the same
188
+ zero-macros failure this exists to fix (make-open-data, holdout round 2:
189
+ macro-paths: ["5_macros"], every macro call rendered to nothing)."""
190
+ if not project_yml.is_file():
191
+ return ["macros"]
192
+ text = project_yml.read_text(errors="replace", encoding="utf-8")
193
+ try:
194
+ import yaml
195
+
196
+ data = yaml.safe_load(text) or {}
197
+ names = data.get("macro-paths")
198
+ if isinstance(names, list):
199
+ cleaned = [n for n in names if isinstance(n, str) and n.strip()]
200
+ if cleaned:
201
+ return cleaned
202
+ except Exception:
203
+ match = re.search(r"macro-paths:\s*\[([^\]]*)\]", text)
204
+ if match:
205
+ names = [
206
+ part.strip().strip("'\"") for part in match.group(1).split(",") if part.strip()
207
+ ]
208
+ if names:
209
+ return names
210
+ return ["macros"]
211
+
212
+
213
+ def _macro_sources(root: Path) -> list[tuple[str | None, str]]:
214
+ """(package, source) for the project's own macros and vendored packages.
215
+
216
+ The project's own macros carry the dbt_project.yml name as their package
217
+ so both bare and self-namespaced calls ({{ ga4.unnest_key(...) }} inside
218
+ the ga4 package itself) resolve. macro-paths is honored for the root
219
+ project and every vendored package alike.
220
+ """
221
+ root_package = None
222
+ project_yml = root / "dbt_project.yml"
223
+ if project_yml.is_file():
224
+ match = re.search(
225
+ r"^name:\s*['\"]?([A-Za-z0-9_]+)",
226
+ project_yml.read_text(errors="replace", encoding="utf-8"),
227
+ re.M,
228
+ )
229
+ if match:
230
+ root_package = match.group(1)
231
+ pairs: list[tuple[str | None, str]] = []
232
+ seen_files: set[Path] = set()
233
+
234
+ def _load_dirs(base: Path, dir_names: list[str], package: str | None) -> None:
235
+ for dir_name in dir_names:
236
+ macros_dir = base / dir_name
237
+ if not macros_dir.is_dir():
238
+ continue
239
+ for macro_file in sorted(macros_dir.rglob("*.sql")):
240
+ resolved = macro_file.resolve()
241
+ if resolved in seen_files:
242
+ continue
243
+ seen_files.add(resolved)
244
+ try:
245
+ pairs.append(
246
+ (package, macro_file.read_text(errors="replace", encoding="utf-8"))
247
+ )
248
+ except OSError:
249
+ continue
250
+
251
+ _load_dirs(root, _macro_dir_names(project_yml), root_package)
252
+ for packages_dir_name in ("dbt_packages", "dbt_modules"):
253
+ packages_dir = root / packages_dir_name
254
+ if not packages_dir.is_dir():
255
+ continue
256
+ for package_dir in sorted(p for p in packages_dir.iterdir() if p.is_dir()):
257
+ _load_dirs(
258
+ package_dir,
259
+ _macro_dir_names(package_dir / "dbt_project.yml"),
260
+ package_dir.name,
261
+ )
262
+ return pairs
263
+
264
+
265
+ def _dbt_model_dirs(root: Path) -> list[Path]:
266
+ dirs = []
267
+ project_yml = (root / "dbt_project.yml").read_text(errors="replace", encoding="utf-8")
268
+ # cheap YAML-free extraction of model-paths; dbt defaults to ["models"]
269
+ match = re.search(r"model-paths:\s*\[([^\]]*)\]", project_yml)
270
+ if match:
271
+ names = [part.strip().strip("'\"") for part in match.group(1).split(",") if part.strip()]
272
+ else:
273
+ names = re.findall(r"model-paths:\s*\n((?:\s*-\s*.+\n?)+)", project_yml)
274
+ names = re.findall(r"-\s*['\"]?([^'\"\n]+)", names[0]) if names else ["models"]
275
+ for name in names or ["models"]:
276
+ candidate = root / name.strip()
277
+ if candidate.is_dir():
278
+ dirs.append(candidate)
279
+ return dirs or ([root / "models"] if (root / "models").is_dir() else [])
280
+
281
+
282
+ def _seed_dirs(project_dir: Path) -> list[Path]:
283
+ """The configured seed-paths (default seeds/ and data/) that exist."""
284
+ names = ["seeds", "data"]
285
+ yml = project_dir / "dbt_project.yml"
286
+ if yml.is_file():
287
+ try:
288
+ import yaml
289
+
290
+ loaded = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
291
+ value = (loaded or {}).get("seed-paths") if isinstance(loaded, dict) else None
292
+ if isinstance(value, str):
293
+ value = [value]
294
+ if isinstance(value, list) and value:
295
+ names = [v for v in value if isinstance(v, str) and v.strip()]
296
+ except Exception:
297
+ pass
298
+ return [project_dir / n for n in names if (project_dir / n).is_dir()]
299
+
300
+
301
+ def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
302
+ """The directories a dbt project owns: its configured model/macro/test/
303
+ seed/snapshot/analysis paths plus dbt's machinery. Anything else under
304
+ the project dir is the team's own SQL, not dbt's. Configured spellings
305
+ are read from dbt_project.yml (review: analysis-paths: ["queries"]
306
+ must not be swept into raw models); the defaults stay excluded too,
307
+ since stale default dirs linger on disk."""
308
+ defaults = {
309
+ "analysis-paths": ["analyses", "analysis"],
310
+ "test-paths": ["tests"],
311
+ "seed-paths": ["seeds", "data"],
312
+ "snapshot-paths": ["snapshots"],
313
+ "docs-paths": [],
314
+ "asset-paths": [],
315
+ "target-path": ["target"],
316
+ "packages-install-path": ["dbt_packages", "dbt_modules"],
317
+ "log-path": ["logs"],
318
+ }
319
+ config: dict = {}
320
+ yml = project_dir / "dbt_project.yml"
321
+ if yml.is_file():
322
+ try:
323
+ import yaml
324
+
325
+ loaded = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
326
+ config = loaded if isinstance(loaded, dict) else {}
327
+ except Exception:
328
+ config = {}
329
+ names: set[str] = set()
330
+ for key, fallback in defaults.items():
331
+ value = config.get(key, [])
332
+ if isinstance(value, str):
333
+ value = [value]
334
+ names.update(v for v in (value or []) if isinstance(v, str) and v.strip())
335
+ names.update(fallback)
336
+ owned = [project_dir / name for name in sorted(names)]
337
+ owned += _dbt_model_dirs(project_dir)
338
+ owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
339
+ return [d for d in owned if d.is_dir()]
@@ -0,0 +1,328 @@
1
+ """Cross-model identity: collisions, suffix lookups, ingested schemas."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ from ripple.loaders.types import Model, Project
8
+
9
+ _WHOLE_PLACEHOLDER_RE = re.compile(r"^(\{\{[^}]*\}\}|\{[^{}]*\})$")
10
+
11
+
12
+ def _split_outside_braces(name: str) -> list[str]:
13
+ parts, depth, cur = [], 0, []
14
+ for ch in name:
15
+ if ch == "{":
16
+ depth += 1
17
+ elif ch == "}":
18
+ depth = max(depth - 1, 0)
19
+ if ch == "." and depth == 0:
20
+ parts.append("".join(cur))
21
+ cur = []
22
+ else:
23
+ cur.append(ch)
24
+ parts.append("".join(cur))
25
+ return parts
26
+
27
+
28
+ def strip_placeholders(name: str) -> str:
29
+ """{hscic}.ccgs -> ccgs, {project}.hscic.bnf -> hscic.bnf,
30
+ {{params.dataset_name}}.blocks -> blocks. A qualifier that is wholly a
31
+ placeholder is deploy config, not identity (openprescribing,
32
+ bitcoin_etl); the relation a person ingests is what is left. A segment
33
+ that mixes text and a placeholder (blocks{{params.ds_postfix}}) is a
34
+ templated NAME, a different table each deploy, and is returned unchanged
35
+ so it stays cited as written (round 11)."""
36
+ parts = _split_outside_braces(name)
37
+ kept = [p for p in parts if p and not _WHOLE_PLACEHOLDER_RE.match(p)]
38
+ if any("{" in p for p in kept):
39
+ return name
40
+ return ".".join(kept)
41
+
42
+
43
+ def _name_suffixes(name: str) -> set[str]:
44
+ """Every suffix a qualified name answers to: db.schema.t -> {db.schema.t,
45
+ schema.t, t}."""
46
+ parts = name.split(".")
47
+ return {".".join(parts[i:]) for i in range(len(parts))}
48
+
49
+
50
+ def _qualifiers_by_bare_name(project: Project) -> dict[str, set[str]]:
51
+ """bare relation name -> every qualified form the project's SQL writes.
52
+
53
+ Only the ingest path needs this, so it parses lazily and only when a
54
+ schemas.json exists. Two entries for one bare name means the name does not
55
+ identify a relation in this repo, whatever a warehouse returned for it.
56
+ """
57
+ import sqlglot
58
+ from sqlglot import exp
59
+
60
+ from ripple.engine.column_ref import qualified_table_name
61
+
62
+ seen: dict[str, set[str]] = {}
63
+ for model in project.models:
64
+ if not model.sql.strip():
65
+ continue
66
+ try:
67
+ parsed = sqlglot.parse(
68
+ model.sql, dialect=model.dialect or project.dialect, error_level=None
69
+ )
70
+ except Exception:
71
+ continue
72
+ for statement in parsed:
73
+ if statement is None:
74
+ continue
75
+ for table in statement.find_all(exp.Table):
76
+ if not table.name or not (table.db or table.catalog):
77
+ continue
78
+ # placeholder qualifiers are deploy config, not identity;
79
+ # counting {{ params.db }}.s.t and s.t as two relations
80
+ # refused bare ingested schemas as falsely ambiguous
81
+ qualified = strip_placeholders(qualified_table_name(table).lower())
82
+ if "." not in qualified:
83
+ continue
84
+ seen.setdefault(table.name.lower(), set()).add(qualified)
85
+ return seen
86
+
87
+
88
+ def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
89
+ """Expose .ripple/schemas.json tables as known external relations.
90
+
91
+ A name matching an existing schema-less relation (a dbt source, a base
92
+ table with no columns) attaches its columns there; anything else becomes
93
+ a new source with declared columns. A model the project actually derives
94
+ keeps its own analysis: ingested schemas fill gaps, they never override.
95
+ """
96
+ from ripple.schemas import load_schemas
97
+
98
+ # the CLI and MCP write .ripple/schemas.json at the repo root they were
99
+ # run from; a dbt project in a subfolder (balboa's transform/) has its
100
+ # own root, and reading only there made every ingest a silent no-op
101
+ tables: dict = {}
102
+ for root in dict.fromkeys([project.root, *extra_roots]):
103
+ tables.update(load_schemas(root) or {})
104
+ if not tables:
105
+ return
106
+ # every relation a spelling names, not the first: a monorepo declares the
107
+ # same source in several projects (mattermost-dwh), and columns attached
108
+ # to one left the readers of the other star_only
109
+ index: dict[str, list[Model]] = {}
110
+ for m in [*project.models, *project.sources]:
111
+ for alias in {m.name, *m.aliases}:
112
+ bucket = index.setdefault(alias.lower(), [])
113
+ # identity, not equality: two projects' declarations of one
114
+ # source compare equal field by field and are distinct objects
115
+ if not any(existing is m for existing in bucket):
116
+ bucket.append(m)
117
+ qualifiers = _qualifiers_by_bare_name(project)
118
+ # a bare spelling a model owns is that model's: an ingested raw.orders
119
+ # granting itself "orders" made model_columns("orders") answer for the
120
+ # source (same rule dbt sources already follow)
121
+ owned = {m.name.lower() for m in project.models if m.sql.strip()}
122
+ for name, columns in tables.items():
123
+ if "." not in name:
124
+ # A bare ingested name cannot answer for two different relations.
125
+ # raw.orders and staging.orders both matched one `orders` entry and
126
+ # every edge came out high_confidence, when the same edges are
127
+ # review_required without the ingest. Refuse rather than pick one:
128
+ # the fix is to re-ingest qualified, and only the user knows which
129
+ # relation the columns came from.
130
+ seen = qualifiers.get(name.lower(), set())
131
+ if len(seen) > 1:
132
+ project.warnings.append(
133
+ f"Ingested schema '{name}' matches {len(seen)} relations "
134
+ f"({', '.join(sorted(seen))}); it was not attached. "
135
+ "Re-ingest using qualified names."
136
+ )
137
+ continue
138
+ candidates = sorted(_name_suffixes(name), key=len, reverse=True)
139
+ parts = name.split(".")
140
+ if len(parts) >= 2:
141
+ candidates.append(f"{parts[-2]}__{parts[-1]}")
142
+ # only a schema-less relation can take these columns; a bare suffix
143
+ # that names a model with SQL (Mozilla's events_stream beside an
144
+ # ingest of mdn_fred.events_stream) must not swallow the ingest
145
+ targets = next(
146
+ (
147
+ [t for t in index[k] if not t.sql.strip()]
148
+ for k in candidates
149
+ if k in index and any(not t.sql.strip() for t in index[k])
150
+ ),
151
+ None,
152
+ )
153
+ if targets:
154
+ for target in targets:
155
+ known = {c.lower() for c in target.declared_columns}
156
+ target.declared_columns.extend(c for c in columns if c not in known)
157
+ target.aliases |= {a for a in _name_suffixes(name) if a.lower() not in owned}
158
+ else:
159
+ source = Model(
160
+ name=name,
161
+ sql="",
162
+ path="",
163
+ aliases={a for a in _name_suffixes(name) if a.lower() not in owned},
164
+ declared_columns=list(columns),
165
+ is_source=True,
166
+ evidence="ingested",
167
+ )
168
+ project.sources.append(source)
169
+ for alias in {source.name, *source.aliases}:
170
+ index.setdefault(alias.lower(), []).append(source)
171
+
172
+
173
+ def _qualify_collisions(project: Project) -> None:
174
+ """Two models with the same display name must never merge into one node.
175
+
176
+ Colliding models get path-qualified canonical names; the bare name stays
177
+ as an alias on both, so a reference to it resolves ambiguously (which the
178
+ graph layer flags review_required) instead of silently binding.
179
+ """
180
+ by_name: dict[str, list[Model]] = {}
181
+ for model in project.models:
182
+ # resolution is case-insensitive (name_candidates lowercases), so
183
+ # collision detection in a sql dir must fold case the same way
184
+ key = model.name.lower() if project.mode == "sql-dir" else model.name
185
+ by_name.setdefault(key, []).append(model)
186
+ taken = {(m.name.lower() if project.mode == "sql-dir" else m.name) for m in project.models}
187
+ for name, group in by_name.items():
188
+ if len(group) < 2:
189
+ continue
190
+ if project.mode == "sql-dir":
191
+ group = _collapse_sql_dir_tables(project, group)
192
+ if len(group) < 2:
193
+ continue
194
+ project.warnings.append(
195
+ f"{len(group)} models share the name '{name}'; qualified by path. "
196
+ "References to the bare name are flagged for review."
197
+ )
198
+ # dbt compiles exactly one model per stem; a swept twin outside the
199
+ # dbt-owned paths (pudl's schema_inputs/ fixtures, holdout round 6)
200
+ # is not deployed under that name. When one member is the dbt model,
201
+ # the bare identity is provably its and only the twins rename.
202
+ dbt_owned = [m for m in group if m.evidence in ("manifest", "raw_jinja")]
203
+ keeper = dbt_owned[0] if len(dbt_owned) == 1 else None
204
+ for model in group:
205
+ if model is keeper:
206
+ continue
207
+ if project.mode == "sql-dir" and not model.sql.strip():
208
+ continue # a base table keeps its table-level name
209
+ base = re.sub(r"\.sql$", "", model.path).replace("/", "__") or model.name
210
+ suffix = model.name.lower()
211
+ if base.lower() == suffix or base.lower().endswith("__" + suffix):
212
+ qualified = base
213
+ else:
214
+ # a multi-relation file: the path alone names the FILE, and
215
+ # holdout round 4 collapsed eleven FAC views into one node
216
+ # this way; the relation keeps its identity inside the path
217
+ qualified = f"{base}__{model.name}"
218
+ # the rename must never assign one canonical name to two
219
+ # relations (review: foo__bar.sql defining bar AND
220
+ # foo__bar); fall through to ever-longer spellings until unique
221
+ key = qualified.lower() if project.mode == "sql-dir" else qualified
222
+ if key in taken and qualified != model.name:
223
+ qualified = f"{base}__{model.name}"
224
+ key = qualified.lower() if project.mode == "sql-dir" else qualified
225
+ while key in taken and key != (
226
+ model.name.lower() if project.mode == "sql-dir" else model.name
227
+ ):
228
+ qualified = f"{qualified}__{model.name}"
229
+ key = qualified.lower() if project.mode == "sql-dir" else qualified
230
+ model.aliases.add(name)
231
+ model.name = qualified
232
+ taken.add(key)
233
+
234
+
235
+ def _collapse_sql_dir_tables(project: Project, group: list[Model]) -> list[Model]:
236
+ """In a plain sql dir the same CREATE TABLE name in several dump files is
237
+ one table-level relation (one dump = one database), not a collision worth
238
+ a path-qualified name. Schema-only definitions merge; models carrying a
239
+ real derivation stay distinct, EXCEPT prior-state derivations: a table
240
+ built by CREATE plus UPDATEs across files (nycdb's add_columns.sql and
241
+ full_text.sql both writing dobjobs, the gap named in PR #28) is one
242
+ relation whose lineage is the union of its statements."""
243
+ tables = [m for m in group if not m.sql.strip()]
244
+ if len(tables) >= 2:
245
+ # prefer the dialect-folded (lowercase) spelling as the surviving name
246
+ keep = next((m for m in tables if m.name == m.name.lower()), tables[0])
247
+ for other in tables:
248
+ if other is keep:
249
+ continue
250
+ for column in other.declared_columns:
251
+ if column not in keep.declared_columns:
252
+ keep.declared_columns.append(column)
253
+ keep.aliases |= other.aliases
254
+ project.models.remove(other)
255
+ group = [keep, *(m for m in group if m.sql.strip())]
256
+ return _merge_self_read_derivations(project, group)
257
+
258
+
259
+ def _qualified_spellings(model: Model) -> set[str]:
260
+ return {a for a in model.aliases if "." in a}
261
+
262
+
263
+ def _merge_self_read_derivations(project: Project, group: list[Model]) -> list[Model]:
264
+ """Fold UPDATE-derived models into the one model that owns the table.
265
+
266
+ Only prior-state (self_read) derivations merge, only when at most one
267
+ creator exists (two creators is a real collision), and only when the
268
+ qualified spellings agree: sales.summary and hr.summary stay two
269
+ relations whatever the bare name says (the round-3 identity rule)."""
270
+ creators = [m for m in group if m.sql.strip() and not m.self_read]
271
+ updates = [m for m in group if m.sql.strip() and m.self_read]
272
+ if not updates or len(creators) > 1:
273
+ return group
274
+ bases = [m for m in group if not m.sql.strip()]
275
+ anchor = creators[0] if creators else (bases[0] if bases else updates[0])
276
+ kept = [anchor, *(m for m in bases if m is not anchor)]
277
+ for update in updates:
278
+ if update is anchor:
279
+ continue
280
+ anchor_q = _qualified_spellings(anchor)
281
+ update_q = _qualified_spellings(update)
282
+ # a bare updater against a schema-qualified creator proves nothing:
283
+ # the update could run under any search_path (review of
284
+ # cycle 6); merging needs both bare or an agreeing qualification
285
+ if (anchor_q or update_q) and not (anchor_q & update_q):
286
+ kept.append(update)
287
+ continue
288
+ if not anchor.sql.strip():
289
+ anchor.sql = update.sql
290
+ else:
291
+ anchor.extra_sqls.append(update.sql)
292
+ anchor.extra_sqls.extend(update.extra_sqls)
293
+ anchor.self_read = True
294
+ anchor.declared_parents |= update.declared_parents
295
+ anchor.aliases |= update.aliases
296
+ anchor.stem_aliases |= update.stem_aliases
297
+ for column in update.declared_columns:
298
+ if column not in anchor.declared_columns:
299
+ anchor.declared_columns.append(column)
300
+ project.models.remove(update)
301
+ return kept
302
+
303
+
304
+ ROLE_STEMS = frozenset({"query", "view", "model", "table", "select", "main", "definition"})
305
+
306
+
307
+ def directory_identity(rel: str) -> tuple[str, set[str]] | None:
308
+ """(name, dotted aliases) for a file whose stem names its role, not the relation.
309
+
310
+ Mozilla's bigquery-etl keeps 887 query.sql and 803 view.sql files at
311
+ sql/<project>/<dataset>/<table>/ and references each relation as
312
+ `project.dataset.table`. Named by stem, every model was "query", the
313
+ collision handler mangled them into path names no reference could match,
314
+ and a 2,096-model repo came out with 0 verified edges. The directory is
315
+ the relation; its two nearest ancestors are the qualified spellings.
316
+ Returns None when the stem carries identity as usual. checks.sql and
317
+ script.sql are deliberately not roles: a check reads the relation, it
318
+ does not define it, and claiming the directory would collide with the
319
+ real definition beside it.
320
+ """
321
+ from pathlib import Path
322
+
323
+ parts = Path(rel).parts
324
+ if len(parts) < 2 or Path(parts[-1]).stem.lower() not in ROLE_STEMS:
325
+ return None
326
+ dirs = parts[:-1]
327
+ aliases = {".".join(dirs[-n:]) for n in (2, 3) if len(dirs) >= n}
328
+ return dirs[-1], aliases