ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/render_shims.py ADDED
@@ -0,0 +1,209 @@
1
+ """Builtin shims for macros whose bodies a repo clone never contains.
2
+
3
+ dbt-core's cross-database builtins (dbt.date_trunc, dbt.concat, ...) ship
4
+ with dbt itself; dbt_utils/fivetran_utils bodies live in dbt_packages/,
5
+ which a raw clone lacks. The generic unknown-macro stub renders NULL, which
6
+ parses but erases every value argument — holdout round 6 lost five dbt_jira
7
+ cases through dbt.date_trunc alone, and dbt_utils.date_spine's NULL parsed
8
+ as a relation literally named null.
9
+
10
+ Only lineage-faithful shapes belong here: each shim either reproduces the
11
+ macro's documented SQL closely enough that the value arguments survive, or
12
+ degrades to a NULL literal (never an empty string — 'select as x' parses
13
+ as a bare column and fabricates an edge). Every entry is pinned by a test
14
+ in tests/test_dbt_builtin_macros.py naming the real-repo miss behind it.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import re
20
+
21
+
22
+ def _sql(value) -> str:
23
+ """Arguments arrive as jinja strings, numbers, or undefined stubs."""
24
+ text = str(value).strip()
25
+ return text if text else "null"
26
+
27
+
28
+ def _part(datepart) -> str:
29
+ """Dateparts arrive quoted or bare; emit one canonical quoted spelling."""
30
+ return str(datepart).strip().strip("'\"").lower() or "day"
31
+
32
+
33
+ def fallback_group_by(n=0, **kwargs):
34
+ # dbt_utils.group_by(n) emits the whole clause; ordinals reference
35
+ # already-listed select items, so lineage is unchanged
36
+ try:
37
+ count = int(kwargs.get("n", n))
38
+ except (TypeError, ValueError):
39
+ return ""
40
+ if count <= 0:
41
+ return ""
42
+ return "group by " + ", ".join(str(i) for i in range(1, count + 1))
43
+
44
+
45
+ def fallback_empty(*args, **kwargs) -> str:
46
+ return ""
47
+
48
+
49
+ def fallback_listagg(
50
+ measure="", delimiter_text="','", order_by_clause="", limit_num=None, **kwargs
51
+ ):
52
+ # the measure expression carries the lineage; the ordering clause is
53
+ # structural (same rule as a window ORDER BY). Cost holdout round 3
54
+ # two dbt_hubspot cases.
55
+ if not measure:
56
+ return "null"
57
+ return f"listagg({measure}, {delimiter_text})"
58
+
59
+
60
+ def fallback_surrogate_key(field_list=None, **kwargs):
61
+ # the fields carry the lineage, the hash contributes nothing beyond
62
+ # them. Cost holdout round 4 all ten danish_democracy_data misses.
63
+ if isinstance(field_list, str):
64
+ field_list = [field_list]
65
+ parts = [
66
+ f"coalesce(cast({field} as varchar), '_')"
67
+ for field in (field_list or [])
68
+ if isinstance(field, str) and field.strip()
69
+ ]
70
+ if not parts:
71
+ return "null"
72
+ return "md5(" + " || '-' || ".join(parts) + ")"
73
+
74
+
75
+ def fallback_date_trunc(datepart="day", date="", **kwargs):
76
+ # dbt_jira lost updated_at_week / open_until through the NULL stub
77
+ # (holdout round 6, five cases)
78
+ return f"date_trunc('{_part(datepart)}', {_sql(date)})"
79
+
80
+
81
+ def fallback_dateadd(datepart="day", interval=0, from_date_or_timestamp="", **kwargs):
82
+ # the datepart is quoted: bare `day` parses as a COLUMN in
83
+ # bigquery/postgres/duckdb and fabricated t.day sources (review
84
+ # of cycle 7)
85
+ return f"dateadd('{_part(datepart)}', {_sql(interval)}, {_sql(from_date_or_timestamp)})"
86
+
87
+
88
+ def fallback_datediff(first_date="", second_date="", datepart="day", **kwargs):
89
+ return f"datediff('{_part(datepart)}', {_sql(first_date)}, {_sql(second_date)})"
90
+
91
+
92
+ def fallback_current_timestamp(**kwargs) -> str:
93
+ return "current_timestamp"
94
+
95
+
96
+ def fallback_concat(fields=None, **kwargs):
97
+ if isinstance(fields, str):
98
+ fields = [fields]
99
+ parts = [f for f in (fields or []) if isinstance(f, str) and f.strip()]
100
+ return " || ".join(parts) if parts else "null"
101
+
102
+
103
+ def fallback_safe_cast(field="", type="varchar", **kwargs): # noqa: A002 - dbt's arg name
104
+ return f"cast({_sql(field)} as {_sql(type)})"
105
+
106
+
107
+ def fallback_split_part(string_text="", delimiter_text="','", part_number=1, **kwargs):
108
+ return f"split_part({_sql(string_text)}, {_sql(delimiter_text)}, {_sql(part_number)})"
109
+
110
+
111
+ def fallback_hash(field="", **kwargs):
112
+ return f"md5(cast({_sql(field)} as varchar))"
113
+
114
+
115
+ def fallback_string_agg(field_to_agg="", delimiter="','", **kwargs):
116
+ # fivetran_utils.string_agg's NULL stub severed dbt_jira's whole
117
+ # multiselect field_value chain (holdout round 6)
118
+ if not field_to_agg:
119
+ return "null"
120
+ return f"string_agg({_sql(field_to_agg)}, {_sql(delimiter)})"
121
+
122
+
123
+ def fallback_pass_through(
124
+ pass_through_variable=None, identifier=None, transform="", _var=None, **kwargs
125
+ ) -> str:
126
+ """fivetran_utils persist/fill_pass_through_columns: renders the
127
+ project-configured pass-through columns as ', ident.col as alias'
128
+ entries, nothing when unconfigured. Entries are strings or
129
+ {name, alias, transform_sql} dicts per the fivetran contract; a
130
+ transform_sql's own expression is opaque, so the column itself is the
131
+ honest value source."""
132
+ entries = _var(pass_through_variable, []) if (_var and pass_through_variable) else []
133
+ parts: list[str] = []
134
+ for entry in entries if isinstance(entries, list) else []:
135
+ if isinstance(entry, str):
136
+ name, alias = entry, None
137
+ elif isinstance(entry, dict):
138
+ name, alias = entry.get("name"), entry.get("alias")
139
+ else:
140
+ continue
141
+ if not name:
142
+ continue
143
+ expr = f"{identifier}.{name}" if identifier else str(name)
144
+ parts.append(f", {expr} as {alias or name}")
145
+ return "".join(parts)
146
+
147
+
148
+ fallback_pass_through.wants_var = True
149
+
150
+
151
+ def fallback_slugify(string="", **kwargs) -> str:
152
+ # runs at jinja level (feeds alias names, not SQL); the NULL stub made
153
+ # dbt_jira's pivot emit a column literally named null
154
+ slug = re.sub(r"[^a-z0-9_]+", "_", str(string).lower())
155
+ return f"_{slug}" if slug[:1].isdigit() else slug
156
+
157
+
158
+ def fallback_date_spine(datepart="day", start_date="", end_date="", **kwargs) -> str:
159
+ # the spine is generated data: its column has no upstream, and the
160
+ # bounds are literals or clock functions. Rendering NULL inside
161
+ # from (...) parsed as a relation named null (holdout round 6,
162
+ # dbt_jira issue_day_id). The real macro names the column
163
+ # date_{datepart} (the cycle-7 review: 'hour' spines exist).
164
+ part = _part(datepart)
165
+ kind = "date" if part in ("day", "week", "month", "quarter", "year") else "timestamp"
166
+ return f"select cast(null as {kind}) as date_{part}"
167
+
168
+
169
+ def _type(name: str):
170
+ return lambda **kwargs: name
171
+
172
+
173
+ # Only lineage-safe shapes belong here; the NULL stub covers everything else.
174
+ BUILTIN_PACKAGE_MACROS = {
175
+ ("dbt_utils", "group_by"): fallback_group_by,
176
+ # emits ", source_relation" only in multi-source unions; an empty
177
+ # partition arg keeps OVER (PARTITION BY x) parseable and feeds no column
178
+ ("fivetran_utils", "partition_by_source_relation"): fallback_empty,
179
+ # emits ", col, ..." only when the pass-through var is configured;
180
+ # unconfigured (the offline default) it emits nothing. The NULL stub
181
+ # rendered comma-less NULLs after the last select item and killed
182
+ # whole dbt_salesforce models (holdout round 7). When the project DOES
183
+ # configure the var, the columns render (the cycle-8 review:
184
+ # returning empty erased configured lineage)
185
+ ("fivetran_utils", "persist_pass_through_columns"): fallback_pass_through,
186
+ ("fivetran_utils", "fill_pass_through_columns"): fallback_pass_through,
187
+ ("dbt", "listagg"): fallback_listagg,
188
+ ("dbt_utils", "generate_surrogate_key"): fallback_surrogate_key,
189
+ # the pre-1.0 spelling of the same macro
190
+ ("dbt_utils", "surrogate_key"): fallback_surrogate_key,
191
+ ("dbt", "date_trunc"): fallback_date_trunc,
192
+ ("dbt", "dateadd"): fallback_dateadd,
193
+ ("dbt", "datediff"): fallback_datediff,
194
+ ("dbt", "current_timestamp"): fallback_current_timestamp,
195
+ ("dbt", "concat"): fallback_concat,
196
+ ("dbt", "safe_cast"): fallback_safe_cast,
197
+ ("dbt", "split_part"): fallback_split_part,
198
+ ("dbt", "hash"): fallback_hash,
199
+ ("dbt", "type_string"): _type("varchar"),
200
+ ("dbt", "type_timestamp"): _type("timestamp"),
201
+ ("dbt", "type_int"): _type("integer"),
202
+ ("dbt", "type_bigint"): _type("bigint"),
203
+ ("dbt", "type_float"): _type("float"),
204
+ ("dbt", "type_numeric"): _type("numeric(28,6)"),
205
+ ("dbt", "type_boolean"): _type("boolean"),
206
+ ("fivetran_utils", "string_agg"): fallback_string_agg,
207
+ ("dbt_utils", "slugify"): fallback_slugify,
208
+ ("dbt_utils", "date_spine"): fallback_date_spine,
209
+ }
ripple/schemas.py ADDED
@@ -0,0 +1,155 @@
1
+ """Agent-ingested warehouse schemas.
2
+
3
+ Ripple never connects to a warehouse. When a project reads tables it does
4
+ not define, the user's own agent (or the user, via the CLI) fetches those
5
+ tables' columns from information_schema and hands them over. They persist
6
+ in .ripple/schemas.json at the project root: committable, sorted keys,
7
+ stable order, so the file diffs cleanly and can be shared with a team.
8
+
9
+ Format:
10
+ {"tables": {"<name>": {"columns": ["a", "b"], "origin": "ingested"}}}
11
+
12
+ Names may be bare or schema/database qualified; matching is case
13
+ insensitive. project.py exposes loaded tables as known external relations.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import csv
19
+ import io
20
+ import json
21
+ from pathlib import Path
22
+
23
+
24
+ def schemas_path(root: str | Path) -> Path:
25
+ return Path(root) / ".ripple" / "schemas.json"
26
+
27
+
28
+ def load_schemas(root: str | Path) -> dict[str, list[str]]:
29
+ """Table name (lowercased) -> column list (lowercased). {} when the file
30
+ is absent or unreadable; a broken file must never break the project load."""
31
+ path = schemas_path(root)
32
+ if not path.is_file():
33
+ return {}
34
+ try:
35
+ doc = json.loads(path.read_text(errors="replace", encoding="utf-8"))
36
+ except (json.JSONDecodeError, OSError):
37
+ return {}
38
+ try:
39
+ tables = normalize_tables(doc)
40
+ except ValueError:
41
+ return {}
42
+ return {name.lower(): [c.lower() for c in cols] for name, cols in tables.items()}
43
+
44
+
45
+ def normalize_tables(payload: dict) -> dict[str, list[str]]:
46
+ """Accept the shapes agents and files actually produce:
47
+
48
+ - {"tables": {name: {"columns": [...]}}} the canonical file format
49
+ - {"tables": {name: [...]}} shorthand
50
+ - {name: [...]} or {name: {"columns": [...]}} bare mapping
51
+ - {"rows": [{"table": ..., "column": ...}]} tabular query results
52
+ """
53
+ if not isinstance(payload, dict):
54
+ raise ValueError("expected a JSON object")
55
+ if isinstance(payload.get("rows"), list):
56
+ tables: dict[str, list[str]] = {}
57
+ for row in payload["rows"]:
58
+ if not isinstance(row, dict):
59
+ continue
60
+ table = row.get("table") or row.get("table_name")
61
+ column = row.get("column") or row.get("column_name")
62
+ if not table or not column:
63
+ continue
64
+ bucket = tables.setdefault(str(table), [])
65
+ if str(column) not in bucket:
66
+ bucket.append(str(column))
67
+ if not tables:
68
+ raise ValueError("no usable rows: each row needs 'table' and 'column'")
69
+ return tables
70
+ mapping = payload.get("tables") if isinstance(payload.get("tables"), dict) else payload
71
+ tables = {}
72
+ for name, entry in mapping.items():
73
+ columns = entry.get("columns") if isinstance(entry, dict) else entry
74
+ if not isinstance(columns, list) or not all(isinstance(c, str) for c in columns):
75
+ raise ValueError(f"table '{name}': expected a list of column names")
76
+ deduped: list[str] = []
77
+ for column in columns:
78
+ if column not in deduped:
79
+ deduped.append(column)
80
+ tables[str(name)] = deduped
81
+ if not tables:
82
+ raise ValueError("no tables found in payload")
83
+ return tables
84
+
85
+
86
+ def parse_csv(text: str) -> dict[str, list[str]]:
87
+ """CSV with a table,column header (information_schema exports)."""
88
+ reader = csv.DictReader(io.StringIO(text))
89
+ fields = {(f or "").strip().lower(): f for f in (reader.fieldnames or [])}
90
+ table_key = fields.get("table") or fields.get("table_name")
91
+ column_key = fields.get("column") or fields.get("column_name")
92
+ if not table_key or not column_key:
93
+ raise ValueError("CSV needs 'table' and 'column' header fields")
94
+ tables: dict[str, list[str]] = {}
95
+ for row in reader:
96
+ table = (row.get(table_key) or "").strip()
97
+ column = (row.get(column_key) or "").strip()
98
+ if not table or not column:
99
+ continue
100
+ bucket = tables.setdefault(table, [])
101
+ if column not in bucket:
102
+ bucket.append(column)
103
+ return tables
104
+
105
+
106
+ def merge_schemas(root: str | Path, tables: dict[str, list[str]]) -> dict:
107
+ """Merge new tables into .ripple/schemas.json and report what changed.
108
+
109
+ Matching is case insensitive; the file keeps lowercase names so the same
110
+ table ingested twice with different casing stays one entry. New columns
111
+ append after existing ones, so a re-ingest diffs as pure additions.
112
+ """
113
+ path = schemas_path(root)
114
+ existing: dict[str, dict] = {}
115
+ if path.is_file():
116
+ try:
117
+ doc = json.loads(path.read_text(errors="replace", encoding="utf-8"))
118
+ except (json.JSONDecodeError, OSError):
119
+ doc = {}
120
+ raw = doc.get("tables") if isinstance(doc, dict) else None
121
+ if isinstance(raw, dict):
122
+ for name, entry in raw.items():
123
+ columns = entry.get("columns") if isinstance(entry, dict) else entry
124
+ if isinstance(columns, list):
125
+ existing[name.lower()] = {
126
+ "columns": [str(c).lower() for c in columns],
127
+ "origin": (entry.get("origin") if isinstance(entry, dict) else None)
128
+ or "ingested",
129
+ }
130
+
131
+ added: list[str] = []
132
+ updated: list[str] = []
133
+ columns_added = 0
134
+ for name, columns in tables.items():
135
+ key = name.lower()
136
+ entry = existing.get(key)
137
+ if entry is None:
138
+ entry = {"columns": [], "origin": "ingested"}
139
+ existing[key] = entry
140
+ added.append(key)
141
+ new = [c.lower() for c in columns if c.lower() not in entry["columns"]]
142
+ if new and key not in added:
143
+ updated.append(key)
144
+ entry["columns"].extend(new)
145
+ columns_added += len(new)
146
+
147
+ path.parent.mkdir(parents=True, exist_ok=True)
148
+ ordered = {name: existing[name] for name in sorted(existing)}
149
+ path.write_text(json.dumps({"tables": ordered}, indent=2) + "\n", encoding="utf-8")
150
+ return {
151
+ "tables_added": sorted(added),
152
+ "tables_updated": sorted(updated),
153
+ "columns_added": columns_added,
154
+ "path": str(path),
155
+ }
ripple/semantic.py ADDED
@@ -0,0 +1,232 @@
1
+ """Downstream reach into the dbt semantic layer, metrics, and exposures.
2
+
3
+ The BI layer, where it is declared as code, is just more edges. A dbt metric is a
4
+ node that derives from a measure, which is a column of a model Ripple already
5
+ mapped. A dbt exposure is a node-level leaf hanging off the models it reads. So a
6
+ column rename can be traced to the metric and the dashboard it breaks, offline,
7
+ with no warehouse connection.
8
+
9
+ This reads the dbt schema YAML directly (the same files a raw dbt project ships),
10
+ so it works without a compiled manifest. Nothing here fabricates a mapping: a
11
+ metric whose measure we cannot resolve to a real column is dropped, not guessed.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ from dataclasses import dataclass
18
+ from pathlib import Path
19
+
20
+ import yaml
21
+
22
+ from ripple.project import IGNORE_DIRS
23
+
24
+ _REF = re.compile(r"ref\(\s*['\"]([^'\"]+)['\"]\s*(?:,\s*['\"]([^'\"]+)['\"])?\s*\)")
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class DownstreamEdge:
29
+ """A model column (or the model itself) feeding a BI node declared in code."""
30
+
31
+ src_model: str
32
+ src_column: str # "*" means node-level (the whole model)
33
+ dst_model: str # "metric:x" | "exposure:y" | "semantic:orders.customer"
34
+ dst_column: str
35
+ trust: str
36
+ reason: str = ""
37
+
38
+
39
+ def _ref_model(value) -> str | None:
40
+ """The model name a `model: ref('orders')` points at."""
41
+ if not isinstance(value, str):
42
+ return None
43
+ m = _REF.search(value)
44
+ if m:
45
+ return m.group(2) or m.group(1)
46
+ return value.strip() or None
47
+
48
+
49
+ def _iter_schema_docs(root: Path):
50
+ for path in sorted(root.rglob("*.yml")) + sorted(root.rglob("*.yaml")):
51
+ parts = path.relative_to(root).parts[:-1]
52
+ if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
53
+ continue
54
+ try:
55
+ doc = yaml.safe_load(path.read_text(errors="replace", encoding="utf-8"))
56
+ except (yaml.YAMLError, OSError):
57
+ continue
58
+ if not isinstance(doc, dict):
59
+ continue
60
+ if "semantic_models" in doc or "metrics" in doc or "exposures" in doc:
61
+ yield doc
62
+ elif any(
63
+ isinstance(m, dict) and ("semantic_model" in m or "metrics" in m)
64
+ for m in (doc.get("models") or [])
65
+ if m
66
+ ):
67
+ # dbt's newer inline syntax: semantic_model nested under a model
68
+ yield doc
69
+
70
+
71
+ def _measure_name(spec) -> str | None:
72
+ if isinstance(spec, str):
73
+ return spec
74
+ if isinstance(spec, dict):
75
+ return spec.get("name")
76
+ return None
77
+
78
+
79
+ def load_downstream(root: Path, known_models: set[str]) -> list[DownstreamEdge]:
80
+ """Every BI edge we can resolve from the project's declared-as-code layer."""
81
+ known = {m.lower() for m in known_models}
82
+ # global measure name -> (model, column). Measure names are unique per project.
83
+ measures: dict[str, tuple[str, str]] = {}
84
+ edges: list[DownstreamEdge] = []
85
+
86
+ docs = list(_iter_schema_docs(root))
87
+
88
+ # pass 0: dbt's inline syntax (jaffle-shop migrated to it): the semantic
89
+ # model is declared under the model entry itself, dimensions/entities on
90
+ # its columns, metrics on the model with the column as name or expr
91
+ for doc in docs:
92
+ for entry in doc.get("models") or []:
93
+ if not isinstance(entry, dict):
94
+ continue
95
+ sm = entry.get("semantic_model")
96
+ has_inline = isinstance(sm, dict) or entry.get("metrics")
97
+ if not has_inline or (isinstance(sm, dict) and sm.get("enabled") is False):
98
+ continue
99
+ model = entry.get("name")
100
+ if not model or model.lower() not in known:
101
+ continue
102
+ for col in entry.get("columns") or []:
103
+ if not (isinstance(col, dict) and col.get("name")):
104
+ continue
105
+ cname = col["name"]
106
+ for kind in ("dimension", "entity"):
107
+ spec = col.get(kind)
108
+ if not isinstance(spec, dict):
109
+ continue
110
+ dname = spec.get("name") or cname
111
+ edges.append(
112
+ DownstreamEdge(
113
+ src_model=model,
114
+ src_column=cname,
115
+ dst_model=f"semantic:{model}.{dname}",
116
+ dst_column=dname,
117
+ trust="verified",
118
+ )
119
+ )
120
+ measure = col.get("measure")
121
+ if isinstance(measure, dict) and measure.get("name"):
122
+ measures[measure["name"]] = (model, cname)
123
+ for metric in entry.get("metrics") or []:
124
+ if not (isinstance(metric, dict) and metric.get("name")):
125
+ continue
126
+ expr = metric.get("expr", metric["name"])
127
+ if isinstance(expr, str) and expr.isidentifier():
128
+ edges.append(
129
+ DownstreamEdge(
130
+ src_model=model,
131
+ src_column=expr,
132
+ dst_model=f"metric:{metric['name']}",
133
+ dst_column="value",
134
+ trust="verified",
135
+ )
136
+ )
137
+ # a constant or computed expr has no single source column:
138
+ # dropped, never guessed
139
+
140
+ # pass 1: semantic models give measures/dimensions/entities their columns
141
+ for doc in docs:
142
+ for sm in doc.get("semantic_models") or []:
143
+ if not isinstance(sm, dict):
144
+ continue
145
+ model = _ref_model(sm.get("model"))
146
+ if not model or model.lower() not in known:
147
+ continue
148
+ sm_name = sm.get("name", model)
149
+ for m in sm.get("measures") or []:
150
+ if isinstance(m, dict) and m.get("name"):
151
+ col = m.get("expr") or m["name"]
152
+ if isinstance(col, str) and col.isidentifier():
153
+ measures[m["name"]] = (model, col)
154
+ for kind in ("dimensions", "entities"):
155
+ for d in sm.get(kind) or []:
156
+ if not (isinstance(d, dict) and d.get("name")):
157
+ continue
158
+ col = d.get("expr") or d["name"]
159
+ if not (isinstance(col, str) and col.isidentifier()):
160
+ continue
161
+ edges.append(
162
+ DownstreamEdge(
163
+ src_model=model,
164
+ src_column=col,
165
+ dst_model=f"semantic:{sm_name}.{d['name']}",
166
+ dst_column=d["name"],
167
+ trust="verified",
168
+ )
169
+ )
170
+
171
+ # pass 2: metrics resolve through their measure(s) to the underlying column
172
+ for doc in docs:
173
+ for metric in doc.get("metrics") or []:
174
+ if not (isinstance(metric, dict) and metric.get("name")):
175
+ continue
176
+ params = metric.get("type_params") or {}
177
+ names: list[str] = []
178
+ for key in ("measure", "numerator", "denominator"):
179
+ n = _measure_name(params.get(key))
180
+ if n:
181
+ names.append(n)
182
+ for m in params.get("measures") or []:
183
+ n = _measure_name(m)
184
+ if n:
185
+ names.append(n)
186
+ for n in names:
187
+ loc = measures.get(n)
188
+ if not loc:
189
+ continue # unresolved measure: drop, never fake
190
+ edges.append(
191
+ DownstreamEdge(
192
+ src_model=loc[0],
193
+ src_column=loc[1],
194
+ dst_model=f"metric:{metric['name']}",
195
+ dst_column="value",
196
+ trust="verified",
197
+ )
198
+ )
199
+
200
+ # pass 3: exposures are node-level (a dashboard reads the whole model)
201
+ for doc in docs:
202
+ for exp in doc.get("exposures") or []:
203
+ if not (isinstance(exp, dict) and exp.get("name")):
204
+ continue
205
+ deps = exp.get("depends_on") or exp.get("refs") or []
206
+ if isinstance(deps, str):
207
+ deps = [deps]
208
+ for dep in deps:
209
+ model = (
210
+ _ref_model(dep)
211
+ if isinstance(dep, str)
212
+ else _ref_model((dep or {}).get("name") if isinstance(dep, dict) else None)
213
+ )
214
+ if model and model.lower() in known:
215
+ edges.append(
216
+ DownstreamEdge(
217
+ src_model=model,
218
+ src_column="*",
219
+ dst_model=f"exposure:{exp['name']}",
220
+ dst_column="*",
221
+ trust="review_required",
222
+ reason=f"declared {exp.get('type', 'exposure')}, node-level",
223
+ )
224
+ )
225
+
226
+ # dedupe
227
+ seen, out = set(), []
228
+ for e in edges:
229
+ if e not in seen:
230
+ seen.add(e)
231
+ out.append(e)
232
+ return out