ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/render_shims.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Builtin shims for macros whose bodies a repo clone never contains.
|
|
2
|
+
|
|
3
|
+
dbt-core's cross-database builtins (dbt.date_trunc, dbt.concat, ...) ship
|
|
4
|
+
with dbt itself; dbt_utils/fivetran_utils bodies live in dbt_packages/,
|
|
5
|
+
which a raw clone lacks. The generic unknown-macro stub renders NULL, which
|
|
6
|
+
parses but erases every value argument — holdout round 6 lost five dbt_jira
|
|
7
|
+
cases through dbt.date_trunc alone, and dbt_utils.date_spine's NULL parsed
|
|
8
|
+
as a relation literally named null.
|
|
9
|
+
|
|
10
|
+
Only lineage-faithful shapes belong here: each shim either reproduces the
|
|
11
|
+
macro's documented SQL closely enough that the value arguments survive, or
|
|
12
|
+
degrades to a NULL literal (never an empty string — 'select as x' parses
|
|
13
|
+
as a bare column and fabricates an edge). Every entry is pinned by a test
|
|
14
|
+
in tests/test_dbt_builtin_macros.py naming the real-repo miss behind it.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _sql(value) -> str:
|
|
23
|
+
"""Arguments arrive as jinja strings, numbers, or undefined stubs."""
|
|
24
|
+
text = str(value).strip()
|
|
25
|
+
return text if text else "null"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _part(datepart) -> str:
|
|
29
|
+
"""Dateparts arrive quoted or bare; emit one canonical quoted spelling."""
|
|
30
|
+
return str(datepart).strip().strip("'\"").lower() or "day"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def fallback_group_by(n=0, **kwargs):
|
|
34
|
+
# dbt_utils.group_by(n) emits the whole clause; ordinals reference
|
|
35
|
+
# already-listed select items, so lineage is unchanged
|
|
36
|
+
try:
|
|
37
|
+
count = int(kwargs.get("n", n))
|
|
38
|
+
except (TypeError, ValueError):
|
|
39
|
+
return ""
|
|
40
|
+
if count <= 0:
|
|
41
|
+
return ""
|
|
42
|
+
return "group by " + ", ".join(str(i) for i in range(1, count + 1))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def fallback_empty(*args, **kwargs) -> str:
|
|
46
|
+
return ""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def fallback_listagg(
|
|
50
|
+
measure="", delimiter_text="','", order_by_clause="", limit_num=None, **kwargs
|
|
51
|
+
):
|
|
52
|
+
# the measure expression carries the lineage; the ordering clause is
|
|
53
|
+
# structural (same rule as a window ORDER BY). Cost holdout round 3
|
|
54
|
+
# two dbt_hubspot cases.
|
|
55
|
+
if not measure:
|
|
56
|
+
return "null"
|
|
57
|
+
return f"listagg({measure}, {delimiter_text})"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def fallback_surrogate_key(field_list=None, **kwargs):
|
|
61
|
+
# the fields carry the lineage, the hash contributes nothing beyond
|
|
62
|
+
# them. Cost holdout round 4 all ten danish_democracy_data misses.
|
|
63
|
+
if isinstance(field_list, str):
|
|
64
|
+
field_list = [field_list]
|
|
65
|
+
parts = [
|
|
66
|
+
f"coalesce(cast({field} as varchar), '_')"
|
|
67
|
+
for field in (field_list or [])
|
|
68
|
+
if isinstance(field, str) and field.strip()
|
|
69
|
+
]
|
|
70
|
+
if not parts:
|
|
71
|
+
return "null"
|
|
72
|
+
return "md5(" + " || '-' || ".join(parts) + ")"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def fallback_date_trunc(datepart="day", date="", **kwargs):
|
|
76
|
+
# dbt_jira lost updated_at_week / open_until through the NULL stub
|
|
77
|
+
# (holdout round 6, five cases)
|
|
78
|
+
return f"date_trunc('{_part(datepart)}', {_sql(date)})"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def fallback_dateadd(datepart="day", interval=0, from_date_or_timestamp="", **kwargs):
|
|
82
|
+
# the datepart is quoted: bare `day` parses as a COLUMN in
|
|
83
|
+
# bigquery/postgres/duckdb and fabricated t.day sources (review
|
|
84
|
+
# of cycle 7)
|
|
85
|
+
return f"dateadd('{_part(datepart)}', {_sql(interval)}, {_sql(from_date_or_timestamp)})"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def fallback_datediff(first_date="", second_date="", datepart="day", **kwargs):
|
|
89
|
+
return f"datediff('{_part(datepart)}', {_sql(first_date)}, {_sql(second_date)})"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def fallback_current_timestamp(**kwargs) -> str:
|
|
93
|
+
return "current_timestamp"
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def fallback_concat(fields=None, **kwargs):
|
|
97
|
+
if isinstance(fields, str):
|
|
98
|
+
fields = [fields]
|
|
99
|
+
parts = [f for f in (fields or []) if isinstance(f, str) and f.strip()]
|
|
100
|
+
return " || ".join(parts) if parts else "null"
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def fallback_safe_cast(field="", type="varchar", **kwargs): # noqa: A002 - dbt's arg name
|
|
104
|
+
return f"cast({_sql(field)} as {_sql(type)})"
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def fallback_split_part(string_text="", delimiter_text="','", part_number=1, **kwargs):
|
|
108
|
+
return f"split_part({_sql(string_text)}, {_sql(delimiter_text)}, {_sql(part_number)})"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def fallback_hash(field="", **kwargs):
|
|
112
|
+
return f"md5(cast({_sql(field)} as varchar))"
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def fallback_string_agg(field_to_agg="", delimiter="','", **kwargs):
|
|
116
|
+
# fivetran_utils.string_agg's NULL stub severed dbt_jira's whole
|
|
117
|
+
# multiselect field_value chain (holdout round 6)
|
|
118
|
+
if not field_to_agg:
|
|
119
|
+
return "null"
|
|
120
|
+
return f"string_agg({_sql(field_to_agg)}, {_sql(delimiter)})"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def fallback_pass_through(
|
|
124
|
+
pass_through_variable=None, identifier=None, transform="", _var=None, **kwargs
|
|
125
|
+
) -> str:
|
|
126
|
+
"""fivetran_utils persist/fill_pass_through_columns: renders the
|
|
127
|
+
project-configured pass-through columns as ', ident.col as alias'
|
|
128
|
+
entries, nothing when unconfigured. Entries are strings or
|
|
129
|
+
{name, alias, transform_sql} dicts per the fivetran contract; a
|
|
130
|
+
transform_sql's own expression is opaque, so the column itself is the
|
|
131
|
+
honest value source."""
|
|
132
|
+
entries = _var(pass_through_variable, []) if (_var and pass_through_variable) else []
|
|
133
|
+
parts: list[str] = []
|
|
134
|
+
for entry in entries if isinstance(entries, list) else []:
|
|
135
|
+
if isinstance(entry, str):
|
|
136
|
+
name, alias = entry, None
|
|
137
|
+
elif isinstance(entry, dict):
|
|
138
|
+
name, alias = entry.get("name"), entry.get("alias")
|
|
139
|
+
else:
|
|
140
|
+
continue
|
|
141
|
+
if not name:
|
|
142
|
+
continue
|
|
143
|
+
expr = f"{identifier}.{name}" if identifier else str(name)
|
|
144
|
+
parts.append(f", {expr} as {alias or name}")
|
|
145
|
+
return "".join(parts)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
fallback_pass_through.wants_var = True
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def fallback_slugify(string="", **kwargs) -> str:
|
|
152
|
+
# runs at jinja level (feeds alias names, not SQL); the NULL stub made
|
|
153
|
+
# dbt_jira's pivot emit a column literally named null
|
|
154
|
+
slug = re.sub(r"[^a-z0-9_]+", "_", str(string).lower())
|
|
155
|
+
return f"_{slug}" if slug[:1].isdigit() else slug
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def fallback_date_spine(datepart="day", start_date="", end_date="", **kwargs) -> str:
|
|
159
|
+
# the spine is generated data: its column has no upstream, and the
|
|
160
|
+
# bounds are literals or clock functions. Rendering NULL inside
|
|
161
|
+
# from (...) parsed as a relation named null (holdout round 6,
|
|
162
|
+
# dbt_jira issue_day_id). The real macro names the column
|
|
163
|
+
# date_{datepart} (the cycle-7 review: 'hour' spines exist).
|
|
164
|
+
part = _part(datepart)
|
|
165
|
+
kind = "date" if part in ("day", "week", "month", "quarter", "year") else "timestamp"
|
|
166
|
+
return f"select cast(null as {kind}) as date_{part}"
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _type(name: str):
|
|
170
|
+
return lambda **kwargs: name
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
# Only lineage-safe shapes belong here; the NULL stub covers everything else.
|
|
174
|
+
BUILTIN_PACKAGE_MACROS = {
|
|
175
|
+
("dbt_utils", "group_by"): fallback_group_by,
|
|
176
|
+
# emits ", source_relation" only in multi-source unions; an empty
|
|
177
|
+
# partition arg keeps OVER (PARTITION BY x) parseable and feeds no column
|
|
178
|
+
("fivetran_utils", "partition_by_source_relation"): fallback_empty,
|
|
179
|
+
# emits ", col, ..." only when the pass-through var is configured;
|
|
180
|
+
# unconfigured (the offline default) it emits nothing. The NULL stub
|
|
181
|
+
# rendered comma-less NULLs after the last select item and killed
|
|
182
|
+
# whole dbt_salesforce models (holdout round 7). When the project DOES
|
|
183
|
+
# configure the var, the columns render (the cycle-8 review:
|
|
184
|
+
# returning empty erased configured lineage)
|
|
185
|
+
("fivetran_utils", "persist_pass_through_columns"): fallback_pass_through,
|
|
186
|
+
("fivetran_utils", "fill_pass_through_columns"): fallback_pass_through,
|
|
187
|
+
("dbt", "listagg"): fallback_listagg,
|
|
188
|
+
("dbt_utils", "generate_surrogate_key"): fallback_surrogate_key,
|
|
189
|
+
# the pre-1.0 spelling of the same macro
|
|
190
|
+
("dbt_utils", "surrogate_key"): fallback_surrogate_key,
|
|
191
|
+
("dbt", "date_trunc"): fallback_date_trunc,
|
|
192
|
+
("dbt", "dateadd"): fallback_dateadd,
|
|
193
|
+
("dbt", "datediff"): fallback_datediff,
|
|
194
|
+
("dbt", "current_timestamp"): fallback_current_timestamp,
|
|
195
|
+
("dbt", "concat"): fallback_concat,
|
|
196
|
+
("dbt", "safe_cast"): fallback_safe_cast,
|
|
197
|
+
("dbt", "split_part"): fallback_split_part,
|
|
198
|
+
("dbt", "hash"): fallback_hash,
|
|
199
|
+
("dbt", "type_string"): _type("varchar"),
|
|
200
|
+
("dbt", "type_timestamp"): _type("timestamp"),
|
|
201
|
+
("dbt", "type_int"): _type("integer"),
|
|
202
|
+
("dbt", "type_bigint"): _type("bigint"),
|
|
203
|
+
("dbt", "type_float"): _type("float"),
|
|
204
|
+
("dbt", "type_numeric"): _type("numeric(28,6)"),
|
|
205
|
+
("dbt", "type_boolean"): _type("boolean"),
|
|
206
|
+
("fivetran_utils", "string_agg"): fallback_string_agg,
|
|
207
|
+
("dbt_utils", "slugify"): fallback_slugify,
|
|
208
|
+
("dbt_utils", "date_spine"): fallback_date_spine,
|
|
209
|
+
}
|
ripple/schemas.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""Agent-ingested warehouse schemas.
|
|
2
|
+
|
|
3
|
+
Ripple never connects to a warehouse. When a project reads tables it does
|
|
4
|
+
not define, the user's own agent (or the user, via the CLI) fetches those
|
|
5
|
+
tables' columns from information_schema and hands them over. They persist
|
|
6
|
+
in .ripple/schemas.json at the project root: committable, sorted keys,
|
|
7
|
+
stable order, so the file diffs cleanly and can be shared with a team.
|
|
8
|
+
|
|
9
|
+
Format:
|
|
10
|
+
{"tables": {"<name>": {"columns": ["a", "b"], "origin": "ingested"}}}
|
|
11
|
+
|
|
12
|
+
Names may be bare or schema/database qualified; matching is case
|
|
13
|
+
insensitive. project.py exposes loaded tables as known external relations.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import csv
|
|
19
|
+
import io
|
|
20
|
+
import json
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def schemas_path(root: str | Path) -> Path:
|
|
25
|
+
return Path(root) / ".ripple" / "schemas.json"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def load_schemas(root: str | Path) -> dict[str, list[str]]:
|
|
29
|
+
"""Table name (lowercased) -> column list (lowercased). {} when the file
|
|
30
|
+
is absent or unreadable; a broken file must never break the project load."""
|
|
31
|
+
path = schemas_path(root)
|
|
32
|
+
if not path.is_file():
|
|
33
|
+
return {}
|
|
34
|
+
try:
|
|
35
|
+
doc = json.loads(path.read_text(errors="replace", encoding="utf-8"))
|
|
36
|
+
except (json.JSONDecodeError, OSError):
|
|
37
|
+
return {}
|
|
38
|
+
try:
|
|
39
|
+
tables = normalize_tables(doc)
|
|
40
|
+
except ValueError:
|
|
41
|
+
return {}
|
|
42
|
+
return {name.lower(): [c.lower() for c in cols] for name, cols in tables.items()}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def normalize_tables(payload: dict) -> dict[str, list[str]]:
|
|
46
|
+
"""Accept the shapes agents and files actually produce:
|
|
47
|
+
|
|
48
|
+
- {"tables": {name: {"columns": [...]}}} the canonical file format
|
|
49
|
+
- {"tables": {name: [...]}} shorthand
|
|
50
|
+
- {name: [...]} or {name: {"columns": [...]}} bare mapping
|
|
51
|
+
- {"rows": [{"table": ..., "column": ...}]} tabular query results
|
|
52
|
+
"""
|
|
53
|
+
if not isinstance(payload, dict):
|
|
54
|
+
raise ValueError("expected a JSON object")
|
|
55
|
+
if isinstance(payload.get("rows"), list):
|
|
56
|
+
tables: dict[str, list[str]] = {}
|
|
57
|
+
for row in payload["rows"]:
|
|
58
|
+
if not isinstance(row, dict):
|
|
59
|
+
continue
|
|
60
|
+
table = row.get("table") or row.get("table_name")
|
|
61
|
+
column = row.get("column") or row.get("column_name")
|
|
62
|
+
if not table or not column:
|
|
63
|
+
continue
|
|
64
|
+
bucket = tables.setdefault(str(table), [])
|
|
65
|
+
if str(column) not in bucket:
|
|
66
|
+
bucket.append(str(column))
|
|
67
|
+
if not tables:
|
|
68
|
+
raise ValueError("no usable rows: each row needs 'table' and 'column'")
|
|
69
|
+
return tables
|
|
70
|
+
mapping = payload.get("tables") if isinstance(payload.get("tables"), dict) else payload
|
|
71
|
+
tables = {}
|
|
72
|
+
for name, entry in mapping.items():
|
|
73
|
+
columns = entry.get("columns") if isinstance(entry, dict) else entry
|
|
74
|
+
if not isinstance(columns, list) or not all(isinstance(c, str) for c in columns):
|
|
75
|
+
raise ValueError(f"table '{name}': expected a list of column names")
|
|
76
|
+
deduped: list[str] = []
|
|
77
|
+
for column in columns:
|
|
78
|
+
if column not in deduped:
|
|
79
|
+
deduped.append(column)
|
|
80
|
+
tables[str(name)] = deduped
|
|
81
|
+
if not tables:
|
|
82
|
+
raise ValueError("no tables found in payload")
|
|
83
|
+
return tables
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def parse_csv(text: str) -> dict[str, list[str]]:
|
|
87
|
+
"""CSV with a table,column header (information_schema exports)."""
|
|
88
|
+
reader = csv.DictReader(io.StringIO(text))
|
|
89
|
+
fields = {(f or "").strip().lower(): f for f in (reader.fieldnames or [])}
|
|
90
|
+
table_key = fields.get("table") or fields.get("table_name")
|
|
91
|
+
column_key = fields.get("column") or fields.get("column_name")
|
|
92
|
+
if not table_key or not column_key:
|
|
93
|
+
raise ValueError("CSV needs 'table' and 'column' header fields")
|
|
94
|
+
tables: dict[str, list[str]] = {}
|
|
95
|
+
for row in reader:
|
|
96
|
+
table = (row.get(table_key) or "").strip()
|
|
97
|
+
column = (row.get(column_key) or "").strip()
|
|
98
|
+
if not table or not column:
|
|
99
|
+
continue
|
|
100
|
+
bucket = tables.setdefault(table, [])
|
|
101
|
+
if column not in bucket:
|
|
102
|
+
bucket.append(column)
|
|
103
|
+
return tables
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def merge_schemas(root: str | Path, tables: dict[str, list[str]]) -> dict:
|
|
107
|
+
"""Merge new tables into .ripple/schemas.json and report what changed.
|
|
108
|
+
|
|
109
|
+
Matching is case insensitive; the file keeps lowercase names so the same
|
|
110
|
+
table ingested twice with different casing stays one entry. New columns
|
|
111
|
+
append after existing ones, so a re-ingest diffs as pure additions.
|
|
112
|
+
"""
|
|
113
|
+
path = schemas_path(root)
|
|
114
|
+
existing: dict[str, dict] = {}
|
|
115
|
+
if path.is_file():
|
|
116
|
+
try:
|
|
117
|
+
doc = json.loads(path.read_text(errors="replace", encoding="utf-8"))
|
|
118
|
+
except (json.JSONDecodeError, OSError):
|
|
119
|
+
doc = {}
|
|
120
|
+
raw = doc.get("tables") if isinstance(doc, dict) else None
|
|
121
|
+
if isinstance(raw, dict):
|
|
122
|
+
for name, entry in raw.items():
|
|
123
|
+
columns = entry.get("columns") if isinstance(entry, dict) else entry
|
|
124
|
+
if isinstance(columns, list):
|
|
125
|
+
existing[name.lower()] = {
|
|
126
|
+
"columns": [str(c).lower() for c in columns],
|
|
127
|
+
"origin": (entry.get("origin") if isinstance(entry, dict) else None)
|
|
128
|
+
or "ingested",
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
added: list[str] = []
|
|
132
|
+
updated: list[str] = []
|
|
133
|
+
columns_added = 0
|
|
134
|
+
for name, columns in tables.items():
|
|
135
|
+
key = name.lower()
|
|
136
|
+
entry = existing.get(key)
|
|
137
|
+
if entry is None:
|
|
138
|
+
entry = {"columns": [], "origin": "ingested"}
|
|
139
|
+
existing[key] = entry
|
|
140
|
+
added.append(key)
|
|
141
|
+
new = [c.lower() for c in columns if c.lower() not in entry["columns"]]
|
|
142
|
+
if new and key not in added:
|
|
143
|
+
updated.append(key)
|
|
144
|
+
entry["columns"].extend(new)
|
|
145
|
+
columns_added += len(new)
|
|
146
|
+
|
|
147
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
148
|
+
ordered = {name: existing[name] for name in sorted(existing)}
|
|
149
|
+
path.write_text(json.dumps({"tables": ordered}, indent=2) + "\n", encoding="utf-8")
|
|
150
|
+
return {
|
|
151
|
+
"tables_added": sorted(added),
|
|
152
|
+
"tables_updated": sorted(updated),
|
|
153
|
+
"columns_added": columns_added,
|
|
154
|
+
"path": str(path),
|
|
155
|
+
}
|
ripple/semantic.py
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
"""Downstream reach into the dbt semantic layer, metrics, and exposures.
|
|
2
|
+
|
|
3
|
+
The BI layer, where it is declared as code, is just more edges. A dbt metric is a
|
|
4
|
+
node that derives from a measure, which is a column of a model Ripple already
|
|
5
|
+
mapped. A dbt exposure is a node-level leaf hanging off the models it reads. So a
|
|
6
|
+
column rename can be traced to the metric and the dashboard it breaks, offline,
|
|
7
|
+
with no warehouse connection.
|
|
8
|
+
|
|
9
|
+
This reads the dbt schema YAML directly (the same files a raw dbt project ships),
|
|
10
|
+
so it works without a compiled manifest. Nothing here fabricates a mapping: a
|
|
11
|
+
metric whose measure we cannot resolve to a real column is dropped, not guessed.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
import yaml
|
|
21
|
+
|
|
22
|
+
from ripple.project import IGNORE_DIRS
|
|
23
|
+
|
|
24
|
+
_REF = re.compile(r"ref\(\s*['\"]([^'\"]+)['\"]\s*(?:,\s*['\"]([^'\"]+)['\"])?\s*\)")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class DownstreamEdge:
|
|
29
|
+
"""A model column (or the model itself) feeding a BI node declared in code."""
|
|
30
|
+
|
|
31
|
+
src_model: str
|
|
32
|
+
src_column: str # "*" means node-level (the whole model)
|
|
33
|
+
dst_model: str # "metric:x" | "exposure:y" | "semantic:orders.customer"
|
|
34
|
+
dst_column: str
|
|
35
|
+
trust: str
|
|
36
|
+
reason: str = ""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _ref_model(value) -> str | None:
|
|
40
|
+
"""The model name a `model: ref('orders')` points at."""
|
|
41
|
+
if not isinstance(value, str):
|
|
42
|
+
return None
|
|
43
|
+
m = _REF.search(value)
|
|
44
|
+
if m:
|
|
45
|
+
return m.group(2) or m.group(1)
|
|
46
|
+
return value.strip() or None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _iter_schema_docs(root: Path):
|
|
50
|
+
for path in sorted(root.rglob("*.yml")) + sorted(root.rglob("*.yaml")):
|
|
51
|
+
parts = path.relative_to(root).parts[:-1]
|
|
52
|
+
if any(p in IGNORE_DIRS or p.startswith(".") for p in parts):
|
|
53
|
+
continue
|
|
54
|
+
try:
|
|
55
|
+
doc = yaml.safe_load(path.read_text(errors="replace", encoding="utf-8"))
|
|
56
|
+
except (yaml.YAMLError, OSError):
|
|
57
|
+
continue
|
|
58
|
+
if not isinstance(doc, dict):
|
|
59
|
+
continue
|
|
60
|
+
if "semantic_models" in doc or "metrics" in doc or "exposures" in doc:
|
|
61
|
+
yield doc
|
|
62
|
+
elif any(
|
|
63
|
+
isinstance(m, dict) and ("semantic_model" in m or "metrics" in m)
|
|
64
|
+
for m in (doc.get("models") or [])
|
|
65
|
+
if m
|
|
66
|
+
):
|
|
67
|
+
# dbt's newer inline syntax: semantic_model nested under a model
|
|
68
|
+
yield doc
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _measure_name(spec) -> str | None:
|
|
72
|
+
if isinstance(spec, str):
|
|
73
|
+
return spec
|
|
74
|
+
if isinstance(spec, dict):
|
|
75
|
+
return spec.get("name")
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def load_downstream(root: Path, known_models: set[str]) -> list[DownstreamEdge]:
|
|
80
|
+
"""Every BI edge we can resolve from the project's declared-as-code layer."""
|
|
81
|
+
known = {m.lower() for m in known_models}
|
|
82
|
+
# global measure name -> (model, column). Measure names are unique per project.
|
|
83
|
+
measures: dict[str, tuple[str, str]] = {}
|
|
84
|
+
edges: list[DownstreamEdge] = []
|
|
85
|
+
|
|
86
|
+
docs = list(_iter_schema_docs(root))
|
|
87
|
+
|
|
88
|
+
# pass 0: dbt's inline syntax (jaffle-shop migrated to it): the semantic
|
|
89
|
+
# model is declared under the model entry itself, dimensions/entities on
|
|
90
|
+
# its columns, metrics on the model with the column as name or expr
|
|
91
|
+
for doc in docs:
|
|
92
|
+
for entry in doc.get("models") or []:
|
|
93
|
+
if not isinstance(entry, dict):
|
|
94
|
+
continue
|
|
95
|
+
sm = entry.get("semantic_model")
|
|
96
|
+
has_inline = isinstance(sm, dict) or entry.get("metrics")
|
|
97
|
+
if not has_inline or (isinstance(sm, dict) and sm.get("enabled") is False):
|
|
98
|
+
continue
|
|
99
|
+
model = entry.get("name")
|
|
100
|
+
if not model or model.lower() not in known:
|
|
101
|
+
continue
|
|
102
|
+
for col in entry.get("columns") or []:
|
|
103
|
+
if not (isinstance(col, dict) and col.get("name")):
|
|
104
|
+
continue
|
|
105
|
+
cname = col["name"]
|
|
106
|
+
for kind in ("dimension", "entity"):
|
|
107
|
+
spec = col.get(kind)
|
|
108
|
+
if not isinstance(spec, dict):
|
|
109
|
+
continue
|
|
110
|
+
dname = spec.get("name") or cname
|
|
111
|
+
edges.append(
|
|
112
|
+
DownstreamEdge(
|
|
113
|
+
src_model=model,
|
|
114
|
+
src_column=cname,
|
|
115
|
+
dst_model=f"semantic:{model}.{dname}",
|
|
116
|
+
dst_column=dname,
|
|
117
|
+
trust="verified",
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
measure = col.get("measure")
|
|
121
|
+
if isinstance(measure, dict) and measure.get("name"):
|
|
122
|
+
measures[measure["name"]] = (model, cname)
|
|
123
|
+
for metric in entry.get("metrics") or []:
|
|
124
|
+
if not (isinstance(metric, dict) and metric.get("name")):
|
|
125
|
+
continue
|
|
126
|
+
expr = metric.get("expr", metric["name"])
|
|
127
|
+
if isinstance(expr, str) and expr.isidentifier():
|
|
128
|
+
edges.append(
|
|
129
|
+
DownstreamEdge(
|
|
130
|
+
src_model=model,
|
|
131
|
+
src_column=expr,
|
|
132
|
+
dst_model=f"metric:{metric['name']}",
|
|
133
|
+
dst_column="value",
|
|
134
|
+
trust="verified",
|
|
135
|
+
)
|
|
136
|
+
)
|
|
137
|
+
# a constant or computed expr has no single source column:
|
|
138
|
+
# dropped, never guessed
|
|
139
|
+
|
|
140
|
+
# pass 1: semantic models give measures/dimensions/entities their columns
|
|
141
|
+
for doc in docs:
|
|
142
|
+
for sm in doc.get("semantic_models") or []:
|
|
143
|
+
if not isinstance(sm, dict):
|
|
144
|
+
continue
|
|
145
|
+
model = _ref_model(sm.get("model"))
|
|
146
|
+
if not model or model.lower() not in known:
|
|
147
|
+
continue
|
|
148
|
+
sm_name = sm.get("name", model)
|
|
149
|
+
for m in sm.get("measures") or []:
|
|
150
|
+
if isinstance(m, dict) and m.get("name"):
|
|
151
|
+
col = m.get("expr") or m["name"]
|
|
152
|
+
if isinstance(col, str) and col.isidentifier():
|
|
153
|
+
measures[m["name"]] = (model, col)
|
|
154
|
+
for kind in ("dimensions", "entities"):
|
|
155
|
+
for d in sm.get(kind) or []:
|
|
156
|
+
if not (isinstance(d, dict) and d.get("name")):
|
|
157
|
+
continue
|
|
158
|
+
col = d.get("expr") or d["name"]
|
|
159
|
+
if not (isinstance(col, str) and col.isidentifier()):
|
|
160
|
+
continue
|
|
161
|
+
edges.append(
|
|
162
|
+
DownstreamEdge(
|
|
163
|
+
src_model=model,
|
|
164
|
+
src_column=col,
|
|
165
|
+
dst_model=f"semantic:{sm_name}.{d['name']}",
|
|
166
|
+
dst_column=d["name"],
|
|
167
|
+
trust="verified",
|
|
168
|
+
)
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
# pass 2: metrics resolve through their measure(s) to the underlying column
|
|
172
|
+
for doc in docs:
|
|
173
|
+
for metric in doc.get("metrics") or []:
|
|
174
|
+
if not (isinstance(metric, dict) and metric.get("name")):
|
|
175
|
+
continue
|
|
176
|
+
params = metric.get("type_params") or {}
|
|
177
|
+
names: list[str] = []
|
|
178
|
+
for key in ("measure", "numerator", "denominator"):
|
|
179
|
+
n = _measure_name(params.get(key))
|
|
180
|
+
if n:
|
|
181
|
+
names.append(n)
|
|
182
|
+
for m in params.get("measures") or []:
|
|
183
|
+
n = _measure_name(m)
|
|
184
|
+
if n:
|
|
185
|
+
names.append(n)
|
|
186
|
+
for n in names:
|
|
187
|
+
loc = measures.get(n)
|
|
188
|
+
if not loc:
|
|
189
|
+
continue # unresolved measure: drop, never fake
|
|
190
|
+
edges.append(
|
|
191
|
+
DownstreamEdge(
|
|
192
|
+
src_model=loc[0],
|
|
193
|
+
src_column=loc[1],
|
|
194
|
+
dst_model=f"metric:{metric['name']}",
|
|
195
|
+
dst_column="value",
|
|
196
|
+
trust="verified",
|
|
197
|
+
)
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
# pass 3: exposures are node-level (a dashboard reads the whole model)
|
|
201
|
+
for doc in docs:
|
|
202
|
+
for exp in doc.get("exposures") or []:
|
|
203
|
+
if not (isinstance(exp, dict) and exp.get("name")):
|
|
204
|
+
continue
|
|
205
|
+
deps = exp.get("depends_on") or exp.get("refs") or []
|
|
206
|
+
if isinstance(deps, str):
|
|
207
|
+
deps = [deps]
|
|
208
|
+
for dep in deps:
|
|
209
|
+
model = (
|
|
210
|
+
_ref_model(dep)
|
|
211
|
+
if isinstance(dep, str)
|
|
212
|
+
else _ref_model((dep or {}).get("name") if isinstance(dep, dict) else None)
|
|
213
|
+
)
|
|
214
|
+
if model and model.lower() in known:
|
|
215
|
+
edges.append(
|
|
216
|
+
DownstreamEdge(
|
|
217
|
+
src_model=model,
|
|
218
|
+
src_column="*",
|
|
219
|
+
dst_model=f"exposure:{exp['name']}",
|
|
220
|
+
dst_column="*",
|
|
221
|
+
trust="review_required",
|
|
222
|
+
reason=f"declared {exp.get('type', 'exposure')}, node-level",
|
|
223
|
+
)
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
# dedupe
|
|
227
|
+
seen, out = set(), []
|
|
228
|
+
for e in edges:
|
|
229
|
+
if e not in seen:
|
|
230
|
+
seen.add(e)
|
|
231
|
+
out.append(e)
|
|
232
|
+
return out
|