ripple-sql 0.1.4__tar.gz → 0.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/CHANGELOG.md +13 -1
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/PKG-INFO +1 -1
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/pyproject.toml +1 -1
- ripple_sql-0.1.6/src/ripple/engine/generators.py +149 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/schema_qualification.py +12 -2
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/statement.py +1 -1
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/home.py +103 -34
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/render.py +21 -10
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer.css +6 -2
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer.html +34 -13
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/explore.js +86 -55
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/find.js +3 -3
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/focus.js +47 -33
- ripple_sql-0.1.4/src/ripple/engine/generators.py +0 -61
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/.gitignore +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/LICENSE +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/README.md +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/__init__.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/answer.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/answer_page.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/cache.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/ci.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/ci_signature.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/cli.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/doctor.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/__init__.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/budget.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/column_lineage.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/column_ref.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/cte_tracing.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dependencies.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dialect.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dispatch.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/extraction.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/jinja.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/json_sources.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/macro_source.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/pipeline.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/preprocess.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/safe_gen.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/scope.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/select_sources.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/sql_script.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tech_debt.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_catalog.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_scalar_vars.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_tvf.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_xml.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/types.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/unused_deps.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/validation.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/graph.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/__init__.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/dbt.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/dbt_config.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/identity.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/sidecar.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/sqldir.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/types.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/lookml.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/mcp_server.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/names.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/project.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/py.typed +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/render_shims.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/schemas.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/semantic.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/server.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/sourcefiles.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/star_resolution.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer_twin.js +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/__init__.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/cli.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/collect.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/discover.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/ingest.py +0 -0
- {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/report.py +0 -0
|
@@ -6,6 +6,16 @@ Notable changes to Ripple. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.6] - 2026-09-10
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
- Thirty-six findings from three rounds of a two-model review of the 0.1.3 to 0.1.5 work: the engine and map fixes each pinned by a test, the page fixes checked with headless browser probes (the page has no automated tests). Rounds two and three also settled INLINE over NULL and non-struct elements, name-matched fields, `AS (xx, yy)` alias lists, maps behind parentheses and casts, a cap before `as_native` parses, kind-prefixed group identities, the loader's own say on which folders are tables, and a cycle twin that stays visible while its other copy is folded. Engine: the generator names (`INLINE`, `EXPLODE`, `POSEXPLODE`) survive schema qualification, which reparsed the original SQL and threw them away; each `INLINE` output now takes lineage from its own struct field, not every field; `EXPLODE` over a map yields `key` and `value`; `as_number` keeps floats and `as_native` reads Python literals as dbt documents. Page: hovering a group heading threw instead of tinting (the group adjacency was walked as arrays); a group or model named `constructor` or `__proto__` crashed the page; a model's badge and a group's reader count included dashboard numbers while the key said models; a column suggestion showed its model's reader count instead of its own; two folders ending in the same name, or a schema and a folder of one name, merged into one row; a monorepo model named `models__marts__a` lost its folder; the filter went stale across focus and home and fought the fold; a late answer could overwrite newer navigation; Escape did not leave focus from the question box; a cycle drew a model on both sides; the typeahead left a stale `aria-activedescendant`.
|
|
13
|
+
|
|
14
|
+
## [0.1.5] - 2026-09-09
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
- Every column under a model on the page carries the number of models that read it directly, the same number a model chip carries, so the cost of a change shows before the question is asked. A column no link touches (a constant, or one the engine could not trace) is drawn loose, dashed and faint, with "no link found" in the key; r/dataengineering asked for "a clear view of mapped vs unmapped fields (with some kind of legend)", and a traced-looking constant would have been a lie.
|
|
18
|
+
|
|
9
19
|
## [0.1.4] - 2026-09-09
|
|
10
20
|
|
|
11
21
|
### Changed
|
|
@@ -249,7 +259,9 @@ First public release.
|
|
|
249
259
|
### Removed
|
|
250
260
|
- The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
|
|
251
261
|
|
|
252
|
-
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.
|
|
262
|
+
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.6...HEAD
|
|
263
|
+
[0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
|
|
264
|
+
[0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
|
|
253
265
|
[0.1.4]: https://github.com/bteh/ripple/releases/tag/v0.1.4
|
|
254
266
|
[0.1.3]: https://github.com/bteh/ripple/releases/tag/v0.1.3
|
|
255
267
|
[0.1.2]: https://github.com/bteh/ripple/releases/tag/v0.1.2
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Bare table-generating functions name their own output columns.
|
|
2
|
+
|
|
3
|
+
Spark's `SELECT INLINE(ARRAY(STRUCT('a' AS platform)))` has no alias and
|
|
4
|
+
no column reference; its output columns are the struct's fields. A bare
|
|
5
|
+
`EXPLODE(x)` is a column called `col`, `POSEXPLODE(x)` is `pos` and `col`,
|
|
6
|
+
and over a map they are `key` and `value`. sqlglot leaves such items
|
|
7
|
+
nameless, and the projection walks named them by their SQL text or dropped
|
|
8
|
+
them, so a lookup table built this way reported no columns at all
|
|
9
|
+
(Wikimedia dbt-jobs, platforms_ephemeral).
|
|
10
|
+
|
|
11
|
+
The rewrite gives each output its default name as an alias over the
|
|
12
|
+
expression that feeds it, before any projection walk runs: a struct field
|
|
13
|
+
feeds its own output only, so a change to one field does not ripple to
|
|
14
|
+
every column of the table.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from sqlglot import exp
|
|
20
|
+
|
|
21
|
+
Output = tuple[str, exp.Expression]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def generator_outputs(item: exp.Expression) -> list[Output] | None:
|
|
25
|
+
"""(name, expression) per output column of a bare generator, None for
|
|
26
|
+
anything else. `INLINE(...) AS (xx, yy)` is the same generator with its
|
|
27
|
+
outputs named by position."""
|
|
28
|
+
if isinstance(item, exp.Aliases):
|
|
29
|
+
inner = generator_outputs(item.this)
|
|
30
|
+
names = [alias.name for alias in item.expressions]
|
|
31
|
+
if inner is None or len(names) != len(inner):
|
|
32
|
+
return None
|
|
33
|
+
return [(name, expr) for name, (_, expr) in zip(names, inner, strict=True)]
|
|
34
|
+
if isinstance(item, exp.Inline):
|
|
35
|
+
return _inline_outputs(item)
|
|
36
|
+
if isinstance(item, (exp.Explode, exp.Posexplode)):
|
|
37
|
+
return _explode_outputs(item)
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _struct_fields(element: exp.Expression) -> list[Output] | None:
|
|
42
|
+
if not isinstance(element, exp.Struct):
|
|
43
|
+
return None
|
|
44
|
+
fields: list[Output] = []
|
|
45
|
+
for field in element.expressions:
|
|
46
|
+
if isinstance(field, exp.PropertyEQ):
|
|
47
|
+
fields.append((field.this.name, field.expression))
|
|
48
|
+
elif isinstance(field, exp.Alias):
|
|
49
|
+
fields.append((field.alias, field.this))
|
|
50
|
+
elif isinstance(field, exp.Column):
|
|
51
|
+
fields.append((field.name, field))
|
|
52
|
+
else:
|
|
53
|
+
return None
|
|
54
|
+
return fields or None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _inline_outputs(item: exp.Inline) -> list[Output] | None:
|
|
58
|
+
"""INLINE over a literal array: the first struct names the outputs and
|
|
59
|
+
field i of every struct feeds output i. That is Spark's own rule: the
|
|
60
|
+
array has one element type, and a later struct must carry the same
|
|
61
|
+
field names at the same positions, case-insensitively, or the query
|
|
62
|
+
fails analysis (TypeCoercion.findTypeForComplex). A struct whose names
|
|
63
|
+
are a permutation of the first's would not run in Spark; it is read by
|
|
64
|
+
name, as written. A NULL element adds nothing, and an element that is
|
|
65
|
+
not a struct literal (a struct column, a CASE) feeds every output, since
|
|
66
|
+
its fields cannot be told apart."""
|
|
67
|
+
array = item.this
|
|
68
|
+
if not (isinstance(array, exp.Array) and array.expressions):
|
|
69
|
+
return None
|
|
70
|
+
elements = [e for e in array.expressions if not isinstance(e, exp.Null)]
|
|
71
|
+
structs = [(e, _struct_fields(e)) for e in elements]
|
|
72
|
+
first = next((fields for _, fields in structs if fields), None)
|
|
73
|
+
if first is None:
|
|
74
|
+
return None
|
|
75
|
+
names = [name for name, _ in first]
|
|
76
|
+
feeds: list[list[exp.Expression]] = [[] for _ in names]
|
|
77
|
+
for element, fields in structs:
|
|
78
|
+
if fields is None:
|
|
79
|
+
for fed_by in feeds:
|
|
80
|
+
fed_by.append(element.copy())
|
|
81
|
+
continue
|
|
82
|
+
matched = _match_fields(first, fields)
|
|
83
|
+
if matched is None:
|
|
84
|
+
return None
|
|
85
|
+
for fed_by, expr in zip(feeds, matched, strict=True):
|
|
86
|
+
fed_by.append(expr.copy())
|
|
87
|
+
return [
|
|
88
|
+
(name, fed_by[0] if len(fed_by) == 1 else exp.Array(expressions=fed_by))
|
|
89
|
+
for name, fed_by in zip(names, feeds, strict=True)
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _match_fields(first: list[Output], fields: list[Output]) -> list[exp.Expression] | None:
|
|
94
|
+
"""The expression of `fields` feeding each output of `first`, in order:
|
|
95
|
+
by position when the names line up (the only shape Spark accepts, and
|
|
96
|
+
the only reading when two unnamed fields share a name, `a.id, b.id`);
|
|
97
|
+
by name when they are a permutation; None when the shapes disagree."""
|
|
98
|
+
if len(fields) != len(first):
|
|
99
|
+
return None
|
|
100
|
+
want = [name.lower() for name, _ in first]
|
|
101
|
+
have = [name.lower() for name, _ in fields]
|
|
102
|
+
if want == have or len(set(have)) != len(have) or set(want) != set(have):
|
|
103
|
+
return [expr for _, expr in fields]
|
|
104
|
+
by_name = {name.lower(): expr for name, expr in fields}
|
|
105
|
+
return [by_name[name] for name in want]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _is_map(source: exp.Expression) -> bool:
|
|
109
|
+
if isinstance(source, (exp.Map, exp.VarMap)):
|
|
110
|
+
return True
|
|
111
|
+
to = source.args.get("to") if isinstance(source, exp.Cast) else None
|
|
112
|
+
return isinstance(to, exp.DataType) and to.this == exp.DataType.Type.MAP
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _explode_outputs(item: exp.Expression) -> list[Output]:
|
|
116
|
+
"""EXPLODE and POSEXPLODE: `col` fed by the whole call over an array;
|
|
117
|
+
`key` and `value` fed by the map's keys and values when the input is
|
|
118
|
+
visibly a map, or both by the whole map when only its type is visible
|
|
119
|
+
(a CAST to MAP). `pos` comes from the whole call either way."""
|
|
120
|
+
source = item.this
|
|
121
|
+
while isinstance(source, exp.Paren):
|
|
122
|
+
source = source.this
|
|
123
|
+
if isinstance(source, (exp.Map, exp.VarMap)):
|
|
124
|
+
outputs: list[Output] = [("key", source.args["keys"]), ("value", source.args["values"])]
|
|
125
|
+
elif _is_map(source):
|
|
126
|
+
outputs = [("key", source), ("value", source)]
|
|
127
|
+
else:
|
|
128
|
+
outputs = [("col", item)]
|
|
129
|
+
if isinstance(item, exp.Posexplode):
|
|
130
|
+
outputs = [("pos", item)] + outputs
|
|
131
|
+
return [(name, expr.copy()) for name, expr in outputs]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def name_generator_outputs(tree: exp.Expression) -> exp.Expression:
|
|
135
|
+
"""Alias every bare generator in every select list with its default
|
|
136
|
+
output names, in place, and return the tree."""
|
|
137
|
+
for select in list(tree.find_all(exp.Select)):
|
|
138
|
+
items = select.expressions
|
|
139
|
+
if not any(generator_outputs(item) for item in items):
|
|
140
|
+
continue
|
|
141
|
+
renamed: list[exp.Expression] = []
|
|
142
|
+
for item in items:
|
|
143
|
+
outputs = generator_outputs(item)
|
|
144
|
+
if not outputs:
|
|
145
|
+
renamed.append(item)
|
|
146
|
+
continue
|
|
147
|
+
renamed.extend(exp.alias_(expr, name) for name, expr in outputs)
|
|
148
|
+
select.set("expressions", renamed)
|
|
149
|
+
return tree
|
|
@@ -92,9 +92,14 @@ def qualify_sql_with_schema(
|
|
|
92
92
|
sql: str,
|
|
93
93
|
dialect: str,
|
|
94
94
|
warehouse_columns: WarehouseColumns,
|
|
95
|
+
expression: "exp.Expression | None" = None,
|
|
95
96
|
) -> tuple["exp.Expression | None", bool]:
|
|
96
97
|
"""Qualify unqualified columns in SQL using warehouse schema.
|
|
97
98
|
|
|
99
|
+
Pass `expression` when the caller has already normalized the tree (bare
|
|
100
|
+
generators named, for one): re-parsing the text here would throw that
|
|
101
|
+
work away, and the caller's names would come back as _col_0.
|
|
102
|
+
|
|
98
103
|
Uses SQLGlot's optimizer.qualify to add table qualifiers to columns.
|
|
99
104
|
This resolves ambiguous column references like SELECT col FROM a, b
|
|
100
105
|
into SELECT a.col FROM a, b when 'col' only exists in table 'a'.
|
|
@@ -120,8 +125,13 @@ def qualify_sql_with_schema(
|
|
|
120
125
|
return None, False
|
|
121
126
|
|
|
122
127
|
schema_dict = build_sqlglot_schema(warehouse_columns)
|
|
128
|
+
|
|
129
|
+
def fresh() -> "exp.Expression":
|
|
130
|
+
# qualify() mutates, so each attempt starts from its own copy
|
|
131
|
+
return expression.copy() if expression is not None else sqlglot.parse_one(sql, read=dialect)
|
|
132
|
+
|
|
123
133
|
try:
|
|
124
|
-
parsed =
|
|
134
|
+
parsed = fresh()
|
|
125
135
|
qualified = qualify(parsed, schema=schema_dict, dialect=dialect)
|
|
126
136
|
return qualified, True
|
|
127
137
|
except Exception as e:
|
|
@@ -130,7 +140,7 @@ def qualify_sql_with_schema(
|
|
|
130
140
|
# what the schema resolves, leave the rest as written instead of throwing
|
|
131
141
|
# away every resolved column with it.
|
|
132
142
|
try:
|
|
133
|
-
parsed =
|
|
143
|
+
parsed = fresh()
|
|
134
144
|
qualified = qualify(
|
|
135
145
|
parsed,
|
|
136
146
|
schema=schema_dict,
|
|
@@ -247,7 +247,7 @@ def extract_column_lineage_fast(
|
|
|
247
247
|
# return the AST directly to avoid double-parsing
|
|
248
248
|
if warehouse_columns:
|
|
249
249
|
qualified_ast, was_qualified = qualify_sql_with_schema(
|
|
250
|
-
cleaned_sql, dialect, warehouse_columns
|
|
250
|
+
cleaned_sql, dialect, warehouse_columns, expression=parsed
|
|
251
251
|
)
|
|
252
252
|
if was_qualified and qualified_ast is not None:
|
|
253
253
|
parsed = qualified_ast
|
|
@@ -115,6 +115,9 @@ def model_map(nodes: list[dict], edges) -> dict:
|
|
|
115
115
|
it, a dashboard number past a dashboard prefix, and a model otherwise.
|
|
116
116
|
Star and placeholder columns are links, never columns."""
|
|
117
117
|
members: dict[str, dict] = {}
|
|
118
|
+
# the loader's identity says whether a folder is the table; kept aside so
|
|
119
|
+
# the page payload does not carry it
|
|
120
|
+
uids = {n["name"]: n.get("id") or "" for n in nodes}
|
|
118
121
|
for n in nodes:
|
|
119
122
|
source = n["type"] == "source"
|
|
120
123
|
columns = [c for c in n.get("columns") or [] if _is_column(c)]
|
|
@@ -130,10 +133,19 @@ def model_map(nodes: list[dict], edges) -> dict:
|
|
|
130
133
|
}
|
|
131
134
|
fed = {e.dst_model for e in edges}
|
|
132
135
|
seen: dict[str, dict[str, None]] = {}
|
|
136
|
+
readers: dict[tuple[str, str], set[str]] = {}
|
|
137
|
+
linked: set[tuple[str, str]] = set()
|
|
133
138
|
for e in edges:
|
|
134
139
|
for model, column in ((e.src_model, e.src_column), (e.dst_model, e.dst_column)):
|
|
135
140
|
if _is_column(column):
|
|
136
141
|
seen.setdefault(model, {})[column] = None
|
|
142
|
+
if e.kind != "value" or not (_is_column(e.src_column) and _is_column(e.dst_column)):
|
|
143
|
+
continue
|
|
144
|
+
linked.add((e.src_model, e.src_column))
|
|
145
|
+
linked.add((e.dst_model, e.dst_column))
|
|
146
|
+
# a dashboard number is not a reader; only models count (as in starters)
|
|
147
|
+
if not e.dst_model.startswith(DASHBOARD_PREFIXES):
|
|
148
|
+
readers.setdefault((e.src_model, e.src_column), set()).add(e.dst_model)
|
|
137
149
|
for name in sorted({e.src_model for e in edges} | fed):
|
|
138
150
|
if name in members:
|
|
139
151
|
known = members[name]["columns"]
|
|
@@ -151,22 +163,33 @@ def model_map(nodes: list[dict], edges) -> dict:
|
|
|
151
163
|
"status": None,
|
|
152
164
|
"path": None,
|
|
153
165
|
}
|
|
166
|
+
for m in members.values():
|
|
167
|
+
# per column, aligned with `columns`: how many models read it directly;
|
|
168
|
+
# -1 when no link touches it at all (a constant, or one the engine
|
|
169
|
+
# could not trace), so the page never draws such a column as traced
|
|
170
|
+
m["reads"] = [
|
|
171
|
+
len(readers.get((m["id"], c), ())) if (m["id"], c) in linked else -1
|
|
172
|
+
for c in m["columns"]
|
|
173
|
+
]
|
|
154
174
|
links = _links(edges)
|
|
155
175
|
layers = _layers(members, links)
|
|
156
176
|
models = [{**m, "layer": layers[m["id"]]} for m in members.values()]
|
|
157
|
-
_name_groups(models)
|
|
177
|
+
labels = _name_groups(models, uids)
|
|
158
178
|
ordered = _ordered(models, links)
|
|
159
|
-
return {"models": ordered, "links": links, "groups": _groups(ordered, links)}
|
|
179
|
+
return {"models": ordered, "links": links, "groups": _groups(ordered, links, labels)}
|
|
160
180
|
|
|
161
181
|
|
|
162
|
-
def group_of(member: dict) -> str:
|
|
163
|
-
"""Where a model lives,
|
|
164
|
-
|
|
182
|
+
def group_of(member: dict, uid: str = "", table_folders: frozenset[str] = frozenset()) -> str:
|
|
183
|
+
"""Where a model lives, unshortened: the folder of a model, the schema of
|
|
184
|
+
a source table, "dashboards" for a dashboard number.
|
|
165
185
|
|
|
166
|
-
A file
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
186
|
+
A model file sits in its folder (`models/marts/orders.sql` is in
|
|
187
|
+
`models/marts`), whatever the model is called: a monorepo may name it
|
|
188
|
+
`models__marts__orders`, and a dbt model may be called query. When the
|
|
189
|
+
loader itself took a folder as the table (`telemetry_derived/clients_v1/
|
|
190
|
+
query.sql`, a schema.yaml sidecar) the group is the folder above, for
|
|
191
|
+
that model and for every other file in the table's folder (a script.sql
|
|
192
|
+
beside the query). A model with no path is in "models"."""
|
|
170
193
|
if member["kind"] == "dashboard":
|
|
171
194
|
return "dashboards"
|
|
172
195
|
if member["kind"] == "source":
|
|
@@ -175,55 +198,95 @@ def group_of(member: dict) -> str:
|
|
|
175
198
|
path = member.get("path")
|
|
176
199
|
if not path:
|
|
177
200
|
return "models"
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
folders = parts[:-1]
|
|
181
|
-
if folders and stem != member["id"]:
|
|
201
|
+
folders = path.replace("\\", "/").split("/")[:-1]
|
|
202
|
+
if table_folder(member, uid) or "/".join(folders).lower() in table_folders:
|
|
182
203
|
folders = folders[:-1]
|
|
183
204
|
return "/".join(folders) or "models"
|
|
184
205
|
|
|
185
206
|
|
|
207
|
+
def table_folder(member: dict, uid: str = "") -> str | None:
|
|
208
|
+
"""The folder a model's loader took as the table, lowercased, or None.
|
|
209
|
+
|
|
210
|
+
That happened when the model's name, or the relation in its uid
|
|
211
|
+
(path::relation, possibly dataset.table), is the last folder and the
|
|
212
|
+
file is called something else (query.sql, schema.yaml). A file that is
|
|
213
|
+
the folder's namesake, or any file in a dbt project, keeps its folder."""
|
|
214
|
+
path = member.get("path")
|
|
215
|
+
if not path or member["kind"] != "model":
|
|
216
|
+
return None
|
|
217
|
+
parts = path.replace("\\", "/").split("/")
|
|
218
|
+
folders, stem = parts[:-1], parts[-1].rsplit(".", 1)[0]
|
|
219
|
+
if not folders:
|
|
220
|
+
return None
|
|
221
|
+
folder = folders[-1].lower()
|
|
222
|
+
relation = uid.rpartition("::")[-1].rpartition(".")[-1].lower()
|
|
223
|
+
name = member["id"].lower()
|
|
224
|
+
named_after_folder = folder in (name, relation) or name.endswith("__" + folder)
|
|
225
|
+
if named_after_folder and stem.lower() != folder:
|
|
226
|
+
return "/".join(folders).lower()
|
|
227
|
+
return None
|
|
228
|
+
|
|
229
|
+
|
|
186
230
|
def _shorten(names: set[str], sep: str) -> dict[str, str]:
|
|
187
|
-
"""
|
|
188
|
-
`models/marts` read as `staging` and `marts`. A lone
|
|
231
|
+
"""Labels: drop the leading segments every deep name shares, so
|
|
232
|
+
`models/staging` and `models/marts` read as `staging` and `marts`. A lone
|
|
233
|
+
name keeps itself, and `models/core` next to `models/staging/core` keeps
|
|
234
|
+
enough to tell them apart."""
|
|
189
235
|
parts = {n: n.split(sep) for n in names}
|
|
190
236
|
while True:
|
|
191
237
|
deep = [p for p in parts.values() if len(p) > 1]
|
|
192
238
|
if len(deep) < 2 or len({p[0] for p in deep}) != 1:
|
|
193
239
|
break
|
|
194
|
-
|
|
240
|
+
nxt = {n: p[1:] if len(p) > 1 else p for n, p in parts.items()}
|
|
241
|
+
# a strip that makes two names read the same stops here
|
|
242
|
+
if len({sep.join(p) for p in nxt.values()}) < len(nxt):
|
|
243
|
+
break
|
|
244
|
+
parts = nxt
|
|
195
245
|
return {n: sep.join(p) for n, p in parts.items()}
|
|
196
246
|
|
|
197
247
|
|
|
198
248
|
SOURCE_SCHEMAS = 6
|
|
199
249
|
|
|
200
250
|
|
|
201
|
-
def _name_groups(models: list[dict]) ->
|
|
202
|
-
"""Set each model's group
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
251
|
+
def _name_groups(models: list[dict], uids: dict[str, str]) -> dict[str, str]:
|
|
252
|
+
"""Set each model's group in place and return the label of every group.
|
|
253
|
+
|
|
254
|
+
The group is an identity, never shown: the kind and the unshortened
|
|
255
|
+
folder or schema (`model:models/marts`, `source:raw`), so a schema and a
|
|
256
|
+
folder of one name, or a folder called dashboards, stay separate rows.
|
|
257
|
+
The label is what the page prints. A project that reads
|
|
258
|
+
from more than a handful of schemas gets one "sources" group: a row per
|
|
259
|
+
schema would be a page of one-table rows, and the chip already carries
|
|
260
|
+
the schema in its name."""
|
|
261
|
+
tables = frozenset(t for m in models if (t := table_folder(m, uids.get(m["id"], ""))))
|
|
262
|
+
raw = {m["id"]: group_of(m, uids.get(m["id"], ""), tables) for m in models}
|
|
263
|
+
labels: dict[str, str] = {}
|
|
264
|
+
for kind, sep in (("model", "/"), ("source", "."), ("dashboard", "/")):
|
|
265
|
+
prefix = kind + ":"
|
|
209
266
|
names = {raw[m["id"]] for m in models if m["kind"] == kind}
|
|
210
267
|
if kind == "source" and len(names) > SOURCE_SCHEMAS:
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
268
|
+
for m in models:
|
|
269
|
+
if m["kind"] == kind:
|
|
270
|
+
raw[m["id"]] = "sources"
|
|
271
|
+
names = {"sources"}
|
|
272
|
+
for name, label in _shorten(names, sep).items():
|
|
273
|
+
labels[prefix + name] = label
|
|
214
274
|
for m in models:
|
|
215
|
-
m["group"] =
|
|
275
|
+
m["group"] = m["kind"] + ":" + raw[m["id"]]
|
|
276
|
+
return labels
|
|
216
277
|
|
|
217
278
|
|
|
218
279
|
GROUP_RANK = {"source": 0, "model": 1, "dashboard": 2}
|
|
219
280
|
|
|
220
281
|
|
|
221
|
-
def _groups(models: list[dict], links: list[dict]) -> list[dict]:
|
|
282
|
+
def _groups(models: list[dict], links: list[dict], labels: dict[str, str]) -> list[dict]:
|
|
222
283
|
"""The groups in the order the data flows: source schemas, then model
|
|
223
284
|
folders by the mean layer of their models, then dashboards. Each carries
|
|
224
|
-
how many models it holds and how many models outside it read
|
|
285
|
+
its label, how many models it holds and how many models outside it read
|
|
286
|
+
one of its; a dashboard number is not a reader."""
|
|
225
287
|
by: dict[str, dict] = {}
|
|
226
288
|
of: dict[str, str] = {}
|
|
289
|
+
kind_of: dict[str, str] = {}
|
|
227
290
|
for m in models:
|
|
228
291
|
g = by.setdefault(
|
|
229
292
|
m["group"],
|
|
@@ -231,19 +294,25 @@ def _groups(models: list[dict], links: list[dict]) -> list[dict]:
|
|
|
231
294
|
)
|
|
232
295
|
g["n"] += 1
|
|
233
296
|
g["layers"].append(m["layer"])
|
|
234
|
-
if GROUP_RANK[m["kind"]] > GROUP_RANK[g["kind"]]:
|
|
235
|
-
g["kind"] = m["kind"]
|
|
236
297
|
of[m["id"]] = m["group"]
|
|
298
|
+
kind_of[m["id"]] = m["kind"]
|
|
237
299
|
for link in links:
|
|
238
300
|
src, dst = of.get(link["src"]), of.get(link["dst"])
|
|
239
|
-
if src and dst and src != dst:
|
|
301
|
+
if src and dst and src != dst and kind_of[link["dst"]] != "dashboard":
|
|
240
302
|
by[src]["readers"].add(link["dst"])
|
|
241
303
|
out = sorted(
|
|
242
304
|
by.values(),
|
|
243
305
|
key=lambda g: (GROUP_RANK[g["kind"]], sum(g["layers"]) / len(g["layers"]), g["id"]),
|
|
244
306
|
)
|
|
245
307
|
return [
|
|
246
|
-
{
|
|
308
|
+
{
|
|
309
|
+
"id": g["id"],
|
|
310
|
+
"label": labels.get(g["id"], g["id"]),
|
|
311
|
+
"kind": g["kind"],
|
|
312
|
+
"n": g["n"],
|
|
313
|
+
"readers": len(g["readers"]),
|
|
314
|
+
}
|
|
315
|
+
for g in out
|
|
247
316
|
]
|
|
248
317
|
|
|
249
318
|
|
|
@@ -18,6 +18,7 @@ matters; unknown macros degrade to NULL (parseable, no fake lineage).
|
|
|
18
18
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
|
+
import ast as _ast
|
|
21
22
|
import datetime as _datetime
|
|
22
23
|
import itertools as _itertools
|
|
23
24
|
import re
|
|
@@ -288,25 +289,35 @@ def _as_bool(value):
|
|
|
288
289
|
|
|
289
290
|
|
|
290
291
|
def _as_number(value):
|
|
292
|
+
"""dbt's as_number: a number stays what it is (1.5 is not 1), a string
|
|
293
|
+
becomes an int or a float, anything else comes back untouched."""
|
|
294
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
295
|
+
return value
|
|
296
|
+
text = str(value).strip()
|
|
291
297
|
try:
|
|
292
|
-
return int(
|
|
293
|
-
except
|
|
298
|
+
return int(text)
|
|
299
|
+
except ValueError:
|
|
294
300
|
pass
|
|
295
301
|
try:
|
|
296
|
-
return float(
|
|
297
|
-
except
|
|
302
|
+
return float(text)
|
|
303
|
+
except ValueError:
|
|
298
304
|
return value
|
|
299
305
|
|
|
300
306
|
|
|
307
|
+
# a literal past this length never reaches the parser: deep enough nesting
|
|
308
|
+
# can overflow the C parser outright, past any except clause, and no dbt
|
|
309
|
+
# var is a tenth of it
|
|
310
|
+
_LITERAL_MAX = 10_000
|
|
311
|
+
|
|
312
|
+
|
|
301
313
|
def _as_native(value):
|
|
302
|
-
"""dbt's as_native: a string read
|
|
303
|
-
|
|
314
|
+
"""dbt's as_native: a string read as a Python literal (a tuple, a dict
|
|
315
|
+
with None in it), the string itself when it is not one."""
|
|
316
|
+
if not isinstance(value, str) or len(value) > _LITERAL_MAX:
|
|
304
317
|
return value
|
|
305
318
|
try:
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
return yaml.safe_load(value)
|
|
309
|
-
except Exception:
|
|
319
|
+
return _ast.literal_eval(value.strip())
|
|
320
|
+
except (ValueError, SyntaxError, TypeError, MemoryError, RecursionError):
|
|
310
321
|
return value
|
|
311
322
|
|
|
312
323
|
|
|
@@ -77,6 +77,7 @@ h1 code{font:500 24px/1 var(--mono); letter-spacing:0}
|
|
|
77
77
|
.key i.r{border-top-style:dashed; border-color:var(--review)}
|
|
78
78
|
.key s{display:inline-block; width:11px; height:11px; border-radius:2px; background:var(--ink); border:1px solid var(--ink); text-decoration:none}
|
|
79
79
|
.key s.src{background:var(--well); border:1px dashed var(--line-strong)}
|
|
80
|
+
.key s.unk{background:transparent; border:1px dashed var(--faint)}
|
|
80
81
|
.key small.n{font:400 11.5px/1 var(--sans); color:var(--faint); border:1px solid var(--line-strong); border-radius:var(--r); padding:2px 5px; background:var(--surface)}
|
|
81
82
|
.hint{margin-left:auto; font-size:12.5px; color:var(--faint)}
|
|
82
83
|
.rows{padding:0 22px 14px; font-size:13.5px; color:var(--muted)}
|
|
@@ -110,6 +111,9 @@ pre{margin:0; padding:16px 22px 18px; overflow-x:auto; font:400 13px/1.65 var(--
|
|
|
110
111
|
.cols{display:flex; flex-direction:column; gap:5px; margin:0 0 6px 10px; padding-left:10px; border-left:1px solid var(--line)}
|
|
111
112
|
.cols .gname{max-width:36ch; white-space:normal; line-height:1.45; padding-top:2px}
|
|
112
113
|
.cols .node{cursor:pointer}
|
|
114
|
+
.cols .node small{font:400 11px/1 var(--sans); color:var(--faint); margin-left:5px}
|
|
115
|
+
.cols .node.unk{color:var(--muted)}
|
|
116
|
+
.cols .node.unk:hover{border-color:var(--accent)}
|
|
113
117
|
.cols .node:hover{border-color:var(--accent); color:var(--accent-strong)}
|
|
114
118
|
.more-chips{font:500 12.5px var(--sans); color:var(--accent); background:none; border:0; padding:4px 0; cursor:pointer; text-align:left}
|
|
115
119
|
.more-chips:hover{text-decoration:underline}
|
|
@@ -140,8 +144,8 @@ pre{margin:0; padding:16px 22px 18px; overflow-x:auto; font:400 13px/1.65 var(--
|
|
|
140
144
|
.gcell:hover{background:var(--well)}
|
|
141
145
|
.grow.sel .gcell{background:var(--accent); color:#fff}
|
|
142
146
|
.grow.sel .gcell small{color:rgba(255,255,255,.72)}
|
|
143
|
-
.grow.up .gcell{background:#E4F1F4;
|
|
144
|
-
.grow.down .gcell{background:var(--well)
|
|
147
|
+
.grow.up .gcell{background:#E4F1F4; color:var(--accent-strong)}
|
|
148
|
+
.grow.down .gcell{background:var(--well)}
|
|
145
149
|
.index .chips{display:flex; flex-wrap:wrap; gap:6px 8px; align-items:center}
|
|
146
150
|
.index .mdl{max-width:100%; overflow-wrap:anywhere; text-align:left; line-height:1.25}
|
|
147
151
|
.index .gname{flex-basis:100%}
|