ripple-sql 0.1.4__tar.gz → 0.1.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/CHANGELOG.md +13 -1
  2. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/PKG-INFO +1 -1
  3. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/pyproject.toml +1 -1
  4. ripple_sql-0.1.6/src/ripple/engine/generators.py +149 -0
  5. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/schema_qualification.py +12 -2
  6. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/statement.py +1 -1
  7. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/home.py +103 -34
  8. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/render.py +21 -10
  9. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer.css +6 -2
  10. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer.html +34 -13
  11. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/explore.js +86 -55
  12. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/find.js +3 -3
  13. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/focus.js +47 -33
  14. ripple_sql-0.1.4/src/ripple/engine/generators.py +0 -61
  15. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/.gitignore +0 -0
  16. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/LICENSE +0 -0
  17. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/README.md +0 -0
  18. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/__init__.py +0 -0
  19. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/answer.py +0 -0
  20. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/answer_page.py +0 -0
  21. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/cache.py +0 -0
  22. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/ci.py +0 -0
  23. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/ci_signature.py +0 -0
  24. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/cli.py +0 -0
  25. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/doctor.py +0 -0
  26. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/__init__.py +0 -0
  27. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/budget.py +0 -0
  28. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/column_lineage.py +0 -0
  29. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/column_ref.py +0 -0
  30. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/cte_tracing.py +0 -0
  31. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dependencies.py +0 -0
  32. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dialect.py +0 -0
  33. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/dispatch.py +0 -0
  34. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/extraction.py +0 -0
  35. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/jinja.py +0 -0
  36. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/json_sources.py +0 -0
  37. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/macro_source.py +0 -0
  38. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/pipeline.py +0 -0
  39. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/preprocess.py +0 -0
  40. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/safe_gen.py +0 -0
  41. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/scope.py +0 -0
  42. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/select_sources.py +0 -0
  43. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/sql_script.py +0 -0
  44. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tech_debt.py +0 -0
  45. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_catalog.py +0 -0
  46. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_scalar_vars.py +0 -0
  47. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_tvf.py +0 -0
  48. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/tsql_xml.py +0 -0
  49. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/types.py +0 -0
  50. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/unused_deps.py +0 -0
  51. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/engine/validation.py +0 -0
  52. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/graph.py +0 -0
  53. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/__init__.py +0 -0
  54. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/dbt.py +0 -0
  55. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/dbt_config.py +0 -0
  56. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/identity.py +0 -0
  57. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/sidecar.py +0 -0
  58. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/sqldir.py +0 -0
  59. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/loaders/types.py +0 -0
  60. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/lookml.py +0 -0
  61. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/mcp_server.py +0 -0
  62. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/names.py +0 -0
  63. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/project.py +0 -0
  64. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/py.typed +0 -0
  65. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/render_shims.py +0 -0
  66. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/schemas.py +0 -0
  67. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/semantic.py +0 -0
  68. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/server.py +0 -0
  69. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/sourcefiles.py +0 -0
  70. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/star_resolution.py +0 -0
  71. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/static/answer_twin.js +0 -0
  72. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/__init__.py +0 -0
  73. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/cli.py +0 -0
  74. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/collect.py +0 -0
  75. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/discover.py +0 -0
  76. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/ingest.py +0 -0
  77. {ripple_sql-0.1.4 → ripple_sql-0.1.6}/src/ripple/usage/report.py +0 -0
@@ -6,6 +6,16 @@ Notable changes to Ripple. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.6] - 2026-09-10
10
+
11
+ ### Fixed
12
+ - Thirty-six findings from three rounds of a two-model review of the 0.1.3 to 0.1.5 work: the engine and map fixes each pinned by a test, the page fixes checked with headless browser probes (the page has no automated tests). Rounds two and three also settled INLINE over NULL and non-struct elements, name-matched fields, `AS (xx, yy)` alias lists, maps behind parentheses and casts, a cap before `as_native` parses, kind-prefixed group identities, the loader's own say on which folders are tables, and a cycle twin that stays visible while its other copy is folded. Engine: the generator names (`INLINE`, `EXPLODE`, `POSEXPLODE`) survive schema qualification, which reparsed the original SQL and threw them away; each `INLINE` output now takes lineage from its own struct field, not every field; `EXPLODE` over a map yields `key` and `value`; `as_number` keeps floats and `as_native` reads Python literals as dbt documents. Page: hovering a group heading threw instead of tinting (the group adjacency was walked as arrays); a group or model named `constructor` or `__proto__` crashed the page; a model's badge and a group's reader count included dashboard numbers while the key said models; a column suggestion showed its model's reader count instead of its own; two folders ending in the same name, or a schema and a folder of one name, merged into one row; a monorepo model named `models__marts__a` lost its folder; the filter went stale across focus and home and fought the fold; a late answer could overwrite newer navigation; Escape did not leave focus from the question box; a cycle drew a model on both sides; the typeahead left a stale `aria-activedescendant`.
13
+
14
+ ## [0.1.5] - 2026-09-09
15
+
16
+ ### Added
17
+ - Every column under a model on the page carries the number of models that read it directly, the same number a model chip carries, so the cost of a change shows before the question is asked. A column no link touches (a constant, or one the engine could not trace) is drawn loose, dashed and faint, with "no link found" in the key; r/dataengineering asked for "a clear view of mapped vs unmapped fields (with some kind of legend)", and a traced-looking constant would have been a lie.
18
+
9
19
  ## [0.1.4] - 2026-09-09
10
20
 
11
21
  ### Changed
@@ -249,7 +259,9 @@ First public release.
249
259
  ### Removed
250
260
  - The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
251
261
 
252
- [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.4...HEAD
262
+ [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.6...HEAD
263
+ [0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
264
+ [0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
253
265
  [0.1.4]: https://github.com/bteh/ripple/releases/tag/v0.1.4
254
266
  [0.1.3]: https://github.com/bteh/ripple/releases/tag/v0.1.3
255
267
  [0.1.2]: https://github.com/bteh/ripple/releases/tag/v0.1.2
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ripple-sql
3
- Version: 0.1.4
3
+ Version: 0.1.6
4
4
  Summary: Offline column-level SQL lineage. See what breaks before you merge.
5
5
  Project-URL: Homepage, https://github.com/bteh/ripple
6
6
  Project-URL: Repository, https://github.com/bteh/ripple
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ripple-sql"
3
- version = "0.1.4"
3
+ version = "0.1.6"
4
4
  description = "Offline column-level SQL lineage. See what breaks before you merge."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -0,0 +1,149 @@
1
+ """Bare table-generating functions name their own output columns.
2
+
3
+ Spark's `SELECT INLINE(ARRAY(STRUCT('a' AS platform)))` has no alias and
4
+ no column reference; its output columns are the struct's fields. A bare
5
+ `EXPLODE(x)` is a column called `col`, `POSEXPLODE(x)` is `pos` and `col`,
6
+ and over a map they are `key` and `value`. sqlglot leaves such items
7
+ nameless, and the projection walks named them by their SQL text or dropped
8
+ them, so a lookup table built this way reported no columns at all
9
+ (Wikimedia dbt-jobs, platforms_ephemeral).
10
+
11
+ The rewrite gives each output its default name as an alias over the
12
+ expression that feeds it, before any projection walk runs: a struct field
13
+ feeds its own output only, so a change to one field does not ripple to
14
+ every column of the table.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from sqlglot import exp
20
+
21
+ Output = tuple[str, exp.Expression]
22
+
23
+
24
+ def generator_outputs(item: exp.Expression) -> list[Output] | None:
25
+ """(name, expression) per output column of a bare generator, None for
26
+ anything else. `INLINE(...) AS (xx, yy)` is the same generator with its
27
+ outputs named by position."""
28
+ if isinstance(item, exp.Aliases):
29
+ inner = generator_outputs(item.this)
30
+ names = [alias.name for alias in item.expressions]
31
+ if inner is None or len(names) != len(inner):
32
+ return None
33
+ return [(name, expr) for name, (_, expr) in zip(names, inner, strict=True)]
34
+ if isinstance(item, exp.Inline):
35
+ return _inline_outputs(item)
36
+ if isinstance(item, (exp.Explode, exp.Posexplode)):
37
+ return _explode_outputs(item)
38
+ return None
39
+
40
+
41
+ def _struct_fields(element: exp.Expression) -> list[Output] | None:
42
+ if not isinstance(element, exp.Struct):
43
+ return None
44
+ fields: list[Output] = []
45
+ for field in element.expressions:
46
+ if isinstance(field, exp.PropertyEQ):
47
+ fields.append((field.this.name, field.expression))
48
+ elif isinstance(field, exp.Alias):
49
+ fields.append((field.alias, field.this))
50
+ elif isinstance(field, exp.Column):
51
+ fields.append((field.name, field))
52
+ else:
53
+ return None
54
+ return fields or None
55
+
56
+
57
+ def _inline_outputs(item: exp.Inline) -> list[Output] | None:
58
+ """INLINE over a literal array: the first struct names the outputs and
59
+ field i of every struct feeds output i. That is Spark's own rule: the
60
+ array has one element type, and a later struct must carry the same
61
+ field names at the same positions, case-insensitively, or the query
62
+ fails analysis (TypeCoercion.findTypeForComplex). A struct whose names
63
+ are a permutation of the first's would not run in Spark; it is read by
64
+ name, as written. A NULL element adds nothing, and an element that is
65
+ not a struct literal (a struct column, a CASE) feeds every output, since
66
+ its fields cannot be told apart."""
67
+ array = item.this
68
+ if not (isinstance(array, exp.Array) and array.expressions):
69
+ return None
70
+ elements = [e for e in array.expressions if not isinstance(e, exp.Null)]
71
+ structs = [(e, _struct_fields(e)) for e in elements]
72
+ first = next((fields for _, fields in structs if fields), None)
73
+ if first is None:
74
+ return None
75
+ names = [name for name, _ in first]
76
+ feeds: list[list[exp.Expression]] = [[] for _ in names]
77
+ for element, fields in structs:
78
+ if fields is None:
79
+ for fed_by in feeds:
80
+ fed_by.append(element.copy())
81
+ continue
82
+ matched = _match_fields(first, fields)
83
+ if matched is None:
84
+ return None
85
+ for fed_by, expr in zip(feeds, matched, strict=True):
86
+ fed_by.append(expr.copy())
87
+ return [
88
+ (name, fed_by[0] if len(fed_by) == 1 else exp.Array(expressions=fed_by))
89
+ for name, fed_by in zip(names, feeds, strict=True)
90
+ ]
91
+
92
+
93
+ def _match_fields(first: list[Output], fields: list[Output]) -> list[exp.Expression] | None:
94
+ """The expression of `fields` feeding each output of `first`, in order:
95
+ by position when the names line up (the only shape Spark accepts, and
96
+ the only reading when two unnamed fields share a name, `a.id, b.id`);
97
+ by name when they are a permutation; None when the shapes disagree."""
98
+ if len(fields) != len(first):
99
+ return None
100
+ want = [name.lower() for name, _ in first]
101
+ have = [name.lower() for name, _ in fields]
102
+ if want == have or len(set(have)) != len(have) or set(want) != set(have):
103
+ return [expr for _, expr in fields]
104
+ by_name = {name.lower(): expr for name, expr in fields}
105
+ return [by_name[name] for name in want]
106
+
107
+
108
+ def _is_map(source: exp.Expression) -> bool:
109
+ if isinstance(source, (exp.Map, exp.VarMap)):
110
+ return True
111
+ to = source.args.get("to") if isinstance(source, exp.Cast) else None
112
+ return isinstance(to, exp.DataType) and to.this == exp.DataType.Type.MAP
113
+
114
+
115
+ def _explode_outputs(item: exp.Expression) -> list[Output]:
116
+ """EXPLODE and POSEXPLODE: `col` fed by the whole call over an array;
117
+ `key` and `value` fed by the map's keys and values when the input is
118
+ visibly a map, or both by the whole map when only its type is visible
119
+ (a CAST to MAP). `pos` comes from the whole call either way."""
120
+ source = item.this
121
+ while isinstance(source, exp.Paren):
122
+ source = source.this
123
+ if isinstance(source, (exp.Map, exp.VarMap)):
124
+ outputs: list[Output] = [("key", source.args["keys"]), ("value", source.args["values"])]
125
+ elif _is_map(source):
126
+ outputs = [("key", source), ("value", source)]
127
+ else:
128
+ outputs = [("col", item)]
129
+ if isinstance(item, exp.Posexplode):
130
+ outputs = [("pos", item)] + outputs
131
+ return [(name, expr.copy()) for name, expr in outputs]
132
+
133
+
134
+ def name_generator_outputs(tree: exp.Expression) -> exp.Expression:
135
+ """Alias every bare generator in every select list with its default
136
+ output names, in place, and return the tree."""
137
+ for select in list(tree.find_all(exp.Select)):
138
+ items = select.expressions
139
+ if not any(generator_outputs(item) for item in items):
140
+ continue
141
+ renamed: list[exp.Expression] = []
142
+ for item in items:
143
+ outputs = generator_outputs(item)
144
+ if not outputs:
145
+ renamed.append(item)
146
+ continue
147
+ renamed.extend(exp.alias_(expr, name) for name, expr in outputs)
148
+ select.set("expressions", renamed)
149
+ return tree
@@ -92,9 +92,14 @@ def qualify_sql_with_schema(
92
92
  sql: str,
93
93
  dialect: str,
94
94
  warehouse_columns: WarehouseColumns,
95
+ expression: "exp.Expression | None" = None,
95
96
  ) -> tuple["exp.Expression | None", bool]:
96
97
  """Qualify unqualified columns in SQL using warehouse schema.
97
98
 
99
+ Pass `expression` when the caller has already normalized the tree (bare
100
+ generators named, for one): re-parsing the text here would throw that
101
+ work away, and the caller's names would come back as _col_0.
102
+
98
103
  Uses SQLGlot's optimizer.qualify to add table qualifiers to columns.
99
104
  This resolves ambiguous column references like SELECT col FROM a, b
100
105
  into SELECT a.col FROM a, b when 'col' only exists in table 'a'.
@@ -120,8 +125,13 @@ def qualify_sql_with_schema(
120
125
  return None, False
121
126
 
122
127
  schema_dict = build_sqlglot_schema(warehouse_columns)
128
+
129
+ def fresh() -> "exp.Expression":
130
+ # qualify() mutates, so each attempt starts from its own copy
131
+ return expression.copy() if expression is not None else sqlglot.parse_one(sql, read=dialect)
132
+
123
133
  try:
124
- parsed = sqlglot.parse_one(sql, read=dialect)
134
+ parsed = fresh()
125
135
  qualified = qualify(parsed, schema=schema_dict, dialect=dialect)
126
136
  return qualified, True
127
137
  except Exception as e:
@@ -130,7 +140,7 @@ def qualify_sql_with_schema(
130
140
  # what the schema resolves, leave the rest as written instead of throwing
131
141
  # away every resolved column with it.
132
142
  try:
133
- parsed = sqlglot.parse_one(sql, read=dialect)
143
+ parsed = fresh()
134
144
  qualified = qualify(
135
145
  parsed,
136
146
  schema=schema_dict,
@@ -247,7 +247,7 @@ def extract_column_lineage_fast(
247
247
  # return the AST directly to avoid double-parsing
248
248
  if warehouse_columns:
249
249
  qualified_ast, was_qualified = qualify_sql_with_schema(
250
- cleaned_sql, dialect, warehouse_columns
250
+ cleaned_sql, dialect, warehouse_columns, expression=parsed
251
251
  )
252
252
  if was_qualified and qualified_ast is not None:
253
253
  parsed = qualified_ast
@@ -115,6 +115,9 @@ def model_map(nodes: list[dict], edges) -> dict:
115
115
  it, a dashboard number past a dashboard prefix, and a model otherwise.
116
116
  Star and placeholder columns are links, never columns."""
117
117
  members: dict[str, dict] = {}
118
+ # the loader's identity says whether a folder is the table; kept aside so
119
+ # the page payload does not carry it
120
+ uids = {n["name"]: n.get("id") or "" for n in nodes}
118
121
  for n in nodes:
119
122
  source = n["type"] == "source"
120
123
  columns = [c for c in n.get("columns") or [] if _is_column(c)]
@@ -130,10 +133,19 @@ def model_map(nodes: list[dict], edges) -> dict:
130
133
  }
131
134
  fed = {e.dst_model for e in edges}
132
135
  seen: dict[str, dict[str, None]] = {}
136
+ readers: dict[tuple[str, str], set[str]] = {}
137
+ linked: set[tuple[str, str]] = set()
133
138
  for e in edges:
134
139
  for model, column in ((e.src_model, e.src_column), (e.dst_model, e.dst_column)):
135
140
  if _is_column(column):
136
141
  seen.setdefault(model, {})[column] = None
142
+ if e.kind != "value" or not (_is_column(e.src_column) and _is_column(e.dst_column)):
143
+ continue
144
+ linked.add((e.src_model, e.src_column))
145
+ linked.add((e.dst_model, e.dst_column))
146
+ # a dashboard number is not a reader; only models count (as in starters)
147
+ if not e.dst_model.startswith(DASHBOARD_PREFIXES):
148
+ readers.setdefault((e.src_model, e.src_column), set()).add(e.dst_model)
137
149
  for name in sorted({e.src_model for e in edges} | fed):
138
150
  if name in members:
139
151
  known = members[name]["columns"]
@@ -151,22 +163,33 @@ def model_map(nodes: list[dict], edges) -> dict:
151
163
  "status": None,
152
164
  "path": None,
153
165
  }
166
+ for m in members.values():
167
+ # per column, aligned with `columns`: how many models read it directly;
168
+ # -1 when no link touches it at all (a constant, or one the engine
169
+ # could not trace), so the page never draws such a column as traced
170
+ m["reads"] = [
171
+ len(readers.get((m["id"], c), ())) if (m["id"], c) in linked else -1
172
+ for c in m["columns"]
173
+ ]
154
174
  links = _links(edges)
155
175
  layers = _layers(members, links)
156
176
  models = [{**m, "layer": layers[m["id"]]} for m in members.values()]
157
- _name_groups(models)
177
+ labels = _name_groups(models, uids)
158
178
  ordered = _ordered(models, links)
159
- return {"models": ordered, "links": links, "groups": _groups(ordered, links)}
179
+ return {"models": ordered, "links": links, "groups": _groups(ordered, links, labels)}
160
180
 
161
181
 
162
- def group_of(member: dict) -> str:
163
- """Where a model lives, as a person would say it: the folder of a model,
164
- the schema of a source table, "dashboards" for a dashboard number.
182
+ def group_of(member: dict, uid: str = "", table_folders: frozenset[str] = frozenset()) -> str:
183
+ """Where a model lives, unshortened: the folder of a model, the schema of
184
+ a source table, "dashboards" for a dashboard number.
165
185
 
166
- A file named after its model sits in its folder (`models/marts/orders.sql`
167
- is in `models/marts`). A file with a generic name names its folder as the
168
- table (`telemetry_derived/clients_v1/query.sql`), so the group is the
169
- folder above. A model with no path is in "models"."""
186
+ A model file sits in its folder (`models/marts/orders.sql` is in
187
+ `models/marts`), whatever the model is called: a monorepo may name it
188
+ `models__marts__orders`, and a dbt model may be called query. When the
189
+ loader itself took a folder as the table (`telemetry_derived/clients_v1/
190
+ query.sql`, a schema.yaml sidecar) the group is the folder above, for
191
+ that model and for every other file in the table's folder (a script.sql
192
+ beside the query). A model with no path is in "models"."""
170
193
  if member["kind"] == "dashboard":
171
194
  return "dashboards"
172
195
  if member["kind"] == "source":
@@ -175,55 +198,95 @@ def group_of(member: dict) -> str:
175
198
  path = member.get("path")
176
199
  if not path:
177
200
  return "models"
178
- parts = path.replace("\\", "/").split("/")
179
- stem = parts[-1].rsplit(".", 1)[0]
180
- folders = parts[:-1]
181
- if folders and stem != member["id"]:
201
+ folders = path.replace("\\", "/").split("/")[:-1]
202
+ if table_folder(member, uid) or "/".join(folders).lower() in table_folders:
182
203
  folders = folders[:-1]
183
204
  return "/".join(folders) or "models"
184
205
 
185
206
 
207
+ def table_folder(member: dict, uid: str = "") -> str | None:
208
+ """The folder a model's loader took as the table, lowercased, or None.
209
+
210
+ That happened when the model's name, or the relation in its uid
211
+ (path::relation, possibly dataset.table), is the last folder and the
212
+ file is called something else (query.sql, schema.yaml). A file that is
213
+ the folder's namesake, or any file in a dbt project, keeps its folder."""
214
+ path = member.get("path")
215
+ if not path or member["kind"] != "model":
216
+ return None
217
+ parts = path.replace("\\", "/").split("/")
218
+ folders, stem = parts[:-1], parts[-1].rsplit(".", 1)[0]
219
+ if not folders:
220
+ return None
221
+ folder = folders[-1].lower()
222
+ relation = uid.rpartition("::")[-1].rpartition(".")[-1].lower()
223
+ name = member["id"].lower()
224
+ named_after_folder = folder in (name, relation) or name.endswith("__" + folder)
225
+ if named_after_folder and stem.lower() != folder:
226
+ return "/".join(folders).lower()
227
+ return None
228
+
229
+
186
230
  def _shorten(names: set[str], sep: str) -> dict[str, str]:
187
- """Drop the leading segments every name shares, so `models/staging` and
188
- `models/marts` read as `staging` and `marts`. A lone name keeps itself."""
231
+ """Labels: drop the leading segments every deep name shares, so
232
+ `models/staging` and `models/marts` read as `staging` and `marts`. A lone
233
+ name keeps itself, and `models/core` next to `models/staging/core` keeps
234
+ enough to tell them apart."""
189
235
  parts = {n: n.split(sep) for n in names}
190
236
  while True:
191
237
  deep = [p for p in parts.values() if len(p) > 1]
192
238
  if len(deep) < 2 or len({p[0] for p in deep}) != 1:
193
239
  break
194
- parts = {n: p[1:] if len(p) > 1 else p for n, p in parts.items()}
240
+ nxt = {n: p[1:] if len(p) > 1 else p for n, p in parts.items()}
241
+ # a strip that makes two names read the same stops here
242
+ if len({sep.join(p) for p in nxt.values()}) < len(nxt):
243
+ break
244
+ parts = nxt
195
245
  return {n: sep.join(p) for n, p in parts.items()}
196
246
 
197
247
 
198
248
  SOURCE_SCHEMAS = 6
199
249
 
200
250
 
201
- def _name_groups(models: list[dict]) -> None:
202
- """Set each model's group, shortened per kind, in place. A project that
203
- reads from more than a handful of schemas gets one "sources" group: a
204
- row per schema would be a page of one-table rows, and the chip already
205
- carries the schema in its name."""
206
- raw = {m["id"]: group_of(m) for m in models}
207
- short: dict[str, str] = {}
208
- for kind, sep in (("model", "/"), ("source", ".")):
251
+ def _name_groups(models: list[dict], uids: dict[str, str]) -> dict[str, str]:
252
+ """Set each model's group in place and return the label of every group.
253
+
254
+ The group is an identity, never shown: the kind and the unshortened
255
+ folder or schema (`model:models/marts`, `source:raw`), so a schema and a
256
+ folder of one name, or a folder called dashboards, stay separate rows.
257
+ The label is what the page prints. A project that reads
258
+ from more than a handful of schemas gets one "sources" group: a row per
259
+ schema would be a page of one-table rows, and the chip already carries
260
+ the schema in its name."""
261
+ tables = frozenset(t for m in models if (t := table_folder(m, uids.get(m["id"], ""))))
262
+ raw = {m["id"]: group_of(m, uids.get(m["id"], ""), tables) for m in models}
263
+ labels: dict[str, str] = {}
264
+ for kind, sep in (("model", "/"), ("source", "."), ("dashboard", "/")):
265
+ prefix = kind + ":"
209
266
  names = {raw[m["id"]] for m in models if m["kind"] == kind}
210
267
  if kind == "source" and len(names) > SOURCE_SCHEMAS:
211
- short.update({n: "sources" for n in names})
212
- continue
213
- short.update(_shorten(names, sep))
268
+ for m in models:
269
+ if m["kind"] == kind:
270
+ raw[m["id"]] = "sources"
271
+ names = {"sources"}
272
+ for name, label in _shorten(names, sep).items():
273
+ labels[prefix + name] = label
214
274
  for m in models:
215
- m["group"] = short.get(raw[m["id"]], raw[m["id"]])
275
+ m["group"] = m["kind"] + ":" + raw[m["id"]]
276
+ return labels
216
277
 
217
278
 
218
279
  GROUP_RANK = {"source": 0, "model": 1, "dashboard": 2}
219
280
 
220
281
 
221
- def _groups(models: list[dict], links: list[dict]) -> list[dict]:
282
+ def _groups(models: list[dict], links: list[dict], labels: dict[str, str]) -> list[dict]:
222
283
  """The groups in the order the data flows: source schemas, then model
223
284
  folders by the mean layer of their models, then dashboards. Each carries
224
- how many models it holds and how many models outside it read one of its."""
285
+ its label, how many models it holds and how many models outside it read
286
+ one of its; a dashboard number is not a reader."""
225
287
  by: dict[str, dict] = {}
226
288
  of: dict[str, str] = {}
289
+ kind_of: dict[str, str] = {}
227
290
  for m in models:
228
291
  g = by.setdefault(
229
292
  m["group"],
@@ -231,19 +294,25 @@ def _groups(models: list[dict], links: list[dict]) -> list[dict]:
231
294
  )
232
295
  g["n"] += 1
233
296
  g["layers"].append(m["layer"])
234
- if GROUP_RANK[m["kind"]] > GROUP_RANK[g["kind"]]:
235
- g["kind"] = m["kind"]
236
297
  of[m["id"]] = m["group"]
298
+ kind_of[m["id"]] = m["kind"]
237
299
  for link in links:
238
300
  src, dst = of.get(link["src"]), of.get(link["dst"])
239
- if src and dst and src != dst:
301
+ if src and dst and src != dst and kind_of[link["dst"]] != "dashboard":
240
302
  by[src]["readers"].add(link["dst"])
241
303
  out = sorted(
242
304
  by.values(),
243
305
  key=lambda g: (GROUP_RANK[g["kind"]], sum(g["layers"]) / len(g["layers"]), g["id"]),
244
306
  )
245
307
  return [
246
- {"id": g["id"], "kind": g["kind"], "n": g["n"], "readers": len(g["readers"])} for g in out
308
+ {
309
+ "id": g["id"],
310
+ "label": labels.get(g["id"], g["id"]),
311
+ "kind": g["kind"],
312
+ "n": g["n"],
313
+ "readers": len(g["readers"]),
314
+ }
315
+ for g in out
247
316
  ]
248
317
 
249
318
 
@@ -18,6 +18,7 @@ matters; unknown macros degrade to NULL (parseable, no fake lineage).
18
18
 
19
19
  from __future__ import annotations
20
20
 
21
+ import ast as _ast
21
22
  import datetime as _datetime
22
23
  import itertools as _itertools
23
24
  import re
@@ -288,25 +289,35 @@ def _as_bool(value):
288
289
 
289
290
 
290
291
  def _as_number(value):
292
+ """dbt's as_number: a number stays what it is (1.5 is not 1), a string
293
+ becomes an int or a float, anything else comes back untouched."""
294
+ if isinstance(value, (int, float)) and not isinstance(value, bool):
295
+ return value
296
+ text = str(value).strip()
291
297
  try:
292
- return int(value)
293
- except (TypeError, ValueError):
298
+ return int(text)
299
+ except ValueError:
294
300
  pass
295
301
  try:
296
- return float(value)
297
- except (TypeError, ValueError):
302
+ return float(text)
303
+ except ValueError:
298
304
  return value
299
305
 
300
306
 
307
+ # a literal past this length never reaches the parser: deep enough nesting
308
+ # can overflow the C parser outright, past any except clause, and no dbt
309
+ # var is a tenth of it
310
+ _LITERAL_MAX = 10_000
311
+
312
+
301
313
  def _as_native(value):
302
- """dbt's as_native: a string read back as YAML, anything else untouched."""
303
- if not isinstance(value, str):
314
+ """dbt's as_native: a string read as a Python literal (a tuple, a dict
315
+ with None in it), the string itself when it is not one."""
316
+ if not isinstance(value, str) or len(value) > _LITERAL_MAX:
304
317
  return value
305
318
  try:
306
- import yaml
307
-
308
- return yaml.safe_load(value)
309
- except Exception:
319
+ return _ast.literal_eval(value.strip())
320
+ except (ValueError, SyntaxError, TypeError, MemoryError, RecursionError):
310
321
  return value
311
322
 
312
323
 
@@ -77,6 +77,7 @@ h1 code{font:500 24px/1 var(--mono); letter-spacing:0}
77
77
  .key i.r{border-top-style:dashed; border-color:var(--review)}
78
78
  .key s{display:inline-block; width:11px; height:11px; border-radius:2px; background:var(--ink); border:1px solid var(--ink); text-decoration:none}
79
79
  .key s.src{background:var(--well); border:1px dashed var(--line-strong)}
80
+ .key s.unk{background:transparent; border:1px dashed var(--faint)}
80
81
  .key small.n{font:400 11.5px/1 var(--sans); color:var(--faint); border:1px solid var(--line-strong); border-radius:var(--r); padding:2px 5px; background:var(--surface)}
81
82
  .hint{margin-left:auto; font-size:12.5px; color:var(--faint)}
82
83
  .rows{padding:0 22px 14px; font-size:13.5px; color:var(--muted)}
@@ -110,6 +111,9 @@ pre{margin:0; padding:16px 22px 18px; overflow-x:auto; font:400 13px/1.65 var(--
110
111
  .cols{display:flex; flex-direction:column; gap:5px; margin:0 0 6px 10px; padding-left:10px; border-left:1px solid var(--line)}
111
112
  .cols .gname{max-width:36ch; white-space:normal; line-height:1.45; padding-top:2px}
112
113
  .cols .node{cursor:pointer}
114
+ .cols .node small{font:400 11px/1 var(--sans); color:var(--faint); margin-left:5px}
115
+ .cols .node.unk{color:var(--muted)}
116
+ .cols .node.unk:hover{border-color:var(--accent)}
113
117
  .cols .node:hover{border-color:var(--accent); color:var(--accent-strong)}
114
118
  .more-chips{font:500 12.5px var(--sans); color:var(--accent); background:none; border:0; padding:4px 0; cursor:pointer; text-align:left}
115
119
  .more-chips:hover{text-decoration:underline}
@@ -140,8 +144,8 @@ pre{margin:0; padding:16px 22px 18px; overflow-x:auto; font:400 13px/1.65 var(--
140
144
  .gcell:hover{background:var(--well)}
141
145
  .grow.sel .gcell{background:var(--accent); color:#fff}
142
146
  .grow.sel .gcell small{color:rgba(255,255,255,.72)}
143
- .grow.up .gcell{background:#E4F1F4; box-shadow:inset 3px 0 0 var(--accent)}
144
- .grow.down .gcell{background:var(--well); box-shadow:inset 3px 0 0 var(--ink)}
147
+ .grow.up .gcell{background:#E4F1F4; color:var(--accent-strong)}
148
+ .grow.down .gcell{background:var(--well)}
145
149
  .index .chips{display:flex; flex-wrap:wrap; gap:6px 8px; align-items:center}
146
150
  .index .mdl{max-width:100%; overflow-wrap:anywhere; text-align:left; line-height:1.25}
147
151
  .index .gname{flex-basis:100%}