ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,83 @@
1
+ """Documented output columns of SQL Server system objects.
2
+
3
+ The sys schema is platform surface, not user warehouse: its DMVs and
4
+ table-valued functions have fixed, documented columns, so an unqualified
5
+ read across joined sys objects can be resolved offline the same way an
6
+ ingested warehouse schema would resolve it (InvestigateWaits joins
7
+ sys.dm_xe_session_targets to sys.dm_xe_sessions and reads target_data
8
+ bare; sqlserver_kit, holdout round 8).
9
+
10
+ Special-case area, one object per rule, each pinned by a test naming the
11
+ motivating case. At about a dozen objects, replace with a generated
12
+ catalog instead of growing this map.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ _SYS_OBJECT_COLUMNS: dict[str, frozenset[str]] = {
18
+ "sys.dm_xe_sessions": frozenset(
19
+ {
20
+ "address",
21
+ "name",
22
+ "pending_buffers",
23
+ "total_regular_buffers",
24
+ "regular_buffer_size",
25
+ "total_large_buffers",
26
+ "large_buffer_size",
27
+ "total_buffer_size",
28
+ "buffer_policy_flags",
29
+ "buffer_policy_desc",
30
+ "flags",
31
+ "flag_desc",
32
+ "dropped_event_count",
33
+ "dropped_buffer_count",
34
+ "blocked_event_fire_time",
35
+ "create_time",
36
+ "largest_event_dropped_size",
37
+ "session_source",
38
+ }
39
+ ),
40
+ "sys.dm_xe_session_targets": frozenset(
41
+ {
42
+ "event_session_address",
43
+ "target_name",
44
+ "target_package_guid",
45
+ "execution_count",
46
+ "execution_duration_ms",
47
+ "target_data",
48
+ "bytes_written",
49
+ }
50
+ ),
51
+ "sys.fn_xe_file_target_read_file": frozenset(
52
+ {
53
+ "module_guid",
54
+ "package_guid",
55
+ "object_name",
56
+ "event_data",
57
+ "file_name",
58
+ "file_offset",
59
+ "timestamp_utc",
60
+ }
61
+ ),
62
+ }
63
+
64
+
65
+ def system_catalog_owner(column: str, candidates: list[str], dialect: str) -> str | None:
66
+ """The one candidate relation whose documented columns include column.
67
+
68
+ Fires only when EVERY candidate is a documented sys object: a single
69
+ unknown relation makes ownership unprovable and the ambiguous refusal
70
+ stands. The catalog is SQL Server platform surface, so any other
71
+ dialect gets None (cycle-9 review, F1).
72
+ """
73
+ if (dialect or "").lower() != "tsql":
74
+ return None
75
+ col = (column or "").lower()
76
+ owners: list[str] = []
77
+ for name in dict.fromkeys(candidates):
78
+ columns = _SYS_OBJECT_COLUMNS.get((name or "").lower())
79
+ if columns is None:
80
+ return None
81
+ if col in columns:
82
+ owners.append(name)
83
+ return owners[0] if len(owners) == 1 else None
@@ -0,0 +1,248 @@
1
+ """Order-aware T-SQL scalar variables in plain scripts.
2
+
3
+ A variable used in an INSERT (a VALUES cell or a SELECT item, alone or
4
+ inside a larger expression) traces to its single reaching definition:
5
+ the last unconditional assignment before the use, with no conditional
6
+ assignment after it. Anything else refuses: no assignment yet (the
7
+ variable is still NULL), an IF-branch assignment that may or may not
8
+ have run, an opaque SET, or a self-referential loop. SELECT @v = expr,
9
+ UPDATE ... SET @v = expr, and DECLARE @v = default are all assignment
10
+ sites; variable-to-variable definitions resolve recursively with a
11
+ cycle guard; multi-row VALUES unions its rows so every row's source is
12
+ cited (ceds_generate metadata scripts, holdout round 12; cycle-13
13
+ review, F6/F7/F12/F13/F14).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+
20
+ from sqlglot import exp
21
+
22
+ from ripple.engine.safe_gen import safe_sql
23
+
24
+ _OPAQUE_ASSIGN = re.compile(r"@([\w$]+)\s*[-+*/%&|^]?=")
25
+
26
+
27
+ class _Def:
28
+ __slots__ = ("pos", "value", "ctx", "conditional")
29
+
30
+ def __init__(self, pos, value, ctx, conditional):
31
+ self.pos = pos
32
+ self.value = value
33
+ self.ctx = ctx
34
+ self.conditional = conditional
35
+
36
+
37
+ def _is_conditional(stmt: exp.Expression) -> bool:
38
+ meta = getattr(stmt, "_meta", None)
39
+ return bool(meta and meta.get("ripple_conditional"))
40
+
41
+
42
+ def _update_ctx(stmt: exp.Update) -> exp.Select:
43
+ """A FROM scope for expressions assigned inside an UPDATE, so their
44
+ column reads stay anchored the way the SET clause saw them."""
45
+ sel = exp.Select()
46
+ upd_from = stmt.args.get("from_") or stmt.args.get("from")
47
+ if upd_from is not None:
48
+ sel.set("from_", upd_from.copy())
49
+ elif isinstance(stmt.this, exp.Table):
50
+ sel.set("from_", exp.From(this=stmt.this.copy()))
51
+ return sel
52
+
53
+
54
+ class _ScalarDefs:
55
+ def __init__(self, statements: list[exp.Expression], dialect: str | None):
56
+ self.defs: dict[str, list[_Def]] = {}
57
+ for pos, stmt in enumerate(statements):
58
+ cond = _is_conditional(stmt)
59
+ if isinstance(stmt, exp.Select):
60
+ eqs = [
61
+ e
62
+ for e in stmt.expressions
63
+ if isinstance(e, exp.EQ) and isinstance(e.this, exp.Parameter)
64
+ ]
65
+ if not eqs or len(eqs) != len(stmt.expressions):
66
+ continue
67
+ for eq in eqs:
68
+ self._add(pos, eq.this.name, eq.expression, stmt, cond)
69
+ elif isinstance(stmt, exp.Update):
70
+ for eq in stmt.expressions:
71
+ if (
72
+ isinstance(eq, exp.EQ)
73
+ and isinstance(eq.this, exp.Parameter)
74
+ and eq.expression is not None
75
+ ):
76
+ self._add(pos, eq.this.name, eq.expression, _update_ctx(stmt), cond)
77
+ elif isinstance(stmt, exp.Declare):
78
+ for item in stmt.expressions:
79
+ if not isinstance(item, exp.DeclareItem):
80
+ continue
81
+ default = item.args.get("default")
82
+ if not isinstance(default, exp.Expression):
83
+ continue
84
+ for param in item.this if isinstance(item.this, list) else [item.this]:
85
+ if isinstance(param, exp.Parameter) and param.name:
86
+ self._add(pos, param.name, default, None, cond)
87
+ elif isinstance(stmt, (exp.Set, exp.Command)):
88
+ text = safe_sql(stmt, dialect) or ""
89
+ for m in _OPAQUE_ASSIGN.finditer(text):
90
+ self._add(pos, m.group(1), None, None, cond)
91
+
92
+ def _add(self, pos, var, value, ctx, conditional) -> None:
93
+ self.defs.setdefault(var.lower(), []).append(_Def(pos, value, ctx, conditional))
94
+
95
+ def resolve(self, var: str, use_pos: int, seen: frozenset[str]) -> exp.Expression | None:
96
+ """The variable's value at use_pos as a self-contained expression
97
+ (typically a scalar subquery), or None when no single reaching
98
+ definition exists."""
99
+ var = var.lower()
100
+ if var in seen:
101
+ return None
102
+ before = [d for d in self.defs.get(var, []) if d.pos < use_pos]
103
+ if not before:
104
+ return None
105
+ last_uncond = None
106
+ for idx, d in enumerate(before):
107
+ if not d.conditional:
108
+ last_uncond = idx
109
+ # reaching definitions: the last unconditional one plus any
110
+ # conditional ones after it; several possibly-live definitions
111
+ # (or none that surely ran) refuse rather than guess
112
+ if last_uncond is None or len(before) - last_uncond != 1:
113
+ return None
114
+ d = before[last_uncond]
115
+ if d.value is None:
116
+ return None
117
+ return self._anchored(d, seen | {var})
118
+
119
+ def _anchored(self, d: _Def, seen: frozenset[str]) -> exp.Expression | None:
120
+ value = d.value.copy()
121
+ if isinstance(value, exp.Parameter):
122
+ return self.resolve(value.name, d.pos, seen)
123
+ for param in list(value.find_all(exp.Parameter)):
124
+ sub = self.resolve(param.name, d.pos, seen)
125
+ if sub is None:
126
+ return None
127
+ param.replace(sub)
128
+ ctx = d.ctx
129
+ if ctx is not None and (ctx.args.get("from_") or ctx.args.get("from")):
130
+ inner = ctx.copy()
131
+ inner.set("expressions", [value])
132
+ return exp.Subquery(this=inner)
133
+ for col in value.find_all(exp.Column):
134
+ owner = _owning_select_within(col, value)
135
+ if owner is None or not (owner.args.get("from_") or owner.args.get("from")):
136
+ return None # a bare column with no scope to anchor it
137
+ return value
138
+
139
+
140
+ def _owning_select_within(col: exp.Expression, root: exp.Expression) -> exp.Select | None:
141
+ node = col.parent
142
+ while node is not None:
143
+ if isinstance(node, exp.Select):
144
+ return node
145
+ if node is root:
146
+ return None
147
+ node = node.parent
148
+ return None
149
+
150
+
151
+ def _carries_sources(node: exp.Expression) -> bool:
152
+ return next(node.find_all(exp.Column), None) is not None
153
+
154
+
155
+ def _substituted_cell(
156
+ cell: exp.Expression, defs: _ScalarDefs, use_pos: int
157
+ ) -> tuple[exp.Expression, bool]:
158
+ """The cell with each resolvable variable replaced by its reaching
159
+ definition's value; True when a replacement carries real sources."""
160
+ if isinstance(cell, exp.Parameter):
161
+ sub = defs.resolve(cell.name, use_pos, frozenset())
162
+ if sub is None:
163
+ return cell.copy(), False
164
+ return sub, _carries_sources(sub)
165
+ out = cell.copy()
166
+ traced = False
167
+ for param in list(out.find_all(exp.Parameter)):
168
+ sub = defs.resolve(param.name, use_pos, frozenset())
169
+ if sub is None:
170
+ continue
171
+ param.replace(sub)
172
+ traced = traced or _carries_sources(sub)
173
+ return out, traced
174
+
175
+
176
+ def inline_scalar_params(statements: list[exp.Expression], dialect: str | None) -> None:
177
+ """Replace resolvable scalar variables inside INSERT ... SELECT items
178
+ in place, so the generic INSERT path traces the variable's sources
179
+ through whatever expression surrounds it (cycle-13 review, F13)."""
180
+ defs = _ScalarDefs(statements, dialect)
181
+ if not defs.defs:
182
+ return
183
+ for pos, stmt in enumerate(statements):
184
+ if not isinstance(stmt, exp.Insert):
185
+ continue
186
+ inner = stmt.expression
187
+ while isinstance(inner, (exp.Subquery, exp.Paren)):
188
+ inner = inner.this
189
+ if not isinstance(inner, exp.Select):
190
+ continue
191
+ for item in list(inner.expressions):
192
+ for param in list(item.find_all(exp.Parameter)):
193
+ sub = defs.resolve(param.name, pos, frozenset())
194
+ if sub is not None:
195
+ param.replace(sub)
196
+
197
+
198
+ def scalar_value_insert_models(statements: list[exp.Expression], dialect: str | None):
199
+ """ScriptModels for INSERT ... VALUES targets whose rows carry
200
+ resolvable scalar variables. Each row becomes one union branch with
201
+ every cell kept, so multi-row inserts cite every row's sources
202
+ instead of last-row-wins (cycle-13 review, F12)."""
203
+ from ripple.engine.column_ref import qualified_table_name
204
+ from ripple.engine.sql_script import ScriptModel, _referenced_tables
205
+
206
+ defs = _ScalarDefs(statements, dialect)
207
+ if not defs.defs:
208
+ return []
209
+ models: list[ScriptModel] = []
210
+ for pos, stmt in enumerate(statements):
211
+ if not isinstance(stmt, exp.Insert) or not isinstance(stmt.expression, exp.Values):
212
+ continue
213
+ schema = stmt.this
214
+ if not isinstance(schema, exp.Schema) or not isinstance(schema.this, exp.Table):
215
+ continue
216
+ if isinstance(schema.this.this, exp.Parameter):
217
+ continue
218
+ idents = schema.expressions
219
+ if not idents or not all(isinstance(c, exp.Identifier) for c in idents):
220
+ continue
221
+ branches: list[exp.Select] = []
222
+ for row in stmt.expression.expressions:
223
+ if not isinstance(row, exp.Tuple) or len(row.expressions) != len(idents):
224
+ continue
225
+ items: list[exp.Alias] = []
226
+ traced = False
227
+ for ident, cell in zip(idents, row.expressions, strict=True):
228
+ new_cell, cell_traced = _substituted_cell(cell, defs, pos)
229
+ traced = traced or cell_traced
230
+ items.append(exp.Alias(this=new_cell, alias=ident.copy()))
231
+ if traced:
232
+ branches.append(exp.Select(expressions=items))
233
+ if not branches:
234
+ continue
235
+ combined: exp.Expression = branches[0]
236
+ for branch in branches[1:]:
237
+ combined = exp.union(combined, branch, distinct=False, copy=False)
238
+ derived_sql = safe_sql(combined, dialect)
239
+ if derived_sql is None:
240
+ continue
241
+ models.append(
242
+ ScriptModel(
243
+ name=qualified_table_name(schema.this) or schema.this.name,
244
+ sql=derived_sql,
245
+ parents=_referenced_tables(combined),
246
+ )
247
+ )
248
+ return models