ripple-sql 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. ripple/__init__.py +31 -0
  2. ripple/answer.py +473 -0
  3. ripple/answer_page.py +214 -0
  4. ripple/cache.py +80 -0
  5. ripple/ci.py +422 -0
  6. ripple/ci_signature.py +374 -0
  7. ripple/cli.py +733 -0
  8. ripple/doctor.py +225 -0
  9. ripple/engine/__init__.py +111 -0
  10. ripple/engine/budget.py +86 -0
  11. ripple/engine/column_lineage.py +112 -0
  12. ripple/engine/column_ref.py +818 -0
  13. ripple/engine/cte_tracing.py +1309 -0
  14. ripple/engine/dependencies.py +466 -0
  15. ripple/engine/dialect.py +132 -0
  16. ripple/engine/dispatch.py +12 -0
  17. ripple/engine/extraction.py +27 -0
  18. ripple/engine/jinja.py +282 -0
  19. ripple/engine/json_sources.py +241 -0
  20. ripple/engine/macro_source.py +127 -0
  21. ripple/engine/pipeline.py +265 -0
  22. ripple/engine/preprocess.py +174 -0
  23. ripple/engine/safe_gen.py +21 -0
  24. ripple/engine/schema_qualification.py +151 -0
  25. ripple/engine/scope.py +488 -0
  26. ripple/engine/select_sources.py +1038 -0
  27. ripple/engine/sql_script.py +729 -0
  28. ripple/engine/statement.py +449 -0
  29. ripple/engine/tech_debt.py +169 -0
  30. ripple/engine/tsql_catalog.py +83 -0
  31. ripple/engine/tsql_scalar_vars.py +248 -0
  32. ripple/engine/tsql_tvf.py +653 -0
  33. ripple/engine/tsql_xml.py +97 -0
  34. ripple/engine/types.py +167 -0
  35. ripple/engine/unused_deps.py +555 -0
  36. ripple/engine/validation.py +158 -0
  37. ripple/graph.py +1499 -0
  38. ripple/home.py +232 -0
  39. ripple/loaders/__init__.py +7 -0
  40. ripple/loaders/dbt.py +359 -0
  41. ripple/loaders/dbt_config.py +339 -0
  42. ripple/loaders/identity.py +328 -0
  43. ripple/loaders/sidecar.py +65 -0
  44. ripple/loaders/sqldir.py +262 -0
  45. ripple/loaders/types.py +197 -0
  46. ripple/lookml.py +163 -0
  47. ripple/mcp_server.py +600 -0
  48. ripple/names.py +40 -0
  49. ripple/project.py +167 -0
  50. ripple/py.typed +0 -0
  51. ripple/render.py +426 -0
  52. ripple/render_shims.py +209 -0
  53. ripple/schemas.py +155 -0
  54. ripple/semantic.py +232 -0
  55. ripple/server.py +184 -0
  56. ripple/sourcefiles.py +64 -0
  57. ripple/star_resolution.py +100 -0
  58. ripple/static/answer.css +146 -0
  59. ripple/static/answer.html +358 -0
  60. ripple/static/answer_twin.js +299 -0
  61. ripple/static/explore.js +133 -0
  62. ripple/usage/__init__.py +18 -0
  63. ripple/usage/cli.py +78 -0
  64. ripple/usage/collect.py +315 -0
  65. ripple/usage/discover.py +190 -0
  66. ripple/usage/ingest.py +414 -0
  67. ripple/usage/report.py +131 -0
  68. ripple_sql-0.1.0.dist-info/METADATA +285 -0
  69. ripple_sql-0.1.0.dist-info/RECORD +72 -0
  70. ripple_sql-0.1.0.dist-info/WHEEL +4 -0
  71. ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
  72. ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,97 @@
1
+ """T-SQL XML method calls, normalized for lineage.
2
+
3
+ col.value('xpath', 'type'), col.query(...), col.exist(...) and
4
+ col.nodes('xpath') parse in sqlglot's tsql as a Dot chain ending in an
5
+ Anonymous call whose base is bare Identifiers, so the column sweep sees
6
+ no Column at all and a .nodes() rowset never reaches the expansion
7
+ machinery (sqlserver_kit DarkQueries and InvestigateWaits, holdout
8
+ round 8). Each such Dot is rewritten into an ordinary call with the base
9
+ column as the first argument: value(col, 'xpath', 'type'). The produced
10
+ value derives from the shredded column exactly like any function
11
+ argument, and .nodes() in a lateral becomes a set-returning function the
12
+ scope machinery already expands.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+
19
+ from sqlglot import exp
20
+
21
+ XML_METHODS = frozenset({"value", "query", "exist", "modify", "nodes"})
22
+
23
+ # sql:column("t.id") exposes a relational column inside an XQuery literal.
24
+ # Conservative on purpose: double-quoted, plain dotted identifiers only
25
+ # (brackets stripped); anything fancier stays an opaque literal.
26
+ _SQL_COLUMN_REF = re.compile(r'sql:column\s*\(\s*"([^"]+)"\s*\)', re.IGNORECASE)
27
+
28
+
29
+ def rewrite_xml_method_calls(parsed: exp.Expression) -> exp.Expression:
30
+ """Rewrite every XML-method Dot in the tree, outermost first."""
31
+ for dot in list(parsed.find_all(exp.Dot)):
32
+ method = dot.expression
33
+ if not isinstance(method, exp.Anonymous):
34
+ continue
35
+ if method.name.lower() not in XML_METHODS:
36
+ continue
37
+ base = _base_column(dot.this)
38
+ if base is None:
39
+ continue
40
+ args = [e.copy() for e in method.expressions or []]
41
+ dot.replace(
42
+ exp.Anonymous(
43
+ this=method.name.lower(),
44
+ expressions=[base, *args, *_embedded_column_reads(args)],
45
+ )
46
+ )
47
+ return parsed
48
+
49
+
50
+ def _embedded_column_reads(args: list[exp.Expression]) -> list[exp.Column]:
51
+ """Columns the XQuery string arguments read through sql:column(), each
52
+ emitted as an ordinary argument of the rewritten call (cycle-9
53
+ review, F7)."""
54
+ columns: list[exp.Column] = []
55
+ for arg in args:
56
+ if not isinstance(arg, exp.Literal) or not arg.is_string:
57
+ continue
58
+ for ref in _SQL_COLUMN_REF.findall(arg.this):
59
+ parts = [p.strip('[]" ') for p in ref.split(".")]
60
+ parts = [p for p in parts if p]
61
+ if not parts or len(parts) > 4:
62
+ continue
63
+ slots = ("this", "table", "db", "catalog")
64
+ columns.append(
65
+ exp.Column(
66
+ **{
67
+ slot: exp.Identifier(this=part, quoted=False)
68
+ for slot, part in zip(slots, reversed(parts), strict=False)
69
+ }
70
+ )
71
+ )
72
+ return columns
73
+
74
+
75
+ def _base_column(node: exp.Expression) -> exp.Column | None:
76
+ """The method's receiver as a Column, or None when it is not a plain
77
+ (optionally qualified) column reference."""
78
+ if isinstance(node, exp.Column):
79
+ return node.copy()
80
+ parts = _identifier_parts(node)
81
+ if not parts or len(parts) > 4:
82
+ return None
83
+ slots = ("this", "table", "db", "catalog")
84
+ return exp.Column(
85
+ **{slot: ident.copy() for slot, ident in zip(slots, reversed(parts), strict=False)}
86
+ )
87
+
88
+
89
+ def _identifier_parts(node: exp.Expression) -> list[exp.Identifier] | None:
90
+ if isinstance(node, exp.Identifier):
91
+ return [node]
92
+ if isinstance(node, exp.Dot):
93
+ left = _identifier_parts(node.this)
94
+ right = _identifier_parts(node.expression)
95
+ if left is not None and right is not None:
96
+ return left + right
97
+ return None
ripple/engine/types.py ADDED
@@ -0,0 +1,167 @@
1
+ """Column lineage type definitions.
2
+
3
+ This module defines the core types used across the column lineage system.
4
+ Centralizing types here enables clean imports and avoids circular dependencies.
5
+
6
+ Type Categories:
7
+ - Trust levels: Confidence indicators for lineage correctness
8
+ - Data structures: Dataclasses for column dependencies and lineage results
9
+ - Type aliases: Common type patterns for warehouse schemas
10
+ """
11
+
12
+ from dataclasses import dataclass, field
13
+ from typing import Any, Literal
14
+
15
+ # Type aliases for clarity
16
+ TrustLevel = Literal["verified", "high_confidence", "moderate", "review_required"]
17
+ """Trust level indicating lineage confidence.
18
+
19
+ - verified: 1.0 confidence - warehouse-qualified or exact match
20
+ - high_confidence: 0.8-0.99 - strong evidence
21
+ - moderate: 0.5-0.79 - heuristic match
22
+ - review_required: <0.5 - ambiguity or collision detected
23
+ """
24
+
25
+ WarehouseColumns = dict[str, dict[str, str]]
26
+ """Warehouse schema format: {table_name: {column_name: data_type}}"""
27
+
28
+
29
+ # Size limits for DoS protection
30
+ MAX_WAREHOUSE_TABLES = 10000 # arrives from outside via ingest_schema
31
+ # real_tables is derived from the caller's own repo, after Ripple has already
32
+ # read every one of those files off disk, so a low cap guards nothing and
33
+ # turns a big monorepo into silence: Dune's spellbook needs 11,962 and used to
34
+ # fail all 7,439 models. Kept as a runaway guard, set far above any real repo.
35
+ MAX_REAL_TABLES = 1_000_000
36
+ MAX_SQL_LENGTH = 5_000_000 # 5MB max SQL size
37
+
38
+ # Internal metadata keys carried inside extraction result dicts. The NUL
39
+ # prefix cannot appear in a SQL identifier, so a real output column named
40
+ # _name (or even _warnings) can never collide with engine metadata.
41
+ META_PREFIX = "\x00ripple:"
42
+ META_WARNINGS = META_PREFIX + "warnings"
43
+ META_EXTRACTION_ERROR = META_PREFIX + "extraction_error"
44
+ META_EXTRACTION_ERROR_TYPE = META_PREFIX + "extraction_error_type"
45
+ META_QUALIFIED_VIA_SCHEMA = META_PREFIX + "qualified_via_schema"
46
+ META_FALLBACK_DIALECT = META_PREFIX + "fallback_dialect"
47
+ META_NEEDS_EXPANSION = META_PREFIX + "needs_expansion"
48
+ META_UNION_SOURCES = META_PREFIX + "union_sources"
49
+ META_UNION_VALIDATION = META_PREFIX + "union_validation"
50
+ META_CTE_COLLISIONS = META_PREFIX + "cte_collisions"
51
+ # Per-CTE metadata carried inside a CTE's own column map (cte_columns[name]):
52
+ # the relations its FROM scope reads, whether the name is defined more than
53
+ # once in the statement (nested WITH shadowing), and whether it is a
54
+ # recursive CTE. Consumers iterating column names must skip META_PREFIX keys.
55
+ META_CTE_RELATIONS = META_PREFIX + "cte_relations"
56
+ META_CTE_SHADOWED = META_PREFIX + "cte_shadowed"
57
+ META_CTE_RECURSIVE = META_PREFIX + "cte_recursive"
58
+
59
+ # Dialects where a select item may reference an alias defined earlier in the
60
+ # same select list (lateral column aliases). In these dialects a bare
61
+ # reference matching an earlier alias resolves to that alias's expression,
62
+ # not to a column of the upstream relation.
63
+ LATERAL_ALIAS_DIALECTS = frozenset(
64
+ {"snowflake", "redshift", "databricks", "spark", "spark2", "duckdb", "clickhouse", "teradata"}
65
+ )
66
+
67
+
68
+ @dataclass
69
+ class FilterDependency:
70
+ """A column used in WHERE, HAVING, or QUALIFY clauses.
71
+
72
+ These are C_ref columns (referenced but not contributing to output)
73
+ following the LineageX paper terminology.
74
+
75
+ Attributes:
76
+ column: Column name used in the filter
77
+ table: Source table (None if unqualified)
78
+ predicate_type: Type of filter clause (where, having, qualify, on,
79
+ window for a PARTITION BY / ORDER BY key)
80
+ operator: Comparison operator if detectable (=, IN, BETWEEN, etc.)
81
+ """
82
+
83
+ column: str
84
+ table: str | None
85
+ predicate_type: Literal["where", "having", "qualify", "on", "window"]
86
+ operator: str | None = None
87
+
88
+
89
+ @dataclass
90
+ class JoinKeyPair:
91
+ """A pair of columns in a JOIN condition.
92
+
93
+ Tracks the relationship between columns used in JOIN ON clauses,
94
+ which is critical for understanding data cardinality and integrity.
95
+
96
+ Attributes:
97
+ left_column: Column from the left side of the join
98
+ left_table: Table/alias for the left column
99
+ right_column: Column from the right side of the join
100
+ right_table: Table/alias for the right column
101
+ join_type: Type of join (inner, left, right, full, cross)
102
+ """
103
+
104
+ left_column: str
105
+ left_table: str
106
+ right_column: str
107
+ right_table: str
108
+ join_type: Literal["inner", "left", "right", "full", "cross"]
109
+
110
+
111
+ @dataclass
112
+ class ColumnDependencies:
113
+ """C_ref: Columns that affect row selection but not output values.
114
+
115
+ This complements C_con (contributing columns) to provide complete
116
+ lineage following the LineageX model.
117
+
118
+ Attributes:
119
+ filter_columns: Columns used in WHERE/HAVING/QUALIFY
120
+ join_keys: Column pairs used in JOIN ON conditions
121
+ window_keys: Columns in a window's PARTITION BY / ORDER BY
122
+ warnings: Any issues encountered during extraction
123
+ """
124
+
125
+ filter_columns: list[FilterDependency] = field(default_factory=list)
126
+ join_keys: list[JoinKeyPair] = field(default_factory=list)
127
+ window_keys: list[FilterDependency] = field(default_factory=list)
128
+ warnings: list[str] = field(default_factory=list)
129
+
130
+
131
+ @dataclass
132
+ class UnifiedLineageResult:
133
+ """Complete lineage following the LineageX C_con + C_ref model.
134
+
135
+ Carries both:
136
+ - C_con: Columns that contribute to output values (SELECT sources)
137
+ - C_ref: Columns referenced in WHERE/JOIN (affect row selection)
138
+
139
+ Attributes:
140
+ contributing: Dict mapping output columns to their sources
141
+ filter_columns: Columns used in WHERE/HAVING/QUALIFY
142
+ join_keys: Column pairs used in JOIN conditions
143
+ window_keys: Columns in a window's PARTITION BY / ORDER BY
144
+ warnings: Any issues encountered during extraction
145
+ qualified_via_schema: True if warehouse schema was used
146
+ """
147
+
148
+ contributing: dict[str, list[dict[str, Any]]]
149
+ filter_columns: list[FilterDependency] = field(default_factory=list)
150
+ join_keys: list[JoinKeyPair] = field(default_factory=list)
151
+ window_keys: list[FilterDependency] = field(default_factory=list)
152
+ warnings: list[dict[str, Any]] = field(default_factory=list)
153
+ qualified_via_schema: bool = False
154
+
155
+
156
+ @dataclass
157
+ class CTEColumnMapping:
158
+ """Column mapping for a single CTE.
159
+
160
+ Tracks which columns a CTE produces and their sources,
161
+ enabling recursive tracing through nested CTEs.
162
+ """
163
+
164
+ cte_name: str
165
+ columns: dict[str, list[dict[str, Any]]]
166
+ has_star: bool = False
167
+ is_union: bool = False