ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""T-SQL XML method calls, normalized for lineage.
|
|
2
|
+
|
|
3
|
+
col.value('xpath', 'type'), col.query(...), col.exist(...) and
|
|
4
|
+
col.nodes('xpath') parse in sqlglot's tsql as a Dot chain ending in an
|
|
5
|
+
Anonymous call whose base is bare Identifiers, so the column sweep sees
|
|
6
|
+
no Column at all and a .nodes() rowset never reaches the expansion
|
|
7
|
+
machinery (sqlserver_kit DarkQueries and InvestigateWaits, holdout
|
|
8
|
+
round 8). Each such Dot is rewritten into an ordinary call with the base
|
|
9
|
+
column as the first argument: value(col, 'xpath', 'type'). The produced
|
|
10
|
+
value derives from the shredded column exactly like any function
|
|
11
|
+
argument, and .nodes() in a lateral becomes a set-returning function the
|
|
12
|
+
scope machinery already expands.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
from sqlglot import exp
|
|
20
|
+
|
|
21
|
+
XML_METHODS = frozenset({"value", "query", "exist", "modify", "nodes"})
|
|
22
|
+
|
|
23
|
+
# sql:column("t.id") exposes a relational column inside an XQuery literal.
|
|
24
|
+
# Conservative on purpose: double-quoted, plain dotted identifiers only
|
|
25
|
+
# (brackets stripped); anything fancier stays an opaque literal.
|
|
26
|
+
_SQL_COLUMN_REF = re.compile(r'sql:column\s*\(\s*"([^"]+)"\s*\)', re.IGNORECASE)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def rewrite_xml_method_calls(parsed: exp.Expression) -> exp.Expression:
|
|
30
|
+
"""Rewrite every XML-method Dot in the tree, outermost first."""
|
|
31
|
+
for dot in list(parsed.find_all(exp.Dot)):
|
|
32
|
+
method = dot.expression
|
|
33
|
+
if not isinstance(method, exp.Anonymous):
|
|
34
|
+
continue
|
|
35
|
+
if method.name.lower() not in XML_METHODS:
|
|
36
|
+
continue
|
|
37
|
+
base = _base_column(dot.this)
|
|
38
|
+
if base is None:
|
|
39
|
+
continue
|
|
40
|
+
args = [e.copy() for e in method.expressions or []]
|
|
41
|
+
dot.replace(
|
|
42
|
+
exp.Anonymous(
|
|
43
|
+
this=method.name.lower(),
|
|
44
|
+
expressions=[base, *args, *_embedded_column_reads(args)],
|
|
45
|
+
)
|
|
46
|
+
)
|
|
47
|
+
return parsed
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _embedded_column_reads(args: list[exp.Expression]) -> list[exp.Column]:
|
|
51
|
+
"""Columns the XQuery string arguments read through sql:column(), each
|
|
52
|
+
emitted as an ordinary argument of the rewritten call (cycle-9
|
|
53
|
+
review, F7)."""
|
|
54
|
+
columns: list[exp.Column] = []
|
|
55
|
+
for arg in args:
|
|
56
|
+
if not isinstance(arg, exp.Literal) or not arg.is_string:
|
|
57
|
+
continue
|
|
58
|
+
for ref in _SQL_COLUMN_REF.findall(arg.this):
|
|
59
|
+
parts = [p.strip('[]" ') for p in ref.split(".")]
|
|
60
|
+
parts = [p for p in parts if p]
|
|
61
|
+
if not parts or len(parts) > 4:
|
|
62
|
+
continue
|
|
63
|
+
slots = ("this", "table", "db", "catalog")
|
|
64
|
+
columns.append(
|
|
65
|
+
exp.Column(
|
|
66
|
+
**{
|
|
67
|
+
slot: exp.Identifier(this=part, quoted=False)
|
|
68
|
+
for slot, part in zip(slots, reversed(parts), strict=False)
|
|
69
|
+
}
|
|
70
|
+
)
|
|
71
|
+
)
|
|
72
|
+
return columns
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _base_column(node: exp.Expression) -> exp.Column | None:
|
|
76
|
+
"""The method's receiver as a Column, or None when it is not a plain
|
|
77
|
+
(optionally qualified) column reference."""
|
|
78
|
+
if isinstance(node, exp.Column):
|
|
79
|
+
return node.copy()
|
|
80
|
+
parts = _identifier_parts(node)
|
|
81
|
+
if not parts or len(parts) > 4:
|
|
82
|
+
return None
|
|
83
|
+
slots = ("this", "table", "db", "catalog")
|
|
84
|
+
return exp.Column(
|
|
85
|
+
**{slot: ident.copy() for slot, ident in zip(slots, reversed(parts), strict=False)}
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _identifier_parts(node: exp.Expression) -> list[exp.Identifier] | None:
|
|
90
|
+
if isinstance(node, exp.Identifier):
|
|
91
|
+
return [node]
|
|
92
|
+
if isinstance(node, exp.Dot):
|
|
93
|
+
left = _identifier_parts(node.this)
|
|
94
|
+
right = _identifier_parts(node.expression)
|
|
95
|
+
if left is not None and right is not None:
|
|
96
|
+
return left + right
|
|
97
|
+
return None
|
ripple/engine/types.py
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Column lineage type definitions.
|
|
2
|
+
|
|
3
|
+
This module defines the core types used across the column lineage system.
|
|
4
|
+
Centralizing types here enables clean imports and avoids circular dependencies.
|
|
5
|
+
|
|
6
|
+
Type Categories:
|
|
7
|
+
- Trust levels: Confidence indicators for lineage correctness
|
|
8
|
+
- Data structures: Dataclasses for column dependencies and lineage results
|
|
9
|
+
- Type aliases: Common type patterns for warehouse schemas
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Any, Literal
|
|
14
|
+
|
|
15
|
+
# Type aliases for clarity
|
|
16
|
+
TrustLevel = Literal["verified", "high_confidence", "moderate", "review_required"]
|
|
17
|
+
"""Trust level indicating lineage confidence.
|
|
18
|
+
|
|
19
|
+
- verified: 1.0 confidence - warehouse-qualified or exact match
|
|
20
|
+
- high_confidence: 0.8-0.99 - strong evidence
|
|
21
|
+
- moderate: 0.5-0.79 - heuristic match
|
|
22
|
+
- review_required: <0.5 - ambiguity or collision detected
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
WarehouseColumns = dict[str, dict[str, str]]
|
|
26
|
+
"""Warehouse schema format: {table_name: {column_name: data_type}}"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Size limits for DoS protection
|
|
30
|
+
MAX_WAREHOUSE_TABLES = 10000 # arrives from outside via ingest_schema
|
|
31
|
+
# real_tables is derived from the caller's own repo, after Ripple has already
|
|
32
|
+
# read every one of those files off disk, so a low cap guards nothing and
|
|
33
|
+
# turns a big monorepo into silence: Dune's spellbook needs 11,962 and used to
|
|
34
|
+
# fail all 7,439 models. Kept as a runaway guard, set far above any real repo.
|
|
35
|
+
MAX_REAL_TABLES = 1_000_000
|
|
36
|
+
MAX_SQL_LENGTH = 5_000_000 # 5MB max SQL size
|
|
37
|
+
|
|
38
|
+
# Internal metadata keys carried inside extraction result dicts. The NUL
|
|
39
|
+
# prefix cannot appear in a SQL identifier, so a real output column named
|
|
40
|
+
# _name (or even _warnings) can never collide with engine metadata.
|
|
41
|
+
META_PREFIX = "\x00ripple:"
|
|
42
|
+
META_WARNINGS = META_PREFIX + "warnings"
|
|
43
|
+
META_EXTRACTION_ERROR = META_PREFIX + "extraction_error"
|
|
44
|
+
META_EXTRACTION_ERROR_TYPE = META_PREFIX + "extraction_error_type"
|
|
45
|
+
META_QUALIFIED_VIA_SCHEMA = META_PREFIX + "qualified_via_schema"
|
|
46
|
+
META_FALLBACK_DIALECT = META_PREFIX + "fallback_dialect"
|
|
47
|
+
META_NEEDS_EXPANSION = META_PREFIX + "needs_expansion"
|
|
48
|
+
META_UNION_SOURCES = META_PREFIX + "union_sources"
|
|
49
|
+
META_UNION_VALIDATION = META_PREFIX + "union_validation"
|
|
50
|
+
META_CTE_COLLISIONS = META_PREFIX + "cte_collisions"
|
|
51
|
+
# Per-CTE metadata carried inside a CTE's own column map (cte_columns[name]):
|
|
52
|
+
# the relations its FROM scope reads, whether the name is defined more than
|
|
53
|
+
# once in the statement (nested WITH shadowing), and whether it is a
|
|
54
|
+
# recursive CTE. Consumers iterating column names must skip META_PREFIX keys.
|
|
55
|
+
META_CTE_RELATIONS = META_PREFIX + "cte_relations"
|
|
56
|
+
META_CTE_SHADOWED = META_PREFIX + "cte_shadowed"
|
|
57
|
+
META_CTE_RECURSIVE = META_PREFIX + "cte_recursive"
|
|
58
|
+
|
|
59
|
+
# Dialects where a select item may reference an alias defined earlier in the
|
|
60
|
+
# same select list (lateral column aliases). In these dialects a bare
|
|
61
|
+
# reference matching an earlier alias resolves to that alias's expression,
|
|
62
|
+
# not to a column of the upstream relation.
|
|
63
|
+
LATERAL_ALIAS_DIALECTS = frozenset(
|
|
64
|
+
{"snowflake", "redshift", "databricks", "spark", "spark2", "duckdb", "clickhouse", "teradata"}
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class FilterDependency:
|
|
70
|
+
"""A column used in WHERE, HAVING, or QUALIFY clauses.
|
|
71
|
+
|
|
72
|
+
These are C_ref columns (referenced but not contributing to output)
|
|
73
|
+
following the LineageX paper terminology.
|
|
74
|
+
|
|
75
|
+
Attributes:
|
|
76
|
+
column: Column name used in the filter
|
|
77
|
+
table: Source table (None if unqualified)
|
|
78
|
+
predicate_type: Type of filter clause (where, having, qualify, on,
|
|
79
|
+
window for a PARTITION BY / ORDER BY key)
|
|
80
|
+
operator: Comparison operator if detectable (=, IN, BETWEEN, etc.)
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
column: str
|
|
84
|
+
table: str | None
|
|
85
|
+
predicate_type: Literal["where", "having", "qualify", "on", "window"]
|
|
86
|
+
operator: str | None = None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass
|
|
90
|
+
class JoinKeyPair:
|
|
91
|
+
"""A pair of columns in a JOIN condition.
|
|
92
|
+
|
|
93
|
+
Tracks the relationship between columns used in JOIN ON clauses,
|
|
94
|
+
which is critical for understanding data cardinality and integrity.
|
|
95
|
+
|
|
96
|
+
Attributes:
|
|
97
|
+
left_column: Column from the left side of the join
|
|
98
|
+
left_table: Table/alias for the left column
|
|
99
|
+
right_column: Column from the right side of the join
|
|
100
|
+
right_table: Table/alias for the right column
|
|
101
|
+
join_type: Type of join (inner, left, right, full, cross)
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
left_column: str
|
|
105
|
+
left_table: str
|
|
106
|
+
right_column: str
|
|
107
|
+
right_table: str
|
|
108
|
+
join_type: Literal["inner", "left", "right", "full", "cross"]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass
|
|
112
|
+
class ColumnDependencies:
|
|
113
|
+
"""C_ref: Columns that affect row selection but not output values.
|
|
114
|
+
|
|
115
|
+
This complements C_con (contributing columns) to provide complete
|
|
116
|
+
lineage following the LineageX model.
|
|
117
|
+
|
|
118
|
+
Attributes:
|
|
119
|
+
filter_columns: Columns used in WHERE/HAVING/QUALIFY
|
|
120
|
+
join_keys: Column pairs used in JOIN ON conditions
|
|
121
|
+
window_keys: Columns in a window's PARTITION BY / ORDER BY
|
|
122
|
+
warnings: Any issues encountered during extraction
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
filter_columns: list[FilterDependency] = field(default_factory=list)
|
|
126
|
+
join_keys: list[JoinKeyPair] = field(default_factory=list)
|
|
127
|
+
window_keys: list[FilterDependency] = field(default_factory=list)
|
|
128
|
+
warnings: list[str] = field(default_factory=list)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class UnifiedLineageResult:
|
|
133
|
+
"""Complete lineage following the LineageX C_con + C_ref model.
|
|
134
|
+
|
|
135
|
+
Carries both:
|
|
136
|
+
- C_con: Columns that contribute to output values (SELECT sources)
|
|
137
|
+
- C_ref: Columns referenced in WHERE/JOIN (affect row selection)
|
|
138
|
+
|
|
139
|
+
Attributes:
|
|
140
|
+
contributing: Dict mapping output columns to their sources
|
|
141
|
+
filter_columns: Columns used in WHERE/HAVING/QUALIFY
|
|
142
|
+
join_keys: Column pairs used in JOIN conditions
|
|
143
|
+
window_keys: Columns in a window's PARTITION BY / ORDER BY
|
|
144
|
+
warnings: Any issues encountered during extraction
|
|
145
|
+
qualified_via_schema: True if warehouse schema was used
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
contributing: dict[str, list[dict[str, Any]]]
|
|
149
|
+
filter_columns: list[FilterDependency] = field(default_factory=list)
|
|
150
|
+
join_keys: list[JoinKeyPair] = field(default_factory=list)
|
|
151
|
+
window_keys: list[FilterDependency] = field(default_factory=list)
|
|
152
|
+
warnings: list[dict[str, Any]] = field(default_factory=list)
|
|
153
|
+
qualified_via_schema: bool = False
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class CTEColumnMapping:
|
|
158
|
+
"""Column mapping for a single CTE.
|
|
159
|
+
|
|
160
|
+
Tracks which columns a CTE produces and their sources,
|
|
161
|
+
enabling recursive tracing through nested CTEs.
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
cte_name: str
|
|
165
|
+
columns: dict[str, list[dict[str, Any]]]
|
|
166
|
+
has_star: bool = False
|
|
167
|
+
is_union: bool = False
|