sqlscope-rs 0.1.0__cp39-abi3-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sqlscope/__init__.py +269 -0
- sqlscope/_native.pyd +0 -0
- sqlscope/_native.pyi +39 -0
- sqlscope/py.typed +0 -0
- sqlscope_rs-0.1.0.dist-info/METADATA +38 -0
- sqlscope_rs-0.1.0.dist-info/RECORD +8 -0
- sqlscope_rs-0.1.0.dist-info/WHEEL +4 -0
- sqlscope_rs-0.1.0.dist-info/sboms/sqlscope-python.cyclonedx.json +1579 -0
sqlscope/__init__.py
ADDED
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""Scope-aware SQL analysis and rewriting.
|
|
2
|
+
|
|
3
|
+
===================== ==========================================================
|
|
4
|
+
Function Purpose
|
|
5
|
+
===================== ==========================================================
|
|
6
|
+
apply_row_filter Filter every in-scope table by a predicate.
|
|
7
|
+
inject_ctes Prepend CTE definitions to a query's root ``WITH``.
|
|
8
|
+
rewrite_tables Replace table references with derived tables.
|
|
9
|
+
column_origins Source columns whose values reach the result (lineage).
|
|
10
|
+
output_columns Names of the columns a statement outputs.
|
|
11
|
+
referenced_columns Columns referenced anywhere, per table.
|
|
12
|
+
column_usages Column references with the clause they appear in.
|
|
13
|
+
===================== ==========================================================
|
|
14
|
+
|
|
15
|
+
Every function accepts ``dialect`` (default ``"trino"``). Functions release the
|
|
16
|
+
GIL while they run and are safe to call from several threads.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from enum import Enum
|
|
23
|
+
from typing import Dict, Iterable, List, Mapping, Optional, Sequence, Tuple, Union
|
|
24
|
+
|
|
25
|
+
from . import _native
|
|
26
|
+
from ._native import (
|
|
27
|
+
Error,
|
|
28
|
+
InternalError,
|
|
29
|
+
InvalidArgumentError,
|
|
30
|
+
ParseError,
|
|
31
|
+
UnsupportedError,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
__version__: str = _native.__version__
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"Clause",
|
|
38
|
+
"ColumnUsage",
|
|
39
|
+
"CteDef",
|
|
40
|
+
"Error",
|
|
41
|
+
"InternalError",
|
|
42
|
+
"InvalidArgumentError",
|
|
43
|
+
"ParseError",
|
|
44
|
+
"Schema",
|
|
45
|
+
"TableRef",
|
|
46
|
+
"TableRewrite",
|
|
47
|
+
"UnionRewrite",
|
|
48
|
+
"UnsupportedError",
|
|
49
|
+
"apply_row_filter",
|
|
50
|
+
"column_origins",
|
|
51
|
+
"column_usages",
|
|
52
|
+
"inject_ctes",
|
|
53
|
+
"output_columns",
|
|
54
|
+
"referenced_columns",
|
|
55
|
+
"rewrite_tables",
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
Schema = Mapping[str, Sequence[str]]
|
|
59
|
+
"""Table name (bare, ``schema.table`` or ``catalog.schema.table``) -> ordered column names."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class CteDef:
|
|
64
|
+
"""A common table expression ``name AS (query)``.
|
|
65
|
+
|
|
66
|
+
``name`` is an identifier value, not SQL text; it is quoted as needed.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
name: str
|
|
70
|
+
query: str
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True)
|
|
74
|
+
class TableRef:
|
|
75
|
+
"""A table name, optionally schema- and catalog-qualified."""
|
|
76
|
+
|
|
77
|
+
table: str
|
|
78
|
+
schema: Optional[str] = None
|
|
79
|
+
catalog: Optional[str] = None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class UnionRewrite:
|
|
84
|
+
"""A derived table backed by ``UNION DISTINCT`` over ``branches``."""
|
|
85
|
+
|
|
86
|
+
table_alias: str
|
|
87
|
+
columns: Sequence[str]
|
|
88
|
+
branches: Sequence[TableRef]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True)
|
|
92
|
+
class TableRewrite:
|
|
93
|
+
"""Replaces references to ``match_key`` (``schema.table``).
|
|
94
|
+
|
|
95
|
+
Set exactly one of ``inline`` (``(SELECT * FROM inline)``) or ``union``.
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
match_key: str
|
|
99
|
+
inline: Optional[TableRef] = None
|
|
100
|
+
union: Optional[UnionRewrite] = None
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class Clause(str, Enum):
|
|
104
|
+
"""The SQL clause that contains a column reference."""
|
|
105
|
+
|
|
106
|
+
SELECT = "SELECT"
|
|
107
|
+
FROM = "FROM"
|
|
108
|
+
JOIN_ON = "JOIN_ON"
|
|
109
|
+
JOIN_USING = "JOIN_USING"
|
|
110
|
+
WHERE = "WHERE"
|
|
111
|
+
GROUP_BY = "GROUP_BY"
|
|
112
|
+
HAVING = "HAVING"
|
|
113
|
+
QUALIFY = "QUALIFY"
|
|
114
|
+
WINDOW = "WINDOW"
|
|
115
|
+
ORDER_BY = "ORDER_BY"
|
|
116
|
+
SORT_BY = "SORT_BY"
|
|
117
|
+
DISTRIBUTE_BY = "DISTRIBUTE_BY"
|
|
118
|
+
CLUSTER_BY = "CLUSTER_BY"
|
|
119
|
+
CONNECT_BY = "CONNECT_BY"
|
|
120
|
+
LATERAL_VIEW = "LATERAL_VIEW"
|
|
121
|
+
UPDATE_SET_TARGET = "UPDATE_SET_TARGET"
|
|
122
|
+
UPDATE_SET_VALUE = "UPDATE_SET_VALUE"
|
|
123
|
+
MERGE_ON = "MERGE_ON"
|
|
124
|
+
MERGE_WHEN = "MERGE_WHEN"
|
|
125
|
+
|
|
126
|
+
def __str__(self) -> str:
|
|
127
|
+
return self.value
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass(frozen=True, order=True)
|
|
131
|
+
class ColumnUsage:
|
|
132
|
+
"""One distinct use of a root table column in a clause."""
|
|
133
|
+
|
|
134
|
+
table: str
|
|
135
|
+
column: str
|
|
136
|
+
clause: Clause = field(compare=True)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _schema(schema: Optional[Schema]) -> Optional[Dict[str, List[str]]]:
|
|
140
|
+
if schema is None:
|
|
141
|
+
return None
|
|
142
|
+
return {str(table): [str(column) for column in columns] for table, columns in schema.items()}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _list(values: Optional[Iterable[str]]) -> Optional[List[str]]:
|
|
146
|
+
if values is None:
|
|
147
|
+
return None
|
|
148
|
+
if isinstance(values, str):
|
|
149
|
+
return [values]
|
|
150
|
+
return list(values)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def apply_row_filter(
|
|
154
|
+
sql: str,
|
|
155
|
+
predicate: str,
|
|
156
|
+
*,
|
|
157
|
+
dialect: Optional[str] = None,
|
|
158
|
+
table_names: Optional[Iterable[str]] = None,
|
|
159
|
+
table_patterns: Optional[Iterable[str]] = None,
|
|
160
|
+
default_db: Optional[str] = None,
|
|
161
|
+
) -> str:
|
|
162
|
+
"""Wrap every in-scope table of a query in a derived table filtered by ``predicate``.
|
|
163
|
+
|
|
164
|
+
``SELECT * FROM a`` becomes ``SELECT * FROM (SELECT * FROM a WHERE <predicate>) AS a``.
|
|
165
|
+
By default every physical table is filtered; ``table_names`` (bare,
|
|
166
|
+
``schema.table`` or ``catalog.schema.table``), ``table_patterns`` (regular
|
|
167
|
+
expressions) and ``default_db`` restrict the scope. CTE references, table
|
|
168
|
+
functions and ``DUAL`` are never wrapped. ``predicate`` is parsed as one
|
|
169
|
+
boolean expression; bind or escape its values before calling. Returns the
|
|
170
|
+
input unchanged when no table is in scope.
|
|
171
|
+
"""
|
|
172
|
+
return _native.apply_row_filter(
|
|
173
|
+
sql,
|
|
174
|
+
predicate,
|
|
175
|
+
dialect=dialect,
|
|
176
|
+
table_names=_list(table_names),
|
|
177
|
+
table_patterns=_list(table_patterns),
|
|
178
|
+
default_db=default_db,
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def inject_ctes(
|
|
183
|
+
sql: str,
|
|
184
|
+
ctes: Iterable[Union[CteDef, Tuple[str, str]]],
|
|
185
|
+
*,
|
|
186
|
+
dialect: Optional[str] = None,
|
|
187
|
+
) -> str:
|
|
188
|
+
"""Add CTE definitions to the root ``WITH`` of a query, before existing CTEs.
|
|
189
|
+
|
|
190
|
+
Definitions keep their order, so each may use earlier ones. A name that
|
|
191
|
+
repeats another definition or an existing root CTE is rejected.
|
|
192
|
+
"""
|
|
193
|
+
pairs = [(cte.name, cte.query) if isinstance(cte, CteDef) else (cte[0], cte[1]) for cte in ctes]
|
|
194
|
+
return _native.inject_ctes(sql, pairs, dialect=dialect)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _table(ref: TableRef) -> Tuple[str, Optional[str], Optional[str]]:
|
|
198
|
+
return (ref.table, ref.schema, ref.catalog)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def rewrite_tables(
|
|
202
|
+
sql: str,
|
|
203
|
+
rewrites: Iterable[TableRewrite],
|
|
204
|
+
*,
|
|
205
|
+
dialect: Optional[str] = None,
|
|
206
|
+
strip_catalogs: Optional[Iterable[str]] = None,
|
|
207
|
+
) -> str:
|
|
208
|
+
"""Replace physical table references according to ``rewrites``.
|
|
209
|
+
|
|
210
|
+
Matched references become derived tables that keep their alias (or bare
|
|
211
|
+
name); qualified column references are rebound onto it. Catalogs listed in
|
|
212
|
+
``strip_catalogs`` are transparent while matching. Returns the input
|
|
213
|
+
unchanged when nothing matches.
|
|
214
|
+
"""
|
|
215
|
+
plan = [
|
|
216
|
+
(
|
|
217
|
+
rewrite.match_key,
|
|
218
|
+
_table(rewrite.inline) if rewrite.inline is not None else None,
|
|
219
|
+
(
|
|
220
|
+
rewrite.union.table_alias,
|
|
221
|
+
list(rewrite.union.columns),
|
|
222
|
+
[_table(branch) for branch in rewrite.union.branches],
|
|
223
|
+
)
|
|
224
|
+
if rewrite.union is not None
|
|
225
|
+
else None,
|
|
226
|
+
)
|
|
227
|
+
for rewrite in rewrites
|
|
228
|
+
]
|
|
229
|
+
return _native.rewrite_tables(sql, plan, dialect=dialect, strip_catalogs=_list(strip_catalogs))
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def column_origins(
|
|
233
|
+
sql: str, *, dialect: Optional[str] = None, schema: Optional[Schema] = None
|
|
234
|
+
) -> Dict[str, List[str]]:
|
|
235
|
+
"""Source columns whose values flow into the result, keyed by root table.
|
|
236
|
+
|
|
237
|
+
Filter-only positions (WHERE, JOIN ON, GROUP BY, HAVING, ORDER BY) and the
|
|
238
|
+
right side of INTERSECT / EXCEPT are excluded. Every table read is present,
|
|
239
|
+
possibly with an empty list.
|
|
240
|
+
"""
|
|
241
|
+
return _native.column_origins(sql, dialect=dialect, schema=_schema(schema))
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def output_columns(
|
|
245
|
+
sql: str, *, dialect: Optional[str] = None, schema: Optional[Schema] = None
|
|
246
|
+
) -> Optional[List[str]]:
|
|
247
|
+
"""Names of the columns a statement outputs, in order.
|
|
248
|
+
|
|
249
|
+
Unaliased expressions are named ``_col{i}``; ``*`` expands from ``schema``
|
|
250
|
+
or stays ``"*"``. Returns ``None`` for statements without columns.
|
|
251
|
+
"""
|
|
252
|
+
return _native.output_columns(sql, dialect=dialect, schema=_schema(schema))
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def referenced_columns(
|
|
256
|
+
sql: str, *, dialect: Optional[str] = None, schema: Optional[Schema] = None
|
|
257
|
+
) -> Dict[str, List[str]]:
|
|
258
|
+
"""Columns referenced anywhere in a statement (including filters), keyed by root table."""
|
|
259
|
+
return _native.referenced_columns(sql, dialect=dialect, schema=_schema(schema))
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def column_usages(
|
|
263
|
+
sql: str, *, dialect: Optional[str] = None, schema: Optional[Schema] = None
|
|
264
|
+
) -> List[ColumnUsage]:
|
|
265
|
+
"""Every distinct ``(table, column, clause)`` use in a statement, sorted."""
|
|
266
|
+
return [
|
|
267
|
+
ColumnUsage(table, column, Clause(clause))
|
|
268
|
+
for table, column, clause in _native.column_usages(sql, dialect=dialect, schema=_schema(schema))
|
|
269
|
+
]
|
sqlscope/_native.pyd
ADDED
|
Binary file
|
sqlscope/_native.pyi
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from typing import Dict, List, Optional, Tuple
|
|
2
|
+
|
|
3
|
+
__version__: str
|
|
4
|
+
|
|
5
|
+
class Error(Exception): ...
|
|
6
|
+
class InvalidArgumentError(Error): ...
|
|
7
|
+
class ParseError(Error): ...
|
|
8
|
+
class UnsupportedError(Error): ...
|
|
9
|
+
class InternalError(Error): ...
|
|
10
|
+
|
|
11
|
+
_Table = Tuple[str, Optional[str], Optional[str]]
|
|
12
|
+
|
|
13
|
+
def apply_row_filter(
|
|
14
|
+
sql: str,
|
|
15
|
+
predicate: str,
|
|
16
|
+
dialect: Optional[str] = ...,
|
|
17
|
+
table_names: Optional[List[str]] = ...,
|
|
18
|
+
table_patterns: Optional[List[str]] = ...,
|
|
19
|
+
default_db: Optional[str] = ...,
|
|
20
|
+
) -> str: ...
|
|
21
|
+
def inject_ctes(sql: str, ctes: List[Tuple[str, str]], dialect: Optional[str] = ...) -> str: ...
|
|
22
|
+
def rewrite_tables(
|
|
23
|
+
sql: str,
|
|
24
|
+
rewrites: List[Tuple[str, Optional[_Table], Optional[Tuple[str, List[str], List[_Table]]]]],
|
|
25
|
+
dialect: Optional[str] = ...,
|
|
26
|
+
strip_catalogs: Optional[List[str]] = ...,
|
|
27
|
+
) -> str: ...
|
|
28
|
+
def column_origins(
|
|
29
|
+
sql: str, dialect: Optional[str] = ..., schema: Optional[Dict[str, List[str]]] = ...
|
|
30
|
+
) -> Dict[str, List[str]]: ...
|
|
31
|
+
def output_columns(
|
|
32
|
+
sql: str, dialect: Optional[str] = ..., schema: Optional[Dict[str, List[str]]] = ...
|
|
33
|
+
) -> Optional[List[str]]: ...
|
|
34
|
+
def referenced_columns(
|
|
35
|
+
sql: str, dialect: Optional[str] = ..., schema: Optional[Dict[str, List[str]]] = ...
|
|
36
|
+
) -> Dict[str, List[str]]: ...
|
|
37
|
+
def column_usages(
|
|
38
|
+
sql: str, dialect: Optional[str] = ..., schema: Optional[Dict[str, List[str]]] = ...
|
|
39
|
+
) -> List[Tuple[str, str, str]]: ...
|
sqlscope/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sqlscope-rs
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Classifier: Programming Language :: Python :: 3
|
|
5
|
+
Classifier: Programming Language :: Rust
|
|
6
|
+
Classifier: Topic :: Database
|
|
7
|
+
Classifier: Typing :: Typed
|
|
8
|
+
Summary: Scope-aware SQL analysis and rewriting: row filters, CTE injection, table rewrites, column lineage.
|
|
9
|
+
Keywords: sql,lineage,row-level-security,parser
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
Requires-Python: >=3.9
|
|
12
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
13
|
+
Project-URL: Repository, https://github.com/dcalsky/sqlscope
|
|
14
|
+
|
|
15
|
+
# sqlscope (Python)
|
|
16
|
+
|
|
17
|
+
Scope-aware SQL analysis and rewriting, backed by the Rust
|
|
18
|
+
[sqlscope](https://github.com/dcalsky/sqlscope) crate.
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
pip install sqlscope-rs
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
import sqlscope
|
|
26
|
+
|
|
27
|
+
sqlscope.apply_row_filter("SELECT id FROM orders", "tenant_id = 7", dialect="postgres")
|
|
28
|
+
# 'SELECT id FROM (SELECT * FROM orders WHERE tenant_id = 7) AS orders'
|
|
29
|
+
|
|
30
|
+
sqlscope.column_origins(
|
|
31
|
+
"SELECT o.id, p.amount FROM orders o JOIN payments p ON o.id = p.order_id WHERE o.status = 'PAID'"
|
|
32
|
+
)
|
|
33
|
+
# {'orders': ['id'], 'payments': ['amount']}
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
See the [project README](https://github.com/dcalsky/sqlscope#readme) for the
|
|
37
|
+
full API.
|
|
38
|
+
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
sqlscope/__init__.py,sha256=Gg4nRMgnge-uYVd-_LKoZwf7qZl0bzKL9CxOqiIZp0A,8686
|
|
2
|
+
sqlscope/_native.pyd,sha256=LnR5_g6oV8FQaS7wQ-L9wRFkhH0e-I41MYlfzujNamY,16678912
|
|
3
|
+
sqlscope/_native.pyi,sha256=ucOBAAwzGXujAw69NXvl3sJagS4BuNe0v_dBNVDmiCM,1443
|
|
4
|
+
sqlscope/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
sqlscope_rs-0.1.0.dist-info/METADATA,sha256=7CmBd7jtovbZJG8EDclKNm7z6BxRp3sb6ChknDeFhgk,1188
|
|
6
|
+
sqlscope_rs-0.1.0.dist-info/WHEEL,sha256=xe4_tbg8wYdeh4O7nLbnWckzIQtCR4nODDgwke_TKy8,95
|
|
7
|
+
sqlscope_rs-0.1.0.dist-info/sboms/sqlscope-python.cyclonedx.json,sha256=ZsG9PZRuQJjHkkZChjHtX6JjC8lOZyese8IkWxc0Me8,49224
|
|
8
|
+
sqlscope_rs-0.1.0.dist-info/RECORD,,
|