sqlglotc 30.13.0__tar.gz → 30.15.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sqlglotc-30.13.0/sqlglotc.egg-info → sqlglotc-30.15.0}/PKG-INFO +2 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/setup.py +1 -0
- sqlglotc-30.15.0/sqlglot/anonymize.py +307 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/array.py +4 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/core.py +1 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/datatypes.py +12 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/ddl.py +1 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/dml.py +1 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/math.py +8 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/properties.py +9 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/query.py +20 -3
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/string.py +4 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/temporal.py +1 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generator.py +86 -31
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/bigquery.py +7 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/clickhouse.py +29 -3
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/doris.py +7 -3
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/duckdb.py +124 -26
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/exasol.py +9 -4
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/fabric.py +2 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/hive.py +24 -13
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/mysql.py +21 -30
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/oracle.py +2 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/postgres.py +15 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/python.py +7 -5
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/singlestore.py +3 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/spark2.py +7 -23
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/sqlite.py +2 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/starrocks.py +7 -3
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/trino.py +16 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/tsql.py +9 -3
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/annotate_types.py +101 -82
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/canonicalize_internal_names.py +21 -7
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/qualify.py +1 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/qualify_columns.py +137 -26
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/qualify_tables.py +2 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/scope.py +1 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/simplify.py +75 -26
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parser.py +183 -43
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/bigquery.py +33 -19
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/clickhouse.py +56 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/dremio.py +6 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/duckdb.py +1 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/mysql.py +16 -6
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/spark.py +0 -1
- sqlglotc-30.15.0/sqlglot/parsers/trino.py +158 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/tsql.py +16 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/schema.py +2 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/tokenizer_core.py +0 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0/sqlglotc.egg-info}/PKG-INFO +2 -2
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglotc.egg-info/SOURCES.txt +1 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglotc.egg-info/requires.txt +1 -1
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglotc.egg-info/scm_file_list.json +2 -2
- sqlglotc-30.15.0/sqlglotc.egg-info/scm_version.json +8 -0
- sqlglotc-30.13.0/sqlglot/parsers/trino.py +0 -63
- sqlglotc-30.13.0/sqlglotc.egg-info/scm_version.json +0 -8
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/MANIFEST.in +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/pyproject.toml +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/setup.cfg +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/errors.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/executor/table.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/aggregate.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/builders.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/constraints.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/functions.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/expressions/json.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/athena.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/databricks.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/dax.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/dremio.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/drill.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/druid.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/dune.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/materialize.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/presto.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/prql.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/redshift.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/risingwave.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/snowflake.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/solr.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/spark.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/tableau.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/generators/teradata.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/helper.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/lineage.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/isolate_table_selects.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/normalize_identifiers.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/optimizer/resolver.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/athena.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/base.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/databricks.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/dax.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/doris.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/drill.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/druid.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/dune.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/exasol.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/fabric.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/hive.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/materialize.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/oracle.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/postgres.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/presto.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/prql.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/redshift.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/risingwave.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/singlestore.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/snowflake.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/solr.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/spark2.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/sqlite.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/starrocks.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/tableau.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/parsers/teradata.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/serde.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/time.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglot/trie.py +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglotc.egg-info/dependency_links.txt +0 -0
- {sqlglotc-30.13.0 → sqlglotc-30.15.0}/sqlglotc.egg-info/top_level.txt +0 -0
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sqlglotc
|
|
3
|
-
Version: 30.
|
|
3
|
+
Version: 30.15.0
|
|
4
4
|
Summary: mypyc-compiled extensions for sqlglot
|
|
5
5
|
Author-email: Toby Mao <toby.mao@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://sqlglot.com/
|
|
8
8
|
Project-URL: Repository, https://github.com/tobymao/sqlglot
|
|
9
9
|
Requires-Python: >=3.10
|
|
10
|
-
Requires-Dist: sqlglot==30.
|
|
10
|
+
Requires-Dist: sqlglot==30.15.0
|
|
11
11
|
Provides-Extra: dev
|
|
12
12
|
Requires-Dist: setuptools>=61.0; extra == "dev"
|
|
13
13
|
Requires-Dist: setuptools_scm; extra == "dev"
|
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import string
|
|
4
|
+
|
|
5
|
+
from sqlglot.dialects.dialect import Dialect, DialectType
|
|
6
|
+
from sqlglot.errors import TokenError
|
|
7
|
+
from sqlglot.tokens import Token, TokenType
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
ALPHABET = string.ascii_lowercase
|
|
11
|
+
ALPHABET_SIZE = len(ALPHABET)
|
|
12
|
+
ANONYMIZED_TYPES = {
|
|
13
|
+
TokenType.BIT_STRING,
|
|
14
|
+
TokenType.BYTE_STRING,
|
|
15
|
+
TokenType.HEX_STRING,
|
|
16
|
+
TokenType.HEREDOC_STRING,
|
|
17
|
+
TokenType.IDENTIFIER,
|
|
18
|
+
TokenType.NATIONAL_STRING,
|
|
19
|
+
TokenType.NUMBER,
|
|
20
|
+
TokenType.RAW_STRING,
|
|
21
|
+
TokenType.STRING,
|
|
22
|
+
TokenType.UNICODE_STRING,
|
|
23
|
+
TokenType.VAR,
|
|
24
|
+
}
|
|
25
|
+
QUOTED_TYPES = ANONYMIZED_TYPES - {TokenType.NUMBER, TokenType.VAR}
|
|
26
|
+
REWRITTEN_TYPES = {TokenType.HINT, TokenType.UNKNOWN}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def anonymize(
|
|
30
|
+
sql_or_tokens: list[Token] | str,
|
|
31
|
+
dialect: DialectType = None,
|
|
32
|
+
) -> list[Token]:
|
|
33
|
+
"""Replaces sensitive tokens (identifiers, strings, numbers) with fixed-width,
|
|
34
|
+
length-preserving, consistent aliases, and blanks out comments and hint bodies. When a
|
|
35
|
+
SQL string is given, it is tokenized with `dialect` first; any un-tokenized remainder
|
|
36
|
+
(e.g. an unterminated literal) is appended as a blanked UNKNOWN token. Mutates and
|
|
37
|
+
returns `sql_or_tokens`.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
sql_or_tokens: The SQL string to anonymize, or its token list.
|
|
41
|
+
dialect: The dialect used to tokenize a SQL string.
|
|
42
|
+
"""
|
|
43
|
+
dialect = Dialect.get_or_raise(dialect)
|
|
44
|
+
tokenizer_class = dialect.tokenizer_class
|
|
45
|
+
parser_class = dialect.parser_class
|
|
46
|
+
|
|
47
|
+
errored = False
|
|
48
|
+
if isinstance(sql_or_tokens, str):
|
|
49
|
+
sql = sql_or_tokens
|
|
50
|
+
tokenizer = dialect.tokenizer()
|
|
51
|
+
try:
|
|
52
|
+
tokens = tokenizer.tokenize(sql)
|
|
53
|
+
except TokenError:
|
|
54
|
+
tokens = tokenizer.tokens
|
|
55
|
+
errored = True
|
|
56
|
+
else:
|
|
57
|
+
tokens = sql_or_tokens
|
|
58
|
+
sql = None
|
|
59
|
+
|
|
60
|
+
hint_start = tokenizer_class.HINT_START
|
|
61
|
+
hint_end = tokenizer_class._COMMENTS.get(hint_start)
|
|
62
|
+
nested = tokenizer_class.NESTED_COMMENTS
|
|
63
|
+
|
|
64
|
+
seen: dict[tuple[bool, str], str] = {}
|
|
65
|
+
counter = 0
|
|
66
|
+
|
|
67
|
+
for i, token in enumerate(tokens):
|
|
68
|
+
token.comments = [_blank(comment) for comment in token.comments]
|
|
69
|
+
|
|
70
|
+
if token.token_type == TokenType.HINT:
|
|
71
|
+
# A hint's text is the whole /*+ ... */ comment, so its body is blanked as well
|
|
72
|
+
text, stop = _blank_comment(token.text, 0, hint_start, hint_end, nested)
|
|
73
|
+
token.text = text + _blank(token.text[stop:])
|
|
74
|
+
continue
|
|
75
|
+
|
|
76
|
+
if (
|
|
77
|
+
token.token_type == TokenType.VAR
|
|
78
|
+
and i + 1 < len(tokens)
|
|
79
|
+
and tokens[i + 1].token_type == TokenType.L_PAREN
|
|
80
|
+
):
|
|
81
|
+
# A function name can live in either registry, e.g. JSON_OBJECT is only in
|
|
82
|
+
# FUNCTION_PARSERS. They're consulted separately to avoid building their union
|
|
83
|
+
name = token.text.upper()
|
|
84
|
+
if name in parser_class.FUNCTIONS or name in parser_class.FUNCTION_PARSERS:
|
|
85
|
+
continue
|
|
86
|
+
if token.token_type not in ANONYMIZED_TYPES:
|
|
87
|
+
continue
|
|
88
|
+
if not token.text:
|
|
89
|
+
continue
|
|
90
|
+
|
|
91
|
+
is_number = token.token_type == TokenType.NUMBER
|
|
92
|
+
key = (is_number, token.text)
|
|
93
|
+
alias = seen.get(key)
|
|
94
|
+
if alias is None:
|
|
95
|
+
seen[key] = (
|
|
96
|
+
_number_alias(counter, token.text) if is_number else _alias(counter, token.text)
|
|
97
|
+
)
|
|
98
|
+
counter += 1
|
|
99
|
+
token.text = seen[key]
|
|
100
|
+
|
|
101
|
+
if sql is not None and errored:
|
|
102
|
+
start = tokens[-1].end + 1 if tokens else 0
|
|
103
|
+
length = len(sql)
|
|
104
|
+
while start < length and sql[start].isspace():
|
|
105
|
+
start += 1
|
|
106
|
+
if start < length:
|
|
107
|
+
# The first two characters are kept so that the delimiter the tokenizer choked
|
|
108
|
+
# on is still visible, e.g. 'u, /*, $$, `u
|
|
109
|
+
tokens.append(
|
|
110
|
+
Token(
|
|
111
|
+
TokenType.UNKNOWN,
|
|
112
|
+
sql[start : start + 2] + "." * (length - start - 2),
|
|
113
|
+
start=start,
|
|
114
|
+
end=length - 1,
|
|
115
|
+
line=sql.count("\n", 0, start) + 1,
|
|
116
|
+
col=start - sql.rfind("\n", 0, start),
|
|
117
|
+
)
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
return tokens
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def render(sql: str, tokens: list[Token], dialect: DialectType = None) -> str:
|
|
124
|
+
"""Recreates the (anonymized) SQL string from the original `sql` and token positions.
|
|
125
|
+
|
|
126
|
+
Every token is rendered over its own source span, so the result is always as long as
|
|
127
|
+
`sql`: a token that wasn't anonymized is emitted verbatim, keeping the spelling the
|
|
128
|
+
tokenizer normalized away, and an anonymized one has its alias fitted between the
|
|
129
|
+
quotes of its span. The gaps between tokens hold only whitespace and comments, so
|
|
130
|
+
they're redacted rather than reconstructed: comment markers and whitespace survive,
|
|
131
|
+
everything else is blanked. That covers anything the tokenizer didn't reach as well,
|
|
132
|
+
which is a trailing gap whenever `anonymize` wasn't the one to tokenize `sql`.
|
|
133
|
+
|
|
134
|
+
Args:
|
|
135
|
+
sql: The original SQL string.
|
|
136
|
+
tokens: The anonymized tokens to render.
|
|
137
|
+
dialect: The dialect used to identify comments and quotes in `sql`.
|
|
138
|
+
"""
|
|
139
|
+
tokenizer_class = Dialect.get_or_raise(dialect).tokenizer_class
|
|
140
|
+
comments = sorted(tokenizer_class._COMMENTS.items(), key=lambda c: len(c[0]), reverse=True)
|
|
141
|
+
nested = tokenizer_class.NESTED_COMMENTS
|
|
142
|
+
quotes = sorted(
|
|
143
|
+
{
|
|
144
|
+
**tokenizer_class._QUOTES,
|
|
145
|
+
**{start: end for start, (end, _) in tokenizer_class._FORMAT_STRINGS.items()},
|
|
146
|
+
**tokenizer_class._IDENTIFIERS,
|
|
147
|
+
}.items(),
|
|
148
|
+
key=lambda quote: (len(quote[0]), len(quote[1])),
|
|
149
|
+
reverse=True,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
result = []
|
|
153
|
+
prev = 0
|
|
154
|
+
|
|
155
|
+
for token in tokens:
|
|
156
|
+
result.append(_redact(sql[prev : token.start], comments, nested))
|
|
157
|
+
prev = token.end + 1
|
|
158
|
+
span = sql[token.start : prev]
|
|
159
|
+
if token.token_type in REWRITTEN_TYPES:
|
|
160
|
+
# Blanked in place by `anonymize`, or synthesized by it, so already span-shaped
|
|
161
|
+
result.append(token.text)
|
|
162
|
+
elif token.token_type in ANONYMIZED_TYPES:
|
|
163
|
+
result.append(_fit(span, token.text, quotes, token.token_type))
|
|
164
|
+
else:
|
|
165
|
+
result.append(span)
|
|
166
|
+
|
|
167
|
+
result.append(_redact(sql[prev:], comments, nested))
|
|
168
|
+
|
|
169
|
+
return "".join(result)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _alias(counter: int, text: str) -> str:
|
|
173
|
+
digits = []
|
|
174
|
+
while counter:
|
|
175
|
+
counter, digit = divmod(counter, ALPHABET_SIZE)
|
|
176
|
+
digits.append(ALPHABET[digit])
|
|
177
|
+
|
|
178
|
+
letters = "".join(reversed(digits)).rjust(sum(not char.isspace() for char in text), "a")
|
|
179
|
+
|
|
180
|
+
alias = []
|
|
181
|
+
i = 0
|
|
182
|
+
for char in text:
|
|
183
|
+
if char.isspace():
|
|
184
|
+
alias.append(char)
|
|
185
|
+
else:
|
|
186
|
+
alias.append(letters[i])
|
|
187
|
+
i += 1
|
|
188
|
+
|
|
189
|
+
return "".join(alias)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _number_alias(counter: int, text: str) -> str:
|
|
193
|
+
if len(text) > 4000:
|
|
194
|
+
digits_seen = False
|
|
195
|
+
blanked = []
|
|
196
|
+
for char in text:
|
|
197
|
+
if char.isdigit():
|
|
198
|
+
blanked.append("0" if digits_seen else "1")
|
|
199
|
+
digits_seen = True
|
|
200
|
+
else:
|
|
201
|
+
blanked.append(char)
|
|
202
|
+
return "".join(blanked)
|
|
203
|
+
|
|
204
|
+
sep = "e" if "e" in text else ("E" if "E" in text else "")
|
|
205
|
+
mantissa, _, exponent = text.partition(sep) if sep else (text, "", "")
|
|
206
|
+
sign = ""
|
|
207
|
+
if exponent.startswith(("-", "+")):
|
|
208
|
+
sign, exponent = exponent[0], exponent[1:]
|
|
209
|
+
exponent_length = len(exponent)
|
|
210
|
+
|
|
211
|
+
integer, dot, fraction = mantissa.partition(".")
|
|
212
|
+
integer_length = len(integer)
|
|
213
|
+
digits = integer_length + len(fraction)
|
|
214
|
+
mantissa_value = 10 ** (digits - 1) + counter % (9 * 10 ** (digits - 1))
|
|
215
|
+
result = str(mantissa_value)
|
|
216
|
+
if dot:
|
|
217
|
+
result = result[:integer_length] + "." + result[integer_length:]
|
|
218
|
+
if sep:
|
|
219
|
+
# The exponent marker is kept even when there are no digits after it, e.g. 1e
|
|
220
|
+
result += sep + sign
|
|
221
|
+
if exponent_length:
|
|
222
|
+
exponent_value = 10 ** (exponent_length - 1) + (counter // (9 * 10 ** (digits - 1))) % (
|
|
223
|
+
9 * 10 ** (exponent_length - 1)
|
|
224
|
+
)
|
|
225
|
+
result += str(exponent_value)
|
|
226
|
+
|
|
227
|
+
return result
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _fit(span: str, alias: str, quotes: list[tuple[str, str]], token_type: TokenType) -> str:
|
|
231
|
+
"""Fits `alias` into the quoted region of `span`, so the two are always the same length."""
|
|
232
|
+
start = end = 0
|
|
233
|
+
|
|
234
|
+
if token_type in QUOTED_TYPES:
|
|
235
|
+
for open_quote, close_quote in quotes:
|
|
236
|
+
if (
|
|
237
|
+
len(span) >= len(open_quote) + len(close_quote)
|
|
238
|
+
and span.startswith(open_quote)
|
|
239
|
+
and span.endswith(close_quote)
|
|
240
|
+
):
|
|
241
|
+
start, end = len(open_quote), len(close_quote)
|
|
242
|
+
|
|
243
|
+
if token_type == TokenType.HEREDOC_STRING:
|
|
244
|
+
# A heredoc's tag is part of its delimiter, e.g. $tag$body$tag$
|
|
245
|
+
open_end = span.find(close_quote, start)
|
|
246
|
+
close_start = span.rfind(open_quote, 0, len(span) - end)
|
|
247
|
+
if start <= open_end < close_start:
|
|
248
|
+
start, end = open_end + len(close_quote), len(span) - close_start
|
|
249
|
+
|
|
250
|
+
break
|
|
251
|
+
|
|
252
|
+
width = len(span) - start - end
|
|
253
|
+
pad = "0" if token_type == TokenType.NUMBER else "a"
|
|
254
|
+
|
|
255
|
+
return span[:start] + alias[:width].rjust(width, pad) + span[len(span) - end :]
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _blank(sql: str) -> str:
|
|
259
|
+
return "".join(char if char.isspace() else "." for char in sql)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _blank_comment(sql: str, i: int, start: str, end: str | None, nested: bool) -> tuple[str, int]:
|
|
263
|
+
"""Blanks the body of the comment at `i`, returning its text and the index past it."""
|
|
264
|
+
body = i + len(start)
|
|
265
|
+
|
|
266
|
+
if not end:
|
|
267
|
+
stop = sql.find("\n", body)
|
|
268
|
+
stop = len(sql) if stop == -1 else stop
|
|
269
|
+
return start + _blank(sql[body:stop]), stop
|
|
270
|
+
|
|
271
|
+
depth = 1
|
|
272
|
+
j = body
|
|
273
|
+
while j < len(sql):
|
|
274
|
+
if nested and sql.startswith(start, j):
|
|
275
|
+
depth += 1
|
|
276
|
+
j += len(start)
|
|
277
|
+
elif sql.startswith(end, j):
|
|
278
|
+
j += len(end)
|
|
279
|
+
depth -= 1
|
|
280
|
+
if not depth:
|
|
281
|
+
return start + _blank(sql[body : j - len(end)]) + end, j
|
|
282
|
+
else:
|
|
283
|
+
j += 1
|
|
284
|
+
|
|
285
|
+
return start + _blank(sql[body:]), len(sql)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _redact(sql: str, comments: list[tuple[str, str | None]], nested: bool) -> str:
|
|
289
|
+
if not sql or sql.isspace():
|
|
290
|
+
return sql
|
|
291
|
+
|
|
292
|
+
result = []
|
|
293
|
+
i = 0
|
|
294
|
+
length = len(sql)
|
|
295
|
+
|
|
296
|
+
while i < length:
|
|
297
|
+
for start, end in comments:
|
|
298
|
+
if sql.startswith(start, i):
|
|
299
|
+
text, i = _blank_comment(sql, i, start, end, nested)
|
|
300
|
+
result.append(text)
|
|
301
|
+
break
|
|
302
|
+
else:
|
|
303
|
+
char = sql[i]
|
|
304
|
+
result.append(char if char.isspace() else ".")
|
|
305
|
+
i += 1
|
|
306
|
+
|
|
307
|
+
return "".join(result)
|
|
@@ -220,10 +220,22 @@ class DataType(Expression):
|
|
|
220
220
|
DType.NCHAR,
|
|
221
221
|
DType.NVARCHAR,
|
|
222
222
|
DType.TEXT,
|
|
223
|
+
DType.TINYTEXT,
|
|
224
|
+
DType.MEDIUMTEXT,
|
|
225
|
+
DType.LONGTEXT,
|
|
223
226
|
DType.VARCHAR,
|
|
224
227
|
DType.NAME,
|
|
225
228
|
}
|
|
226
229
|
|
|
230
|
+
BINARY_TYPES: t.ClassVar[set[DType]] = {
|
|
231
|
+
DType.BINARY,
|
|
232
|
+
DType.VARBINARY,
|
|
233
|
+
DType.TINYBLOB,
|
|
234
|
+
DType.BLOB,
|
|
235
|
+
DType.MEDIUMBLOB,
|
|
236
|
+
DType.LONGBLOB,
|
|
237
|
+
}
|
|
238
|
+
|
|
227
239
|
SIGNED_INTEGER_TYPES: t.ClassVar[set[DType]] = {
|
|
228
240
|
DType.BIGINT,
|
|
229
241
|
DType.INT,
|
|
@@ -164,6 +164,14 @@ class Log(Expression, Func):
|
|
|
164
164
|
arg_types = {"this": True, "expression": False}
|
|
165
165
|
|
|
166
166
|
|
|
167
|
+
class Negative(Expression, Func):
|
|
168
|
+
pass
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class Nanvl(Expression, Func):
|
|
172
|
+
arg_types = {"this": True, "expression": True}
|
|
173
|
+
|
|
174
|
+
|
|
167
175
|
class Pi(Expression, Func):
|
|
168
176
|
arg_types = {}
|
|
169
177
|
|
|
@@ -42,7 +42,15 @@ class AutoIncrementProperty(Property):
|
|
|
42
42
|
|
|
43
43
|
|
|
44
44
|
class AutoRefreshProperty(Property):
|
|
45
|
-
arg_types = {
|
|
45
|
+
arg_types = {
|
|
46
|
+
"this": False,
|
|
47
|
+
"cadence": False,
|
|
48
|
+
"offset": False,
|
|
49
|
+
"randomize": False,
|
|
50
|
+
"expressions": False,
|
|
51
|
+
"settings": False,
|
|
52
|
+
"append": False,
|
|
53
|
+
}
|
|
46
54
|
|
|
47
55
|
|
|
48
56
|
class BackupProperty(Property):
|
|
@@ -444,7 +444,7 @@ class RecursiveWithSearch(Expression):
|
|
|
444
444
|
|
|
445
445
|
|
|
446
446
|
class With(Expression):
|
|
447
|
-
arg_types = {"expressions":
|
|
447
|
+
arg_types = {"expressions": False, "recursive": False, "search": False, "udfs": False}
|
|
448
448
|
|
|
449
449
|
@property
|
|
450
450
|
def recursive(self) -> bool:
|
|
@@ -1761,6 +1761,7 @@ class Pivot(Expression):
|
|
|
1761
1761
|
"identify_pivot_strings": False,
|
|
1762
1762
|
"prefixed_pivot_columns": False,
|
|
1763
1763
|
"pivot_column_naming": False,
|
|
1764
|
+
"value_columns_first": False,
|
|
1764
1765
|
}
|
|
1765
1766
|
|
|
1766
1767
|
@property
|
|
@@ -1814,7 +1815,13 @@ class Pivot(Expression):
|
|
|
1814
1815
|
for ident in (e.expressions if isinstance(e, Tuple) else [e])
|
|
1815
1816
|
if isinstance(ident, Identifier)
|
|
1816
1817
|
]
|
|
1817
|
-
|
|
1818
|
+
# T-SQL emits the value column(s) ahead of the name column, everyone else emits them after it
|
|
1819
|
+
ordered = (
|
|
1820
|
+
value_columns + name_columns
|
|
1821
|
+
if self.args.get("value_columns_first")
|
|
1822
|
+
else name_columns + value_columns
|
|
1823
|
+
)
|
|
1824
|
+
outputs = [i.name for i in ordered]
|
|
1818
1825
|
else:
|
|
1819
1826
|
excluded = {c.output_name for c in self.find_all(Column)}
|
|
1820
1827
|
outputs = [c.output_name for c in self.args.get("columns") or []]
|
|
@@ -2105,7 +2112,7 @@ class StoredProcedure(Expression):
|
|
|
2105
2112
|
|
|
2106
2113
|
|
|
2107
2114
|
class Block(Expression):
|
|
2108
|
-
arg_types = {"expressions": True}
|
|
2115
|
+
arg_types = {"expressions": True, "begin": False}
|
|
2109
2116
|
|
|
2110
2117
|
|
|
2111
2118
|
class IfBlock(Expression):
|
|
@@ -2120,6 +2127,16 @@ class EndStatement(Expression):
|
|
|
2120
2127
|
arg_types = {}
|
|
2121
2128
|
|
|
2122
2129
|
|
|
2130
|
+
# https://trino.io/docs/current/udf.html
|
|
2131
|
+
class FunctionSpecification(Expression):
|
|
2132
|
+
arg_types = {
|
|
2133
|
+
"this": True,
|
|
2134
|
+
"characteristics": False,
|
|
2135
|
+
"properties": False,
|
|
2136
|
+
"expression": True,
|
|
2137
|
+
}
|
|
2138
|
+
|
|
2139
|
+
|
|
2123
2140
|
UNWRAPPED_QUERIES = (Select, SetOperation)
|
|
2124
2141
|
|
|
2125
2142
|
|
|
@@ -501,7 +501,7 @@ class TsOrDiToDi(Expression, Func):
|
|
|
501
501
|
|
|
502
502
|
|
|
503
503
|
class TsOrDsToDate(Expression, Func):
|
|
504
|
-
arg_types = {"this": True, "format": False, "safe": False}
|
|
504
|
+
arg_types = {"this": True, "format": False, "safe": False, "default_date": False}
|
|
505
505
|
|
|
506
506
|
|
|
507
507
|
class TsOrDsToDateStr(Expression, Func):
|