sqlglotc 30.14.0__tar.gz → 30.16.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. {sqlglotc-30.14.0/sqlglotc.egg-info → sqlglotc-30.16.0}/PKG-INFO +2 -2
  2. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/setup.py +1 -0
  3. sqlglotc-30.16.0/sqlglot/anonymize.py +307 -0
  4. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/errors.py +15 -1
  5. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/array.py +6 -4
  6. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/core.py +20 -14
  7. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/datatypes.py +12 -0
  8. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/ddl.py +2 -0
  9. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/dml.py +1 -0
  10. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/json.py +4 -3
  11. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/math.py +4 -0
  12. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/properties.py +9 -1
  13. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/query.py +24 -3
  14. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/string.py +7 -3
  15. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/temporal.py +1 -1
  16. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generator.py +134 -40
  17. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/bigquery.py +8 -1
  18. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/clickhouse.py +32 -5
  19. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/doris.py +12 -5
  20. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/duckdb.py +88 -48
  21. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/exasol.py +9 -4
  22. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/fabric.py +2 -1
  23. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/hive.py +11 -5
  24. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/mysql.py +32 -36
  25. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/oracle.py +2 -0
  26. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/postgres.py +23 -1
  27. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/python.py +4 -4
  28. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/singlestore.py +5 -3
  29. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/snowflake.py +1 -1
  30. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/spark2.py +12 -24
  31. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/sqlite.py +9 -2
  32. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/starrocks.py +10 -5
  33. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/trino.py +50 -0
  34. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/tsql.py +7 -1
  35. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/lineage.py +78 -33
  36. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/annotate_types.py +101 -82
  37. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/canonicalize_internal_names.py +19 -6
  38. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/qualify_columns.py +95 -17
  39. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/qualify_tables.py +11 -3
  40. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/scope.py +7 -1
  41. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/simplify.py +66 -24
  42. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parser.py +223 -56
  43. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/bigquery.py +33 -19
  44. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/clickhouse.py +80 -17
  45. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/dremio.py +6 -0
  46. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/duckdb.py +11 -0
  47. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/mysql.py +17 -7
  48. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/postgres.py +16 -1
  49. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/prql.py +1 -1
  50. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/snowflake.py +11 -4
  51. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/spark.py +0 -1
  52. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/sqlite.py +79 -0
  53. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/teradata.py +1 -1
  54. sqlglotc-30.16.0/sqlglot/parsers/trino.py +222 -0
  55. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/tsql.py +24 -3
  56. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/schema.py +2 -0
  57. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/tokenizer_core.py +1 -3
  58. {sqlglotc-30.14.0 → sqlglotc-30.16.0/sqlglotc.egg-info}/PKG-INFO +2 -2
  59. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglotc.egg-info/SOURCES.txt +1 -0
  60. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglotc.egg-info/requires.txt +1 -1
  61. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglotc.egg-info/scm_file_list.json +2 -2
  62. sqlglotc-30.16.0/sqlglotc.egg-info/scm_version.json +8 -0
  63. sqlglotc-30.14.0/sqlglot/parsers/trino.py +0 -63
  64. sqlglotc-30.14.0/sqlglotc.egg-info/scm_version.json +0 -8
  65. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/MANIFEST.in +0 -0
  66. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/pyproject.toml +0 -0
  67. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/setup.cfg +0 -0
  68. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/executor/table.py +0 -0
  69. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/aggregate.py +0 -0
  70. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/builders.py +0 -0
  71. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/constraints.py +0 -0
  72. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/expressions/functions.py +0 -0
  73. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/athena.py +0 -0
  74. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/databricks.py +0 -0
  75. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/dax.py +0 -0
  76. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/dremio.py +0 -0
  77. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/drill.py +0 -0
  78. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/druid.py +0 -0
  79. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/dune.py +0 -0
  80. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/materialize.py +0 -0
  81. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/presto.py +0 -0
  82. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/prql.py +0 -0
  83. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/redshift.py +0 -0
  84. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/risingwave.py +0 -0
  85. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/solr.py +0 -0
  86. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/spark.py +0 -0
  87. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/tableau.py +0 -0
  88. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/generators/teradata.py +0 -0
  89. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/helper.py +0 -0
  90. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/isolate_table_selects.py +0 -0
  91. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/normalize_identifiers.py +0 -0
  92. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/qualify.py +0 -0
  93. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/optimizer/resolver.py +0 -0
  94. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/athena.py +0 -0
  95. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/base.py +0 -0
  96. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/databricks.py +0 -0
  97. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/dax.py +0 -0
  98. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/doris.py +0 -0
  99. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/drill.py +0 -0
  100. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/druid.py +0 -0
  101. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/dune.py +0 -0
  102. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/exasol.py +0 -0
  103. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/fabric.py +0 -0
  104. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/hive.py +0 -0
  105. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/materialize.py +0 -0
  106. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/oracle.py +0 -0
  107. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/presto.py +0 -0
  108. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/redshift.py +0 -0
  109. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/risingwave.py +0 -0
  110. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/singlestore.py +0 -0
  111. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/solr.py +0 -0
  112. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/spark2.py +0 -0
  113. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/starrocks.py +0 -0
  114. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/parsers/tableau.py +0 -0
  115. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/serde.py +0 -0
  116. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/time.py +0 -0
  117. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglot/trie.py +0 -0
  118. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglotc.egg-info/dependency_links.txt +0 -0
  119. {sqlglotc-30.14.0 → sqlglotc-30.16.0}/sqlglotc.egg-info/top_level.txt +0 -0
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlglotc
3
- Version: 30.14.0
3
+ Version: 30.16.0
4
4
  Summary: mypyc-compiled extensions for sqlglot
5
5
  Author-email: Toby Mao <toby.mao@gmail.com>
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://sqlglot.com/
8
8
  Project-URL: Repository, https://github.com/tobymao/sqlglot
9
9
  Requires-Python: >=3.10
10
- Requires-Dist: sqlglot==30.14.0
10
+ Requires-Dist: sqlglot==30.16.0
11
11
  Provides-Extra: dev
12
12
  Requires-Dist: setuptools>=61.0; extra == "dev"
13
13
  Requires-Dist: setuptools_scm; extra == "dev"
@@ -43,6 +43,7 @@ def _subpkg_files(src_dir, subpkg, files=None):
43
43
 
44
44
  def _source_files(src_dir):
45
45
  return [
46
+ "anonymize.py",
46
47
  "errors.py",
47
48
  "generator.py",
48
49
  "helper.py",
@@ -0,0 +1,307 @@
1
+ from __future__ import annotations
2
+
3
+ import string
4
+
5
+ from sqlglot.dialects.dialect import Dialect, DialectType
6
+ from sqlglot.errors import TokenError
7
+ from sqlglot.tokens import Token, TokenType
8
+
9
+
10
+ ALPHABET = string.ascii_lowercase
11
+ ALPHABET_SIZE = len(ALPHABET)
12
+ ANONYMIZED_TYPES = {
13
+ TokenType.BIT_STRING,
14
+ TokenType.BYTE_STRING,
15
+ TokenType.HEX_STRING,
16
+ TokenType.HEREDOC_STRING,
17
+ TokenType.IDENTIFIER,
18
+ TokenType.NATIONAL_STRING,
19
+ TokenType.NUMBER,
20
+ TokenType.RAW_STRING,
21
+ TokenType.STRING,
22
+ TokenType.UNICODE_STRING,
23
+ TokenType.VAR,
24
+ }
25
+ QUOTED_TYPES = ANONYMIZED_TYPES - {TokenType.NUMBER, TokenType.VAR}
26
+ REWRITTEN_TYPES = {TokenType.HINT, TokenType.UNKNOWN}
27
+
28
+
29
+ def anonymize(
30
+ sql_or_tokens: list[Token] | str,
31
+ dialect: DialectType = None,
32
+ ) -> list[Token]:
33
+ """Replaces sensitive tokens (identifiers, strings, numbers) with fixed-width,
34
+ length-preserving, consistent aliases, and blanks out comments and hint bodies. When a
35
+ SQL string is given, it is tokenized with `dialect` first; any un-tokenized remainder
36
+ (e.g. an unterminated literal) is appended as a blanked UNKNOWN token. Mutates and
37
+ returns `sql_or_tokens`.
38
+
39
+ Args:
40
+ sql_or_tokens: The SQL string to anonymize, or its token list.
41
+ dialect: The dialect used to tokenize a SQL string.
42
+ """
43
+ dialect = Dialect.get_or_raise(dialect)
44
+ tokenizer_class = dialect.tokenizer_class
45
+ parser_class = dialect.parser_class
46
+
47
+ errored = False
48
+ if isinstance(sql_or_tokens, str):
49
+ sql = sql_or_tokens
50
+ tokenizer = dialect.tokenizer()
51
+ try:
52
+ tokens = tokenizer.tokenize(sql)
53
+ except TokenError:
54
+ tokens = tokenizer.tokens
55
+ errored = True
56
+ else:
57
+ tokens = sql_or_tokens
58
+ sql = None
59
+
60
+ hint_start = tokenizer_class.HINT_START
61
+ hint_end = tokenizer_class._COMMENTS.get(hint_start)
62
+ nested = tokenizer_class.NESTED_COMMENTS
63
+
64
+ seen: dict[tuple[bool, str], str] = {}
65
+ counter = 0
66
+
67
+ for i, token in enumerate(tokens):
68
+ token.comments = [_blank(comment) for comment in token.comments]
69
+
70
+ if token.token_type == TokenType.HINT:
71
+ # A hint's text is the whole /*+ ... */ comment, so its body is blanked as well
72
+ text, stop = _blank_comment(token.text, 0, hint_start, hint_end, nested)
73
+ token.text = text + _blank(token.text[stop:])
74
+ continue
75
+
76
+ if (
77
+ token.token_type == TokenType.VAR
78
+ and i + 1 < len(tokens)
79
+ and tokens[i + 1].token_type == TokenType.L_PAREN
80
+ ):
81
+ # A function name can live in either registry, e.g. JSON_OBJECT is only in
82
+ # FUNCTION_PARSERS. They're consulted separately to avoid building their union
83
+ name = token.text.upper()
84
+ if name in parser_class.FUNCTIONS or name in parser_class.FUNCTION_PARSERS:
85
+ continue
86
+ if token.token_type not in ANONYMIZED_TYPES:
87
+ continue
88
+ if not token.text:
89
+ continue
90
+
91
+ is_number = token.token_type == TokenType.NUMBER
92
+ key = (is_number, token.text)
93
+ alias = seen.get(key)
94
+ if alias is None:
95
+ seen[key] = (
96
+ _number_alias(counter, token.text) if is_number else _alias(counter, token.text)
97
+ )
98
+ counter += 1
99
+ token.text = seen[key]
100
+
101
+ if sql is not None and errored:
102
+ start = tokens[-1].end + 1 if tokens else 0
103
+ length = len(sql)
104
+ while start < length and sql[start].isspace():
105
+ start += 1
106
+ if start < length:
107
+ # The first two characters are kept so that the delimiter the tokenizer choked
108
+ # on is still visible, e.g. 'u, /*, $$, `u
109
+ tokens.append(
110
+ Token(
111
+ TokenType.UNKNOWN,
112
+ sql[start : start + 2] + "." * (length - start - 2),
113
+ start=start,
114
+ end=length - 1,
115
+ line=sql.count("\n", 0, start) + 1,
116
+ col=start - sql.rfind("\n", 0, start),
117
+ )
118
+ )
119
+
120
+ return tokens
121
+
122
+
123
+ def render(sql: str, tokens: list[Token], dialect: DialectType = None) -> str:
124
+ """Recreates the (anonymized) SQL string from the original `sql` and token positions.
125
+
126
+ Every token is rendered over its own source span, so the result is always as long as
127
+ `sql`: a token that wasn't anonymized is emitted verbatim, keeping the spelling the
128
+ tokenizer normalized away, and an anonymized one has its alias fitted between the
129
+ quotes of its span. The gaps between tokens hold only whitespace and comments, so
130
+ they're redacted rather than reconstructed: comment markers and whitespace survive,
131
+ everything else is blanked. That covers anything the tokenizer didn't reach as well,
132
+ which is a trailing gap whenever `anonymize` wasn't the one to tokenize `sql`.
133
+
134
+ Args:
135
+ sql: The original SQL string.
136
+ tokens: The anonymized tokens to render.
137
+ dialect: The dialect used to identify comments and quotes in `sql`.
138
+ """
139
+ tokenizer_class = Dialect.get_or_raise(dialect).tokenizer_class
140
+ comments = sorted(tokenizer_class._COMMENTS.items(), key=lambda c: len(c[0]), reverse=True)
141
+ nested = tokenizer_class.NESTED_COMMENTS
142
+ quotes = sorted(
143
+ {
144
+ **tokenizer_class._QUOTES,
145
+ **{start: end for start, (end, _) in tokenizer_class._FORMAT_STRINGS.items()},
146
+ **tokenizer_class._IDENTIFIERS,
147
+ }.items(),
148
+ key=lambda quote: (len(quote[0]), len(quote[1])),
149
+ reverse=True,
150
+ )
151
+
152
+ result = []
153
+ prev = 0
154
+
155
+ for token in tokens:
156
+ result.append(_redact(sql[prev : token.start], comments, nested))
157
+ prev = token.end + 1
158
+ span = sql[token.start : prev]
159
+ if token.token_type in REWRITTEN_TYPES:
160
+ # Blanked in place by `anonymize`, or synthesized by it, so already span-shaped
161
+ result.append(token.text)
162
+ elif token.token_type in ANONYMIZED_TYPES:
163
+ result.append(_fit(span, token.text, quotes, token.token_type))
164
+ else:
165
+ result.append(span)
166
+
167
+ result.append(_redact(sql[prev:], comments, nested))
168
+
169
+ return "".join(result)
170
+
171
+
172
+ def _alias(counter: int, text: str) -> str:
173
+ digits = []
174
+ while counter:
175
+ counter, digit = divmod(counter, ALPHABET_SIZE)
176
+ digits.append(ALPHABET[digit])
177
+
178
+ letters = "".join(reversed(digits)).rjust(sum(not char.isspace() for char in text), "a")
179
+
180
+ alias = []
181
+ i = 0
182
+ for char in text:
183
+ if char.isspace():
184
+ alias.append(char)
185
+ else:
186
+ alias.append(letters[i])
187
+ i += 1
188
+
189
+ return "".join(alias)
190
+
191
+
192
+ def _number_alias(counter: int, text: str) -> str:
193
+ if len(text) > 4000:
194
+ digits_seen = False
195
+ blanked = []
196
+ for char in text:
197
+ if char.isdigit():
198
+ blanked.append("0" if digits_seen else "1")
199
+ digits_seen = True
200
+ else:
201
+ blanked.append(char)
202
+ return "".join(blanked)
203
+
204
+ sep = "e" if "e" in text else ("E" if "E" in text else "")
205
+ mantissa, _, exponent = text.partition(sep) if sep else (text, "", "")
206
+ sign = ""
207
+ if exponent.startswith(("-", "+")):
208
+ sign, exponent = exponent[0], exponent[1:]
209
+ exponent_length = len(exponent)
210
+
211
+ integer, dot, fraction = mantissa.partition(".")
212
+ integer_length = len(integer)
213
+ digits = integer_length + len(fraction)
214
+ mantissa_value = 10 ** (digits - 1) + counter % (9 * 10 ** (digits - 1))
215
+ result = str(mantissa_value)
216
+ if dot:
217
+ result = result[:integer_length] + "." + result[integer_length:]
218
+ if sep:
219
+ # The exponent marker is kept even when there are no digits after it, e.g. 1e
220
+ result += sep + sign
221
+ if exponent_length:
222
+ exponent_value = 10 ** (exponent_length - 1) + (counter // (9 * 10 ** (digits - 1))) % (
223
+ 9 * 10 ** (exponent_length - 1)
224
+ )
225
+ result += str(exponent_value)
226
+
227
+ return result
228
+
229
+
230
+ def _fit(span: str, alias: str, quotes: list[tuple[str, str]], token_type: TokenType) -> str:
231
+ """Fits `alias` into the quoted region of `span`, so the two are always the same length."""
232
+ start = end = 0
233
+
234
+ if token_type in QUOTED_TYPES:
235
+ for open_quote, close_quote in quotes:
236
+ if (
237
+ len(span) >= len(open_quote) + len(close_quote)
238
+ and span.startswith(open_quote)
239
+ and span.endswith(close_quote)
240
+ ):
241
+ start, end = len(open_quote), len(close_quote)
242
+
243
+ if token_type == TokenType.HEREDOC_STRING:
244
+ # A heredoc's tag is part of its delimiter, e.g. $tag$body$tag$
245
+ open_end = span.find(close_quote, start)
246
+ close_start = span.rfind(open_quote, 0, len(span) - end)
247
+ if start <= open_end < close_start:
248
+ start, end = open_end + len(close_quote), len(span) - close_start
249
+
250
+ break
251
+
252
+ width = len(span) - start - end
253
+ pad = "0" if token_type == TokenType.NUMBER else "a"
254
+
255
+ return span[:start] + alias[:width].rjust(width, pad) + span[len(span) - end :]
256
+
257
+
258
+ def _blank(sql: str) -> str:
259
+ return "".join(char if char.isspace() else "." for char in sql)
260
+
261
+
262
+ def _blank_comment(sql: str, i: int, start: str, end: str | None, nested: bool) -> tuple[str, int]:
263
+ """Blanks the body of the comment at `i`, returning its text and the index past it."""
264
+ body = i + len(start)
265
+
266
+ if not end:
267
+ stop = sql.find("\n", body)
268
+ stop = len(sql) if stop == -1 else stop
269
+ return start + _blank(sql[body:stop]), stop
270
+
271
+ depth = 1
272
+ j = body
273
+ while j < len(sql):
274
+ if nested and sql.startswith(start, j):
275
+ depth += 1
276
+ j += len(start)
277
+ elif sql.startswith(end, j):
278
+ j += len(end)
279
+ depth -= 1
280
+ if not depth:
281
+ return start + _blank(sql[body : j - len(end)]) + end, j
282
+ else:
283
+ j += 1
284
+
285
+ return start + _blank(sql[body:]), len(sql)
286
+
287
+
288
+ def _redact(sql: str, comments: list[tuple[str, str | None]], nested: bool) -> str:
289
+ if not sql or sql.isspace():
290
+ return sql
291
+
292
+ result = []
293
+ i = 0
294
+ length = len(sql)
295
+
296
+ while i < length:
297
+ for start, end in comments:
298
+ if sql.startswith(start, i):
299
+ text, i = _blank_comment(sql, i, start, end, nested)
300
+ result.append(text)
301
+ break
302
+ else:
303
+ char = sql[i]
304
+ result.append(char if char.isspace() else ".")
305
+ i += 1
306
+
307
+ return "".join(result)
@@ -72,7 +72,21 @@ class ParseError(SqlglotError):
72
72
 
73
73
 
74
74
  class TokenError(SqlglotError):
75
- pass
75
+ """Error raised when tokenizing fails.
76
+
77
+ When available, `start` and `end` are the offsets in the source SQL of the context
78
+ snippet quoted in the message, i.e. the snippet is `sql[start:end]`.
79
+ """
80
+
81
+ def __init__(
82
+ self,
83
+ message: str,
84
+ start: int | None = None,
85
+ end: int | None = None,
86
+ ):
87
+ super().__init__(message)
88
+ self.start = start
89
+ self.end = end
76
90
 
77
91
 
78
92
  class OptimizeError(SqlglotError):
@@ -8,6 +8,7 @@ from sqlglot.expressions.core import (
8
8
  Expr,
9
9
  Func,
10
10
  Binary,
11
+ Predicate,
11
12
  to_identifier,
12
13
  )
13
14
  from sqlglot.helper import trait
@@ -109,16 +110,16 @@ class ArrayAny(Expression, Func):
109
110
  arg_types = {"this": True, "expression": True}
110
111
 
111
112
 
112
- class ArrayContains(Expression, Binary, Func):
113
+ class ArrayContains(Expression, Binary, Predicate, Func):
113
114
  arg_types = {"this": True, "expression": True, "ensure_variant": False, "check_null": False}
114
115
  _sql_names = ["ARRAY_CONTAINS", "ARRAY_HAS"]
115
116
 
116
117
 
117
- class ArrayContainsAll(Expression, Binary, Func):
118
+ class ArrayContainsAll(Expression, Binary, Predicate, Func):
118
119
  _sql_names = ["ARRAY_CONTAINS_ALL", "ARRAY_HAS_ALL"]
119
120
 
120
121
 
121
- class ArrayContainedBy(Expression, Binary, Func):
122
+ class ArrayContainedBy(Expression, Binary, Predicate, Func):
122
123
  pass
123
124
 
124
125
 
@@ -132,7 +133,7 @@ class ArrayIntersect(Expression, Func):
132
133
  _sql_names = ["ARRAY_INTERSECT", "ARRAY_INTERSECTION"]
133
134
 
134
135
 
135
- class ArrayOverlaps(Expression, Binary, Func):
136
+ class ArrayOverlaps(Expression, Binary, Predicate, Func):
136
137
  arg_types = {"this": True, "expression": True, "null_safe": False}
137
138
 
138
139
 
@@ -343,6 +344,7 @@ class ToMap(Expression, Func):
343
344
  class VarMap(Expression, Func):
344
345
  arg_types = {"keys": True, "values": True}
345
346
  is_var_len_args = True
347
+ var_len_arg_key = "values"
346
348
 
347
349
  @property
348
350
  def keys(self) -> list[Expr]:
@@ -14,7 +14,7 @@ from builtins import type as Type
14
14
  from collections import deque
15
15
  from collections.abc import Collection, Iterator, Mapping, MutableMapping, Sequence
16
16
  from copy import deepcopy
17
- from decimal import Decimal
17
+ from decimal import Decimal, InvalidOperation
18
18
  from functools import reduce
19
19
 
20
20
  from sqlglot._typing import E, GeneratorNoDialectArgs, ParserNoDialectArgs, T
@@ -85,6 +85,7 @@ class Expr:
85
85
  arg_types: t.ClassVar[dict[str, bool]] = {"this": True}
86
86
  required_args: t.ClassVar[set[str]] = {"this"}
87
87
  is_var_len_args: t.ClassVar[bool] = False
88
+ var_len_arg_key: t.ClassVar[str] = "expressions"
88
89
  _hash_raw_args: t.ClassVar[bool] = False
89
90
  is_subquery: t.ClassVar[bool] = False
90
91
  is_cast: t.ClassVar[bool] = False
@@ -1577,7 +1578,7 @@ class Condition(Expr):
1577
1578
 
1578
1579
  @trait
1579
1580
  class Predicate(Condition):
1580
- """Relationships like x = y, x > 1, x >= y."""
1581
+ """Any condition that evaluates to a boolean, e.g. x = y, x LIKE 'a%', a @> b."""
1581
1582
 
1582
1583
 
1583
1584
  class Cache(Expression):
@@ -1643,8 +1644,11 @@ class Func(Condition):
1643
1644
  The base class for all function expressions.
1644
1645
 
1645
1646
  Attributes:
1646
- is_var_len_args (bool): if set to True the last argument defined in arg_types will be
1647
+ is_var_len_args (bool): if set to True the argument identified by var_len_arg_key will be
1647
1648
  treated as a variable length argument and the argument's value will be stored as a list.
1649
+ var_len_arg_key (str): the arg_types key that collects the variable length arguments.
1650
+ Arguments preceding it in arg_types are filled positionally; those following it (e.g.
1651
+ dialect flags) are never populated by from_arg_list.
1648
1652
  _sql_names (list): the SQL name (1st item in the list) and aliases (subsequent items) for this
1649
1653
  function expression. These values are used to map this node to a name during parsing as
1650
1654
  well as to provide the function's name during SQL string generation. By default the SQL
@@ -1652,18 +1656,17 @@ class Func(Condition):
1652
1656
  """
1653
1657
 
1654
1658
  is_var_len_args: t.ClassVar[bool] = False
1659
+ var_len_arg_key: t.ClassVar[str] = "expressions"
1655
1660
  _sql_names: t.ClassVar[list[str]] = []
1656
1661
 
1657
1662
  @classmethod
1658
1663
  def from_arg_list(cls, args: Sequence[object]) -> Self:
1659
1664
  if cls.is_var_len_args:
1660
1665
  all_arg_keys = tuple(cls.arg_types)
1661
- # If this function supports variable length argument treat the last argument as such.
1662
- non_var_len_arg_keys = all_arg_keys[:-1] if cls.is_var_len_args else all_arg_keys
1663
- num_non_var = len(non_var_len_arg_keys)
1666
+ var_len_index = all_arg_keys.index(cls.var_len_arg_key)
1664
1667
 
1665
- args_dict = {arg_key: arg for arg, arg_key in zip(args, non_var_len_arg_keys)}
1666
- args_dict[all_arg_keys[-1]] = args[num_non_var:]
1668
+ args_dict = {arg_key: arg for arg, arg_key in zip(args, all_arg_keys[:var_len_index])}
1669
+ args_dict[cls.var_len_arg_key] = args[var_len_index:]
1667
1670
  else:
1668
1671
  args_dict = {arg_key: arg for arg, arg_key in zip(args, cls.arg_types)}
1669
1672
 
@@ -1764,7 +1767,10 @@ class Literal(Expression, Condition):
1764
1767
  try:
1765
1768
  return int(self.this)
1766
1769
  except ValueError:
1767
- return Decimal(self.this)
1770
+ try:
1771
+ return Decimal(self.this)
1772
+ except InvalidOperation as e:
1773
+ raise ValueError(f"Invalid numeric literal: {self.this!r}") from e
1768
1774
  return self.this
1769
1775
 
1770
1776
 
@@ -2119,15 +2125,15 @@ class Div(Expression, Binary):
2119
2125
  arg_types = {"this": True, "expression": True, "typed": False, "safe": False}
2120
2126
 
2121
2127
 
2122
- class Overlaps(Expression, Binary):
2128
+ class Overlaps(Expression, Binary, Predicate):
2123
2129
  pass
2124
2130
 
2125
2131
 
2126
- class ExtendsLeft(Expression, Binary):
2132
+ class ExtendsLeft(Expression, Binary, Predicate):
2127
2133
  pass
2128
2134
 
2129
2135
 
2130
- class ExtendsRight(Expression, Binary):
2136
+ class ExtendsRight(Expression, Binary, Predicate):
2131
2137
  pass
2132
2138
 
2133
2139
 
@@ -2231,7 +2237,7 @@ class Sub(Expression, Binary):
2231
2237
  pass
2232
2238
 
2233
2239
 
2234
- class Adjacent(Expression, Binary):
2240
+ class Adjacent(Expression, Binary, Predicate):
2235
2241
  pass
2236
2242
 
2237
2243
 
@@ -2317,7 +2323,7 @@ class Pow(Expression, Binary, Func):
2317
2323
  _sql_names = ["POWER", "POW"]
2318
2324
 
2319
2325
 
2320
- class RegexpLike(Expression, Binary, Func):
2326
+ class RegexpLike(Expression, Binary, Predicate, Func):
2321
2327
  arg_types = {"this": True, "expression": True, "flag": False, "full_match": False}
2322
2328
 
2323
2329
 
@@ -220,10 +220,22 @@ class DataType(Expression):
220
220
  DType.NCHAR,
221
221
  DType.NVARCHAR,
222
222
  DType.TEXT,
223
+ DType.TINYTEXT,
224
+ DType.MEDIUMTEXT,
225
+ DType.LONGTEXT,
223
226
  DType.VARCHAR,
224
227
  DType.NAME,
225
228
  }
226
229
 
230
+ BINARY_TYPES: t.ClassVar[set[DType]] = {
231
+ DType.BINARY,
232
+ DType.VARBINARY,
233
+ DType.TINYBLOB,
234
+ DType.BLOB,
235
+ DType.MEDIUMBLOB,
236
+ DType.LONGBLOB,
237
+ }
238
+
227
239
  SIGNED_INTEGER_TYPES: t.ClassVar[set[DType]] = {
228
240
  DType.BIGINT,
229
241
  DType.INT,
@@ -250,6 +250,7 @@ class AlterColumn(Expression):
250
250
  "allow_null": False,
251
251
  "visible": False,
252
252
  "rename_to": False,
253
+ "exists": False,
253
254
  }
254
255
 
255
256
 
@@ -358,6 +359,7 @@ class Drop(Expression):
358
359
  "concurrently": False,
359
360
  "sync": False,
360
361
  "iceberg": False,
362
+ "force": False,
361
363
  }
362
364
 
363
365
  @property
@@ -212,6 +212,7 @@ class Insert(Expression, DDL, DML):
212
212
  "settings": False,
213
213
  "source": False,
214
214
  "default": False,
215
+ "using": False,
215
216
  }
216
217
 
217
218
  def with_(
@@ -46,15 +46,15 @@ class JSONArrayInsert(Expression, Func):
46
46
  _sql_names = ["JSON_ARRAY_INSERT"]
47
47
 
48
48
 
49
- class JSONBContains(Expression, Binary, Func):
49
+ class JSONBContains(Expression, Binary, Predicate, Func):
50
50
  _sql_names = ["JSONB_CONTAINS"]
51
51
 
52
52
 
53
- class JSONBContainsAllTopKeys(Expression, Binary, Func):
53
+ class JSONBContainsAllTopKeys(Expression, Binary, Predicate, Func):
54
54
  pass
55
55
 
56
56
 
57
- class JSONBContainsAnyTopKeys(Expression, Binary, Func):
57
+ class JSONBContainsAnyTopKeys(Expression, Binary, Predicate, Func):
58
58
  pass
59
59
 
60
60
 
@@ -133,6 +133,7 @@ class JSONExtractScalar(Expression, Binary, Func):
133
133
  "expressions": False,
134
134
  "json_type": False,
135
135
  "scalar_only": False,
136
+ "json_subtype": False,
136
137
  }
137
138
  _sql_names = ["JSON_EXTRACT_SCALAR"]
138
139
  is_var_len_args = True
@@ -164,6 +164,10 @@ class Log(Expression, Func):
164
164
  arg_types = {"this": True, "expression": False}
165
165
 
166
166
 
167
+ class Negative(Expression, Func):
168
+ pass
169
+
170
+
167
171
  class Nanvl(Expression, Func):
168
172
  arg_types = {"this": True, "expression": True}
169
173
 
@@ -42,7 +42,15 @@ class AutoIncrementProperty(Property):
42
42
 
43
43
 
44
44
  class AutoRefreshProperty(Property):
45
- arg_types = {"this": True}
45
+ arg_types = {
46
+ "this": False,
47
+ "cadence": False,
48
+ "offset": False,
49
+ "randomize": False,
50
+ "expressions": False,
51
+ "settings": False,
52
+ "append": False,
53
+ }
46
54
 
47
55
 
48
56
  class BackupProperty(Property):
@@ -444,7 +444,7 @@ class RecursiveWithSearch(Expression):
444
444
 
445
445
 
446
446
  class With(Expression):
447
- arg_types = {"expressions": True, "recursive": False, "search": False}
447
+ arg_types = {"expressions": False, "recursive": False, "search": False, "udfs": False}
448
448
 
449
449
  @property
450
450
  def recursive(self) -> bool:
@@ -1761,6 +1761,7 @@ class Pivot(Expression):
1761
1761
  "identify_pivot_strings": False,
1762
1762
  "prefixed_pivot_columns": False,
1763
1763
  "pivot_column_naming": False,
1764
+ "value_columns_first": False,
1764
1765
  }
1765
1766
 
1766
1767
  @property
@@ -1814,7 +1815,13 @@ class Pivot(Expression):
1814
1815
  for ident in (e.expressions if isinstance(e, Tuple) else [e])
1815
1816
  if isinstance(ident, Identifier)
1816
1817
  ]
1817
- outputs = [i.name for i in name_columns + value_columns]
1818
+ # T-SQL emits the value column(s) ahead of the name column, everyone else emits them after it
1819
+ ordered = (
1820
+ value_columns + name_columns
1821
+ if self.args.get("value_columns_first")
1822
+ else name_columns + value_columns
1823
+ )
1824
+ outputs = [i.name for i in ordered]
1818
1825
  else:
1819
1826
  excluded = {c.output_name for c in self.find_all(Column)}
1820
1827
  outputs = [c.output_name for c in self.args.get("columns") or []]
@@ -2105,13 +2112,17 @@ class StoredProcedure(Expression):
2105
2112
 
2106
2113
 
2107
2114
  class Block(Expression):
2108
- arg_types = {"expressions": True}
2115
+ arg_types = {"expressions": True, "begin": False}
2109
2116
 
2110
2117
 
2111
2118
  class IfBlock(Expression):
2112
2119
  arg_types = {"this": True, "true": True, "false": False}
2113
2120
 
2114
2121
 
2122
+ class CaseStatement(Expression):
2123
+ arg_types = {"this": False, "ifs": True, "default": False}
2124
+
2125
+
2115
2126
  class WhileBlock(Expression):
2116
2127
  arg_types = {"this": True, "body": True}
2117
2128
 
@@ -2120,6 +2131,16 @@ class EndStatement(Expression):
2120
2131
  arg_types = {}
2121
2132
 
2122
2133
 
2134
+ # https://trino.io/docs/current/udf.html
2135
+ class FunctionSpecification(Expression):
2136
+ arg_types = {
2137
+ "this": True,
2138
+ "characteristics": False,
2139
+ "properties": False,
2140
+ "expression": True,
2141
+ }
2142
+
2143
+
2123
2144
  UNWRAPPED_QUERIES = (Select, SetOperation)
2124
2145
 
2125
2146