get-objects-lib 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- get_objects_lib/__init__.py +12 -0
- get_objects_lib/dialect/__init__.py +3 -0
- get_objects_lib/dialect/dialect.py +11 -0
- get_objects_lib/dialect/expressions.py +116 -0
- get_objects_lib/dialect/generator.py +231 -0
- get_objects_lib/dialect/parser/__init__.py +3 -0
- get_objects_lib/dialect/parser/base.py +278 -0
- get_objects_lib/dialect/parser/non_semicolon.py +182 -0
- get_objects_lib/dialect/parser/sql_server.py +512 -0
- get_objects_lib/dialect/text_utils.py +22 -0
- get_objects_lib/dialect/tokenizer.py +51 -0
- get_objects_lib/objects/__init__.py +3 -0
- get_objects_lib/objects/dependencies.py +162 -0
- get_objects_lib/objects/header.py +160 -0
- get_objects_lib/objects/standardize.py +72 -0
- get_objects_lib-0.1.0.dist-info/METADATA +231 -0
- get_objects_lib-0.1.0.dist-info/RECORD +18 -0
- get_objects_lib-0.1.0.dist-info/WHEEL +4 -0
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
import typing as t
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
from typing import Generator
|
|
4
|
+
|
|
5
|
+
from sqlglot import Parser, Token, exp, TokenType
|
|
6
|
+
|
|
7
|
+
from get_objects_lib.dialect.parser.base import BaseParser
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class NonSemiColonParser(BaseParser):
|
|
11
|
+
@staticmethod
|
|
12
|
+
def iter_around(items: t.Sequence[Token]) -> Generator[tuple[int, Token | None, Token, Token | None]]:
|
|
13
|
+
for i, item in enumerate(items):
|
|
14
|
+
_prev = items[i - 1] if i > 0 else None
|
|
15
|
+
_next = items[i + 1] if i + 1 < len(items) else None
|
|
16
|
+
yield i, _prev, item, _next
|
|
17
|
+
|
|
18
|
+
def _parse(
|
|
19
|
+
self,
|
|
20
|
+
parse_method: t.Callable[[Parser], exp.Expr | None],
|
|
21
|
+
raw_tokens: list[Token],
|
|
22
|
+
sql: str | None = None,
|
|
23
|
+
) -> list[exp.Expr | None]:
|
|
24
|
+
self.reset()
|
|
25
|
+
self.sql = sql or ""
|
|
26
|
+
|
|
27
|
+
chunks: list[list[Token]] = [[]]
|
|
28
|
+
raw_tokens = list(filter(lambda x: x.token_type != TokenType.SEMICOLON, raw_tokens))
|
|
29
|
+
levels = defaultdict(lambda: False)
|
|
30
|
+
begins = []
|
|
31
|
+
|
|
32
|
+
block_without_begin = False
|
|
33
|
+
# Profundidad de parentesis: un ")" interno no cierra el exterior
|
|
34
|
+
paren_depth = 0
|
|
35
|
+
updating = False
|
|
36
|
+
inserting = False
|
|
37
|
+
added = False
|
|
38
|
+
skip = 0
|
|
39
|
+
case_depth = 0
|
|
40
|
+
prev_closed_case = False
|
|
41
|
+
# Un IF/ELSE/WHILE espera su cuerpo: un BEGIN aqui es ese cuerpo
|
|
42
|
+
pending_body = False
|
|
43
|
+
# Tras WITH cte AS (...) la sentencia principal no abre otra
|
|
44
|
+
cte_pending = False
|
|
45
|
+
|
|
46
|
+
for i, _prev, token, _next in self.iter_around(raw_tokens):
|
|
47
|
+
if skip:
|
|
48
|
+
skip -= 1
|
|
49
|
+
continue
|
|
50
|
+
|
|
51
|
+
# GO cierra el lote: todo lo abierto termina aqui y GO [n] queda
|
|
52
|
+
# como su propia sentencia.
|
|
53
|
+
separator_length = self._batch_separator_length(raw_tokens, i)
|
|
54
|
+
if separator_length:
|
|
55
|
+
chunks.append(raw_tokens[i:i + separator_length])
|
|
56
|
+
chunks.append([])
|
|
57
|
+
skip = separator_length - 1
|
|
58
|
+
levels.clear()
|
|
59
|
+
begins.clear()
|
|
60
|
+
block_without_begin = updating = inserting = added = False
|
|
61
|
+
case_depth = paren_depth = 0
|
|
62
|
+
pending_body = False
|
|
63
|
+
continue
|
|
64
|
+
|
|
65
|
+
# Dentro de un CASE, ELSE y END son parte de la expresion, no
|
|
66
|
+
# cierran ni abren sentencias.
|
|
67
|
+
in_case = bool(case_depth)
|
|
68
|
+
closes_case = bool(case_depth) and self._is_end(token)
|
|
69
|
+
if token.token_type == TokenType.CASE:
|
|
70
|
+
case_depth += 1
|
|
71
|
+
elif closes_case:
|
|
72
|
+
case_depth -= 1
|
|
73
|
+
|
|
74
|
+
# FOR UPDATE de un cursor, el evento UPDATE de un trigger y la funcion
|
|
75
|
+
# UPDATE(columna) no son un UPDATE
|
|
76
|
+
if (
|
|
77
|
+
token.token_type == TokenType.UPDATE
|
|
78
|
+
and not (_prev and _prev.token_type == TokenType.FOR)
|
|
79
|
+
and self._is_open_token(token, _prev, _next)
|
|
80
|
+
):
|
|
81
|
+
updating = True
|
|
82
|
+
|
|
83
|
+
# INSERT ... SELECT / INSERT ... EXEC: el primer SELECT o EXEC es el
|
|
84
|
+
# origen del INSERT, salvo que antes aparezca VALUES.
|
|
85
|
+
if token.token_type == TokenType.INSERT:
|
|
86
|
+
inserting = True
|
|
87
|
+
if token.token_type == TokenType.VALUES:
|
|
88
|
+
inserting = False
|
|
89
|
+
|
|
90
|
+
if token.token_type == TokenType.L_PAREN:
|
|
91
|
+
paren_depth += 1
|
|
92
|
+
|
|
93
|
+
if token.token_type == TokenType.R_PAREN and paren_depth:
|
|
94
|
+
paren_depth -= 1
|
|
95
|
+
|
|
96
|
+
if self._is_end(token) and not in_case:
|
|
97
|
+
levels[len(begins)] = False
|
|
98
|
+
chunks.append([])
|
|
99
|
+
if begins: begins.pop()
|
|
100
|
+
|
|
101
|
+
# Lo que sigue a "AS" es el cuerpo de un VIEW/PROCEDURE/FUNCTION,
|
|
102
|
+
# no una sentencia nueva.
|
|
103
|
+
# Lo mismo despues de FOR (DECLARE c CURSOR FOR SELECT / FOR UPDATE) y
|
|
104
|
+
# de THEN (MERGE ... WHEN MATCHED THEN UPDATE / INSERT / DELETE)
|
|
105
|
+
after_as = _prev is not None and _prev.token_type in (TokenType.ALIAS, TokenType.FOR, TokenType.THEN)
|
|
106
|
+
|
|
107
|
+
# BEGIN que nadie esperaba (ni IF/ELSE/WHILE ni AS): bloque suelto,
|
|
108
|
+
# abre una sentencia nueva
|
|
109
|
+
is_block_begin = self._is_begin(token, _next) and not in_case and not paren_depth
|
|
110
|
+
bare_begin = is_block_begin and not pending_body and not after_as
|
|
111
|
+
starts_cte = self._starts_cte(raw_tokens, i) and not paren_depth and not in_case
|
|
112
|
+
is_open = self._is_open_token(token, _prev, _next) or bare_begin or starts_cte
|
|
113
|
+
|
|
114
|
+
# SELECT/INSERT/UPDATE/DELETE/MERGE que usa el CTE: sigue en su sentencia
|
|
115
|
+
if (
|
|
116
|
+
cte_pending and is_open and not paren_depth and not in_case
|
|
117
|
+
and token.token_type in (TokenType.SELECT, TokenType.INSERT, TokenType.UPDATE, TokenType.DELETE, TokenType.MERGE)
|
|
118
|
+
):
|
|
119
|
+
cte_pending = False
|
|
120
|
+
is_open = False
|
|
121
|
+
if token.token_type == TokenType.UPDATE:
|
|
122
|
+
updating = True
|
|
123
|
+
if token.token_type == TokenType.INSERT:
|
|
124
|
+
inserting = True
|
|
125
|
+
|
|
126
|
+
if is_open and not paren_depth and not after_as and not in_case:
|
|
127
|
+
if block_without_begin:
|
|
128
|
+
block_without_begin = False
|
|
129
|
+
begins.append(token)
|
|
130
|
+
added = True
|
|
131
|
+
|
|
132
|
+
if updating and token.token_type == TokenType.SET:
|
|
133
|
+
updating = False
|
|
134
|
+
elif inserting and token.token_type in (TokenType.SELECT, TokenType.EXECUTE):
|
|
135
|
+
inserting = False
|
|
136
|
+
else:
|
|
137
|
+
if token.token_type != TokenType.INSERT:
|
|
138
|
+
inserting = False
|
|
139
|
+
|
|
140
|
+
level = len(begins)
|
|
141
|
+
if levels[level]:
|
|
142
|
+
# END ELSE de un bloque se queda junto; el END de un CASE no
|
|
143
|
+
block_end = _prev.token_type == TokenType.END and not prev_closed_case
|
|
144
|
+
if not block_end or token.token_type != TokenType.ELSE:
|
|
145
|
+
chunks.append([])
|
|
146
|
+
|
|
147
|
+
levels[level] = True
|
|
148
|
+
|
|
149
|
+
if starts_cte:
|
|
150
|
+
cte_pending = True
|
|
151
|
+
|
|
152
|
+
starts_body = (
|
|
153
|
+
self._is_word(token, 'IF', 'ELSE', 'WHILE')
|
|
154
|
+
and not in_case
|
|
155
|
+
and self._is_open_token(token, _prev, _next) # no el IF de DROP TABLE IF EXISTS
|
|
156
|
+
)
|
|
157
|
+
if starts_body:
|
|
158
|
+
block_without_begin = self._is_block_without_begin(raw_tokens, i)
|
|
159
|
+
|
|
160
|
+
if starts_body:
|
|
161
|
+
pending_body = True
|
|
162
|
+
elif is_block_begin or (is_open and not paren_depth and not in_case):
|
|
163
|
+
pending_body = False
|
|
164
|
+
|
|
165
|
+
chunks[-1].append(token)
|
|
166
|
+
|
|
167
|
+
if self._is_begin(token, _next):
|
|
168
|
+
begins.append(token)
|
|
169
|
+
|
|
170
|
+
prev_closed_case = closes_case
|
|
171
|
+
|
|
172
|
+
_next_next = raw_tokens[i + 2] if i + 2 < len(raw_tokens) else None
|
|
173
|
+
# Si seguimos dentro de un CASE, el siguiente token es parte de el
|
|
174
|
+
if added and _next and not case_depth and (self._is_end(_next) or self._is_open_token(_next, token, _next_next)):
|
|
175
|
+
levels[len(begins)] = False
|
|
176
|
+
added = False
|
|
177
|
+
begins.pop()
|
|
178
|
+
|
|
179
|
+
chunks = list(filter(lambda x: x, chunks))
|
|
180
|
+
self._chunks = chunks
|
|
181
|
+
|
|
182
|
+
return self._parse_batch_statements(parse_method=parse_method, sep_first_statement=False, top_level=True)
|