ch-migrate-cli 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ch_migrate/__init__.py +68 -0
- ch_migrate/authoring.py +163 -0
- ch_migrate/bootstrap.py +379 -0
- ch_migrate/cli.py +1049 -0
- ch_migrate/config.py +116 -0
- ch_migrate/connection.py +83 -0
- ch_migrate/deps.py +123 -0
- ch_migrate/diff.py +232 -0
- ch_migrate/display.py +450 -0
- ch_migrate/downgrade.py +103 -0
- ch_migrate/env.py +226 -0
- ch_migrate/helpers.py +162 -0
- ch_migrate/hooks.py +74 -0
- ch_migrate/introspect.py +732 -0
- ch_migrate/lint.py +601 -0
- ch_migrate/mv_validate.py +554 -0
- ch_migrate/py.typed +0 -0
- ch_migrate/rebase.py +308 -0
- ch_migrate/runner.py +188 -0
- ch_migrate/scaffold.py +253 -0
- ch_migrate/secrets.py +162 -0
- ch_migrate/skills/ch-migrate/SKILL.md +250 -0
- ch_migrate/sql.py +216 -0
- ch_migrate/statements.py +144 -0
- ch_migrate/templates/bootstrap/init_users.sql +56 -0
- ch_migrate/templates/project/alembic.ini.template +43 -0
- ch_migrate/templates/project/config.yaml.template +55 -0
- ch_migrate/templates/project/env.local.example.template +25 -0
- ch_migrate/templates/project/script.py.mako.template +30 -0
- ch_migrate/ui.py +79 -0
- ch_migrate_cli-0.5.0.dist-info/METADATA +448 -0
- ch_migrate_cli-0.5.0.dist-info/RECORD +36 -0
- ch_migrate_cli-0.5.0.dist-info/WHEEL +4 -0
- ch_migrate_cli-0.5.0.dist-info/entry_points.txt +2 -0
- ch_migrate_cli-0.5.0.dist-info/licenses/LICENSE +21 -0
- clickhouse_alembic/__init__.py +59 -0
ch_migrate/introspect.py
ADDED
|
@@ -0,0 +1,732 @@
|
|
|
1
|
+
"""ClickHouse schema introspection: structured DDL parsing and live schema capture."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections import deque
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from enum import Enum
|
|
9
|
+
from typing import Any, Literal
|
|
10
|
+
|
|
11
|
+
# Alembic's bookkeeping table: not part of the user's schema, so snapshot, diff
|
|
12
|
+
# and deps leave it out.
|
|
13
|
+
VERSION_TABLE = "alembic_version"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
# Data models
|
|
18
|
+
# ---------------------------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ColumnDefinition:
|
|
23
|
+
name: str
|
|
24
|
+
type: str
|
|
25
|
+
default_kind: str | None = None # DEFAULT, MATERIALIZED, ALIAS
|
|
26
|
+
default_expr: str | None = None
|
|
27
|
+
codec: str | None = None
|
|
28
|
+
comment: str | None = None
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class TableDefinition:
|
|
33
|
+
name: str
|
|
34
|
+
engine: str
|
|
35
|
+
columns: list[ColumnDefinition] = field(default_factory=list)
|
|
36
|
+
order_by: list[str] = field(default_factory=list)
|
|
37
|
+
partition_by: str | None = None
|
|
38
|
+
ttl: str | None = None
|
|
39
|
+
settings: dict[str, str] = field(default_factory=dict)
|
|
40
|
+
raw_ddl: str = ""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class ViewDefinition:
|
|
45
|
+
name: str
|
|
46
|
+
select_query: str
|
|
47
|
+
raw_ddl: str = ""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class MVDefinition:
|
|
52
|
+
name: str
|
|
53
|
+
target_table: str | None = None
|
|
54
|
+
source_tables: list[str] = field(default_factory=list)
|
|
55
|
+
select_query: str = ""
|
|
56
|
+
engine: str | None = None
|
|
57
|
+
raw_ddl: str = ""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class DictDefinition:
|
|
62
|
+
name: str
|
|
63
|
+
primary_key: str | None = None
|
|
64
|
+
source_type: str | None = None # e.g. "clickhouse", "mysql", "http"
|
|
65
|
+
source_table: str | None = None
|
|
66
|
+
source_db: str | None = None
|
|
67
|
+
source_query: str | None = None
|
|
68
|
+
layout: str | None = None
|
|
69
|
+
lifetime: str | None = None
|
|
70
|
+
structure_keys: list[str] = field(default_factory=list)
|
|
71
|
+
structure_attributes: list[str] = field(default_factory=list)
|
|
72
|
+
raw_ddl: str = ""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class Schema:
|
|
77
|
+
tables: dict[str, TableDefinition] = field(default_factory=dict)
|
|
78
|
+
views: dict[str, ViewDefinition] = field(default_factory=dict)
|
|
79
|
+
materialized_views: dict[str, MVDefinition] = field(default_factory=dict)
|
|
80
|
+
dictionaries: dict[str, DictDefinition] = field(default_factory=dict)
|
|
81
|
+
database: str = ""
|
|
82
|
+
ch_version: str | None = None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# ---------------------------------------------------------------------------
|
|
86
|
+
# Dependency graph
|
|
87
|
+
# ---------------------------------------------------------------------------
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class DepType(str, Enum):
|
|
91
|
+
SCHEMA = "schema"
|
|
92
|
+
DATA_FLOW = "data_flow"
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass
|
|
96
|
+
class DependencyEdge:
|
|
97
|
+
source: str
|
|
98
|
+
target: str
|
|
99
|
+
dep_type: DepType
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@dataclass
|
|
103
|
+
class ObjectNode:
|
|
104
|
+
name: str
|
|
105
|
+
obj_type: Literal["table", "view", "materialized_view", "dictionary"]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass
|
|
109
|
+
class DependencyGraph:
|
|
110
|
+
nodes: dict[str, ObjectNode] = field(default_factory=dict)
|
|
111
|
+
edges: list[DependencyEdge] = field(default_factory=list)
|
|
112
|
+
|
|
113
|
+
def topological_order(self) -> list[str]:
|
|
114
|
+
"""Return names in safe drop/recreate order (leaves first)."""
|
|
115
|
+
in_degree: dict[str, int] = {name: 0 for name in self.nodes}
|
|
116
|
+
adj: dict[str, list[str]] = {name: [] for name in self.nodes}
|
|
117
|
+
for edge in self.edges:
|
|
118
|
+
if edge.target in in_degree and edge.source in adj:
|
|
119
|
+
adj[edge.source].append(edge.target)
|
|
120
|
+
in_degree[edge.target] += 1
|
|
121
|
+
|
|
122
|
+
queue = deque(n for n, d in in_degree.items() if d == 0)
|
|
123
|
+
result: list[str] = []
|
|
124
|
+
while queue:
|
|
125
|
+
node = queue.popleft()
|
|
126
|
+
result.append(node)
|
|
127
|
+
for neighbor in adj.get(node, []):
|
|
128
|
+
in_degree[neighbor] -= 1
|
|
129
|
+
if in_degree[neighbor] == 0:
|
|
130
|
+
queue.append(neighbor)
|
|
131
|
+
|
|
132
|
+
# Append any remaining nodes (cycles) at the end
|
|
133
|
+
for name in self.nodes:
|
|
134
|
+
if name not in result:
|
|
135
|
+
result.append(name)
|
|
136
|
+
|
|
137
|
+
return result
|
|
138
|
+
|
|
139
|
+
def affected_by_drop(self, name: str) -> list[ObjectNode]:
|
|
140
|
+
"""Return direct dependents that would be affected by dropping the given object."""
|
|
141
|
+
seen: set[str] = set()
|
|
142
|
+
affected: list[ObjectNode] = []
|
|
143
|
+
for edge in self.edges:
|
|
144
|
+
if edge.source == name and edge.target in self.nodes and edge.target not in seen:
|
|
145
|
+
seen.add(edge.target)
|
|
146
|
+
affected.append(self.nodes[edge.target])
|
|
147
|
+
return affected
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# ---------------------------------------------------------------------------
|
|
151
|
+
# DDL parsing
|
|
152
|
+
# ---------------------------------------------------------------------------
|
|
153
|
+
|
|
154
|
+
# Regex patterns for parsing CREATE statements
|
|
155
|
+
_RE_CREATE_TABLE = re.compile(
|
|
156
|
+
r"CREATE\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?"
|
|
157
|
+
r"(?:`?(\w+)`?\.)?`?(\w+)`?"
|
|
158
|
+
r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?\(",
|
|
159
|
+
re.IGNORECASE,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
_RE_ENGINE = re.compile(r"ENGINE\s*=\s*(\w+(?:\(.*?\))?)", re.IGNORECASE)
|
|
163
|
+
|
|
164
|
+
_RE_ORDER_BY = re.compile(r"ORDER\s+BY\s+(.+?)(?=\s*(?:PARTITION|TTL|SETTINGS|$))", re.IGNORECASE)
|
|
165
|
+
|
|
166
|
+
_RE_PARTITION_BY = re.compile(
|
|
167
|
+
r"PARTITION\s+BY\s+(.+?)(?=\s*(?:ORDER|TTL|SETTINGS|$))", re.IGNORECASE
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
_RE_TTL = re.compile(r"TTL\s+(.+?)(?=\s*(?:SETTINGS|$))", re.IGNORECASE)
|
|
171
|
+
|
|
172
|
+
_RE_SETTINGS = re.compile(r"SETTINGS\s+(.+)$", re.IGNORECASE | re.MULTILINE)
|
|
173
|
+
|
|
174
|
+
_RE_CREATE_VIEW = re.compile(
|
|
175
|
+
r"CREATE\s+VIEW\s+(?:IF\s+NOT\s+EXISTS\s+)?"
|
|
176
|
+
r"(?:`?(\w+)`?\.)?`?(\w+)`?"
|
|
177
|
+
r"\s+AS\s+",
|
|
178
|
+
re.IGNORECASE,
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
_RE_CREATE_MV = re.compile(
|
|
182
|
+
r"CREATE\s+MATERIALIZED\s+VIEW\s+(?:IF\s+NOT\s+EXISTS\s+)?"
|
|
183
|
+
r"(?:`?(\w+)`?\.)?`?(\w+)`?"
|
|
184
|
+
r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?",
|
|
185
|
+
re.IGNORECASE,
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
_RE_MV_TO = re.compile(r"TO\s+(?:`?(\w+)`?\.)?`?(\w+)`?", re.IGNORECASE)
|
|
189
|
+
|
|
190
|
+
_RE_CREATE_DICT = re.compile(
|
|
191
|
+
r"CREATE\s+(?:OR\s+REPLACE\s+)?DICTIONARY\s+(?:IF\s+NOT\s+EXISTS\s+)?"
|
|
192
|
+
r"(?:`?(\w+)`?\.)?`?(\w+)`?"
|
|
193
|
+
r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?\(",
|
|
194
|
+
re.IGNORECASE,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _parse_columns(columns_block: str) -> list[ColumnDefinition]:
|
|
199
|
+
"""Parse column definitions from the block between CREATE TABLE (...).
|
|
200
|
+
|
|
201
|
+
Handles nested parentheses in types like Nullable(String), Tuple(a UInt8, b String).
|
|
202
|
+
"""
|
|
203
|
+
columns: list[ColumnDefinition] = []
|
|
204
|
+
depth = 0
|
|
205
|
+
current: list[str] = []
|
|
206
|
+
|
|
207
|
+
for char in columns_block:
|
|
208
|
+
if char == "(":
|
|
209
|
+
depth += 1
|
|
210
|
+
current.append(char)
|
|
211
|
+
elif char == ")":
|
|
212
|
+
depth -= 1
|
|
213
|
+
current.append(char)
|
|
214
|
+
elif char == "," and depth == 0:
|
|
215
|
+
line = "".join(current).strip()
|
|
216
|
+
if line:
|
|
217
|
+
col = _parse_single_column(line)
|
|
218
|
+
if col:
|
|
219
|
+
columns.append(col)
|
|
220
|
+
current = []
|
|
221
|
+
else:
|
|
222
|
+
current.append(char)
|
|
223
|
+
|
|
224
|
+
# Last column (no trailing comma)
|
|
225
|
+
line = "".join(current).strip()
|
|
226
|
+
if line:
|
|
227
|
+
col = _parse_single_column(line)
|
|
228
|
+
if col:
|
|
229
|
+
columns.append(col)
|
|
230
|
+
|
|
231
|
+
return columns
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _parse_single_column(line: str) -> ColumnDefinition | None:
|
|
235
|
+
"""Parse a single column definition line."""
|
|
236
|
+
line = line.strip()
|
|
237
|
+
if not line:
|
|
238
|
+
return None
|
|
239
|
+
|
|
240
|
+
# Skip constraints (INDEX, PROJECTION, CONSTRAINT)
|
|
241
|
+
if re.match(r"(?:INDEX|PROJECTION|CONSTRAINT)\s+", line, re.IGNORECASE):
|
|
242
|
+
return None
|
|
243
|
+
|
|
244
|
+
# Match: `name` Type [DEFAULT|MATERIALIZED|ALIAS expr] [CODEC(...)] [COMMENT '...']
|
|
245
|
+
m = re.match(r"`?(\w+)`?\s+(.+)", line)
|
|
246
|
+
if not m:
|
|
247
|
+
return None
|
|
248
|
+
|
|
249
|
+
name = m.group(1)
|
|
250
|
+
rest = m.group(2)
|
|
251
|
+
|
|
252
|
+
# Extract COMMENT
|
|
253
|
+
comment = None
|
|
254
|
+
comment_match = re.search(r"COMMENT\s+'((?:[^'\\]|\\.)*)'", rest, re.IGNORECASE)
|
|
255
|
+
if comment_match:
|
|
256
|
+
comment = comment_match.group(1)
|
|
257
|
+
rest = rest[: comment_match.start()].rstrip()
|
|
258
|
+
|
|
259
|
+
# Extract CODEC
|
|
260
|
+
codec = None
|
|
261
|
+
codec_match = re.search(r"CODEC\s*\((.+?)\)\s*$", rest, re.IGNORECASE)
|
|
262
|
+
if codec_match:
|
|
263
|
+
codec = codec_match.group(1)
|
|
264
|
+
rest = rest[: codec_match.start()].rstrip()
|
|
265
|
+
|
|
266
|
+
# Extract DEFAULT/MATERIALIZED/ALIAS
|
|
267
|
+
default_kind = None
|
|
268
|
+
default_expr = None
|
|
269
|
+
default_match = re.search(
|
|
270
|
+
r"\b(DEFAULT|MATERIALIZED|ALIAS)\s+(.+)$", rest, re.IGNORECASE
|
|
271
|
+
)
|
|
272
|
+
if default_match:
|
|
273
|
+
default_kind = default_match.group(1).upper()
|
|
274
|
+
default_expr = default_match.group(2).strip()
|
|
275
|
+
rest = rest[: default_match.start()].rstrip()
|
|
276
|
+
|
|
277
|
+
col_type = rest.strip()
|
|
278
|
+
|
|
279
|
+
return ColumnDefinition(
|
|
280
|
+
name=name,
|
|
281
|
+
type=col_type,
|
|
282
|
+
default_kind=default_kind,
|
|
283
|
+
default_expr=default_expr,
|
|
284
|
+
codec=codec,
|
|
285
|
+
comment=comment,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _extract_columns_block(ddl: str, start_paren_pos: int) -> str:
|
|
290
|
+
"""Extract the columns block from CREATE TABLE, handling nested parens."""
|
|
291
|
+
depth = 0
|
|
292
|
+
i = start_paren_pos
|
|
293
|
+
while i < len(ddl):
|
|
294
|
+
if ddl[i] == "(":
|
|
295
|
+
depth += 1
|
|
296
|
+
elif ddl[i] == ")":
|
|
297
|
+
depth -= 1
|
|
298
|
+
if depth == 0:
|
|
299
|
+
return ddl[start_paren_pos + 1 : i]
|
|
300
|
+
i += 1
|
|
301
|
+
return ddl[start_paren_pos + 1 :]
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _parse_order_by(expr: str) -> list[str]:
|
|
305
|
+
"""Parse ORDER BY expression into a list of key expressions."""
|
|
306
|
+
expr = expr.strip()
|
|
307
|
+
# Handle tuple syntax: (col1, col2, col3)
|
|
308
|
+
if expr.startswith("("):
|
|
309
|
+
expr = expr.strip("()")
|
|
310
|
+
parts: list[str] = []
|
|
311
|
+
depth = 0
|
|
312
|
+
current: list[str] = []
|
|
313
|
+
for char in expr:
|
|
314
|
+
if char == "(":
|
|
315
|
+
depth += 1
|
|
316
|
+
current.append(char)
|
|
317
|
+
elif char == ")":
|
|
318
|
+
depth -= 1
|
|
319
|
+
current.append(char)
|
|
320
|
+
elif char == "," and depth == 0:
|
|
321
|
+
parts.append("".join(current).strip())
|
|
322
|
+
current = []
|
|
323
|
+
else:
|
|
324
|
+
current.append(char)
|
|
325
|
+
remaining = "".join(current).strip()
|
|
326
|
+
if remaining:
|
|
327
|
+
parts.append(remaining)
|
|
328
|
+
return parts
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _parse_settings(settings_str: str) -> dict[str, str]:
|
|
332
|
+
"""Parse SETTINGS key=value pairs."""
|
|
333
|
+
result: dict[str, str] = {}
|
|
334
|
+
for pair in settings_str.split(","):
|
|
335
|
+
pair = pair.strip()
|
|
336
|
+
if "=" in pair:
|
|
337
|
+
k, v = pair.split("=", 1)
|
|
338
|
+
result[k.strip()] = v.strip()
|
|
339
|
+
return result
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _extract_from_tables(select_query: str) -> list[str]:
|
|
343
|
+
"""Extract table references from a SELECT query (FROM and JOIN clauses)."""
|
|
344
|
+
tables: list[str] = []
|
|
345
|
+
# Match FROM db.table or FROM table (not subqueries or function calls)
|
|
346
|
+
for m in re.finditer(
|
|
347
|
+
r"(?:FROM|JOIN)\s+(?:`?(\w+)`?\.)?`?(\w+)`?(?!\s*\()", select_query, re.IGNORECASE
|
|
348
|
+
):
|
|
349
|
+
db_part = m.group(1)
|
|
350
|
+
table_name = m.group(2)
|
|
351
|
+
# Skip system tables and subquery keywords
|
|
352
|
+
if table_name.upper() in ("SELECT", "LATERAL", "EACH"):
|
|
353
|
+
continue
|
|
354
|
+
full_name = f"{db_part}.{table_name}" if db_part else table_name
|
|
355
|
+
if full_name not in tables:
|
|
356
|
+
tables.append(full_name)
|
|
357
|
+
return tables
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def parse_create_table(ddl: str) -> TableDefinition | None:
|
|
361
|
+
"""Parse a CREATE TABLE statement into a TableDefinition."""
|
|
362
|
+
m = _RE_CREATE_TABLE.search(ddl)
|
|
363
|
+
if not m:
|
|
364
|
+
return None
|
|
365
|
+
|
|
366
|
+
name = m.group(2)
|
|
367
|
+
paren_pos = ddl.index("(", m.end() - 1)
|
|
368
|
+
columns_block = _extract_columns_block(ddl, paren_pos)
|
|
369
|
+
columns = _parse_columns(columns_block)
|
|
370
|
+
|
|
371
|
+
# Everything after the closing paren of columns
|
|
372
|
+
after_columns = ddl[paren_pos + len(columns_block) + 2 :]
|
|
373
|
+
|
|
374
|
+
engine_match = _RE_ENGINE.search(after_columns)
|
|
375
|
+
engine = engine_match.group(1) if engine_match else ""
|
|
376
|
+
|
|
377
|
+
order_by: list[str] = []
|
|
378
|
+
ob_match = _RE_ORDER_BY.search(after_columns)
|
|
379
|
+
if ob_match:
|
|
380
|
+
order_by = _parse_order_by(ob_match.group(1))
|
|
381
|
+
|
|
382
|
+
partition_by = None
|
|
383
|
+
pb_match = _RE_PARTITION_BY.search(after_columns)
|
|
384
|
+
if pb_match:
|
|
385
|
+
partition_by = pb_match.group(1).strip()
|
|
386
|
+
|
|
387
|
+
ttl = None
|
|
388
|
+
ttl_match = _RE_TTL.search(after_columns)
|
|
389
|
+
if ttl_match:
|
|
390
|
+
ttl = ttl_match.group(1).strip()
|
|
391
|
+
|
|
392
|
+
settings: dict[str, str] = {}
|
|
393
|
+
settings_match = _RE_SETTINGS.search(after_columns)
|
|
394
|
+
if settings_match:
|
|
395
|
+
settings = _parse_settings(settings_match.group(1))
|
|
396
|
+
|
|
397
|
+
return TableDefinition(
|
|
398
|
+
name=name,
|
|
399
|
+
engine=engine,
|
|
400
|
+
columns=columns,
|
|
401
|
+
order_by=order_by,
|
|
402
|
+
partition_by=partition_by,
|
|
403
|
+
ttl=ttl,
|
|
404
|
+
settings=settings,
|
|
405
|
+
raw_ddl=ddl,
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def parse_create_view(ddl: str) -> ViewDefinition | None:
|
|
410
|
+
"""Parse a CREATE VIEW statement into a ViewDefinition."""
|
|
411
|
+
m = _RE_CREATE_VIEW.search(ddl)
|
|
412
|
+
if not m:
|
|
413
|
+
return None
|
|
414
|
+
|
|
415
|
+
name = m.group(2)
|
|
416
|
+
select_query = ddl[m.end() :].strip()
|
|
417
|
+
|
|
418
|
+
return ViewDefinition(name=name, select_query=select_query, raw_ddl=ddl)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def parse_create_mv(ddl: str) -> MVDefinition | None:
|
|
422
|
+
"""Parse a CREATE MATERIALIZED VIEW statement into an MVDefinition."""
|
|
423
|
+
m = _RE_CREATE_MV.search(ddl)
|
|
424
|
+
if not m:
|
|
425
|
+
return None
|
|
426
|
+
|
|
427
|
+
name = m.group(2)
|
|
428
|
+
rest = ddl[m.end() :]
|
|
429
|
+
|
|
430
|
+
# Check for TO clause (target table)
|
|
431
|
+
target_table = None
|
|
432
|
+
to_match = _RE_MV_TO.search(rest)
|
|
433
|
+
if to_match:
|
|
434
|
+
target_db = to_match.group(1)
|
|
435
|
+
target_tbl = to_match.group(2)
|
|
436
|
+
target_table = f"{target_db}.{target_tbl}" if target_db else target_tbl
|
|
437
|
+
|
|
438
|
+
# Find the AS SELECT part
|
|
439
|
+
as_match = re.search(r"\bAS\s+(?=SELECT\b)", rest, re.IGNORECASE)
|
|
440
|
+
select_query = ""
|
|
441
|
+
if as_match:
|
|
442
|
+
select_query = rest[as_match.end() :].strip()
|
|
443
|
+
|
|
444
|
+
# Extract engine if present (between TO and AS, or before AS)
|
|
445
|
+
engine = None
|
|
446
|
+
engine_match = re.search(r"ENGINE\s*=\s*(\w+(?:\([^)]*\))?)", rest, re.IGNORECASE)
|
|
447
|
+
if engine_match:
|
|
448
|
+
engine = engine_match.group(1)
|
|
449
|
+
|
|
450
|
+
source_tables = _extract_from_tables(select_query) if select_query else []
|
|
451
|
+
|
|
452
|
+
return MVDefinition(
|
|
453
|
+
name=name,
|
|
454
|
+
target_table=target_table,
|
|
455
|
+
source_tables=source_tables,
|
|
456
|
+
select_query=select_query,
|
|
457
|
+
engine=engine,
|
|
458
|
+
raw_ddl=ddl,
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def parse_create_dictionary(ddl: str) -> DictDefinition | None:
|
|
463
|
+
"""Parse a CREATE DICTIONARY statement into a DictDefinition."""
|
|
464
|
+
m = _RE_CREATE_DICT.search(ddl)
|
|
465
|
+
if not m:
|
|
466
|
+
return None
|
|
467
|
+
|
|
468
|
+
name = m.group(2)
|
|
469
|
+
|
|
470
|
+
# Extract PRIMARY KEY
|
|
471
|
+
pk_match = re.search(r"PRIMARY\s+KEY\s+(\w+)", ddl, re.IGNORECASE)
|
|
472
|
+
primary_key = pk_match.group(1) if pk_match else None
|
|
473
|
+
|
|
474
|
+
# Extract SOURCE
|
|
475
|
+
source_type = None
|
|
476
|
+
source_table = None
|
|
477
|
+
source_db = None
|
|
478
|
+
source_query = None
|
|
479
|
+
source_match = re.search(r"SOURCE\s*\(\s*(\w+)\s*\(", ddl, re.IGNORECASE)
|
|
480
|
+
if source_match:
|
|
481
|
+
source_type = source_match.group(1).lower()
|
|
482
|
+
|
|
483
|
+
# Extract table from SOURCE(CLICKHOUSE(TABLE '...' DB '...'))
|
|
484
|
+
tbl_match = re.search(
|
|
485
|
+
r"TABLE\s+'([^']+)'", ddl[source_match.start() :], re.IGNORECASE
|
|
486
|
+
)
|
|
487
|
+
if tbl_match:
|
|
488
|
+
source_table = tbl_match.group(1)
|
|
489
|
+
|
|
490
|
+
db_match = re.search(
|
|
491
|
+
r"DB\s+'([^']+)'", ddl[source_match.start() :], re.IGNORECASE
|
|
492
|
+
)
|
|
493
|
+
if db_match:
|
|
494
|
+
source_db = db_match.group(1)
|
|
495
|
+
|
|
496
|
+
query_match = re.search(
|
|
497
|
+
r"QUERY\s+'((?:[^'\\]|\\.)*)'",
|
|
498
|
+
ddl[source_match.start() :],
|
|
499
|
+
re.IGNORECASE,
|
|
500
|
+
)
|
|
501
|
+
if query_match:
|
|
502
|
+
source_query = query_match.group(1)
|
|
503
|
+
|
|
504
|
+
# Extract LAYOUT
|
|
505
|
+
layout_match = re.search(r"LAYOUT\s*\(\s*(\w+)", ddl, re.IGNORECASE)
|
|
506
|
+
layout = layout_match.group(1) if layout_match else None
|
|
507
|
+
|
|
508
|
+
# Extract LIFETIME
|
|
509
|
+
lifetime_match = re.search(r"LIFETIME\s*\((.+?)\)", ddl, re.IGNORECASE)
|
|
510
|
+
lifetime = lifetime_match.group(1).strip() if lifetime_match else None
|
|
511
|
+
|
|
512
|
+
return DictDefinition(
|
|
513
|
+
name=name,
|
|
514
|
+
primary_key=primary_key,
|
|
515
|
+
source_type=source_type,
|
|
516
|
+
source_table=source_table,
|
|
517
|
+
source_db=source_db,
|
|
518
|
+
source_query=source_query,
|
|
519
|
+
layout=layout,
|
|
520
|
+
lifetime=lifetime,
|
|
521
|
+
raw_ddl=ddl,
|
|
522
|
+
)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def parse_create_statement(
|
|
526
|
+
ddl: str,
|
|
527
|
+
) -> TableDefinition | ViewDefinition | MVDefinition | DictDefinition | None:
|
|
528
|
+
"""Parse any CREATE statement into the appropriate definition type.
|
|
529
|
+
|
|
530
|
+
Returns None if the DDL cannot be parsed.
|
|
531
|
+
"""
|
|
532
|
+
ddl_stripped = ddl.strip()
|
|
533
|
+
|
|
534
|
+
if re.match(r"CREATE\s+MATERIALIZED\s+VIEW\b", ddl_stripped, re.IGNORECASE):
|
|
535
|
+
return parse_create_mv(ddl_stripped)
|
|
536
|
+
if re.match(r"CREATE\s+(?:OR\s+REPLACE\s+)?DICTIONARY\b", ddl_stripped, re.IGNORECASE):
|
|
537
|
+
return parse_create_dictionary(ddl_stripped)
|
|
538
|
+
if re.match(r"CREATE\s+VIEW\b", ddl_stripped, re.IGNORECASE):
|
|
539
|
+
return parse_create_view(ddl_stripped)
|
|
540
|
+
if re.match(r"CREATE\s+TABLE\b", ddl_stripped, re.IGNORECASE):
|
|
541
|
+
return parse_create_table(ddl_stripped)
|
|
542
|
+
|
|
543
|
+
return None
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
# ---------------------------------------------------------------------------
|
|
547
|
+
# Live schema introspection
|
|
548
|
+
# ---------------------------------------------------------------------------
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def list_objects(
|
|
552
|
+
client: Any, database: str, obj_type: str = "table"
|
|
553
|
+
) -> list[str]:
|
|
554
|
+
"""List objects of a given type in a database.
|
|
555
|
+
|
|
556
|
+
Args:
|
|
557
|
+
client: clickhouse-connect client.
|
|
558
|
+
database: Database name.
|
|
559
|
+
obj_type: One of 'table', 'view', 'materialized_view', 'dictionary'.
|
|
560
|
+
|
|
561
|
+
Returns:
|
|
562
|
+
List of object names.
|
|
563
|
+
"""
|
|
564
|
+
if obj_type == "dictionary":
|
|
565
|
+
result = client.query(
|
|
566
|
+
"SELECT name FROM system.dictionaries WHERE database = {db:String}",
|
|
567
|
+
parameters={"db": database},
|
|
568
|
+
)
|
|
569
|
+
return [row[0] for row in result.result_rows]
|
|
570
|
+
|
|
571
|
+
engine_filter = {
|
|
572
|
+
"table": f"engine NOT IN ('View', 'MaterializedView') AND name != '{VERSION_TABLE}'",
|
|
573
|
+
"view": "engine = 'View'",
|
|
574
|
+
"materialized_view": "engine = 'MaterializedView'",
|
|
575
|
+
}.get(obj_type, "1=1")
|
|
576
|
+
|
|
577
|
+
result = client.query(
|
|
578
|
+
f"SELECT name FROM system.tables "
|
|
579
|
+
f"WHERE database = {{db:String}} AND {engine_filter}",
|
|
580
|
+
parameters={"db": database},
|
|
581
|
+
)
|
|
582
|
+
return [row[0] for row in result.result_rows]
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def get_create_statement(
|
|
586
|
+
client: Any, database: str, name: str, obj_type: str = "table"
|
|
587
|
+
) -> str:
|
|
588
|
+
"""Get the CREATE statement for an object.
|
|
589
|
+
|
|
590
|
+
Args:
|
|
591
|
+
client: clickhouse-connect client.
|
|
592
|
+
database: Database name.
|
|
593
|
+
name: Object name.
|
|
594
|
+
obj_type: One of 'table', 'view', 'materialized_view', 'dictionary'.
|
|
595
|
+
|
|
596
|
+
Returns:
|
|
597
|
+
The CREATE statement as a string.
|
|
598
|
+
"""
|
|
599
|
+
if obj_type == "dictionary":
|
|
600
|
+
show_type = "DICTIONARY"
|
|
601
|
+
else:
|
|
602
|
+
show_type = "TABLE"
|
|
603
|
+
|
|
604
|
+
result = client.query(f"SHOW CREATE {show_type} `{database}`.`{name}`")
|
|
605
|
+
if result.result_rows:
|
|
606
|
+
return result.result_rows[0][0]
|
|
607
|
+
return ""
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def get_live_schema(client: Any, database: str) -> Schema:
|
|
611
|
+
"""Capture the full schema from a live ClickHouse database.
|
|
612
|
+
|
|
613
|
+
Args:
|
|
614
|
+
client: clickhouse-connect client.
|
|
615
|
+
database: Database name.
|
|
616
|
+
|
|
617
|
+
Returns:
|
|
618
|
+
A Schema object with all tables, views, MVs, and dictionaries.
|
|
619
|
+
"""
|
|
620
|
+
# Detect CH version
|
|
621
|
+
version_result = client.query("SELECT version()")
|
|
622
|
+
ch_version = version_result.result_rows[0][0] if version_result.result_rows else None
|
|
623
|
+
|
|
624
|
+
schema = Schema(database=database, ch_version=ch_version)
|
|
625
|
+
|
|
626
|
+
# Tables
|
|
627
|
+
for name in list_objects(client, database, "table"):
|
|
628
|
+
ddl = get_create_statement(client, database, name, "table")
|
|
629
|
+
parsed = parse_create_table(ddl)
|
|
630
|
+
if parsed:
|
|
631
|
+
schema.tables[name] = parsed
|
|
632
|
+
else:
|
|
633
|
+
# Fallback: store raw DDL in a minimal TableDefinition
|
|
634
|
+
schema.tables[name] = TableDefinition(name=name, engine="", raw_ddl=ddl)
|
|
635
|
+
|
|
636
|
+
# Views
|
|
637
|
+
for name in list_objects(client, database, "view"):
|
|
638
|
+
ddl = get_create_statement(client, database, name, "table")
|
|
639
|
+
parsed = parse_create_view(ddl)
|
|
640
|
+
if parsed:
|
|
641
|
+
schema.views[name] = parsed
|
|
642
|
+
else:
|
|
643
|
+
schema.views[name] = ViewDefinition(name=name, select_query="", raw_ddl=ddl)
|
|
644
|
+
|
|
645
|
+
# Materialized views
|
|
646
|
+
for name in list_objects(client, database, "materialized_view"):
|
|
647
|
+
ddl = get_create_statement(client, database, name, "table")
|
|
648
|
+
parsed = parse_create_mv(ddl)
|
|
649
|
+
if parsed:
|
|
650
|
+
schema.materialized_views[name] = parsed
|
|
651
|
+
else:
|
|
652
|
+
schema.materialized_views[name] = MVDefinition(name=name, raw_ddl=ddl)
|
|
653
|
+
|
|
654
|
+
# Dictionaries
|
|
655
|
+
for name in list_objects(client, database, "dictionary"):
|
|
656
|
+
ddl = get_create_statement(client, database, name, "dictionary")
|
|
657
|
+
parsed = parse_create_dictionary(ddl)
|
|
658
|
+
if parsed:
|
|
659
|
+
schema.dictionaries[name] = parsed
|
|
660
|
+
else:
|
|
661
|
+
schema.dictionaries[name] = DictDefinition(name=name, raw_ddl=ddl)
|
|
662
|
+
|
|
663
|
+
return schema
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def get_dependencies(client: Any, database: str) -> DependencyGraph:
|
|
667
|
+
"""Build a dependency graph from the live database.
|
|
668
|
+
|
|
669
|
+
Queries system.tables for materialized views and system.dictionaries for
|
|
670
|
+
dictionary source tables. Edges are typed as schema or data_flow.
|
|
671
|
+
|
|
672
|
+
Args:
|
|
673
|
+
client: clickhouse-connect client.
|
|
674
|
+
database: Database name.
|
|
675
|
+
|
|
676
|
+
Returns:
|
|
677
|
+
A DependencyGraph with nodes and typed edges.
|
|
678
|
+
"""
|
|
679
|
+
graph = DependencyGraph()
|
|
680
|
+
schema = get_live_schema(client, database)
|
|
681
|
+
|
|
682
|
+
# Add all objects as nodes
|
|
683
|
+
for name in schema.tables:
|
|
684
|
+
graph.nodes[name] = ObjectNode(name=name, obj_type="table")
|
|
685
|
+
for name in schema.views:
|
|
686
|
+
graph.nodes[name] = ObjectNode(name=name, obj_type="view")
|
|
687
|
+
for name in schema.materialized_views:
|
|
688
|
+
graph.nodes[name] = ObjectNode(name=name, obj_type="materialized_view")
|
|
689
|
+
for name in schema.dictionaries:
|
|
690
|
+
graph.nodes[name] = ObjectNode(name=name, obj_type="dictionary")
|
|
691
|
+
|
|
692
|
+
# MV dependencies
|
|
693
|
+
for name, mv in schema.materialized_views.items():
|
|
694
|
+
for src in mv.source_tables:
|
|
695
|
+
# Normalize: strip db prefix if it matches current database
|
|
696
|
+
src_name = src.split(".")[-1] if "." in src else src
|
|
697
|
+
if src_name in graph.nodes:
|
|
698
|
+
# Data flow: MV triggers on INSERT to source
|
|
699
|
+
graph.edges.append(
|
|
700
|
+
DependencyEdge(source=src_name, target=name, dep_type=DepType.DATA_FLOW)
|
|
701
|
+
)
|
|
702
|
+
# Schema dependency: MV SELECT references this table
|
|
703
|
+
graph.edges.append(
|
|
704
|
+
DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
|
|
705
|
+
)
|
|
706
|
+
|
|
707
|
+
# TO table dependency
|
|
708
|
+
if mv.target_table:
|
|
709
|
+
tgt_name = mv.target_table.split(".")[-1] if "." in mv.target_table else mv.target_table
|
|
710
|
+
if tgt_name in graph.nodes:
|
|
711
|
+
graph.edges.append(
|
|
712
|
+
DependencyEdge(source=name, target=tgt_name, dep_type=DepType.DATA_FLOW)
|
|
713
|
+
)
|
|
714
|
+
|
|
715
|
+
# Dictionary dependencies
|
|
716
|
+
for name, d in schema.dictionaries.items():
|
|
717
|
+
if d.source_table:
|
|
718
|
+
src_name = d.source_table
|
|
719
|
+
if src_name in graph.nodes:
|
|
720
|
+
graph.edges.append(
|
|
721
|
+
DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
|
|
722
|
+
)
|
|
723
|
+
elif d.source_query:
|
|
724
|
+
# Parse tables from the source query
|
|
725
|
+
for src in _extract_from_tables(d.source_query):
|
|
726
|
+
src_name = src.split(".")[-1] if "." in src else src
|
|
727
|
+
if src_name in graph.nodes:
|
|
728
|
+
graph.edges.append(
|
|
729
|
+
DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
|
|
730
|
+
)
|
|
731
|
+
|
|
732
|
+
return graph
|