ch-migrate-cli 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,732 @@
1
+ """ClickHouse schema introspection: structured DDL parsing and live schema capture."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections import deque
7
+ from dataclasses import dataclass, field
8
+ from enum import Enum
9
+ from typing import Any, Literal
10
+
11
+ # Alembic's bookkeeping table: not part of the user's schema, so snapshot, diff
12
+ # and deps leave it out.
13
+ VERSION_TABLE = "alembic_version"
14
+
15
+
16
+ # ---------------------------------------------------------------------------
17
+ # Data models
18
+ # ---------------------------------------------------------------------------
19
+
20
+
21
+ @dataclass
22
+ class ColumnDefinition:
23
+ name: str
24
+ type: str
25
+ default_kind: str | None = None # DEFAULT, MATERIALIZED, ALIAS
26
+ default_expr: str | None = None
27
+ codec: str | None = None
28
+ comment: str | None = None
29
+
30
+
31
+ @dataclass
32
+ class TableDefinition:
33
+ name: str
34
+ engine: str
35
+ columns: list[ColumnDefinition] = field(default_factory=list)
36
+ order_by: list[str] = field(default_factory=list)
37
+ partition_by: str | None = None
38
+ ttl: str | None = None
39
+ settings: dict[str, str] = field(default_factory=dict)
40
+ raw_ddl: str = ""
41
+
42
+
43
+ @dataclass
44
+ class ViewDefinition:
45
+ name: str
46
+ select_query: str
47
+ raw_ddl: str = ""
48
+
49
+
50
+ @dataclass
51
+ class MVDefinition:
52
+ name: str
53
+ target_table: str | None = None
54
+ source_tables: list[str] = field(default_factory=list)
55
+ select_query: str = ""
56
+ engine: str | None = None
57
+ raw_ddl: str = ""
58
+
59
+
60
+ @dataclass
61
+ class DictDefinition:
62
+ name: str
63
+ primary_key: str | None = None
64
+ source_type: str | None = None # e.g. "clickhouse", "mysql", "http"
65
+ source_table: str | None = None
66
+ source_db: str | None = None
67
+ source_query: str | None = None
68
+ layout: str | None = None
69
+ lifetime: str | None = None
70
+ structure_keys: list[str] = field(default_factory=list)
71
+ structure_attributes: list[str] = field(default_factory=list)
72
+ raw_ddl: str = ""
73
+
74
+
75
+ @dataclass
76
+ class Schema:
77
+ tables: dict[str, TableDefinition] = field(default_factory=dict)
78
+ views: dict[str, ViewDefinition] = field(default_factory=dict)
79
+ materialized_views: dict[str, MVDefinition] = field(default_factory=dict)
80
+ dictionaries: dict[str, DictDefinition] = field(default_factory=dict)
81
+ database: str = ""
82
+ ch_version: str | None = None
83
+
84
+
85
+ # ---------------------------------------------------------------------------
86
+ # Dependency graph
87
+ # ---------------------------------------------------------------------------
88
+
89
+
90
+ class DepType(str, Enum):
91
+ SCHEMA = "schema"
92
+ DATA_FLOW = "data_flow"
93
+
94
+
95
+ @dataclass
96
+ class DependencyEdge:
97
+ source: str
98
+ target: str
99
+ dep_type: DepType
100
+
101
+
102
+ @dataclass
103
+ class ObjectNode:
104
+ name: str
105
+ obj_type: Literal["table", "view", "materialized_view", "dictionary"]
106
+
107
+
108
+ @dataclass
109
+ class DependencyGraph:
110
+ nodes: dict[str, ObjectNode] = field(default_factory=dict)
111
+ edges: list[DependencyEdge] = field(default_factory=list)
112
+
113
+ def topological_order(self) -> list[str]:
114
+ """Return names in safe drop/recreate order (leaves first)."""
115
+ in_degree: dict[str, int] = {name: 0 for name in self.nodes}
116
+ adj: dict[str, list[str]] = {name: [] for name in self.nodes}
117
+ for edge in self.edges:
118
+ if edge.target in in_degree and edge.source in adj:
119
+ adj[edge.source].append(edge.target)
120
+ in_degree[edge.target] += 1
121
+
122
+ queue = deque(n for n, d in in_degree.items() if d == 0)
123
+ result: list[str] = []
124
+ while queue:
125
+ node = queue.popleft()
126
+ result.append(node)
127
+ for neighbor in adj.get(node, []):
128
+ in_degree[neighbor] -= 1
129
+ if in_degree[neighbor] == 0:
130
+ queue.append(neighbor)
131
+
132
+ # Append any remaining nodes (cycles) at the end
133
+ for name in self.nodes:
134
+ if name not in result:
135
+ result.append(name)
136
+
137
+ return result
138
+
139
+ def affected_by_drop(self, name: str) -> list[ObjectNode]:
140
+ """Return direct dependents that would be affected by dropping the given object."""
141
+ seen: set[str] = set()
142
+ affected: list[ObjectNode] = []
143
+ for edge in self.edges:
144
+ if edge.source == name and edge.target in self.nodes and edge.target not in seen:
145
+ seen.add(edge.target)
146
+ affected.append(self.nodes[edge.target])
147
+ return affected
148
+
149
+
150
+ # ---------------------------------------------------------------------------
151
+ # DDL parsing
152
+ # ---------------------------------------------------------------------------
153
+
154
+ # Regex patterns for parsing CREATE statements
155
+ _RE_CREATE_TABLE = re.compile(
156
+ r"CREATE\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?"
157
+ r"(?:`?(\w+)`?\.)?`?(\w+)`?"
158
+ r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?\(",
159
+ re.IGNORECASE,
160
+ )
161
+
162
+ _RE_ENGINE = re.compile(r"ENGINE\s*=\s*(\w+(?:\(.*?\))?)", re.IGNORECASE)
163
+
164
+ _RE_ORDER_BY = re.compile(r"ORDER\s+BY\s+(.+?)(?=\s*(?:PARTITION|TTL|SETTINGS|$))", re.IGNORECASE)
165
+
166
+ _RE_PARTITION_BY = re.compile(
167
+ r"PARTITION\s+BY\s+(.+?)(?=\s*(?:ORDER|TTL|SETTINGS|$))", re.IGNORECASE
168
+ )
169
+
170
+ _RE_TTL = re.compile(r"TTL\s+(.+?)(?=\s*(?:SETTINGS|$))", re.IGNORECASE)
171
+
172
+ _RE_SETTINGS = re.compile(r"SETTINGS\s+(.+)$", re.IGNORECASE | re.MULTILINE)
173
+
174
+ _RE_CREATE_VIEW = re.compile(
175
+ r"CREATE\s+VIEW\s+(?:IF\s+NOT\s+EXISTS\s+)?"
176
+ r"(?:`?(\w+)`?\.)?`?(\w+)`?"
177
+ r"\s+AS\s+",
178
+ re.IGNORECASE,
179
+ )
180
+
181
+ _RE_CREATE_MV = re.compile(
182
+ r"CREATE\s+MATERIALIZED\s+VIEW\s+(?:IF\s+NOT\s+EXISTS\s+)?"
183
+ r"(?:`?(\w+)`?\.)?`?(\w+)`?"
184
+ r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?",
185
+ re.IGNORECASE,
186
+ )
187
+
188
+ _RE_MV_TO = re.compile(r"TO\s+(?:`?(\w+)`?\.)?`?(\w+)`?", re.IGNORECASE)
189
+
190
+ _RE_CREATE_DICT = re.compile(
191
+ r"CREATE\s+(?:OR\s+REPLACE\s+)?DICTIONARY\s+(?:IF\s+NOT\s+EXISTS\s+)?"
192
+ r"(?:`?(\w+)`?\.)?`?(\w+)`?"
193
+ r"\s*(?:ON\s+CLUSTER\s+\S+\s*)?\(",
194
+ re.IGNORECASE,
195
+ )
196
+
197
+
198
+ def _parse_columns(columns_block: str) -> list[ColumnDefinition]:
199
+ """Parse column definitions from the block between CREATE TABLE (...).
200
+
201
+ Handles nested parentheses in types like Nullable(String), Tuple(a UInt8, b String).
202
+ """
203
+ columns: list[ColumnDefinition] = []
204
+ depth = 0
205
+ current: list[str] = []
206
+
207
+ for char in columns_block:
208
+ if char == "(":
209
+ depth += 1
210
+ current.append(char)
211
+ elif char == ")":
212
+ depth -= 1
213
+ current.append(char)
214
+ elif char == "," and depth == 0:
215
+ line = "".join(current).strip()
216
+ if line:
217
+ col = _parse_single_column(line)
218
+ if col:
219
+ columns.append(col)
220
+ current = []
221
+ else:
222
+ current.append(char)
223
+
224
+ # Last column (no trailing comma)
225
+ line = "".join(current).strip()
226
+ if line:
227
+ col = _parse_single_column(line)
228
+ if col:
229
+ columns.append(col)
230
+
231
+ return columns
232
+
233
+
234
+ def _parse_single_column(line: str) -> ColumnDefinition | None:
235
+ """Parse a single column definition line."""
236
+ line = line.strip()
237
+ if not line:
238
+ return None
239
+
240
+ # Skip constraints (INDEX, PROJECTION, CONSTRAINT)
241
+ if re.match(r"(?:INDEX|PROJECTION|CONSTRAINT)\s+", line, re.IGNORECASE):
242
+ return None
243
+
244
+ # Match: `name` Type [DEFAULT|MATERIALIZED|ALIAS expr] [CODEC(...)] [COMMENT '...']
245
+ m = re.match(r"`?(\w+)`?\s+(.+)", line)
246
+ if not m:
247
+ return None
248
+
249
+ name = m.group(1)
250
+ rest = m.group(2)
251
+
252
+ # Extract COMMENT
253
+ comment = None
254
+ comment_match = re.search(r"COMMENT\s+'((?:[^'\\]|\\.)*)'", rest, re.IGNORECASE)
255
+ if comment_match:
256
+ comment = comment_match.group(1)
257
+ rest = rest[: comment_match.start()].rstrip()
258
+
259
+ # Extract CODEC
260
+ codec = None
261
+ codec_match = re.search(r"CODEC\s*\((.+?)\)\s*$", rest, re.IGNORECASE)
262
+ if codec_match:
263
+ codec = codec_match.group(1)
264
+ rest = rest[: codec_match.start()].rstrip()
265
+
266
+ # Extract DEFAULT/MATERIALIZED/ALIAS
267
+ default_kind = None
268
+ default_expr = None
269
+ default_match = re.search(
270
+ r"\b(DEFAULT|MATERIALIZED|ALIAS)\s+(.+)$", rest, re.IGNORECASE
271
+ )
272
+ if default_match:
273
+ default_kind = default_match.group(1).upper()
274
+ default_expr = default_match.group(2).strip()
275
+ rest = rest[: default_match.start()].rstrip()
276
+
277
+ col_type = rest.strip()
278
+
279
+ return ColumnDefinition(
280
+ name=name,
281
+ type=col_type,
282
+ default_kind=default_kind,
283
+ default_expr=default_expr,
284
+ codec=codec,
285
+ comment=comment,
286
+ )
287
+
288
+
289
+ def _extract_columns_block(ddl: str, start_paren_pos: int) -> str:
290
+ """Extract the columns block from CREATE TABLE, handling nested parens."""
291
+ depth = 0
292
+ i = start_paren_pos
293
+ while i < len(ddl):
294
+ if ddl[i] == "(":
295
+ depth += 1
296
+ elif ddl[i] == ")":
297
+ depth -= 1
298
+ if depth == 0:
299
+ return ddl[start_paren_pos + 1 : i]
300
+ i += 1
301
+ return ddl[start_paren_pos + 1 :]
302
+
303
+
304
+ def _parse_order_by(expr: str) -> list[str]:
305
+ """Parse ORDER BY expression into a list of key expressions."""
306
+ expr = expr.strip()
307
+ # Handle tuple syntax: (col1, col2, col3)
308
+ if expr.startswith("("):
309
+ expr = expr.strip("()")
310
+ parts: list[str] = []
311
+ depth = 0
312
+ current: list[str] = []
313
+ for char in expr:
314
+ if char == "(":
315
+ depth += 1
316
+ current.append(char)
317
+ elif char == ")":
318
+ depth -= 1
319
+ current.append(char)
320
+ elif char == "," and depth == 0:
321
+ parts.append("".join(current).strip())
322
+ current = []
323
+ else:
324
+ current.append(char)
325
+ remaining = "".join(current).strip()
326
+ if remaining:
327
+ parts.append(remaining)
328
+ return parts
329
+
330
+
331
+ def _parse_settings(settings_str: str) -> dict[str, str]:
332
+ """Parse SETTINGS key=value pairs."""
333
+ result: dict[str, str] = {}
334
+ for pair in settings_str.split(","):
335
+ pair = pair.strip()
336
+ if "=" in pair:
337
+ k, v = pair.split("=", 1)
338
+ result[k.strip()] = v.strip()
339
+ return result
340
+
341
+
342
+ def _extract_from_tables(select_query: str) -> list[str]:
343
+ """Extract table references from a SELECT query (FROM and JOIN clauses)."""
344
+ tables: list[str] = []
345
+ # Match FROM db.table or FROM table (not subqueries or function calls)
346
+ for m in re.finditer(
347
+ r"(?:FROM|JOIN)\s+(?:`?(\w+)`?\.)?`?(\w+)`?(?!\s*\()", select_query, re.IGNORECASE
348
+ ):
349
+ db_part = m.group(1)
350
+ table_name = m.group(2)
351
+ # Skip system tables and subquery keywords
352
+ if table_name.upper() in ("SELECT", "LATERAL", "EACH"):
353
+ continue
354
+ full_name = f"{db_part}.{table_name}" if db_part else table_name
355
+ if full_name not in tables:
356
+ tables.append(full_name)
357
+ return tables
358
+
359
+
360
+ def parse_create_table(ddl: str) -> TableDefinition | None:
361
+ """Parse a CREATE TABLE statement into a TableDefinition."""
362
+ m = _RE_CREATE_TABLE.search(ddl)
363
+ if not m:
364
+ return None
365
+
366
+ name = m.group(2)
367
+ paren_pos = ddl.index("(", m.end() - 1)
368
+ columns_block = _extract_columns_block(ddl, paren_pos)
369
+ columns = _parse_columns(columns_block)
370
+
371
+ # Everything after the closing paren of columns
372
+ after_columns = ddl[paren_pos + len(columns_block) + 2 :]
373
+
374
+ engine_match = _RE_ENGINE.search(after_columns)
375
+ engine = engine_match.group(1) if engine_match else ""
376
+
377
+ order_by: list[str] = []
378
+ ob_match = _RE_ORDER_BY.search(after_columns)
379
+ if ob_match:
380
+ order_by = _parse_order_by(ob_match.group(1))
381
+
382
+ partition_by = None
383
+ pb_match = _RE_PARTITION_BY.search(after_columns)
384
+ if pb_match:
385
+ partition_by = pb_match.group(1).strip()
386
+
387
+ ttl = None
388
+ ttl_match = _RE_TTL.search(after_columns)
389
+ if ttl_match:
390
+ ttl = ttl_match.group(1).strip()
391
+
392
+ settings: dict[str, str] = {}
393
+ settings_match = _RE_SETTINGS.search(after_columns)
394
+ if settings_match:
395
+ settings = _parse_settings(settings_match.group(1))
396
+
397
+ return TableDefinition(
398
+ name=name,
399
+ engine=engine,
400
+ columns=columns,
401
+ order_by=order_by,
402
+ partition_by=partition_by,
403
+ ttl=ttl,
404
+ settings=settings,
405
+ raw_ddl=ddl,
406
+ )
407
+
408
+
409
+ def parse_create_view(ddl: str) -> ViewDefinition | None:
410
+ """Parse a CREATE VIEW statement into a ViewDefinition."""
411
+ m = _RE_CREATE_VIEW.search(ddl)
412
+ if not m:
413
+ return None
414
+
415
+ name = m.group(2)
416
+ select_query = ddl[m.end() :].strip()
417
+
418
+ return ViewDefinition(name=name, select_query=select_query, raw_ddl=ddl)
419
+
420
+
421
+ def parse_create_mv(ddl: str) -> MVDefinition | None:
422
+ """Parse a CREATE MATERIALIZED VIEW statement into an MVDefinition."""
423
+ m = _RE_CREATE_MV.search(ddl)
424
+ if not m:
425
+ return None
426
+
427
+ name = m.group(2)
428
+ rest = ddl[m.end() :]
429
+
430
+ # Check for TO clause (target table)
431
+ target_table = None
432
+ to_match = _RE_MV_TO.search(rest)
433
+ if to_match:
434
+ target_db = to_match.group(1)
435
+ target_tbl = to_match.group(2)
436
+ target_table = f"{target_db}.{target_tbl}" if target_db else target_tbl
437
+
438
+ # Find the AS SELECT part
439
+ as_match = re.search(r"\bAS\s+(?=SELECT\b)", rest, re.IGNORECASE)
440
+ select_query = ""
441
+ if as_match:
442
+ select_query = rest[as_match.end() :].strip()
443
+
444
+ # Extract engine if present (between TO and AS, or before AS)
445
+ engine = None
446
+ engine_match = re.search(r"ENGINE\s*=\s*(\w+(?:\([^)]*\))?)", rest, re.IGNORECASE)
447
+ if engine_match:
448
+ engine = engine_match.group(1)
449
+
450
+ source_tables = _extract_from_tables(select_query) if select_query else []
451
+
452
+ return MVDefinition(
453
+ name=name,
454
+ target_table=target_table,
455
+ source_tables=source_tables,
456
+ select_query=select_query,
457
+ engine=engine,
458
+ raw_ddl=ddl,
459
+ )
460
+
461
+
462
+ def parse_create_dictionary(ddl: str) -> DictDefinition | None:
463
+ """Parse a CREATE DICTIONARY statement into a DictDefinition."""
464
+ m = _RE_CREATE_DICT.search(ddl)
465
+ if not m:
466
+ return None
467
+
468
+ name = m.group(2)
469
+
470
+ # Extract PRIMARY KEY
471
+ pk_match = re.search(r"PRIMARY\s+KEY\s+(\w+)", ddl, re.IGNORECASE)
472
+ primary_key = pk_match.group(1) if pk_match else None
473
+
474
+ # Extract SOURCE
475
+ source_type = None
476
+ source_table = None
477
+ source_db = None
478
+ source_query = None
479
+ source_match = re.search(r"SOURCE\s*\(\s*(\w+)\s*\(", ddl, re.IGNORECASE)
480
+ if source_match:
481
+ source_type = source_match.group(1).lower()
482
+
483
+ # Extract table from SOURCE(CLICKHOUSE(TABLE '...' DB '...'))
484
+ tbl_match = re.search(
485
+ r"TABLE\s+'([^']+)'", ddl[source_match.start() :], re.IGNORECASE
486
+ )
487
+ if tbl_match:
488
+ source_table = tbl_match.group(1)
489
+
490
+ db_match = re.search(
491
+ r"DB\s+'([^']+)'", ddl[source_match.start() :], re.IGNORECASE
492
+ )
493
+ if db_match:
494
+ source_db = db_match.group(1)
495
+
496
+ query_match = re.search(
497
+ r"QUERY\s+'((?:[^'\\]|\\.)*)'",
498
+ ddl[source_match.start() :],
499
+ re.IGNORECASE,
500
+ )
501
+ if query_match:
502
+ source_query = query_match.group(1)
503
+
504
+ # Extract LAYOUT
505
+ layout_match = re.search(r"LAYOUT\s*\(\s*(\w+)", ddl, re.IGNORECASE)
506
+ layout = layout_match.group(1) if layout_match else None
507
+
508
+ # Extract LIFETIME
509
+ lifetime_match = re.search(r"LIFETIME\s*\((.+?)\)", ddl, re.IGNORECASE)
510
+ lifetime = lifetime_match.group(1).strip() if lifetime_match else None
511
+
512
+ return DictDefinition(
513
+ name=name,
514
+ primary_key=primary_key,
515
+ source_type=source_type,
516
+ source_table=source_table,
517
+ source_db=source_db,
518
+ source_query=source_query,
519
+ layout=layout,
520
+ lifetime=lifetime,
521
+ raw_ddl=ddl,
522
+ )
523
+
524
+
525
+ def parse_create_statement(
526
+ ddl: str,
527
+ ) -> TableDefinition | ViewDefinition | MVDefinition | DictDefinition | None:
528
+ """Parse any CREATE statement into the appropriate definition type.
529
+
530
+ Returns None if the DDL cannot be parsed.
531
+ """
532
+ ddl_stripped = ddl.strip()
533
+
534
+ if re.match(r"CREATE\s+MATERIALIZED\s+VIEW\b", ddl_stripped, re.IGNORECASE):
535
+ return parse_create_mv(ddl_stripped)
536
+ if re.match(r"CREATE\s+(?:OR\s+REPLACE\s+)?DICTIONARY\b", ddl_stripped, re.IGNORECASE):
537
+ return parse_create_dictionary(ddl_stripped)
538
+ if re.match(r"CREATE\s+VIEW\b", ddl_stripped, re.IGNORECASE):
539
+ return parse_create_view(ddl_stripped)
540
+ if re.match(r"CREATE\s+TABLE\b", ddl_stripped, re.IGNORECASE):
541
+ return parse_create_table(ddl_stripped)
542
+
543
+ return None
544
+
545
+
546
+ # ---------------------------------------------------------------------------
547
+ # Live schema introspection
548
+ # ---------------------------------------------------------------------------
549
+
550
+
551
+ def list_objects(
552
+ client: Any, database: str, obj_type: str = "table"
553
+ ) -> list[str]:
554
+ """List objects of a given type in a database.
555
+
556
+ Args:
557
+ client: clickhouse-connect client.
558
+ database: Database name.
559
+ obj_type: One of 'table', 'view', 'materialized_view', 'dictionary'.
560
+
561
+ Returns:
562
+ List of object names.
563
+ """
564
+ if obj_type == "dictionary":
565
+ result = client.query(
566
+ "SELECT name FROM system.dictionaries WHERE database = {db:String}",
567
+ parameters={"db": database},
568
+ )
569
+ return [row[0] for row in result.result_rows]
570
+
571
+ engine_filter = {
572
+ "table": f"engine NOT IN ('View', 'MaterializedView') AND name != '{VERSION_TABLE}'",
573
+ "view": "engine = 'View'",
574
+ "materialized_view": "engine = 'MaterializedView'",
575
+ }.get(obj_type, "1=1")
576
+
577
+ result = client.query(
578
+ f"SELECT name FROM system.tables "
579
+ f"WHERE database = {{db:String}} AND {engine_filter}",
580
+ parameters={"db": database},
581
+ )
582
+ return [row[0] for row in result.result_rows]
583
+
584
+
585
+ def get_create_statement(
586
+ client: Any, database: str, name: str, obj_type: str = "table"
587
+ ) -> str:
588
+ """Get the CREATE statement for an object.
589
+
590
+ Args:
591
+ client: clickhouse-connect client.
592
+ database: Database name.
593
+ name: Object name.
594
+ obj_type: One of 'table', 'view', 'materialized_view', 'dictionary'.
595
+
596
+ Returns:
597
+ The CREATE statement as a string.
598
+ """
599
+ if obj_type == "dictionary":
600
+ show_type = "DICTIONARY"
601
+ else:
602
+ show_type = "TABLE"
603
+
604
+ result = client.query(f"SHOW CREATE {show_type} `{database}`.`{name}`")
605
+ if result.result_rows:
606
+ return result.result_rows[0][0]
607
+ return ""
608
+
609
+
610
+ def get_live_schema(client: Any, database: str) -> Schema:
611
+ """Capture the full schema from a live ClickHouse database.
612
+
613
+ Args:
614
+ client: clickhouse-connect client.
615
+ database: Database name.
616
+
617
+ Returns:
618
+ A Schema object with all tables, views, MVs, and dictionaries.
619
+ """
620
+ # Detect CH version
621
+ version_result = client.query("SELECT version()")
622
+ ch_version = version_result.result_rows[0][0] if version_result.result_rows else None
623
+
624
+ schema = Schema(database=database, ch_version=ch_version)
625
+
626
+ # Tables
627
+ for name in list_objects(client, database, "table"):
628
+ ddl = get_create_statement(client, database, name, "table")
629
+ parsed = parse_create_table(ddl)
630
+ if parsed:
631
+ schema.tables[name] = parsed
632
+ else:
633
+ # Fallback: store raw DDL in a minimal TableDefinition
634
+ schema.tables[name] = TableDefinition(name=name, engine="", raw_ddl=ddl)
635
+
636
+ # Views
637
+ for name in list_objects(client, database, "view"):
638
+ ddl = get_create_statement(client, database, name, "table")
639
+ parsed = parse_create_view(ddl)
640
+ if parsed:
641
+ schema.views[name] = parsed
642
+ else:
643
+ schema.views[name] = ViewDefinition(name=name, select_query="", raw_ddl=ddl)
644
+
645
+ # Materialized views
646
+ for name in list_objects(client, database, "materialized_view"):
647
+ ddl = get_create_statement(client, database, name, "table")
648
+ parsed = parse_create_mv(ddl)
649
+ if parsed:
650
+ schema.materialized_views[name] = parsed
651
+ else:
652
+ schema.materialized_views[name] = MVDefinition(name=name, raw_ddl=ddl)
653
+
654
+ # Dictionaries
655
+ for name in list_objects(client, database, "dictionary"):
656
+ ddl = get_create_statement(client, database, name, "dictionary")
657
+ parsed = parse_create_dictionary(ddl)
658
+ if parsed:
659
+ schema.dictionaries[name] = parsed
660
+ else:
661
+ schema.dictionaries[name] = DictDefinition(name=name, raw_ddl=ddl)
662
+
663
+ return schema
664
+
665
+
666
+ def get_dependencies(client: Any, database: str) -> DependencyGraph:
667
+ """Build a dependency graph from the live database.
668
+
669
+ Queries system.tables for materialized views and system.dictionaries for
670
+ dictionary source tables. Edges are typed as schema or data_flow.
671
+
672
+ Args:
673
+ client: clickhouse-connect client.
674
+ database: Database name.
675
+
676
+ Returns:
677
+ A DependencyGraph with nodes and typed edges.
678
+ """
679
+ graph = DependencyGraph()
680
+ schema = get_live_schema(client, database)
681
+
682
+ # Add all objects as nodes
683
+ for name in schema.tables:
684
+ graph.nodes[name] = ObjectNode(name=name, obj_type="table")
685
+ for name in schema.views:
686
+ graph.nodes[name] = ObjectNode(name=name, obj_type="view")
687
+ for name in schema.materialized_views:
688
+ graph.nodes[name] = ObjectNode(name=name, obj_type="materialized_view")
689
+ for name in schema.dictionaries:
690
+ graph.nodes[name] = ObjectNode(name=name, obj_type="dictionary")
691
+
692
+ # MV dependencies
693
+ for name, mv in schema.materialized_views.items():
694
+ for src in mv.source_tables:
695
+ # Normalize: strip db prefix if it matches current database
696
+ src_name = src.split(".")[-1] if "." in src else src
697
+ if src_name in graph.nodes:
698
+ # Data flow: MV triggers on INSERT to source
699
+ graph.edges.append(
700
+ DependencyEdge(source=src_name, target=name, dep_type=DepType.DATA_FLOW)
701
+ )
702
+ # Schema dependency: MV SELECT references this table
703
+ graph.edges.append(
704
+ DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
705
+ )
706
+
707
+ # TO table dependency
708
+ if mv.target_table:
709
+ tgt_name = mv.target_table.split(".")[-1] if "." in mv.target_table else mv.target_table
710
+ if tgt_name in graph.nodes:
711
+ graph.edges.append(
712
+ DependencyEdge(source=name, target=tgt_name, dep_type=DepType.DATA_FLOW)
713
+ )
714
+
715
+ # Dictionary dependencies
716
+ for name, d in schema.dictionaries.items():
717
+ if d.source_table:
718
+ src_name = d.source_table
719
+ if src_name in graph.nodes:
720
+ graph.edges.append(
721
+ DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
722
+ )
723
+ elif d.source_query:
724
+ # Parse tables from the source query
725
+ for src in _extract_from_tables(d.source_query):
726
+ src_name = src.split(".")[-1] if "." in src else src
727
+ if src_name in graph.nodes:
728
+ graph.edges.append(
729
+ DependencyEdge(source=src_name, target=name, dep_type=DepType.SCHEMA)
730
+ )
731
+
732
+ return graph