sql-dag-flow 0.5.4__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sql_dag_flow-0.5.4/src/sql_dag_flow.egg-info → sql_dag_flow-0.6.0}/PKG-INFO +6 -4
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/README.md +5 -3
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/pyproject.toml +1 -1
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/main.py +11 -8
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/parser.py +105 -4
- sql_dag_flow-0.5.4/src/sql_dag_flow/static/assets/index-BEi2iL86.js → sql_dag_flow-0.6.0/src/sql_dag_flow/static/assets/index-BRMQHj2Z.js +41 -41
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/static/index.html +1 -1
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0/src/sql_dag_flow.egg-info}/PKG-INFO +6 -4
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow.egg-info/SOURCES.txt +1 -1
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/LICENSE +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/MANIFEST.in +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/setup.cfg +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/__init__.py +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/static/vite.svg +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/test_api_endpoints.py +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/test_parser.py +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow/verify_counts.py +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow.egg-info/requires.txt +0 -0
- {sql_dag_flow-0.5.4 → sql_dag_flow-0.6.0}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sql-dag-flow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
|
|
5
5
|
Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -58,8 +58,10 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
58
58
|
|
|
59
59
|
### 🔍 Visualization & Analysis
|
|
60
60
|
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
61
|
+
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
62
|
+
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
61
63
|
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
62
|
-
* **Discovery Mode**: Visualize "Ghost Nodes" (missing files or external tables)
|
|
64
|
+
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
63
65
|
* **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
|
|
64
66
|
* **Smart Layout (New 🧠)**:
|
|
65
67
|
* Powered by **ELK (Eclipse Layout Kernel)**.
|
|
@@ -100,7 +102,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
100
102
|
* **Column-Level Lineage (New in v0.4.9 🆕)**: Traces how each output column derives from source columns. Shows transformation chain (e.g., `order_timestamp ← orders_raw.order_date via CAST(... AS DATETIME)`).
|
|
101
103
|
* **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
|
|
102
104
|
* **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
|
|
103
|
-
* **Performance Optimizations (
|
|
105
|
+
* **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
|
|
104
106
|
* **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
|
|
105
107
|
* **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
|
|
106
108
|
|
|
@@ -147,7 +149,7 @@ Install easily via `pip`:
|
|
|
147
149
|
pip install sql-dag-flow
|
|
148
150
|
```
|
|
149
151
|
|
|
150
|
-
To update to the latest version (**v0.
|
|
152
|
+
To update to the latest version (**v0.6.0**):
|
|
151
153
|
|
|
152
154
|
```bash
|
|
153
155
|
pip install --upgrade sql-dag-flow
|
|
@@ -33,8 +33,10 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
33
33
|
|
|
34
34
|
### 🔍 Visualization & Analysis
|
|
35
35
|
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
36
|
+
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
37
|
+
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
36
38
|
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
37
|
-
* **Discovery Mode**: Visualize "Ghost Nodes" (missing files or external tables)
|
|
39
|
+
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
38
40
|
* **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
|
|
39
41
|
* **Smart Layout (New 🧠)**:
|
|
40
42
|
* Powered by **ELK (Eclipse Layout Kernel)**.
|
|
@@ -75,7 +77,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
75
77
|
* **Column-Level Lineage (New in v0.4.9 🆕)**: Traces how each output column derives from source columns. Shows transformation chain (e.g., `order_timestamp ← orders_raw.order_date via CAST(... AS DATETIME)`).
|
|
76
78
|
* **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
|
|
77
79
|
* **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
|
|
78
|
-
* **Performance Optimizations (
|
|
80
|
+
* **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
|
|
79
81
|
* **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
|
|
80
82
|
* **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
|
|
81
83
|
|
|
@@ -122,7 +124,7 @@ Install easily via `pip`:
|
|
|
122
124
|
pip install sql-dag-flow
|
|
123
125
|
```
|
|
124
126
|
|
|
125
|
-
To update to the latest version (**v0.
|
|
127
|
+
To update to the latest version (**v0.6.0**):
|
|
126
128
|
|
|
127
129
|
```bash
|
|
128
130
|
pip install --upgrade sql-dag-flow
|
|
@@ -40,7 +40,7 @@ DIAGRAM_FILE = "sql_diagram.json"
|
|
|
40
40
|
_parse_cache = {"key": None, "tables": None, "time": 0}
|
|
41
41
|
PARSE_CACHE_TTL = 5 # seconds
|
|
42
42
|
|
|
43
|
-
def _cached_parse(directory, subfolders_tuple, dialect):
|
|
43
|
+
def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None):
|
|
44
44
|
"""Parse with simple TTL cache. Prevents re-parsing on concurrent requests."""
|
|
45
45
|
cache_key = (directory, subfolders_tuple, dialect)
|
|
46
46
|
now = time.time()
|
|
@@ -48,22 +48,23 @@ def _cached_parse(directory, subfolders_tuple, dialect):
|
|
|
48
48
|
return _parse_cache["tables"]
|
|
49
49
|
|
|
50
50
|
subfolders_list = list(subfolders_tuple) if subfolders_tuple else None
|
|
51
|
-
tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect)
|
|
51
|
+
tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids)
|
|
52
52
|
_parse_cache["key"] = cache_key
|
|
53
53
|
_parse_cache["tables"] = tables
|
|
54
54
|
_parse_cache["time"] = now
|
|
55
55
|
return tables
|
|
56
56
|
|
|
57
57
|
@app.get("/graph")
|
|
58
|
-
def get_graph(dialect: str = "bigquery", discovery: bool = False, expanded_nodes: str = ""):
|
|
58
|
+
def get_graph(dialect: str = "bigquery", discovery: bool = False, expanded_nodes: str = "", visible_node_ids: str = "", discovery_filter: str = "all"):
|
|
59
59
|
"""Parses SQL files in the current directory and returns graph data."""
|
|
60
60
|
if not os.path.exists(CURRENT_DIRECTORY):
|
|
61
61
|
return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
|
|
62
62
|
|
|
63
63
|
exp_nodes_list = [n.strip() for n in expanded_nodes.split(",")] if expanded_nodes else []
|
|
64
|
+
visible_list = [n.strip() for n in visible_node_ids.split(",")] if visible_node_ids else None
|
|
64
65
|
try:
|
|
65
|
-
tables = _cached_parse(CURRENT_DIRECTORY, None, dialect)
|
|
66
|
-
nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=exp_nodes_list)
|
|
66
|
+
tables = _cached_parse(CURRENT_DIRECTORY, None, dialect, visible_node_ids=visible_list)
|
|
67
|
+
nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=exp_nodes_list, discovery_filter=discovery_filter)
|
|
67
68
|
return {"nodes": nodes, "edges": edges, "cycles": cycles}
|
|
68
69
|
except Exception as e:
|
|
69
70
|
import traceback
|
|
@@ -117,14 +118,16 @@ def get_filtered_graph(data: dict = Body(...)):
|
|
|
117
118
|
if not os.path.exists(CURRENT_DIRECTORY):
|
|
118
119
|
return {"nodes": [], "edges": [], "error": "Directory not found"}
|
|
119
120
|
|
|
120
|
-
subfolders = data.get("subfolders")
|
|
121
|
+
subfolders = data.get("subfolders")
|
|
121
122
|
dialect = data.get("dialect", "bigquery")
|
|
122
123
|
discovery = data.get("discovery", False)
|
|
123
124
|
expanded_nodes = data.get("expanded_nodes", [])
|
|
125
|
+
visible_node_ids = data.get("visible_node_ids", None)
|
|
126
|
+
discovery_filter = data.get("discovery_filter", "all")
|
|
124
127
|
|
|
125
128
|
try:
|
|
126
|
-
tables = _cached_parse(CURRENT_DIRECTORY, tuple(subfolders) if subfolders else None, dialect)
|
|
127
|
-
nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=expanded_nodes)
|
|
129
|
+
tables = _cached_parse(CURRENT_DIRECTORY, tuple(subfolders) if subfolders else None, dialect, visible_node_ids=visible_node_ids)
|
|
130
|
+
nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=expanded_nodes, discovery_filter=discovery_filter)
|
|
128
131
|
return {"nodes": nodes, "edges": edges, "cycles": cycles}
|
|
129
132
|
except Exception as e:
|
|
130
133
|
import traceback
|
|
@@ -1,12 +1,52 @@
|
|
|
1
1
|
import os
|
|
2
2
|
import re
|
|
3
3
|
import time
|
|
4
|
+
import json
|
|
5
|
+
import hashlib
|
|
4
6
|
import sqlglot
|
|
5
7
|
from sqlglot import exp
|
|
6
8
|
from sqlglot.optimizer.qualify_columns import qualify_columns as sqlglot_qualify_columns
|
|
7
9
|
import networkx as nx
|
|
8
10
|
|
|
9
11
|
|
|
12
|
+
# ===== Persistent File Cache =====
|
|
13
|
+
CACHE_DIR = ".sqldagflow"
|
|
14
|
+
CACHE_FILENAME = "cache.json"
|
|
15
|
+
|
|
16
|
+
def _get_cache_path(directory):
|
|
17
|
+
return os.path.join(directory, CACHE_DIR, CACHE_FILENAME)
|
|
18
|
+
|
|
19
|
+
def _load_cache(directory):
|
|
20
|
+
"""Load parse cache from disk. Returns empty dict if not found."""
|
|
21
|
+
cache_path = _get_cache_path(directory)
|
|
22
|
+
try:
|
|
23
|
+
if os.path.exists(cache_path):
|
|
24
|
+
with open(cache_path, "r", encoding="utf-8") as f:
|
|
25
|
+
return json.load(f)
|
|
26
|
+
except Exception as e:
|
|
27
|
+
print(f" Cache load error (will rebuild): {e}")
|
|
28
|
+
return {}
|
|
29
|
+
|
|
30
|
+
def _save_cache(directory, cache):
|
|
31
|
+
"""Save parse cache to disk."""
|
|
32
|
+
cache_path = _get_cache_path(directory)
|
|
33
|
+
cache_dir = os.path.dirname(cache_path)
|
|
34
|
+
try:
|
|
35
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
36
|
+
with open(cache_path, "w", encoding="utf-8") as f:
|
|
37
|
+
json.dump(cache, f, separators=(',', ':'))
|
|
38
|
+
except Exception as e:
|
|
39
|
+
print(f" Cache save error: {e}")
|
|
40
|
+
|
|
41
|
+
def _file_cache_key(filepath):
|
|
42
|
+
"""Cache key = filepath mtime + size for fast invalidation."""
|
|
43
|
+
try:
|
|
44
|
+
stat = os.stat(filepath)
|
|
45
|
+
return f"{stat.st_mtime_ns}:{stat.st_size}"
|
|
46
|
+
except Exception:
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
10
50
|
def extract_output_columns(parsed, dialect="bigquery"):
|
|
11
51
|
"""
|
|
12
52
|
Extract output column schema from a parsed SQL AST.
|
|
@@ -83,14 +123,23 @@ def extract_output_columns(parsed, dialect="bigquery"):
|
|
|
83
123
|
|
|
84
124
|
return columns
|
|
85
125
|
|
|
86
|
-
def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
126
|
+
def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None):
|
|
87
127
|
"""
|
|
88
128
|
Recursively scans a directory for .sql files and parses them.
|
|
89
129
|
Returns a dictionary mapping table names to their dependencies and metadata.
|
|
130
|
+
|
|
131
|
+
Uses persistent mtime-based cache to skip re-parsing unchanged files.
|
|
132
|
+
If visible_node_ids is provided, qualify_columns and column_lineage
|
|
133
|
+
are only computed for visible nodes (performance optimization).
|
|
90
134
|
"""
|
|
91
135
|
tables = {}
|
|
136
|
+
cache = _load_cache(directory)
|
|
137
|
+
cache_hits = 0
|
|
138
|
+
cache_misses = 0
|
|
92
139
|
|
|
93
140
|
for root, dirs, files in os.walk(directory):
|
|
141
|
+
# Skip hidden config folders like .sqldagflow
|
|
142
|
+
dirs[:] = [d for d in dirs if not d.startswith('.')]
|
|
94
143
|
# Filter subfolders if allowed_subfolders is specified
|
|
95
144
|
if allowed_subfolders is not None:
|
|
96
145
|
# allowed_subfolders contains relative paths like "sub1", "sub1/nested"
|
|
@@ -180,6 +229,17 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
180
229
|
elif "gold" in lower_path:
|
|
181
230
|
layer = "gold"
|
|
182
231
|
|
|
232
|
+
# ===== Cache check: skip re-parsing if file unchanged =====
|
|
233
|
+
file_key = _file_cache_key(filepath)
|
|
234
|
+
cached_entry = cache.get(filepath)
|
|
235
|
+
if cached_entry and cached_entry.get("cache_key") == file_key and cached_entry.get("dialect") == dialect:
|
|
236
|
+
# Cache hit — use stored parse result
|
|
237
|
+
tables[filename_base] = cached_entry["data"]
|
|
238
|
+
cache_hits += 1
|
|
239
|
+
continue
|
|
240
|
+
|
|
241
|
+
cache_misses += 1
|
|
242
|
+
|
|
183
243
|
with open(filepath, "r", encoding="utf-8") as f:
|
|
184
244
|
sql_content = f.read()
|
|
185
245
|
|
|
@@ -545,8 +605,38 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
545
605
|
"error": str(e),
|
|
546
606
|
"content": sql_content
|
|
547
607
|
}
|
|
608
|
+
|
|
609
|
+
# ===== Save to cache after parsing (success or error) =====
|
|
610
|
+
if filename_base in tables:
|
|
611
|
+
file_key = _file_cache_key(filepath)
|
|
612
|
+
if file_key:
|
|
613
|
+
# Store a cache-safe copy (no sets, convert to lists)
|
|
614
|
+
cache_data = {}
|
|
615
|
+
for k, v in tables[filename_base].items():
|
|
616
|
+
if isinstance(v, set):
|
|
617
|
+
cache_data[k] = list(v)
|
|
618
|
+
elif isinstance(v, dict):
|
|
619
|
+
cache_data[k] = {
|
|
620
|
+
dk: list(dv) if isinstance(dv, set) else dv
|
|
621
|
+
for dk, dv in v.items()
|
|
622
|
+
}
|
|
623
|
+
else:
|
|
624
|
+
cache_data[k] = v
|
|
625
|
+
cache[filepath] = {
|
|
626
|
+
"cache_key": file_key,
|
|
627
|
+
"dialect": dialect,
|
|
628
|
+
"data": cache_data
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
# ===== Cache stats & persist =====
|
|
632
|
+
total_files = cache_hits + cache_misses
|
|
633
|
+
if total_files > 0:
|
|
634
|
+
print(f" Parse cache: {cache_hits}/{total_files} hits ({cache_misses} re-parsed)")
|
|
635
|
+
_save_cache(directory, cache)
|
|
636
|
+
|
|
548
637
|
# ===== Second Pass: Qualify Columns & Column Lineage =====
|
|
549
638
|
# Build a global schema dict for qualify_columns and lineage
|
|
639
|
+
# (schema building is always done for ALL tables — it's fast)
|
|
550
640
|
global_schema = {} # {dataset: {table: {col: type}}}
|
|
551
641
|
for tid, tdata in tables.items():
|
|
552
642
|
schema_cols = tdata.get("schema", [])
|
|
@@ -567,8 +657,13 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
567
657
|
|
|
568
658
|
# Re-extract column_references using qualify_columns for precision
|
|
569
659
|
# Protected with per-table time budget to prevent hanging on large projects
|
|
660
|
+
# If visible_node_ids provided, only qualify visible nodes (performance)
|
|
570
661
|
qualify_tables = [(tid, tdata) for tid, tdata in tables.items()
|
|
571
662
|
if not tdata.get("error") and tdata.get("content")]
|
|
663
|
+
if visible_node_ids is not None:
|
|
664
|
+
visible_set = set(visible_node_ids)
|
|
665
|
+
qualify_tables = [(tid, tdata) for tid, tdata in qualify_tables if tid in visible_set]
|
|
666
|
+
print(f" Selective qualify: {len(qualify_tables)} visible of {len(tables)} total")
|
|
572
667
|
total_qualify = len(qualify_tables)
|
|
573
668
|
|
|
574
669
|
for idx, (tid, tdata) in enumerate(qualify_tables):
|
|
@@ -650,12 +745,17 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
650
745
|
# ===== Column-Level Lineage =====
|
|
651
746
|
# For each table, trace how each output column derives from source columns
|
|
652
747
|
# Protected: skip if project is very large (>100 tables) and cap columns per table
|
|
748
|
+
# If visible_node_ids provided, only compute lineage for visible nodes
|
|
653
749
|
MAX_LINEAGE_TABLES = 500
|
|
654
750
|
MAX_COLS_PER_TABLE = 30
|
|
655
751
|
LINEAGE_TIME_BUDGET = 1.5 # seconds per table
|
|
656
752
|
|
|
657
753
|
lineage_tables = [(tid, tdata) for tid, tdata in tables.items()
|
|
658
754
|
if not tdata.get("error") and tdata.get("content") and tdata.get("schema")]
|
|
755
|
+
if visible_node_ids is not None:
|
|
756
|
+
visible_set = set(visible_node_ids)
|
|
757
|
+
lineage_tables = [(tid, tdata) for tid, tdata in lineage_tables if tid in visible_set]
|
|
758
|
+
print(f" Selective lineage: {len(lineage_tables)} visible of {len(tables)} total")
|
|
659
759
|
total_lineage = len(lineage_tables)
|
|
660
760
|
|
|
661
761
|
if total_lineage > MAX_LINEAGE_TABLES:
|
|
@@ -725,12 +825,13 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
725
825
|
return tables
|
|
726
826
|
|
|
727
827
|
|
|
728
|
-
def build_graph(tables, discovery_mode=False, expanded_nodes=None):
|
|
828
|
+
def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_filter='all'):
|
|
729
829
|
"""
|
|
730
830
|
Constructs nodes and edges for React Flow.
|
|
731
831
|
If discovery_mode is True, creates 'ghost' nodes for dependencies
|
|
732
832
|
that are not found in the parsed tables.
|
|
733
833
|
Also creates ghost nodes for any node whose ID is in expanded_nodes list.
|
|
834
|
+
discovery_filter: 'all' | 'external' | 'cte' — controls which ghost types to show.
|
|
734
835
|
"""
|
|
735
836
|
nodes = []
|
|
736
837
|
edges = []
|
|
@@ -790,7 +891,7 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None):
|
|
|
790
891
|
cte_name = ":".join(parts[2:])
|
|
791
892
|
cte_internal_deps = tables[source_id].get("cte_deps", {}).get(cte_name, {})
|
|
792
893
|
|
|
793
|
-
if discovery_mode or (expanded_nodes and source_id in expanded_nodes):
|
|
894
|
+
if (discovery_mode and discovery_filter in ('all', 'cte')) or (expanded_nodes and source_id in expanded_nodes):
|
|
794
895
|
# Discovery Mode or Expanded: Create CTE ghost node with incoming edges
|
|
795
896
|
cte_id = dep
|
|
796
897
|
|
|
@@ -885,7 +986,7 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None):
|
|
|
885
986
|
continue
|
|
886
987
|
|
|
887
988
|
# Handle missing external nodes (discovery mode or expanded node only)
|
|
888
|
-
if (discovery_mode or (expanded_nodes and source_id in expanded_nodes)) and not target_id:
|
|
989
|
+
if ((discovery_mode and discovery_filter in ('all', 'external')) or (expanded_nodes and source_id in expanded_nodes)) and not target_id:
|
|
889
990
|
# Create a unique ID for the missing node
|
|
890
991
|
# Use the full dependency name as the ID
|
|
891
992
|
ghost_id = dep
|