sql-dag-flow 0.6.1__tar.gz → 0.7.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sql_dag_flow-0.6.1/src/sql_dag_flow.egg-info → sql_dag_flow-0.7.1}/PKG-INFO +6 -2
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/README.md +5 -1
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/pyproject.toml +1 -1
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/main.py +154 -7
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/parser.py +51 -18
- sql_dag_flow-0.6.1/src/sql_dag_flow/static/assets/index-KdnHXpnR.js → sql_dag_flow-0.7.1/src/sql_dag_flow/static/assets/index-BqmvUL2G.js +58 -52
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/index.html +1 -1
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1/src/sql_dag_flow.egg-info}/PKG-INFO +6 -2
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/SOURCES.txt +2 -2
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/LICENSE +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/MANIFEST.in +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/setup.cfg +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/__init__.py +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/vite.svg +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/test_api_endpoints.py +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/test_parser.py +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/verify_counts.py +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/requires.txt +0 -0
- {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sql-dag-flow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.1
|
|
4
4
|
Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
|
|
5
5
|
Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -60,6 +60,8 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
60
60
|
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
61
61
|
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
62
62
|
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
63
|
+
* **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
|
|
64
|
+
* **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
|
|
63
65
|
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
64
66
|
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
65
67
|
* **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
|
|
@@ -91,6 +93,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
91
93
|
|
|
92
94
|
### 📊 Discovery & Analysis Tools
|
|
93
95
|
* **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
|
|
96
|
+
* **Git Blast Radius (New in v0.7.0 🌿)**: Highlights the models changed in your git working tree — or versus a base branch — together with every downstream model they affect. The "what does this PR break?" view, rendered directly on the canvas.
|
|
94
97
|
* **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
|
|
95
98
|
* **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
|
|
96
99
|
* **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
|
|
@@ -103,6 +106,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
103
106
|
* **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
|
|
104
107
|
* **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
|
|
105
108
|
* **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
|
|
109
|
+
* **Performance Overhaul (New in v0.7.0 ⚡)**: Scoped parsing keeps the heavy `sqlglot` work proportional to your view; the Data Dictionary export no longer parses the entire project just to export a handful of models; upstream/downstream counts are computed in a single topological pass instead of one graph traversal per node; and the canvas virtualizes off-screen nodes and stops re-rendering every node on each refresh.
|
|
106
110
|
* **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
|
|
107
111
|
* **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
|
|
108
112
|
|
|
@@ -149,7 +153,7 @@ Install easily via `pip`:
|
|
|
149
153
|
pip install sql-dag-flow
|
|
150
154
|
```
|
|
151
155
|
|
|
152
|
-
To update to the latest version (**v0.
|
|
156
|
+
To update to the latest version (**v0.7.1**):
|
|
153
157
|
|
|
154
158
|
```bash
|
|
155
159
|
pip install --upgrade sql-dag-flow
|
|
@@ -35,6 +35,8 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
35
35
|
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
36
36
|
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
37
37
|
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
38
|
+
* **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
|
|
39
|
+
* **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
|
|
38
40
|
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
39
41
|
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
40
42
|
* **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
|
|
@@ -66,6 +68,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
66
68
|
|
|
67
69
|
### 📊 Discovery & Analysis Tools
|
|
68
70
|
* **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
|
|
71
|
+
* **Git Blast Radius (New in v0.7.0 🌿)**: Highlights the models changed in your git working tree — or versus a base branch — together with every downstream model they affect. The "what does this PR break?" view, rendered directly on the canvas.
|
|
69
72
|
* **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
|
|
70
73
|
* **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
|
|
71
74
|
* **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
|
|
@@ -78,6 +81,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
78
81
|
* **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
|
|
79
82
|
* **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
|
|
80
83
|
* **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
|
|
84
|
+
* **Performance Overhaul (New in v0.7.0 ⚡)**: Scoped parsing keeps the heavy `sqlglot` work proportional to your view; the Data Dictionary export no longer parses the entire project just to export a handful of models; upstream/downstream counts are computed in a single topological pass instead of one graph traversal per node; and the canvas virtualizes off-screen nodes and stops re-rendering every node on each refresh.
|
|
81
85
|
* **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
|
|
82
86
|
* **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
|
|
83
87
|
|
|
@@ -124,7 +128,7 @@ Install easily via `pip`:
|
|
|
124
128
|
pip install sql-dag-flow
|
|
125
129
|
```
|
|
126
130
|
|
|
127
|
-
To update to the latest version (**v0.
|
|
131
|
+
To update to the latest version (**v0.7.1**):
|
|
128
132
|
|
|
129
133
|
```bash
|
|
130
134
|
pip install --upgrade sql-dag-flow
|
|
@@ -14,6 +14,7 @@ import time
|
|
|
14
14
|
import socket
|
|
15
15
|
import argparse
|
|
16
16
|
import shutil
|
|
17
|
+
import subprocess
|
|
17
18
|
from .parser import parse_sql_files, build_graph
|
|
18
19
|
|
|
19
20
|
app = FastAPI()
|
|
@@ -40,15 +41,16 @@ DIAGRAM_FILE = "sql_diagram.json"
|
|
|
40
41
|
_parse_cache = {"key": None, "tables": None, "time": 0}
|
|
41
42
|
PARSE_CACHE_TTL = 5 # seconds
|
|
42
43
|
|
|
43
|
-
def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None):
|
|
44
|
+
def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None, target_ids=None):
|
|
44
45
|
"""Parse with simple TTL cache. Prevents re-parsing on concurrent requests."""
|
|
45
|
-
|
|
46
|
+
target_key = tuple(sorted(target_ids)) if target_ids else None
|
|
47
|
+
cache_key = (directory, subfolders_tuple, dialect, target_key)
|
|
46
48
|
now = time.time()
|
|
47
49
|
if _parse_cache["key"] == cache_key and (now - _parse_cache["time"]) < PARSE_CACHE_TTL:
|
|
48
50
|
return _parse_cache["tables"]
|
|
49
|
-
|
|
51
|
+
|
|
50
52
|
subfolders_list = list(subfolders_tuple) if subfolders_tuple else None
|
|
51
|
-
tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids)
|
|
53
|
+
tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids, target_ids=target_ids)
|
|
52
54
|
_parse_cache["key"] = cache_key
|
|
53
55
|
_parse_cache["tables"] = tables
|
|
54
56
|
_parse_cache["time"] = now
|
|
@@ -148,6 +150,143 @@ def get_filtered_graph(data: dict = Body(...)):
|
|
|
148
150
|
traceback.print_exc()
|
|
149
151
|
return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
|
|
150
152
|
|
|
153
|
+
@app.post("/graph/scoped")
|
|
154
|
+
def get_scoped_graph(data: dict = Body(...)):
|
|
155
|
+
"""Parse + build the graph for ONLY the given node ids (a saved view).
|
|
156
|
+
|
|
157
|
+
This is the fast path for reopening/refreshing a curated diagram: files
|
|
158
|
+
outside the scope are never parsed, and newly-added files never appear
|
|
159
|
+
unless the user explicitly adds them (see /scan/new). Any dependency that
|
|
160
|
+
points outside the scope simply stays unresolved rather than flooding the
|
|
161
|
+
canvas.
|
|
162
|
+
"""
|
|
163
|
+
if not os.path.exists(CURRENT_DIRECTORY):
|
|
164
|
+
return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
|
|
165
|
+
|
|
166
|
+
node_ids = data.get("node_ids") or []
|
|
167
|
+
if not node_ids:
|
|
168
|
+
return {"nodes": [], "edges": [], "cycles": []}
|
|
169
|
+
|
|
170
|
+
dialect = data.get("dialect", "bigquery")
|
|
171
|
+
discovery = data.get("discovery", False)
|
|
172
|
+
expanded_nodes = data.get("expanded_nodes", {})
|
|
173
|
+
if isinstance(expanded_nodes, list):
|
|
174
|
+
expanded_nodes = {n: 'all' for n in expanded_nodes}
|
|
175
|
+
discovery_filter = data.get("discovery_filter", "all")
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
tables = _cached_parse(
|
|
179
|
+
CURRENT_DIRECTORY, None, dialect,
|
|
180
|
+
visible_node_ids=node_ids, target_ids=node_ids,
|
|
181
|
+
)
|
|
182
|
+
nodes, edges, cycles = build_graph(
|
|
183
|
+
tables, discovery_mode=discovery,
|
|
184
|
+
expanded_nodes=expanded_nodes, discovery_filter=discovery_filter,
|
|
185
|
+
)
|
|
186
|
+
return {"nodes": nodes, "edges": edges, "cycles": cycles}
|
|
187
|
+
except Exception as e:
|
|
188
|
+
import traceback
|
|
189
|
+
traceback.print_exc()
|
|
190
|
+
return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@app.post("/scan/new")
|
|
194
|
+
def scan_new_files(data: dict = Body(...)):
|
|
195
|
+
"""List .sql models present on disk but not in the caller's known set.
|
|
196
|
+
|
|
197
|
+
Pure filesystem walk — no SQL parsing — so it stays instant even on huge
|
|
198
|
+
projects. Lets the UI say "5 new models found. Add?" instead of silently
|
|
199
|
+
re-indexing everything on refresh.
|
|
200
|
+
"""
|
|
201
|
+
if not os.path.exists(CURRENT_DIRECTORY):
|
|
202
|
+
return {"new": []}
|
|
203
|
+
|
|
204
|
+
known = set(data.get("known_ids") or [])
|
|
205
|
+
new_models = []
|
|
206
|
+
seen = set()
|
|
207
|
+
for root, dirs, files in os.walk(CURRENT_DIRECTORY):
|
|
208
|
+
dirs[:] = [d for d in dirs if not d.startswith('.')]
|
|
209
|
+
for f in files:
|
|
210
|
+
if not f.endswith(".sql"):
|
|
211
|
+
continue
|
|
212
|
+
node_id = os.path.splitext(f)[0]
|
|
213
|
+
if node_id in known or node_id in seen:
|
|
214
|
+
continue
|
|
215
|
+
seen.add(node_id)
|
|
216
|
+
rel = os.path.relpath(os.path.join(root, f), CURRENT_DIRECTORY).replace(os.sep, '/')
|
|
217
|
+
new_models.append({"id": node_id, "path": rel})
|
|
218
|
+
new_models.sort(key=lambda m: m["path"])
|
|
219
|
+
return {"new": new_models}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _run_git(args, timeout=10):
|
|
223
|
+
"""Run a git command inside CURRENT_DIRECTORY, return stdout or None on failure."""
|
|
224
|
+
try:
|
|
225
|
+
r = subprocess.run(
|
|
226
|
+
["git", "-C", CURRENT_DIRECTORY] + args,
|
|
227
|
+
capture_output=True, text=True, timeout=timeout,
|
|
228
|
+
)
|
|
229
|
+
if r.returncode != 0:
|
|
230
|
+
return None
|
|
231
|
+
return r.stdout
|
|
232
|
+
except Exception:
|
|
233
|
+
return None
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@app.get("/git/changes")
|
|
237
|
+
def git_changes(base: str = ""):
|
|
238
|
+
"""Return the .sql models changed in the current git working tree.
|
|
239
|
+
|
|
240
|
+
Maps changed files to node ids (filename base) so the UI can highlight the
|
|
241
|
+
edited models and their downstream blast radius — the core "what does this
|
|
242
|
+
PR affect?" view. If `base` (a branch/ref) is given, also includes files
|
|
243
|
+
that differ from that ref (committed changes on the current branch).
|
|
244
|
+
Degrades gracefully to {is_git:false} outside a repo.
|
|
245
|
+
"""
|
|
246
|
+
inside = _run_git(["rev-parse", "--is-inside-work-tree"])
|
|
247
|
+
if inside is None or inside.strip() != "true":
|
|
248
|
+
return {"is_git": False, "changed": [], "base": base}
|
|
249
|
+
|
|
250
|
+
files = set()
|
|
251
|
+
|
|
252
|
+
# 1. Uncommitted + staged + untracked changes
|
|
253
|
+
porcelain = _run_git(["status", "--porcelain", "--untracked-files=all"]) or ""
|
|
254
|
+
for line in porcelain.splitlines():
|
|
255
|
+
if len(line) < 4:
|
|
256
|
+
continue
|
|
257
|
+
path = line[3:].strip()
|
|
258
|
+
# Renames show as "old -> new"; keep the new path.
|
|
259
|
+
if " -> " in path:
|
|
260
|
+
path = path.split(" -> ")[-1]
|
|
261
|
+
path = path.strip().strip('"')
|
|
262
|
+
files.add(path)
|
|
263
|
+
|
|
264
|
+
# 2. Committed diff vs a base ref (e.g. main) — the PR view
|
|
265
|
+
if base:
|
|
266
|
+
diff = _run_git(["diff", "--name-only", f"{base}...HEAD"])
|
|
267
|
+
if diff is None:
|
|
268
|
+
# Fall back to a two-dot diff if the merge-base form fails
|
|
269
|
+
diff = _run_git(["diff", "--name-only", base]) or ""
|
|
270
|
+
for line in diff.splitlines():
|
|
271
|
+
files.add(line.strip())
|
|
272
|
+
|
|
273
|
+
changed_ids = sorted({
|
|
274
|
+
os.path.splitext(os.path.basename(f))[0]
|
|
275
|
+
for f in files if f.endswith(".sql")
|
|
276
|
+
})
|
|
277
|
+
return {"is_git": True, "changed": changed_ids, "base": base}
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
@app.get("/git/branches")
|
|
281
|
+
def git_branches():
|
|
282
|
+
"""List local branch names (for the base-branch picker). Empty outside a repo."""
|
|
283
|
+
out = _run_git(["branch", "--format=%(refname:short)"])
|
|
284
|
+
if out is None:
|
|
285
|
+
return {"is_git": False, "branches": []}
|
|
286
|
+
branches = [b.strip() for b in out.splitlines() if b.strip()]
|
|
287
|
+
return {"is_git": True, "branches": branches}
|
|
288
|
+
|
|
289
|
+
|
|
151
290
|
@app.get("/config/path")
|
|
152
291
|
def get_path():
|
|
153
292
|
return {"path": CURRENT_DIRECTORY}
|
|
@@ -157,11 +296,19 @@ def export_data_dictionary(data: dict = Body(...)):
|
|
|
157
296
|
"""Generates a Markdown data dictionary report for visible nodes only."""
|
|
158
297
|
dialect = data.get("dialect", "bigquery")
|
|
159
298
|
visible_node_ids = data.get("visible_node_ids", None) # None = export all
|
|
160
|
-
|
|
299
|
+
|
|
161
300
|
if not os.path.exists(CURRENT_DIRECTORY):
|
|
162
301
|
raise HTTPException(status_code=400, detail="Directory not found")
|
|
163
|
-
|
|
164
|
-
|
|
302
|
+
|
|
303
|
+
# Scope the parse to the visible nodes so we don't run the expensive
|
|
304
|
+
# qualify_columns + column-lineage passes over the whole project just to
|
|
305
|
+
# throw most of it away. target_ids skips non-visible files entirely.
|
|
306
|
+
tables = parse_sql_files(
|
|
307
|
+
CURRENT_DIRECTORY,
|
|
308
|
+
dialect=dialect,
|
|
309
|
+
visible_node_ids=visible_node_ids,
|
|
310
|
+
target_ids=visible_node_ids,
|
|
311
|
+
)
|
|
165
312
|
nodes, edges, cycles = build_graph(tables, discovery_mode=False)
|
|
166
313
|
|
|
167
314
|
# Filter to only visible nodes if list provided
|
|
@@ -123,20 +123,26 @@ def extract_output_columns(parsed, dialect="bigquery"):
|
|
|
123
123
|
|
|
124
124
|
return columns
|
|
125
125
|
|
|
126
|
-
def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None):
|
|
126
|
+
def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None, target_ids=None):
|
|
127
127
|
"""
|
|
128
128
|
Recursively scans a directory for .sql files and parses them.
|
|
129
129
|
Returns a dictionary mapping table names to their dependencies and metadata.
|
|
130
|
-
|
|
130
|
+
|
|
131
131
|
Uses persistent mtime-based cache to skip re-parsing unchanged files.
|
|
132
132
|
If visible_node_ids is provided, qualify_columns and column_lineage
|
|
133
133
|
are only computed for visible nodes (performance optimization).
|
|
134
|
+
If target_ids is provided (a set/list of node ids = filename bases), ONLY
|
|
135
|
+
those files are parsed and returned — the walk still runs (cheap) but
|
|
136
|
+
non-scoped files are skipped, so the heavy sqlglot work stays O(scope).
|
|
137
|
+
This is what powers saved "views": reopening one never re-parses the whole
|
|
138
|
+
project nor floods the canvas with newly-added files.
|
|
134
139
|
"""
|
|
135
140
|
tables = {}
|
|
136
141
|
cache = _load_cache(directory)
|
|
137
142
|
cache_hits = 0
|
|
138
143
|
cache_misses = 0
|
|
139
|
-
|
|
144
|
+
target_set = set(target_ids) if target_ids is not None else None
|
|
145
|
+
|
|
140
146
|
for root, dirs, files in os.walk(directory):
|
|
141
147
|
# Skip hidden config folders like .sqldagflow
|
|
142
148
|
dirs[:] = [d for d in dirs if not d.startswith('.')]
|
|
@@ -218,7 +224,11 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
|
|
|
218
224
|
filepath = os.path.join(root, file)
|
|
219
225
|
# Heuristic for table name: filename without extension
|
|
220
226
|
filename_base = os.path.splitext(file)[0]
|
|
221
|
-
|
|
227
|
+
|
|
228
|
+
# Scoped parse: skip any file not in the requested view.
|
|
229
|
+
if target_set is not None and filename_base not in target_set:
|
|
230
|
+
continue
|
|
231
|
+
|
|
222
232
|
# Layer detection based on folder structure first, then filename
|
|
223
233
|
lower_path = filepath.lower()
|
|
224
234
|
layer = "other"
|
|
@@ -1068,22 +1078,45 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
|
|
|
1068
1078
|
"label": consumer_data.get("label", consumer_id)
|
|
1069
1079
|
})
|
|
1070
1080
|
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1081
|
+
# ===== Reachability counts (ancestors / descendants) =====
|
|
1082
|
+
# Compute for ALL nodes in two topological passes instead of running a
|
|
1083
|
+
# separate O(V+E) traversal per node (the old approach was O(V*(V+E))).
|
|
1084
|
+
# Each node reuses its successors'/predecessors' already-computed sets.
|
|
1085
|
+
# Falls back to per-node only when the graph has cycles (rare, capped later).
|
|
1086
|
+
ancestors_count = {}
|
|
1087
|
+
descendants_count = {}
|
|
1088
|
+
try:
|
|
1089
|
+
topo = list(nx.topological_sort(G))
|
|
1090
|
+
desc_sets = {}
|
|
1091
|
+
for n in reversed(topo):
|
|
1092
|
+
s = set()
|
|
1093
|
+
for succ in G.successors(n):
|
|
1094
|
+
s.add(succ)
|
|
1095
|
+
s |= desc_sets[succ]
|
|
1096
|
+
desc_sets[n] = s
|
|
1097
|
+
anc_sets = {}
|
|
1098
|
+
for n in topo:
|
|
1099
|
+
s = set()
|
|
1100
|
+
for pred in G.predecessors(n):
|
|
1101
|
+
s.add(pred)
|
|
1102
|
+
s |= anc_sets[pred]
|
|
1103
|
+
anc_sets[n] = s
|
|
1104
|
+
descendants_count = {n: len(desc_sets[n]) for n in G.nodes()}
|
|
1105
|
+
ancestors_count = {n: len(anc_sets[n]) for n in G.nodes()}
|
|
1106
|
+
except Exception:
|
|
1107
|
+
# Cyclic graph: topological_sort is undefined — fall back per node.
|
|
1108
|
+
for n in G.nodes():
|
|
1083
1109
|
try:
|
|
1084
|
-
|
|
1110
|
+
ancestors_count[n] = len(nx.ancestors(G, n))
|
|
1111
|
+
descendants_count[n] = len(nx.descendants(G, n))
|
|
1085
1112
|
except Exception:
|
|
1086
|
-
|
|
1113
|
+
ancestors_count[n] = 0
|
|
1114
|
+
descendants_count[n] = 0
|
|
1115
|
+
|
|
1116
|
+
for table_name, data in all_nodes_data.items():
|
|
1117
|
+
# Nested = all upstream ancestors; downstream = all descendants.
|
|
1118
|
+
nested_count = ancestors_count.get(table_name, 0)
|
|
1119
|
+
downstream_count = descendants_count.get(table_name, 0)
|
|
1087
1120
|
|
|
1088
1121
|
nodes.append({
|
|
1089
1122
|
"id": table_name,
|