sql-dag-flow 0.6.1__tar.gz → 0.7.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {sql_dag_flow-0.6.1/src/sql_dag_flow.egg-info → sql_dag_flow-0.7.1}/PKG-INFO +6 -2
  2. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/README.md +5 -1
  3. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/pyproject.toml +1 -1
  4. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/main.py +154 -7
  5. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/parser.py +51 -18
  6. sql_dag_flow-0.6.1/src/sql_dag_flow/static/assets/index-KdnHXpnR.js → sql_dag_flow-0.7.1/src/sql_dag_flow/static/assets/index-BqmvUL2G.js +58 -52
  7. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/index.html +1 -1
  8. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1/src/sql_dag_flow.egg-info}/PKG-INFO +6 -2
  9. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/SOURCES.txt +2 -2
  10. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/LICENSE +0 -0
  11. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/MANIFEST.in +0 -0
  12. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/setup.cfg +0 -0
  13. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/__init__.py +0 -0
  14. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
  15. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/static/vite.svg +0 -0
  16. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/test_api_endpoints.py +0 -0
  17. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/test_parser.py +0 -0
  18. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow/verify_counts.py +0 -0
  19. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
  20. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
  21. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/requires.txt +0 -0
  22. {sql_dag_flow-0.6.1 → sql_dag_flow-0.7.1}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sql-dag-flow
3
- Version: 0.6.1
3
+ Version: 0.7.1
4
4
  Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
5
5
  Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
6
6
  License: MIT
@@ -60,6 +60,8 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
60
60
  * **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
61
61
  * **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
62
62
  * **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
63
+ * **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
64
+ * **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
63
65
  * **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
64
66
  * **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
65
67
  * **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
@@ -91,6 +93,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
91
93
 
92
94
  ### 📊 Discovery & Analysis Tools
93
95
  * **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
96
+ * **Git Blast Radius (New in v0.7.0 🌿)**: Highlights the models changed in your git working tree — or versus a base branch — together with every downstream model they affect. The "what does this PR break?" view, rendered directly on the canvas.
94
97
  * **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
95
98
  * **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
96
99
  * **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
@@ -103,6 +106,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
103
106
  * **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
104
107
  * **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
105
108
  * **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
109
+ * **Performance Overhaul (New in v0.7.0 ⚡)**: Scoped parsing keeps the heavy `sqlglot` work proportional to your view; the Data Dictionary export no longer parses the entire project just to export a handful of models; upstream/downstream counts are computed in a single topological pass instead of one graph traversal per node; and the canvas virtualizes off-screen nodes and stops re-rendering every node on each refresh.
106
110
  * **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
107
111
  * **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
108
112
 
@@ -149,7 +153,7 @@ Install easily via `pip`:
149
153
  pip install sql-dag-flow
150
154
  ```
151
155
 
152
- To update to the latest version (**v0.6.0**):
156
+ To update to the latest version (**v0.7.1**):
153
157
 
154
158
  ```bash
155
159
  pip install --upgrade sql-dag-flow
@@ -35,6 +35,8 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
35
35
  * **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
36
36
  * **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
37
37
  * **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
38
+ * **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
39
+ * **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
38
40
  * **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
39
41
  * **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
40
42
  * **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
@@ -66,6 +68,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
66
68
 
67
69
  ### 📊 Discovery & Analysis Tools
68
70
  * **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
71
+ * **Git Blast Radius (New in v0.7.0 🌿)**: Highlights the models changed in your git working tree — or versus a base branch — together with every downstream model they affect. The "what does this PR break?" view, rendered directly on the canvas.
69
72
  * **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
70
73
  * **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
71
74
  * **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
@@ -78,6 +81,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
78
81
  * **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
79
82
  * **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
80
83
  * **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
84
+ * **Performance Overhaul (New in v0.7.0 ⚡)**: Scoped parsing keeps the heavy `sqlglot` work proportional to your view; the Data Dictionary export no longer parses the entire project just to export a handful of models; upstream/downstream counts are computed in a single topological pass instead of one graph traversal per node; and the canvas virtualizes off-screen nodes and stops re-rendering every node on each refresh.
81
85
  * **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
82
86
  * **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
83
87
 
@@ -124,7 +128,7 @@ Install easily via `pip`:
124
128
  pip install sql-dag-flow
125
129
  ```
126
130
 
127
- To update to the latest version (**v0.6.0**):
131
+ To update to the latest version (**v0.7.1**):
128
132
 
129
133
  ```bash
130
134
  pip install --upgrade sql-dag-flow
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sql-dag-flow"
7
- version = "0.6.1"
7
+ version = "0.7.1"
8
8
  description = "A sophisticated SQL lineage visualization tool for Medallion Architectures."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
@@ -14,6 +14,7 @@ import time
14
14
  import socket
15
15
  import argparse
16
16
  import shutil
17
+ import subprocess
17
18
  from .parser import parse_sql_files, build_graph
18
19
 
19
20
  app = FastAPI()
@@ -40,15 +41,16 @@ DIAGRAM_FILE = "sql_diagram.json"
40
41
  _parse_cache = {"key": None, "tables": None, "time": 0}
41
42
  PARSE_CACHE_TTL = 5 # seconds
42
43
 
43
- def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None):
44
+ def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None, target_ids=None):
44
45
  """Parse with simple TTL cache. Prevents re-parsing on concurrent requests."""
45
- cache_key = (directory, subfolders_tuple, dialect)
46
+ target_key = tuple(sorted(target_ids)) if target_ids else None
47
+ cache_key = (directory, subfolders_tuple, dialect, target_key)
46
48
  now = time.time()
47
49
  if _parse_cache["key"] == cache_key and (now - _parse_cache["time"]) < PARSE_CACHE_TTL:
48
50
  return _parse_cache["tables"]
49
-
51
+
50
52
  subfolders_list = list(subfolders_tuple) if subfolders_tuple else None
51
- tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids)
53
+ tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids, target_ids=target_ids)
52
54
  _parse_cache["key"] = cache_key
53
55
  _parse_cache["tables"] = tables
54
56
  _parse_cache["time"] = now
@@ -148,6 +150,143 @@ def get_filtered_graph(data: dict = Body(...)):
148
150
  traceback.print_exc()
149
151
  return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
150
152
 
153
+ @app.post("/graph/scoped")
154
+ def get_scoped_graph(data: dict = Body(...)):
155
+ """Parse + build the graph for ONLY the given node ids (a saved view).
156
+
157
+ This is the fast path for reopening/refreshing a curated diagram: files
158
+ outside the scope are never parsed, and newly-added files never appear
159
+ unless the user explicitly adds them (see /scan/new). Any dependency that
160
+ points outside the scope simply stays unresolved rather than flooding the
161
+ canvas.
162
+ """
163
+ if not os.path.exists(CURRENT_DIRECTORY):
164
+ return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
165
+
166
+ node_ids = data.get("node_ids") or []
167
+ if not node_ids:
168
+ return {"nodes": [], "edges": [], "cycles": []}
169
+
170
+ dialect = data.get("dialect", "bigquery")
171
+ discovery = data.get("discovery", False)
172
+ expanded_nodes = data.get("expanded_nodes", {})
173
+ if isinstance(expanded_nodes, list):
174
+ expanded_nodes = {n: 'all' for n in expanded_nodes}
175
+ discovery_filter = data.get("discovery_filter", "all")
176
+
177
+ try:
178
+ tables = _cached_parse(
179
+ CURRENT_DIRECTORY, None, dialect,
180
+ visible_node_ids=node_ids, target_ids=node_ids,
181
+ )
182
+ nodes, edges, cycles = build_graph(
183
+ tables, discovery_mode=discovery,
184
+ expanded_nodes=expanded_nodes, discovery_filter=discovery_filter,
185
+ )
186
+ return {"nodes": nodes, "edges": edges, "cycles": cycles}
187
+ except Exception as e:
188
+ import traceback
189
+ traceback.print_exc()
190
+ return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
191
+
192
+
193
+ @app.post("/scan/new")
194
+ def scan_new_files(data: dict = Body(...)):
195
+ """List .sql models present on disk but not in the caller's known set.
196
+
197
+ Pure filesystem walk — no SQL parsing — so it stays instant even on huge
198
+ projects. Lets the UI say "5 new models found. Add?" instead of silently
199
+ re-indexing everything on refresh.
200
+ """
201
+ if not os.path.exists(CURRENT_DIRECTORY):
202
+ return {"new": []}
203
+
204
+ known = set(data.get("known_ids") or [])
205
+ new_models = []
206
+ seen = set()
207
+ for root, dirs, files in os.walk(CURRENT_DIRECTORY):
208
+ dirs[:] = [d for d in dirs if not d.startswith('.')]
209
+ for f in files:
210
+ if not f.endswith(".sql"):
211
+ continue
212
+ node_id = os.path.splitext(f)[0]
213
+ if node_id in known or node_id in seen:
214
+ continue
215
+ seen.add(node_id)
216
+ rel = os.path.relpath(os.path.join(root, f), CURRENT_DIRECTORY).replace(os.sep, '/')
217
+ new_models.append({"id": node_id, "path": rel})
218
+ new_models.sort(key=lambda m: m["path"])
219
+ return {"new": new_models}
220
+
221
+
222
+ def _run_git(args, timeout=10):
223
+ """Run a git command inside CURRENT_DIRECTORY, return stdout or None on failure."""
224
+ try:
225
+ r = subprocess.run(
226
+ ["git", "-C", CURRENT_DIRECTORY] + args,
227
+ capture_output=True, text=True, timeout=timeout,
228
+ )
229
+ if r.returncode != 0:
230
+ return None
231
+ return r.stdout
232
+ except Exception:
233
+ return None
234
+
235
+
236
+ @app.get("/git/changes")
237
+ def git_changes(base: str = ""):
238
+ """Return the .sql models changed in the current git working tree.
239
+
240
+ Maps changed files to node ids (filename base) so the UI can highlight the
241
+ edited models and their downstream blast radius — the core "what does this
242
+ PR affect?" view. If `base` (a branch/ref) is given, also includes files
243
+ that differ from that ref (committed changes on the current branch).
244
+ Degrades gracefully to {is_git:false} outside a repo.
245
+ """
246
+ inside = _run_git(["rev-parse", "--is-inside-work-tree"])
247
+ if inside is None or inside.strip() != "true":
248
+ return {"is_git": False, "changed": [], "base": base}
249
+
250
+ files = set()
251
+
252
+ # 1. Uncommitted + staged + untracked changes
253
+ porcelain = _run_git(["status", "--porcelain", "--untracked-files=all"]) or ""
254
+ for line in porcelain.splitlines():
255
+ if len(line) < 4:
256
+ continue
257
+ path = line[3:].strip()
258
+ # Renames show as "old -> new"; keep the new path.
259
+ if " -> " in path:
260
+ path = path.split(" -> ")[-1]
261
+ path = path.strip().strip('"')
262
+ files.add(path)
263
+
264
+ # 2. Committed diff vs a base ref (e.g. main) — the PR view
265
+ if base:
266
+ diff = _run_git(["diff", "--name-only", f"{base}...HEAD"])
267
+ if diff is None:
268
+ # Fall back to a two-dot diff if the merge-base form fails
269
+ diff = _run_git(["diff", "--name-only", base]) or ""
270
+ for line in diff.splitlines():
271
+ files.add(line.strip())
272
+
273
+ changed_ids = sorted({
274
+ os.path.splitext(os.path.basename(f))[0]
275
+ for f in files if f.endswith(".sql")
276
+ })
277
+ return {"is_git": True, "changed": changed_ids, "base": base}
278
+
279
+
280
+ @app.get("/git/branches")
281
+ def git_branches():
282
+ """List local branch names (for the base-branch picker). Empty outside a repo."""
283
+ out = _run_git(["branch", "--format=%(refname:short)"])
284
+ if out is None:
285
+ return {"is_git": False, "branches": []}
286
+ branches = [b.strip() for b in out.splitlines() if b.strip()]
287
+ return {"is_git": True, "branches": branches}
288
+
289
+
151
290
  @app.get("/config/path")
152
291
  def get_path():
153
292
  return {"path": CURRENT_DIRECTORY}
@@ -157,11 +296,19 @@ def export_data_dictionary(data: dict = Body(...)):
157
296
  """Generates a Markdown data dictionary report for visible nodes only."""
158
297
  dialect = data.get("dialect", "bigquery")
159
298
  visible_node_ids = data.get("visible_node_ids", None) # None = export all
160
-
299
+
161
300
  if not os.path.exists(CURRENT_DIRECTORY):
162
301
  raise HTTPException(status_code=400, detail="Directory not found")
163
-
164
- tables = parse_sql_files(CURRENT_DIRECTORY, dialect=dialect)
302
+
303
+ # Scope the parse to the visible nodes so we don't run the expensive
304
+ # qualify_columns + column-lineage passes over the whole project just to
305
+ # throw most of it away. target_ids skips non-visible files entirely.
306
+ tables = parse_sql_files(
307
+ CURRENT_DIRECTORY,
308
+ dialect=dialect,
309
+ visible_node_ids=visible_node_ids,
310
+ target_ids=visible_node_ids,
311
+ )
165
312
  nodes, edges, cycles = build_graph(tables, discovery_mode=False)
166
313
 
167
314
  # Filter to only visible nodes if list provided
@@ -123,20 +123,26 @@ def extract_output_columns(parsed, dialect="bigquery"):
123
123
 
124
124
  return columns
125
125
 
126
- def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None):
126
+ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None, target_ids=None):
127
127
  """
128
128
  Recursively scans a directory for .sql files and parses them.
129
129
  Returns a dictionary mapping table names to their dependencies and metadata.
130
-
130
+
131
131
  Uses persistent mtime-based cache to skip re-parsing unchanged files.
132
132
  If visible_node_ids is provided, qualify_columns and column_lineage
133
133
  are only computed for visible nodes (performance optimization).
134
+ If target_ids is provided (a set/list of node ids = filename bases), ONLY
135
+ those files are parsed and returned — the walk still runs (cheap) but
136
+ non-scoped files are skipped, so the heavy sqlglot work stays O(scope).
137
+ This is what powers saved "views": reopening one never re-parses the whole
138
+ project nor floods the canvas with newly-added files.
134
139
  """
135
140
  tables = {}
136
141
  cache = _load_cache(directory)
137
142
  cache_hits = 0
138
143
  cache_misses = 0
139
-
144
+ target_set = set(target_ids) if target_ids is not None else None
145
+
140
146
  for root, dirs, files in os.walk(directory):
141
147
  # Skip hidden config folders like .sqldagflow
142
148
  dirs[:] = [d for d in dirs if not d.startswith('.')]
@@ -218,7 +224,11 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
218
224
  filepath = os.path.join(root, file)
219
225
  # Heuristic for table name: filename without extension
220
226
  filename_base = os.path.splitext(file)[0]
221
-
227
+
228
+ # Scoped parse: skip any file not in the requested view.
229
+ if target_set is not None and filename_base not in target_set:
230
+ continue
231
+
222
232
  # Layer detection based on folder structure first, then filename
223
233
  lower_path = filepath.lower()
224
234
  layer = "other"
@@ -1068,22 +1078,45 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
1068
1078
  "label": consumer_data.get("label", consumer_id)
1069
1079
  })
1070
1080
 
1071
- for table_name, data in all_nodes_data.items():
1072
- # Calculate nested dependencies (all ancestors in the dependency graph)
1073
- nested_count = 0
1074
- if G.has_node(table_name):
1075
- try:
1076
- nested_count = len(nx.ancestors(G, table_name))
1077
- except Exception:
1078
- pass
1079
-
1080
- # Get downstream impact count
1081
- downstream_count = 0
1082
- if G.has_node(table_name):
1081
+ # ===== Reachability counts (ancestors / descendants) =====
1082
+ # Compute for ALL nodes in two topological passes instead of running a
1083
+ # separate O(V+E) traversal per node (the old approach was O(V*(V+E))).
1084
+ # Each node reuses its successors'/predecessors' already-computed sets.
1085
+ # Falls back to per-node only when the graph has cycles (rare, capped later).
1086
+ ancestors_count = {}
1087
+ descendants_count = {}
1088
+ try:
1089
+ topo = list(nx.topological_sort(G))
1090
+ desc_sets = {}
1091
+ for n in reversed(topo):
1092
+ s = set()
1093
+ for succ in G.successors(n):
1094
+ s.add(succ)
1095
+ s |= desc_sets[succ]
1096
+ desc_sets[n] = s
1097
+ anc_sets = {}
1098
+ for n in topo:
1099
+ s = set()
1100
+ for pred in G.predecessors(n):
1101
+ s.add(pred)
1102
+ s |= anc_sets[pred]
1103
+ anc_sets[n] = s
1104
+ descendants_count = {n: len(desc_sets[n]) for n in G.nodes()}
1105
+ ancestors_count = {n: len(anc_sets[n]) for n in G.nodes()}
1106
+ except Exception:
1107
+ # Cyclic graph: topological_sort is undefined — fall back per node.
1108
+ for n in G.nodes():
1083
1109
  try:
1084
- downstream_count = len(nx.descendants(G, table_name))
1110
+ ancestors_count[n] = len(nx.ancestors(G, n))
1111
+ descendants_count[n] = len(nx.descendants(G, n))
1085
1112
  except Exception:
1086
- pass
1113
+ ancestors_count[n] = 0
1114
+ descendants_count[n] = 0
1115
+
1116
+ for table_name, data in all_nodes_data.items():
1117
+ # Nested = all upstream ancestors; downstream = all descendants.
1118
+ nested_count = ancestors_count.get(table_name, 0)
1119
+ downstream_count = descendants_count.get(table_name, 0)
1087
1120
 
1088
1121
  nodes.append({
1089
1122
  "id": table_name,