sql-dag-flow 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {sql_dag_flow-0.6.0/src/sql_dag_flow.egg-info → sql_dag_flow-0.7.0}/PKG-INFO +2 -2
  2. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/README.md +1 -1
  3. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/pyproject.toml +1 -1
  4. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/main.py +171 -10
  5. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/parser.py +59 -21
  6. sql_dag_flow-0.6.0/src/sql_dag_flow/static/assets/index-BRMQHj2Z.js → sql_dag_flow-0.7.0/src/sql_dag_flow/static/assets/index-BqmvUL2G.js +58 -52
  7. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/static/index.html +1 -1
  8. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0/src/sql_dag_flow.egg-info}/PKG-INFO +2 -2
  9. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow.egg-info/SOURCES.txt +1 -1
  10. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/LICENSE +0 -0
  11. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/MANIFEST.in +0 -0
  12. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/setup.cfg +0 -0
  13. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/__init__.py +0 -0
  14. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
  15. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/static/vite.svg +0 -0
  16. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/test_api_endpoints.py +0 -0
  17. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/test_parser.py +0 -0
  18. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow/verify_counts.py +0 -0
  19. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
  20. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
  21. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow.egg-info/requires.txt +0 -0
  22. {sql_dag_flow-0.6.0 → sql_dag_flow-0.7.0}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sql-dag-flow
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
5
5
  Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
6
6
  License: MIT
@@ -149,7 +149,7 @@ Install easily via `pip`:
149
149
  pip install sql-dag-flow
150
150
  ```
151
151
 
152
- To update to the latest version (**v0.6.0**):
152
+ To update to the latest version (**v0.7.0**):
153
153
 
154
154
  ```bash
155
155
  pip install --upgrade sql-dag-flow
@@ -124,7 +124,7 @@ Install easily via `pip`:
124
124
  pip install sql-dag-flow
125
125
  ```
126
126
 
127
- To update to the latest version (**v0.6.0**):
127
+ To update to the latest version (**v0.7.0**):
128
128
 
129
129
  ```bash
130
130
  pip install --upgrade sql-dag-flow
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sql-dag-flow"
7
- version = "0.6.0"
7
+ version = "0.7.0"
8
8
  description = "A sophisticated SQL lineage visualization tool for Medallion Architectures."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
@@ -14,6 +14,7 @@ import time
14
14
  import socket
15
15
  import argparse
16
16
  import shutil
17
+ import subprocess
17
18
  from .parser import parse_sql_files, build_graph
18
19
 
19
20
  app = FastAPI()
@@ -40,15 +41,16 @@ DIAGRAM_FILE = "sql_diagram.json"
40
41
  _parse_cache = {"key": None, "tables": None, "time": 0}
41
42
  PARSE_CACHE_TTL = 5 # seconds
42
43
 
43
- def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None):
44
+ def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None, target_ids=None):
44
45
  """Parse with simple TTL cache. Prevents re-parsing on concurrent requests."""
45
- cache_key = (directory, subfolders_tuple, dialect)
46
+ target_key = tuple(sorted(target_ids)) if target_ids else None
47
+ cache_key = (directory, subfolders_tuple, dialect, target_key)
46
48
  now = time.time()
47
49
  if _parse_cache["key"] == cache_key and (now - _parse_cache["time"]) < PARSE_CACHE_TTL:
48
50
  return _parse_cache["tables"]
49
-
51
+
50
52
  subfolders_list = list(subfolders_tuple) if subfolders_tuple else None
51
- tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids)
53
+ tables = parse_sql_files(directory, allowed_subfolders=subfolders_list, dialect=dialect, visible_node_ids=visible_node_ids, target_ids=target_ids)
52
54
  _parse_cache["key"] = cache_key
53
55
  _parse_cache["tables"] = tables
54
56
  _parse_cache["time"] = now
@@ -60,11 +62,22 @@ def get_graph(dialect: str = "bigquery", discovery: bool = False, expanded_nodes
60
62
  if not os.path.exists(CURRENT_DIRECTORY):
61
63
  return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
62
64
 
63
- exp_nodes_list = [n.strip() for n in expanded_nodes.split(",")] if expanded_nodes else []
65
+ # Parse expanded_nodes as dict {node_id: mode}
66
+ exp_nodes_dict = {}
67
+ if expanded_nodes:
68
+ for item in expanded_nodes.split(","):
69
+ item = item.strip()
70
+ if not item:
71
+ continue
72
+ parts = item.rsplit(":", 1)
73
+ if len(parts) == 2 and parts[1] in ("all", "external", "cte"):
74
+ exp_nodes_dict[parts[0]] = parts[1]
75
+ else:
76
+ exp_nodes_dict[item] = "all"
64
77
  visible_list = [n.strip() for n in visible_node_ids.split(",")] if visible_node_ids else None
65
78
  try:
66
79
  tables = _cached_parse(CURRENT_DIRECTORY, None, dialect, visible_node_ids=visible_list)
67
- nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=exp_nodes_list, discovery_filter=discovery_filter)
80
+ nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=exp_nodes_dict, discovery_filter=discovery_filter)
68
81
  return {"nodes": nodes, "edges": edges, "cycles": cycles}
69
82
  except Exception as e:
70
83
  import traceback
@@ -121,7 +134,10 @@ def get_filtered_graph(data: dict = Body(...)):
121
134
  subfolders = data.get("subfolders")
122
135
  dialect = data.get("dialect", "bigquery")
123
136
  discovery = data.get("discovery", False)
124
- expanded_nodes = data.get("expanded_nodes", [])
137
+ expanded_nodes = data.get("expanded_nodes", {})
138
+ # Normalize: if frontend sends a list (legacy), convert to dict
139
+ if isinstance(expanded_nodes, list):
140
+ expanded_nodes = {n: 'all' for n in expanded_nodes}
125
141
  visible_node_ids = data.get("visible_node_ids", None)
126
142
  discovery_filter = data.get("discovery_filter", "all")
127
143
 
@@ -134,6 +150,143 @@ def get_filtered_graph(data: dict = Body(...)):
134
150
  traceback.print_exc()
135
151
  return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
136
152
 
153
+ @app.post("/graph/scoped")
154
+ def get_scoped_graph(data: dict = Body(...)):
155
+ """Parse + build the graph for ONLY the given node ids (a saved view).
156
+
157
+ This is the fast path for reopening/refreshing a curated diagram: files
158
+ outside the scope are never parsed, and newly-added files never appear
159
+ unless the user explicitly adds them (see /scan/new). Any dependency that
160
+ points outside the scope simply stays unresolved rather than flooding the
161
+ canvas.
162
+ """
163
+ if not os.path.exists(CURRENT_DIRECTORY):
164
+ return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
165
+
166
+ node_ids = data.get("node_ids") or []
167
+ if not node_ids:
168
+ return {"nodes": [], "edges": [], "cycles": []}
169
+
170
+ dialect = data.get("dialect", "bigquery")
171
+ discovery = data.get("discovery", False)
172
+ expanded_nodes = data.get("expanded_nodes", {})
173
+ if isinstance(expanded_nodes, list):
174
+ expanded_nodes = {n: 'all' for n in expanded_nodes}
175
+ discovery_filter = data.get("discovery_filter", "all")
176
+
177
+ try:
178
+ tables = _cached_parse(
179
+ CURRENT_DIRECTORY, None, dialect,
180
+ visible_node_ids=node_ids, target_ids=node_ids,
181
+ )
182
+ nodes, edges, cycles = build_graph(
183
+ tables, discovery_mode=discovery,
184
+ expanded_nodes=expanded_nodes, discovery_filter=discovery_filter,
185
+ )
186
+ return {"nodes": nodes, "edges": edges, "cycles": cycles}
187
+ except Exception as e:
188
+ import traceback
189
+ traceback.print_exc()
190
+ return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
191
+
192
+
193
+ @app.post("/scan/new")
194
+ def scan_new_files(data: dict = Body(...)):
195
+ """List .sql models present on disk but not in the caller's known set.
196
+
197
+ Pure filesystem walk — no SQL parsing — so it stays instant even on huge
198
+ projects. Lets the UI say "5 new models found. Add?" instead of silently
199
+ re-indexing everything on refresh.
200
+ """
201
+ if not os.path.exists(CURRENT_DIRECTORY):
202
+ return {"new": []}
203
+
204
+ known = set(data.get("known_ids") or [])
205
+ new_models = []
206
+ seen = set()
207
+ for root, dirs, files in os.walk(CURRENT_DIRECTORY):
208
+ dirs[:] = [d for d in dirs if not d.startswith('.')]
209
+ for f in files:
210
+ if not f.endswith(".sql"):
211
+ continue
212
+ node_id = os.path.splitext(f)[0]
213
+ if node_id in known or node_id in seen:
214
+ continue
215
+ seen.add(node_id)
216
+ rel = os.path.relpath(os.path.join(root, f), CURRENT_DIRECTORY).replace(os.sep, '/')
217
+ new_models.append({"id": node_id, "path": rel})
218
+ new_models.sort(key=lambda m: m["path"])
219
+ return {"new": new_models}
220
+
221
+
222
+ def _run_git(args, timeout=10):
223
+ """Run a git command inside CURRENT_DIRECTORY, return stdout or None on failure."""
224
+ try:
225
+ r = subprocess.run(
226
+ ["git", "-C", CURRENT_DIRECTORY] + args,
227
+ capture_output=True, text=True, timeout=timeout,
228
+ )
229
+ if r.returncode != 0:
230
+ return None
231
+ return r.stdout
232
+ except Exception:
233
+ return None
234
+
235
+
236
+ @app.get("/git/changes")
237
+ def git_changes(base: str = ""):
238
+ """Return the .sql models changed in the current git working tree.
239
+
240
+ Maps changed files to node ids (filename base) so the UI can highlight the
241
+ edited models and their downstream blast radius — the core "what does this
242
+ PR affect?" view. If `base` (a branch/ref) is given, also includes files
243
+ that differ from that ref (committed changes on the current branch).
244
+ Degrades gracefully to {is_git:false} outside a repo.
245
+ """
246
+ inside = _run_git(["rev-parse", "--is-inside-work-tree"])
247
+ if inside is None or inside.strip() != "true":
248
+ return {"is_git": False, "changed": [], "base": base}
249
+
250
+ files = set()
251
+
252
+ # 1. Uncommitted + staged + untracked changes
253
+ porcelain = _run_git(["status", "--porcelain", "--untracked-files=all"]) or ""
254
+ for line in porcelain.splitlines():
255
+ if len(line) < 4:
256
+ continue
257
+ path = line[3:].strip()
258
+ # Renames show as "old -> new"; keep the new path.
259
+ if " -> " in path:
260
+ path = path.split(" -> ")[-1]
261
+ path = path.strip().strip('"')
262
+ files.add(path)
263
+
264
+ # 2. Committed diff vs a base ref (e.g. main) — the PR view
265
+ if base:
266
+ diff = _run_git(["diff", "--name-only", f"{base}...HEAD"])
267
+ if diff is None:
268
+ # Fall back to a two-dot diff if the merge-base form fails
269
+ diff = _run_git(["diff", "--name-only", base]) or ""
270
+ for line in diff.splitlines():
271
+ files.add(line.strip())
272
+
273
+ changed_ids = sorted({
274
+ os.path.splitext(os.path.basename(f))[0]
275
+ for f in files if f.endswith(".sql")
276
+ })
277
+ return {"is_git": True, "changed": changed_ids, "base": base}
278
+
279
+
280
+ @app.get("/git/branches")
281
+ def git_branches():
282
+ """List local branch names (for the base-branch picker). Empty outside a repo."""
283
+ out = _run_git(["branch", "--format=%(refname:short)"])
284
+ if out is None:
285
+ return {"is_git": False, "branches": []}
286
+ branches = [b.strip() for b in out.splitlines() if b.strip()]
287
+ return {"is_git": True, "branches": branches}
288
+
289
+
137
290
  @app.get("/config/path")
138
291
  def get_path():
139
292
  return {"path": CURRENT_DIRECTORY}
@@ -143,11 +296,19 @@ def export_data_dictionary(data: dict = Body(...)):
143
296
  """Generates a Markdown data dictionary report for visible nodes only."""
144
297
  dialect = data.get("dialect", "bigquery")
145
298
  visible_node_ids = data.get("visible_node_ids", None) # None = export all
146
-
299
+
147
300
  if not os.path.exists(CURRENT_DIRECTORY):
148
301
  raise HTTPException(status_code=400, detail="Directory not found")
149
-
150
- tables = parse_sql_files(CURRENT_DIRECTORY, dialect=dialect)
302
+
303
+ # Scope the parse to the visible nodes so we don't run the expensive
304
+ # qualify_columns + column-lineage passes over the whole project just to
305
+ # throw most of it away. target_ids skips non-visible files entirely.
306
+ tables = parse_sql_files(
307
+ CURRENT_DIRECTORY,
308
+ dialect=dialect,
309
+ visible_node_ids=visible_node_ids,
310
+ target_ids=visible_node_ids,
311
+ )
151
312
  nodes, edges, cycles = build_graph(tables, discovery_mode=False)
152
313
 
153
314
  # Filter to only visible nodes if list provided
@@ -123,20 +123,26 @@ def extract_output_columns(parsed, dialect="bigquery"):
123
123
 
124
124
  return columns
125
125
 
126
- def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None):
126
+ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visible_node_ids=None, target_ids=None):
127
127
  """
128
128
  Recursively scans a directory for .sql files and parses them.
129
129
  Returns a dictionary mapping table names to their dependencies and metadata.
130
-
130
+
131
131
  Uses persistent mtime-based cache to skip re-parsing unchanged files.
132
132
  If visible_node_ids is provided, qualify_columns and column_lineage
133
133
  are only computed for visible nodes (performance optimization).
134
+ If target_ids is provided (a set/list of node ids = filename bases), ONLY
135
+ those files are parsed and returned — the walk still runs (cheap) but
136
+ non-scoped files are skipped, so the heavy sqlglot work stays O(scope).
137
+ This is what powers saved "views": reopening one never re-parses the whole
138
+ project nor floods the canvas with newly-added files.
134
139
  """
135
140
  tables = {}
136
141
  cache = _load_cache(directory)
137
142
  cache_hits = 0
138
143
  cache_misses = 0
139
-
144
+ target_set = set(target_ids) if target_ids is not None else None
145
+
140
146
  for root, dirs, files in os.walk(directory):
141
147
  # Skip hidden config folders like .sqldagflow
142
148
  dirs[:] = [d for d in dirs if not d.startswith('.')]
@@ -218,7 +224,11 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
218
224
  filepath = os.path.join(root, file)
219
225
  # Heuristic for table name: filename without extension
220
226
  filename_base = os.path.splitext(file)[0]
221
-
227
+
228
+ # Scoped parse: skip any file not in the requested view.
229
+ if target_set is not None and filename_base not in target_set:
230
+ continue
231
+
222
232
  # Layer detection based on folder structure first, then filename
223
233
  lower_path = filepath.lower()
224
234
  layer = "other"
@@ -830,9 +840,12 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
830
840
  Constructs nodes and edges for React Flow.
831
841
  If discovery_mode is True, creates 'ghost' nodes for dependencies
832
842
  that are not found in the parsed tables.
833
- Also creates ghost nodes for any node whose ID is in expanded_nodes list.
843
+ expanded_nodes: dict {node_id: 'all'|'external'|'cte'} or list (legacy, treated as 'all').
834
844
  discovery_filter: 'all' | 'external' | 'cte' — controls which ghost types to show.
835
845
  """
846
+ # Normalize legacy list format to dict
847
+ if isinstance(expanded_nodes, list):
848
+ expanded_nodes = {n: 'all' for n in expanded_nodes}
836
849
  nodes = []
837
850
  edges = []
838
851
 
@@ -891,7 +904,8 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
891
904
  cte_name = ":".join(parts[2:])
892
905
  cte_internal_deps = tables[source_id].get("cte_deps", {}).get(cte_name, {})
893
906
 
894
- if (discovery_mode and discovery_filter in ('all', 'cte')) or (expanded_nodes and source_id in expanded_nodes):
907
+ expand_mode = expanded_nodes.get(source_id, None) if expanded_nodes else None
908
+ if (discovery_mode and discovery_filter in ('all', 'cte')) or (expand_mode and expand_mode in ('all', 'cte')):
895
909
  # Discovery Mode or Expanded: Create CTE ghost node with incoming edges
896
910
  cte_id = dep
897
911
 
@@ -986,7 +1000,8 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
986
1000
  continue
987
1001
 
988
1002
  # Handle missing external nodes (discovery mode or expanded node only)
989
- if ((discovery_mode and discovery_filter in ('all', 'external')) or (expanded_nodes and source_id in expanded_nodes)) and not target_id:
1003
+ expand_mode_ext = expanded_nodes.get(source_id, None) if expanded_nodes else None
1004
+ if ((discovery_mode and discovery_filter in ('all', 'external')) or (expand_mode_ext and expand_mode_ext in ('all', 'external'))) and not target_id:
990
1005
  # Create a unique ID for the missing node
991
1006
  # Use the full dependency name as the ID
992
1007
  ghost_id = dep
@@ -1063,22 +1078,45 @@ def build_graph(tables, discovery_mode=False, expanded_nodes=None, discovery_fil
1063
1078
  "label": consumer_data.get("label", consumer_id)
1064
1079
  })
1065
1080
 
1066
- for table_name, data in all_nodes_data.items():
1067
- # Calculate nested dependencies (all ancestors in the dependency graph)
1068
- nested_count = 0
1069
- if G.has_node(table_name):
1070
- try:
1071
- nested_count = len(nx.ancestors(G, table_name))
1072
- except Exception:
1073
- pass
1074
-
1075
- # Get downstream impact count
1076
- downstream_count = 0
1077
- if G.has_node(table_name):
1081
+ # ===== Reachability counts (ancestors / descendants) =====
1082
+ # Compute for ALL nodes in two topological passes instead of running a
1083
+ # separate O(V+E) traversal per node (the old approach was O(V*(V+E))).
1084
+ # Each node reuses its successors'/predecessors' already-computed sets.
1085
+ # Falls back to per-node only when the graph has cycles (rare, capped later).
1086
+ ancestors_count = {}
1087
+ descendants_count = {}
1088
+ try:
1089
+ topo = list(nx.topological_sort(G))
1090
+ desc_sets = {}
1091
+ for n in reversed(topo):
1092
+ s = set()
1093
+ for succ in G.successors(n):
1094
+ s.add(succ)
1095
+ s |= desc_sets[succ]
1096
+ desc_sets[n] = s
1097
+ anc_sets = {}
1098
+ for n in topo:
1099
+ s = set()
1100
+ for pred in G.predecessors(n):
1101
+ s.add(pred)
1102
+ s |= anc_sets[pred]
1103
+ anc_sets[n] = s
1104
+ descendants_count = {n: len(desc_sets[n]) for n in G.nodes()}
1105
+ ancestors_count = {n: len(anc_sets[n]) for n in G.nodes()}
1106
+ except Exception:
1107
+ # Cyclic graph: topological_sort is undefined — fall back per node.
1108
+ for n in G.nodes():
1078
1109
  try:
1079
- downstream_count = len(nx.descendants(G, table_name))
1110
+ ancestors_count[n] = len(nx.ancestors(G, n))
1111
+ descendants_count[n] = len(nx.descendants(G, n))
1080
1112
  except Exception:
1081
- pass
1113
+ ancestors_count[n] = 0
1114
+ descendants_count[n] = 0
1115
+
1116
+ for table_name, data in all_nodes_data.items():
1117
+ # Nested = all upstream ancestors; downstream = all descendants.
1118
+ nested_count = ancestors_count.get(table_name, 0)
1119
+ downstream_count = descendants_count.get(table_name, 0)
1082
1120
 
1083
1121
  nodes.append({
1084
1122
  "id": table_name,