sql-dag-flow 0.4.8__tar.gz → 0.4.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {sql_dag_flow-0.4.8/src/sql_dag_flow.egg-info → sql_dag_flow-0.4.9}/PKG-INFO +8 -6
  2. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/README.md +7 -5
  3. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/pyproject.toml +1 -1
  4. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/main.py +47 -3
  5. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/parser.py +163 -0
  6. sql_dag_flow-0.4.8/src/sql_dag_flow/static/assets/index-SW0VVE5U.js → sql_dag_flow-0.4.9/src/sql_dag_flow/static/assets/index-ZylJfG6u.js +39 -39
  7. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/static/index.html +1 -1
  8. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9/src/sql_dag_flow.egg-info}/PKG-INFO +8 -6
  9. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow.egg-info/SOURCES.txt +1 -1
  10. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/LICENSE +0 -0
  11. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/MANIFEST.in +0 -0
  12. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/setup.cfg +0 -0
  13. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/__init__.py +0 -0
  14. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/static/assets/index-BM6GPh_j.css +0 -0
  15. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/static/vite.svg +0 -0
  16. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/test_api_endpoints.py +0 -0
  17. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/test_parser.py +0 -0
  18. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow/verify_counts.py +0 -0
  19. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
  20. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
  21. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow.egg-info/requires.txt +0 -0
  22. {sql_dag_flow-0.4.8 → sql_dag_flow-0.4.9}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sql-dag-flow
3
- Version: 0.4.8
3
+ Version: 0.4.9
4
4
  Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
5
5
  Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
6
6
  License: MIT
@@ -89,15 +89,17 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
89
89
  ### 📊 Discovery & Analysis Tools
90
90
  * **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
91
91
  * **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
92
- * **Column Usage Tracking (Improved in v0.4.8 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Now tracks both qualified (`u.name`) and unqualified (`name`) column references for accurate counts.
92
+ * **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
93
93
  * **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
94
94
  * **Business Rule Extraction**: Automatically detects and displays WHERE filters, CASE logic, HAVING clauses, and aggregations from each SQL model.
95
95
  * **Complexity Scoring**: Weighted metric per node (JOINs×3, CTEs×2, Subqueries×3, Filters×1, CASE×2, Aggregations×1, UNIONs×2) with color-coded badges (🟢 Low, 🟡 Medium, 🟠 High, 🔴 Very High). Toggleable via ⚡ button.
96
96
  * **Node Comparison**: Select exactly 2 nodes and compare them side-by-side. Highlights differences across metadata, schema columns (shared vs unique), dependencies (Venn-style), complexity scores (with delta indicators), business rules, and SQL content (synced scroll).
97
97
  * **Statistics Panel**: Centered popup with layer distribution bars, edge/source/sink/orphan counts, project/dataset tree, and architecture health validation (now detects **circular dependencies**).
98
- * **Schema Extraction (Improved in v0.4.8 🔧)**: Backend AST-based extraction via `sqlglot` handles DDL, CTAS, `CREATE VIEW AS`, CTEs, window functions, CASE expressions, and `SELECT *`.
99
- * **Batch Hide from Toolbar (New in v0.4.8 🆕)**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
100
- * **Discovery Mode Fix (v0.4.8 🔧)**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
98
+ * **Schema Extraction**: Backend AST-based extraction via `sqlglot` handles DDL, CTAS, `CREATE VIEW AS`, CTEs, window functions, CASE expressions, and `SELECT *`.
99
+ * **Column-Level Lineage (New in v0.4.9 🆕)**: Traces how each output column derives from source columns. Shows transformation chain (e.g., `order_timestamp ← orders_raw.order_date via CAST(... AS DATETIME)`).
100
+ * **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
101
+ * **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
102
+ * **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
101
103
 
102
104
  ### 🎨 Linear-Inspired UI (New in v0.4.6 ✨)
103
105
  * **Design Token System**: ~80 CSS custom properties for consistent theming across all components.
@@ -142,7 +144,7 @@ Install easily via `pip`:
142
144
  pip install sql-dag-flow
143
145
  ```
144
146
 
145
- To update to the latest version (**v0.4.8**):
147
+ To update to the latest version (**v0.4.9**):
146
148
 
147
149
  ```bash
148
150
  pip install --upgrade sql-dag-flow
@@ -64,15 +64,17 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
64
64
  ### 📊 Discovery & Analysis Tools
65
65
  * **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
66
66
  * **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
67
- * **Column Usage Tracking (Improved in v0.4.8 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Now tracks both qualified (`u.name`) and unqualified (`name`) column references for accurate counts.
67
+ * **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
68
68
  * **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
69
69
  * **Business Rule Extraction**: Automatically detects and displays WHERE filters, CASE logic, HAVING clauses, and aggregations from each SQL model.
70
70
  * **Complexity Scoring**: Weighted metric per node (JOINs×3, CTEs×2, Subqueries×3, Filters×1, CASE×2, Aggregations×1, UNIONs×2) with color-coded badges (🟢 Low, 🟡 Medium, 🟠 High, 🔴 Very High). Toggleable via ⚡ button.
71
71
  * **Node Comparison**: Select exactly 2 nodes and compare them side-by-side. Highlights differences across metadata, schema columns (shared vs unique), dependencies (Venn-style), complexity scores (with delta indicators), business rules, and SQL content (synced scroll).
72
72
  * **Statistics Panel**: Centered popup with layer distribution bars, edge/source/sink/orphan counts, project/dataset tree, and architecture health validation (now detects **circular dependencies**).
73
- * **Schema Extraction (Improved in v0.4.8 🔧)**: Backend AST-based extraction via `sqlglot` handles DDL, CTAS, `CREATE VIEW AS`, CTEs, window functions, CASE expressions, and `SELECT *`.
74
- * **Batch Hide from Toolbar (New in v0.4.8 🆕)**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
75
- * **Discovery Mode Fix (v0.4.8 🔧)**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
73
+ * **Schema Extraction**: Backend AST-based extraction via `sqlglot` handles DDL, CTAS, `CREATE VIEW AS`, CTEs, window functions, CASE expressions, and `SELECT *`.
74
+ * **Column-Level Lineage (New in v0.4.9 🆕)**: Traces how each output column derives from source columns. Shows transformation chain (e.g., `order_timestamp ← orders_raw.order_date via CAST(... AS DATETIME)`).
75
+ * **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
76
+ * **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
77
+ * **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
76
78
 
77
79
  ### 🎨 Linear-Inspired UI (New in v0.4.6 ✨)
78
80
  * **Design Token System**: ~80 CSS custom properties for consistent theming across all components.
@@ -117,7 +119,7 @@ Install easily via `pip`:
117
119
  pip install sql-dag-flow
118
120
  ```
119
121
 
120
- To update to the latest version (**v0.4.8**):
122
+ To update to the latest version (**v0.4.9**):
121
123
 
122
124
  ```bash
123
125
  pip install --upgrade sql-dag-flow
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sql-dag-flow"
7
- version = "0.4.8"
7
+ version = "0.4.9"
8
8
  description = "A sophisticated SQL lineage visualization tool for Medallion Architectures."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
@@ -107,15 +107,24 @@ def get_filtered_graph(data: dict = Body(...)):
107
107
  def get_path():
108
108
  return {"path": CURRENT_DIRECTORY}
109
109
 
110
- @app.get("/export")
111
- def export_data_dictionary(dialect: str = "bigquery"):
112
- """Generates a Markdown data dictionary report."""
110
+ @app.post("/export")
111
+ def export_data_dictionary(data: dict = Body(...)):
112
+ """Generates a Markdown data dictionary report for visible nodes only."""
113
+ dialect = data.get("dialect", "bigquery")
114
+ visible_node_ids = data.get("visible_node_ids", None) # None = export all
115
+
113
116
  if not os.path.exists(CURRENT_DIRECTORY):
114
117
  raise HTTPException(status_code=400, detail="Directory not found")
115
118
 
116
119
  tables = parse_sql_files(CURRENT_DIRECTORY, dialect=dialect)
117
120
  nodes, edges, cycles = build_graph(tables, discovery_mode=False)
118
121
 
122
+ # Filter to only visible nodes if list provided
123
+ if visible_node_ids is not None:
124
+ visible_set = set(visible_node_ids)
125
+ nodes = [n for n in nodes if n['id'] in visible_set]
126
+ edges = [e for e in edges if e['source'] in visible_set and e['target'] in visible_set]
127
+
119
128
  lines = []
120
129
  lines.append(f"# Data Dictionary")
121
130
  lines.append(f"")
@@ -195,6 +204,17 @@ def export_data_dictionary(dialect: str = "bigquery"):
195
204
  lines.extend(rules_items[:6])
196
205
  lines.append(f"")
197
206
 
207
+ # Schema columns
208
+ schema = d.get('schema', [])
209
+ if schema:
210
+ lines.append(f"**Schema ({len(schema)} columns):**")
211
+ lines.append(f"")
212
+ lines.append(f"| Column | Type |")
213
+ lines.append(f"|--------|------|")
214
+ for col in schema:
215
+ lines.append(f"| `{col.get('name', '?')}` | {col.get('type', 'UNKNOWN')} |")
216
+ lines.append(f"")
217
+
198
218
  # Column consumers
199
219
  col_consumers = d.get('column_consumers', {})
200
220
  if col_consumers:
@@ -207,6 +227,30 @@ def export_data_dictionary(dialect: str = "bigquery"):
207
227
  lines.append(f"| `{col}` | {consumer_labels} |")
208
228
  lines.append(f"")
209
229
 
230
+ # Column lineage
231
+ col_lineage = d.get('column_lineage', {})
232
+ if col_lineage:
233
+ lines.append(f"**Column Lineage:**")
234
+ lines.append(f"")
235
+ lines.append(f"| Output Column | Source | Transform |")
236
+ lines.append(f"|---------------|--------|-----------|")
237
+ for col, sources in sorted(col_lineage.items()):
238
+ for src in sources:
239
+ src_ref = f"{src.get('source_table', '')}.{src.get('source_column', '')}" if src.get('source_table') else src.get('source_column', '')
240
+ transform = src.get('transform', '') or '—'
241
+ lines.append(f"| `{col}` | `{src_ref}` | {transform} |")
242
+ lines.append(f"")
243
+
244
+ # Syntax warnings
245
+ syntax_warnings = d.get('syntax_warnings', [])
246
+ if syntax_warnings:
247
+ lines.append(f"**⚠️ Syntax Issues ({len(syntax_warnings)}):**")
248
+ lines.append(f"")
249
+ for w in syntax_warnings:
250
+ loc = f"Line {w.get('line', '?')}, Col {w.get('col', '?')}" if w.get('line') else ""
251
+ lines.append(f"- {w.get('description', 'Unknown error')} {f'({loc})' if loc else ''}")
252
+ lines.append(f"")
253
+
210
254
  lines.append(f"---")
211
255
  lines.append(f"")
212
256
 
@@ -3,6 +3,7 @@ import re
3
3
  import time
4
4
  import sqlglot
5
5
  from sqlglot import exp
6
+ from sqlglot.optimizer.qualify_columns import qualify_columns as sqlglot_qualify_columns
6
7
  import networkx as nx
7
8
 
8
9
 
@@ -506,6 +507,30 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
506
507
  "header_meta": header_meta,
507
508
  "last_modified_days": days_ago
508
509
  }
510
+ except sqlglot.errors.ParseError as pe:
511
+ # Feature 4: Capture structured syntax errors
512
+ syntax_warnings = []
513
+ for err in pe.errors:
514
+ syntax_warnings.append({
515
+ "description": err.get("description", str(pe)),
516
+ "line": err.get("line"),
517
+ "col": err.get("col"),
518
+ "highlight": err.get("highlight", "")
519
+ })
520
+ print(f"Syntax error in {filepath}: {pe}")
521
+ tables[filename_base] = {
522
+ "id": filename_base,
523
+ "label": filename_base,
524
+ "layer": layer,
525
+ "type": "unknown",
526
+ "project": "n/a",
527
+ "dataset": "n/a",
528
+ "path": filepath,
529
+ "dependencies": {},
530
+ "error": str(pe),
531
+ "syntax_warnings": syntax_warnings,
532
+ "content": sql_content
533
+ }
509
534
  except Exception as e:
510
535
  print(f"Error parsing {filepath}: {e}")
511
536
  tables[filename_base] = {
@@ -520,6 +545,144 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
520
545
  "error": str(e),
521
546
  "content": sql_content
522
547
  }
548
+ # ===== Second Pass: Qualify Columns & Column Lineage =====
549
+ # Build a global schema dict for qualify_columns and lineage
550
+ global_schema = {} # {dataset: {table: {col: type}}}
551
+ for tid, tdata in tables.items():
552
+ schema_cols = tdata.get("schema", [])
553
+ if not schema_cols:
554
+ continue
555
+ ds = tdata.get("dataset", "default")
556
+ label = tdata.get("label", tid)
557
+ if ds not in global_schema:
558
+ global_schema[ds] = {}
559
+ col_dict = {}
560
+ for c in schema_cols:
561
+ col_dict[c["name"]] = c.get("type", "UNKNOWN")
562
+ global_schema[ds][label] = col_dict
563
+ # Also add without dataset prefix for simpler lookups
564
+ if "default" not in global_schema:
565
+ global_schema["default"] = {}
566
+ global_schema["default"][label] = col_dict
567
+
568
+ # Re-extract column_references using qualify_columns for precision
569
+ for tid, tdata in tables.items():
570
+ if tdata.get("error") or not tdata.get("content"):
571
+ continue
572
+ try:
573
+ parsed = sqlglot.parse_one(tdata["content"], read=dialect)
574
+
575
+ # Try to qualify columns using the global schema
576
+ try:
577
+ qualified_ast = sqlglot_qualify_columns(parsed.copy(), schema=global_schema, dialect=dialect)
578
+ except Exception:
579
+ qualified_ast = parsed # Fallback to unqualified
580
+
581
+ # Re-extract column references from qualified AST
582
+ alias_map = {}
583
+ defined_ctes = set()
584
+ for cte in qualified_ast.find_all(exp.CTE):
585
+ cn = cte.alias_or_name
586
+ if cn:
587
+ defined_ctes.add(cn)
588
+
589
+ for t in qualified_ast.find_all(exp.Table):
590
+ t_name = t.name
591
+ t_full = t_name
592
+ if t.db:
593
+ t_full = f"{t.db}.{t_name}"
594
+ if t.catalog:
595
+ t_full = f"{t.catalog}.{t.db}.{t_name}"
596
+ if t.alias:
597
+ alias_map[t.alias] = t_full
598
+ alias_map[t_name] = t_full
599
+
600
+ column_references = {}
601
+ target_name = tdata.get("label", tid)
602
+ unqualified_columns = set()
603
+
604
+ for col in qualified_ast.find_all(exp.Column):
605
+ col_name = col.name
606
+ col_table = col.table
607
+ if col_table and col_table in alias_map:
608
+ source = alias_map[col_table]
609
+ if source not in column_references:
610
+ column_references[source] = set()
611
+ column_references[source].add(col_name)
612
+ elif not col_table:
613
+ unqualified_columns.add(col_name)
614
+
615
+ # Fallback for any remaining unqualified columns
616
+ if unqualified_columns:
617
+ source_tables = [
618
+ v for k, v in alias_map.items()
619
+ if v != target_name
620
+ and v.split('.')[-1] != target_name
621
+ and v not in defined_ctes
622
+ and v.split('.')[-1] not in defined_ctes
623
+ ]
624
+ source_tables = list(set(source_tables))
625
+ if source_tables:
626
+ for src in source_tables:
627
+ if src not in column_references:
628
+ column_references[src] = set()
629
+ column_references[src].update(unqualified_columns)
630
+
631
+ column_references = {k: sorted(list(v)) for k, v in column_references.items()}
632
+ tdata["column_references"] = column_references
633
+
634
+ except Exception:
635
+ pass # Keep original column_references
636
+
637
+ # ===== Column-Level Lineage =====
638
+ # For each table, trace how each output column derives from source columns
639
+ for tid, tdata in tables.items():
640
+ if tdata.get("error") or not tdata.get("content"):
641
+ continue
642
+ schema_cols = tdata.get("schema", [])
643
+ if not schema_cols:
644
+ continue
645
+
646
+ column_lineage = {} # col_name -> [{source_table, source_column, transform}]
647
+ sql_content = tdata["content"]
648
+
649
+ for col_info in schema_cols:
650
+ col_name = col_info["name"]
651
+ if col_name == "*":
652
+ continue
653
+ try:
654
+ from sqlglot.lineage import lineage as sqlglot_lineage
655
+ node = sqlglot_lineage(
656
+ col_name, sql_content,
657
+ schema=global_schema,
658
+ dialect=dialect
659
+ )
660
+ sources = []
661
+ for child in node.downstream:
662
+ source_col = child.name
663
+ # Extract the source table from the expression
664
+ source_table = ""
665
+ if child.source and isinstance(child.source, exp.Table):
666
+ source_table = child.source.name
667
+ if child.source.db:
668
+ source_table = f"{child.source.db}.{child.source.name}"
669
+
670
+ transform = ""
671
+ if child.expression and str(child.expression) != source_col:
672
+ transform = str(child.expression)
673
+
674
+ sources.append({
675
+ "source_table": source_table,
676
+ "source_column": source_col,
677
+ "transform": transform
678
+ })
679
+ if sources:
680
+ column_lineage[col_name] = sources
681
+ except Exception:
682
+ pass # Lineage not available for this column
683
+
684
+ if column_lineage:
685
+ tdata["column_lineage"] = column_lineage
523
686
 
524
687
  return tables
525
688