sql-dag-flow 0.9.0__tar.gz → 0.9.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sql_dag_flow-0.9.0/src/sql_dag_flow.egg-info → sql_dag_flow-0.9.1}/PKG-INFO +20 -3
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/README.md +205 -188
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/pyproject.toml +1 -1
- sql_dag_flow-0.9.1/src/sql_dag_flow/__init__.py +15 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow/main.py +9 -1
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow/parser.py +36 -4
- sql_dag_flow-0.9.0/src/sql_dag_flow/static/assets/index-CcBSO8Tg.js → sql_dag_flow-0.9.1/src/sql_dag_flow/static/assets/index-BUchoaRt.js +50 -50
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow/static/index.html +1 -1
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1/src/sql_dag_flow.egg-info}/PKG-INFO +20 -3
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow.egg-info/SOURCES.txt +1 -1
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/tests/test_procedures.py +68 -0
- sql_dag_flow-0.9.0/src/sql_dag_flow/__init__.py +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/LICENSE +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/MANIFEST.in +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/setup.cfg +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow/static/vite.svg +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow.egg-info/requires.txt +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/tests/test_consistency.py +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/tests/test_graph.py +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/tests/test_identity.py +0 -0
- {sql_dag_flow-0.9.0 → sql_dag_flow-0.9.1}/tests/test_schema_scope.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sql-dag-flow
|
|
3
|
-
Version: 0.9.
|
|
3
|
+
Version: 0.9.1
|
|
4
4
|
Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
|
|
5
5
|
Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -66,6 +66,7 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
|
|
|
66
66
|
* **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
|
|
67
67
|
* **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
|
|
68
68
|
* **Stored Procedures (New in v0.9.0 ⚙️)**: Full support for `CREATE PROCEDURE` files. The body between `BEGIN ... END` is parsed statement by statement, so a procedure shows both what it **reads** (`FROM`/`JOIN`) and what it **writes** (`INSERT`, `MERGE`, `UPDATE`, `DELETE`) — writes become outgoing edges, placing the procedure between its inputs and the tables it produces. `CALL` between procedures is tracked too. Multi-statement scripts benefit as well: every statement is inspected, not just the first.
|
|
69
|
+
* **Real BigQuery Procedure Signatures (New in v0.9.1 🔧)**: `OUT` / `INOUT` parameter modes are rejected by the SQL parser and used to fail the *entire* file - a valid procedure showed a bogus `Expecting )` error and lost all of its lineage. Signatures are now normalised before parsing, so procedures with output parameters, backtick-quoted names and dashed project ids work as expected.
|
|
69
70
|
* **20x Faster Column Analysis (New in v0.9.0 ⚡)**: The `qualify_columns` and column-lineage passes used to receive the *entire* project's schema for every model, and re-parse a model once per column. They now get only the tables a model actually reads, and reuse a single parsed AST. On a 300-model project the visible-node analysis went from **457 ms to 23 ms per model** — a full-project run that never finished within two minutes now takes under 7 seconds. The same fix repaired a silent bug: computed columns stored an expression in the type slot, which made `qualify_columns` raise and discard its own work on **every** model. It now runs successfully across the board.
|
|
70
71
|
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
71
72
|
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
@@ -158,7 +159,7 @@ Install easily via `pip`:
|
|
|
158
159
|
pip install sql-dag-flow
|
|
159
160
|
```
|
|
160
161
|
|
|
161
|
-
To update to the latest version (**v0.9.
|
|
162
|
+
To update to the latest version (**v0.9.1**):
|
|
162
163
|
|
|
163
164
|
```bash
|
|
164
165
|
pip install --upgrade sql-dag-flow
|
|
@@ -180,7 +181,23 @@ sql-dag-flow
|
|
|
180
181
|
sql-dag-flow /path/to/my/dbt_project
|
|
181
182
|
```
|
|
182
183
|
|
|
183
|
-
### 2.
|
|
184
|
+
### 2. Check your version
|
|
185
|
+
|
|
186
|
+
```bash
|
|
187
|
+
sql-dag-flow --version # -> sql-dag-flow 0.9.1
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The version is also shown in the bottom-right corner of the app, so you can always tell which build you are looking at. It is read from the installed package metadata, so it can never drift from the release.
|
|
191
|
+
|
|
192
|
+
To force an upgrade to the newest published release:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
pip install --upgrade --force-reinstall sql-dag-flow
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
If you installed from a local checkout (`pip install -e .`), re-run that command after pulling, or the reported version stays frozen at whatever was installed.
|
|
199
|
+
|
|
200
|
+
### 3. Python API
|
|
184
201
|
|
|
185
202
|
Integrate into your workflows:
|
|
186
203
|
|
|
@@ -1,188 +1,205 @@
|
|
|
1
|
-
# SQL DAG Flow
|
|
2
|
-
|
|
3
|
-
> **"Static Data Lineage for Modern Data Engineers. No databases, just code."**
|
|
4
|
-
|
|
5
|
-
**SQL DAG Flow** is a lightweight, open-source Python library designed to transform your SQL code into visual architecture.
|
|
6
|
-
|
|
7
|
-
Unlike traditional lineage tools that require active database connections or query log access, **SQL DAG Flow** performs **static analysis (parsing)** of your local `.sql` files. This allows for instant, secure dependency visualization, bottleneck identification, and Data Lineage documentation without leaving your development environment.
|
|
8
|
-
|
|
9
|
-
Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) and modern stacks (DuckDB, BigQuery, Snowflake), it bridges the gap between the code you write and the architecture you design.
|
|
10
|
-
|
|
11
|
-
## 💡 Philosophy: Why this exists
|
|
12
|
-
|
|
13
|
-
* **Local-First & Zero-Config**: You don't need to configure servers, cloud credentials, or Docker containers. If you have SQL files, you have a diagram.
|
|
14
|
-
* **Security by Design**: By relying on static analysis, your code never leaves your machine and no access to sensitive production data is required.
|
|
15
|
-
* **Living Documentation**: The diagram is generated *from* the code. If the code changes, the documentation updates, eliminating obsolete manually-drawn diagrams.
|
|
16
|
-
|
|
17
|
-
---
|
|
18
|
-
|
|
19
|
-
## 🎯 Objectives & Use Cases
|
|
20
|
-
|
|
21
|
-
* **1. Legacy Code Audit & Refactoring**:
|
|
22
|
-
* *The Problem*: You join a new project with 200+ undocumented SQL scripts. Nobody knows what breaks what.
|
|
23
|
-
* *The Solution*: Run `sql-dag-flow` to instantly map the "spaghetti" dependencies. Identify orphan tables, circular dependencies, and the impact of changing a Silver layer table.
|
|
24
|
-
* *The Solution*: Generate interactive pipeline visualizations (ETL/ELT) to include in your Pull Requests, Wikis, or client deliverables.
|
|
25
|
-
* **3. Medallion Architecture Validation**:
|
|
26
|
-
* *The Problem*: It's hard to verify if the logical separation of layers (Bronze → Silver → Gold) is being respected.
|
|
27
|
-
* *The Solution*: The tool visually groups your scripts by folder structure, allowing you to validate that data flows correctly between quality layers without improper "jumps".
|
|
28
|
-
* **4. Accelerated Onboarding**:
|
|
29
|
-
* *The Problem*: Explaining data flow to new engineers takes hours of whiteboard drawing.
|
|
30
|
-
* *The Solution*: Deliver an interactive map where new team members can explore where data comes from, view associated SQL code, and understand business logic without reading thousands of lines of code.
|
|
31
|
-
|
|
32
|
-
## 🚀 Key Features
|
|
33
|
-
|
|
34
|
-
### 🔍 Visualization & Analysis
|
|
35
|
-
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
36
|
-
* **Trustworthy Lineage (New in v0.8.0 🎯)**: Every `.sql` file appears exactly once, even when two folders hold the same filename (`staging/customers.sql` and `marts/customers.sql` no longer overwrite each other — they're addressed by path). Dependency resolution is strict: a reference naming a dataset will never be matched to a model in a *different* dataset, and an ambiguous name draws no edge at all. Anything unresolved, ambiguous or duplicated is reported in a warnings banner instead of failing silently — a missing edge you can see beats a wrong edge you can't.
|
|
37
|
-
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
38
|
-
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
39
|
-
* **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
|
|
40
|
-
* **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
|
|
41
|
-
* **Stored Procedures (New in v0.9.0 ⚙️)**: Full support for `CREATE PROCEDURE` files. The body between `BEGIN ... END` is parsed statement by statement, so a procedure shows both what it **reads** (`FROM`/`JOIN`) and what it **writes** (`INSERT`, `MERGE`, `UPDATE`, `DELETE`) — writes become outgoing edges, placing the procedure between its inputs and the tables it produces. `CALL` between procedures is tracked too. Multi-statement scripts benefit as well: every statement is inspected, not just the first.
|
|
42
|
-
* **
|
|
43
|
-
* **
|
|
44
|
-
* **
|
|
45
|
-
* **
|
|
46
|
-
* **
|
|
47
|
-
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
* **
|
|
56
|
-
* **
|
|
57
|
-
* **
|
|
58
|
-
|
|
59
|
-
* **
|
|
60
|
-
* **
|
|
61
|
-
* **
|
|
62
|
-
* **
|
|
63
|
-
* **
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
* **
|
|
68
|
-
|
|
69
|
-
* **
|
|
70
|
-
* **
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
* **
|
|
75
|
-
* **
|
|
76
|
-
* **
|
|
77
|
-
* **
|
|
78
|
-
* **
|
|
79
|
-
* **
|
|
80
|
-
* **
|
|
81
|
-
* **
|
|
82
|
-
* **
|
|
83
|
-
* **
|
|
84
|
-
* **
|
|
85
|
-
* **
|
|
86
|
-
* **
|
|
87
|
-
* **Performance
|
|
88
|
-
* **
|
|
89
|
-
* **
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
* **
|
|
94
|
-
* **
|
|
95
|
-
* **
|
|
96
|
-
* **
|
|
97
|
-
* **
|
|
98
|
-
* **
|
|
99
|
-
* **
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
* **
|
|
105
|
-
* **
|
|
106
|
-
* **
|
|
107
|
-
* **Export
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
|
117
|
-
|
|
|
118
|
-
| **
|
|
119
|
-
| **
|
|
120
|
-
| **
|
|
121
|
-
| **
|
|
122
|
-
| **
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
1
|
+
# SQL DAG Flow
|
|
2
|
+
|
|
3
|
+
> **"Static Data Lineage for Modern Data Engineers. No databases, just code."**
|
|
4
|
+
|
|
5
|
+
**SQL DAG Flow** is a lightweight, open-source Python library designed to transform your SQL code into visual architecture.
|
|
6
|
+
|
|
7
|
+
Unlike traditional lineage tools that require active database connections or query log access, **SQL DAG Flow** performs **static analysis (parsing)** of your local `.sql` files. This allows for instant, secure dependency visualization, bottleneck identification, and Data Lineage documentation without leaving your development environment.
|
|
8
|
+
|
|
9
|
+
Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) and modern stacks (DuckDB, BigQuery, Snowflake), it bridges the gap between the code you write and the architecture you design.
|
|
10
|
+
|
|
11
|
+
## 💡 Philosophy: Why this exists
|
|
12
|
+
|
|
13
|
+
* **Local-First & Zero-Config**: You don't need to configure servers, cloud credentials, or Docker containers. If you have SQL files, you have a diagram.
|
|
14
|
+
* **Security by Design**: By relying on static analysis, your code never leaves your machine and no access to sensitive production data is required.
|
|
15
|
+
* **Living Documentation**: The diagram is generated *from* the code. If the code changes, the documentation updates, eliminating obsolete manually-drawn diagrams.
|
|
16
|
+
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
## 🎯 Objectives & Use Cases
|
|
20
|
+
|
|
21
|
+
* **1. Legacy Code Audit & Refactoring**:
|
|
22
|
+
* *The Problem*: You join a new project with 200+ undocumented SQL scripts. Nobody knows what breaks what.
|
|
23
|
+
* *The Solution*: Run `sql-dag-flow` to instantly map the "spaghetti" dependencies. Identify orphan tables, circular dependencies, and the impact of changing a Silver layer table.
|
|
24
|
+
* *The Solution*: Generate interactive pipeline visualizations (ETL/ELT) to include in your Pull Requests, Wikis, or client deliverables.
|
|
25
|
+
* **3. Medallion Architecture Validation**:
|
|
26
|
+
* *The Problem*: It's hard to verify if the logical separation of layers (Bronze → Silver → Gold) is being respected.
|
|
27
|
+
* *The Solution*: The tool visually groups your scripts by folder structure, allowing you to validate that data flows correctly between quality layers without improper "jumps".
|
|
28
|
+
* **4. Accelerated Onboarding**:
|
|
29
|
+
* *The Problem*: Explaining data flow to new engineers takes hours of whiteboard drawing.
|
|
30
|
+
* *The Solution*: Deliver an interactive map where new team members can explore where data comes from, view associated SQL code, and understand business logic without reading thousands of lines of code.
|
|
31
|
+
|
|
32
|
+
## 🚀 Key Features
|
|
33
|
+
|
|
34
|
+
### 🔍 Visualization & Analysis
|
|
35
|
+
* **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
|
|
36
|
+
* **Trustworthy Lineage (New in v0.8.0 🎯)**: Every `.sql` file appears exactly once, even when two folders hold the same filename (`staging/customers.sql` and `marts/customers.sql` no longer overwrite each other — they're addressed by path). Dependency resolution is strict: a reference naming a dataset will never be matched to a model in a *different* dataset, and an ambiguous name draws no edge at all. Anything unresolved, ambiguous or duplicated is reported in a warnings banner instead of failing silently — a missing edge you can see beats a wrong edge you can't.
|
|
37
|
+
* **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
|
|
38
|
+
* **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
|
|
39
|
+
* **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
|
|
40
|
+
* **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
|
|
41
|
+
* **Stored Procedures (New in v0.9.0 ⚙️)**: Full support for `CREATE PROCEDURE` files. The body between `BEGIN ... END` is parsed statement by statement, so a procedure shows both what it **reads** (`FROM`/`JOIN`) and what it **writes** (`INSERT`, `MERGE`, `UPDATE`, `DELETE`) — writes become outgoing edges, placing the procedure between its inputs and the tables it produces. `CALL` between procedures is tracked too. Multi-statement scripts benefit as well: every statement is inspected, not just the first.
|
|
42
|
+
* **Real BigQuery Procedure Signatures (New in v0.9.1 🔧)**: `OUT` / `INOUT` parameter modes are rejected by the SQL parser and used to fail the *entire* file - a valid procedure showed a bogus `Expecting )` error and lost all of its lineage. Signatures are now normalised before parsing, so procedures with output parameters, backtick-quoted names and dashed project ids work as expected.
|
|
43
|
+
* **20x Faster Column Analysis (New in v0.9.0 ⚡)**: The `qualify_columns` and column-lineage passes used to receive the *entire* project's schema for every model, and re-parse a model once per column. They now get only the tables a model actually reads, and reuse a single parsed AST. On a 300-model project the visible-node analysis went from **457 ms to 23 ms per model** — a full-project run that never finished within two minutes now takes under 7 seconds. The same fix repaired a silent bug: computed columns stored an expression in the type slot, which made `qualify_columns` raise and discard its own work on **every** model. It now runs successfully across the board.
|
|
44
|
+
* **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
|
|
45
|
+
* **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
|
|
46
|
+
* **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
|
|
47
|
+
* **Smart Layout (New 🧠)**:
|
|
48
|
+
* Powered by **ELK (Eclipse Layout Kernel)**.
|
|
49
|
+
* Minimizes edge crossings and optimizes flow direction.
|
|
50
|
+
* Intelligent "Port" handling for cleaner connections.
|
|
51
|
+
* **Startup Configuration Selector**: Instantly resume previous sessions by selecting any `.json` configuration file found in your project directory upon launching the app.
|
|
52
|
+
|
|
53
|
+
### 🎮 Interactive Graph
|
|
54
|
+
* **Smart Context Menu**:
|
|
55
|
+
* **Focus Tree**: Isolate a node and its lineage (ancestors + descendants) to declutter the view.
|
|
56
|
+
* **Select Tree**: One-click selection of an entire dependency chain for easy movement.
|
|
57
|
+
* **Hide/Show**: Toggle visibility of individual nodes or full branches.
|
|
58
|
+
* **Advanced Navigation**:
|
|
59
|
+
* **Sidebar**: Grouped list of nodes with toggle between **By Layer** and **By Project/Dataset** views.
|
|
60
|
+
* **Command Palette (Cmd+P)**: Instantly search and navigate to nodes across large projects.
|
|
61
|
+
* **Keyboard Arrow Navigation**: Rapidly explore lineage by moving ← (upstream) and → (downstream) between connected nodes.
|
|
62
|
+
* **Breadcrumb Trail**: Maintain context while drilling down with a visual history of visited nodes.
|
|
63
|
+
* **SQL Content Search**: Search inside SQL file content across all nodes — find WHERE clauses, JOINs, or any keyword.
|
|
64
|
+
* **Details Panel**: View formatted SQL code, schema preview (DDL, CTAS, Views), node configuration, and add **custom descriptions** to document models.
|
|
65
|
+
|
|
66
|
+
### 📝 Notes & Annotations
|
|
67
|
+
* **Center Placement**: New notes spawn exactly in the center of your view.
|
|
68
|
+
* **Rich Styling**:
|
|
69
|
+
* **Markdown Support**: Write rich text notes.
|
|
70
|
+
* **Transparent & Borderless**:Create clean, floating text labels without boxes.
|
|
71
|
+
* **Groups**: Create visual containers to group related nodes.
|
|
72
|
+
|
|
73
|
+
### 📊 Discovery & Analysis Tools
|
|
74
|
+
* **Impact Analysis**: Visualize blast radius before making changes. Highlights downstream models, column usage, and risk levels.
|
|
75
|
+
* **Git Blast Radius (New in v0.7.0 🌿)**: Highlights the models changed in your git working tree — or versus a base branch — together with every downstream model they affect. The "what does this PR break?" view, rendered directly on the canvas.
|
|
76
|
+
* **Diff View on Refresh**: Automatically summarizes added, removed, and modified nodes/edges after code changes.
|
|
77
|
+
* **Column Usage Tracking (Improved in v0.4.9 🔧)**: Schema Preview shows which specific columns are used by downstream consumers. Uses `sqlglot.optimizer.qualify_columns` for precise resolution of unqualified column references.
|
|
78
|
+
* **Staleness Detection**: Automatically flags inactive models (`Last Modified > 90d` = Stale) to help clean up legacy pipelines.
|
|
79
|
+
* **Business Rule Extraction**: Automatically detects and displays WHERE filters, CASE logic, HAVING clauses, and aggregations from each SQL model.
|
|
80
|
+
* **Complexity Scoring**: Weighted metric per node (JOINs×3, CTEs×2, Subqueries×3, Filters×1, CASE×2, Aggregations×1, UNIONs×2) with color-coded badges (🟢 Low, 🟡 Medium, 🟠 High, 🔴 Very High). Toggleable via ⚡ button.
|
|
81
|
+
* **Node Comparison**: Select exactly 2 nodes and compare them side-by-side. Highlights differences across metadata, schema columns (shared vs unique), dependencies (Venn-style), complexity scores (with delta indicators), business rules, and SQL content (synced scroll).
|
|
82
|
+
* **Statistics Panel**: Centered popup with layer distribution bars, edge/source/sink/orphan counts, project/dataset tree, and architecture health validation (now detects **circular dependencies**).
|
|
83
|
+
* **Schema Extraction**: Backend AST-based extraction via `sqlglot` handles DDL, CTAS, `CREATE VIEW AS`, CTEs, window functions, CASE expressions, and `SELECT *`.
|
|
84
|
+
* **Column-Level Lineage (New in v0.4.9 🆕)**: Traces how each output column derives from source columns. Shows transformation chain (e.g., `order_timestamp ← orders_raw.order_date via CAST(... AS DATETIME)`).
|
|
85
|
+
* **SQL Syntax Validation (New in v0.4.9 🆕)**: Detects SQL parse errors and displays structured warnings with line/column references. Shows ⚠️ badge on nodes with syntax issues.
|
|
86
|
+
* **Large DAG Support & Safe Cycle Detection (New in v0.5.1 🚀)**: Optimized cycle detection algorithm prevents backend hanging and "Failed to fetch" browser errors when analyzing massive projects (50+ nodes). Employs quick DAG verifications and bounded iterator loops.
|
|
87
|
+
* **Performance Optimizations (v0.6.0 ⚡)**: Combines a new `.sqldagflow` persistent disk cache with visibility-based selective processing to make large DAG refreshes virtually instantaneous.
|
|
88
|
+
* **Performance Overhaul (New in v0.7.0 ⚡)**: Scoped parsing keeps the heavy `sqlglot` work proportional to your view; the Data Dictionary export no longer parses the entire project just to export a handful of models; upstream/downstream counts are computed in a single topological pass instead of one graph traversal per node; and the canvas virtualizes off-screen nodes and stops re-rendering every node on each refresh.
|
|
89
|
+
* **Batch Hide from Toolbar**: Select multiple nodes → click "Hide" in the selection toolbar to hide them all at once.
|
|
90
|
+
* **Discovery Mode Fix**: Ghost nodes from Discovery Mode are now hidden when their connected source nodes are hidden, preventing orphan ghost nodes.
|
|
91
|
+
|
|
92
|
+
### 🎨 Linear-Inspired UI (New in v0.4.6 ✨)
|
|
93
|
+
* **Design Token System**: ~80 CSS custom properties for consistent theming across all components.
|
|
94
|
+
* **Premium Dark Theme**: Deep `#0d0d0d` canvas with warm white text (`#e8e8e6`), never pure white.
|
|
95
|
+
* **Refined Light Theme**: Warm off-white `#f7f6f3` backgrounds — never harsh pure white.
|
|
96
|
+
* **Glassmorphism Toolbars**: `backdrop-filter: blur(16px)` on all floating panels.
|
|
97
|
+
* **Violet-Indigo Accent**: Premium `#7c6aef` accent color replacing generic blues/greens.
|
|
98
|
+
* **Smooth Animations**: `fadeIn` and `slideUp` micro-animations on popovers and modals.
|
|
99
|
+
* **Custom Scrollbars**: Subtle, styled scrollbars matching the theme.
|
|
100
|
+
* **Focus Rings**: Accessible focus indicators using the accent color.
|
|
101
|
+
|
|
102
|
+
### ⚙️ Customization & Export
|
|
103
|
+
* **Premium UI**:
|
|
104
|
+
* **Themes**: Toggle between Light and Dark modes.
|
|
105
|
+
* **Palettes**: Choose from **Standard**, **Vivid**, **Pastel**, or **Linear** (LCH-inspired tones) color schemes.
|
|
106
|
+
* **Styles**: Switch between "Full" (colored body) and "Minimal" (colored border) node styles.
|
|
107
|
+
* **Export Dictionary**: Generate and download a comprehensive Markdown Data Dictionary report of your entire DAG.
|
|
108
|
+
* **Export Graph**: Save high-resolution **PNG** or vector **SVG** diagrams for documentation.
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## 🎨 Visual Legend & Color Palettes
|
|
113
|
+
|
|
114
|
+
SQL DAG Flow uses distinct colors to identify node types. You can switch between these palettes in the Settings.
|
|
115
|
+
|
|
116
|
+
| Node Type | Layer / Meaning | Standard | Vivid | Pastel | Linear |
|
|
117
|
+
| :--- | :--- | :--- | :--- | :--- | :--- |
|
|
118
|
+
| **Bronze** | Raw Ingestion | 🟤 Brown (`#8B4513`) | 🟠 Warm (`#E8734A`) | 🟤 Sand (`#DCC1B0`) | 🟤 Muted (`#B08968`) |
|
|
119
|
+
| **Silver** | Cleaned / Conformed | ⚪ Gray (`#708090`) | 🔵 Ocean (`#5CA8D3`) | ⚪ Fog (`#B8C5D0`) | ⚪ Slate (`#8E99A4`) |
|
|
120
|
+
| **Gold** | Business Aggregates | 🟡 Gold (`#DAA520`) | 🟡 Amber (`#F0C75E`) | 🟡 Cream (`#F0E4B8`) | 🟡 Warm (`#D4A843`) |
|
|
121
|
+
| **External** | Missing / Ghost Node | 🟠 Rust (`#C06430`) | 🟠 Spice (`#E8943A`) | 🟠 Peach (`#E8D0A8`) | 🟠 Sand (`#CC8B5E`) |
|
|
122
|
+
| **CTE** | Internal Common Table Expression | 💖 Pink (`#E91E63`) | 💜 Rose (`#D45B8C`) | 🌸 Blush (`#DAAFC0`) | 💜 Mauve (`#C77092`) |
|
|
123
|
+
| **Other** | Uncategorized | 🔵 Teal (`#4CA1AF`) | 💠 Aqua (`#4AABB8`) | 🧊 Mist (`#A8D0D8`) | 🔵 Ocean (`#6B9DAD`) |
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
## 📦 Installation
|
|
128
|
+
|
|
129
|
+
Install easily via `pip`:
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
pip install sql-dag-flow
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
To update to the latest version (**v0.9.1**):
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
pip install --upgrade sql-dag-flow
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
---
|
|
142
|
+
|
|
143
|
+
## ▶️ Usage
|
|
144
|
+
|
|
145
|
+
### 1. Command Line Interface (CLI)
|
|
146
|
+
|
|
147
|
+
Run directly from your terminal:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
# Analyze the current directory
|
|
151
|
+
sql-dag-flow
|
|
152
|
+
|
|
153
|
+
# Analyze a specific SQL project
|
|
154
|
+
sql-dag-flow /path/to/my/dbt_project
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
### 2. Check your version
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
sql-dag-flow --version # -> sql-dag-flow 0.9.1
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
The version is also shown in the bottom-right corner of the app, so you can always tell which build you are looking at. It is read from the installed package metadata, so it can never drift from the release.
|
|
164
|
+
|
|
165
|
+
To force an upgrade to the newest published release:
|
|
166
|
+
|
|
167
|
+
```bash
|
|
168
|
+
pip install --upgrade --force-reinstall sql-dag-flow
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
If you installed from a local checkout (`pip install -e .`), re-run that command after pulling, or the reported version stays frozen at whatever was installed.
|
|
172
|
+
|
|
173
|
+
### 3. Python API
|
|
174
|
+
|
|
175
|
+
Integrate into your workflows:
|
|
176
|
+
|
|
177
|
+
```python
|
|
178
|
+
from sql_dag_flow import start
|
|
179
|
+
|
|
180
|
+
# Start the server and open the browser
|
|
181
|
+
start(directory="./my_sql_project")
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
---
|
|
185
|
+
|
|
186
|
+
## 📂 Project Structure Expectations
|
|
187
|
+
|
|
188
|
+
SQL DAG Flow looks for standard Medallion Architecture naming conventions:
|
|
189
|
+
|
|
190
|
+
* **Bronze Layer**: Folders named `bronze`, `raw`, `landing`, or `staging`.
|
|
191
|
+
* **Silver Layer**: Folders named `silver`, `intermediate`, or `conformed`.
|
|
192
|
+
* **Gold Layer**: Folders named `gold`, `mart`, `serving`, or `presentation`.
|
|
193
|
+
* **Other**: Any other folder is categorized as "Other" (Teal).
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
## 🤝 Contributing
|
|
198
|
+
|
|
199
|
+
Contributions are welcome!
|
|
200
|
+
1. Fork the repository.
|
|
201
|
+
2. Create a feature branch.
|
|
202
|
+
3. Submit a Pull Request.
|
|
203
|
+
|
|
204
|
+
---
|
|
205
|
+
*Created by [Flavio Sandoval](https://github.com/dsandovalflavio)*
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""SQL DAG Flow — static data lineage for local .sql files."""
|
|
2
|
+
|
|
3
|
+
try: # Python 3.8+
|
|
4
|
+
from importlib.metadata import PackageNotFoundError, version as _pkg_version
|
|
5
|
+
except ImportError: # pragma: no cover - Python 3.7 fallback
|
|
6
|
+
from importlib_metadata import PackageNotFoundError, version as _pkg_version
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
# Read the version from the installed package metadata rather than hardcoding
|
|
10
|
+
# it, so it can never drift from pyproject.toml.
|
|
11
|
+
__version__ = _pkg_version("sql-dag-flow")
|
|
12
|
+
except PackageNotFoundError: # running from a source checkout without install
|
|
13
|
+
__version__ = "0.0.0+dev"
|
|
14
|
+
|
|
15
|
+
__all__ = ["__version__"]
|
|
@@ -15,6 +15,7 @@ import socket
|
|
|
15
15
|
import argparse
|
|
16
16
|
import shutil
|
|
17
17
|
import subprocess
|
|
18
|
+
from . import __version__
|
|
18
19
|
from .parser import parse_sql_files, build_graph
|
|
19
20
|
|
|
20
21
|
app = FastAPI()
|
|
@@ -294,6 +295,12 @@ def git_branches():
|
|
|
294
295
|
return {"is_git": True, "branches": branches}
|
|
295
296
|
|
|
296
297
|
|
|
298
|
+
@app.get("/version")
|
|
299
|
+
def get_version():
|
|
300
|
+
"""Version of the installed package, so the UI can show which build is running."""
|
|
301
|
+
return {"version": __version__}
|
|
302
|
+
|
|
303
|
+
|
|
297
304
|
@app.get("/config/path")
|
|
298
305
|
def get_path():
|
|
299
306
|
return {"path": CURRENT_DIRECTORY}
|
|
@@ -639,10 +646,11 @@ def start():
|
|
|
639
646
|
# CLI Argument Parsing with argparse
|
|
640
647
|
parser = argparse.ArgumentParser(
|
|
641
648
|
prog='sql-dag-flow',
|
|
642
|
-
description='SQL DAG Flow - Medallion Architecture Visualizer'
|
|
649
|
+
description=f'SQL DAG Flow {__version__} - Medallion Architecture Visualizer'
|
|
643
650
|
)
|
|
644
651
|
parser.add_argument('path', nargs='?', default=None, help='Path to SQL project folder')
|
|
645
652
|
parser.add_argument('--port', '-p', type=int, default=8000, help='Port to run the server on (default: 8000)')
|
|
653
|
+
parser.add_argument('--version', '-V', action='version', version=f'sql-dag-flow {__version__}')
|
|
646
654
|
|
|
647
655
|
# Use parse_known_args to be tolerant of unexpected args
|
|
648
656
|
args, unknown = parser.parse_known_args()
|
|
@@ -95,6 +95,26 @@ _PROC_BODY_RE = re.compile(r"\bBEGIN\b(.*)\bEND\b", re.IGNORECASE | re.DOTALL)
|
|
|
95
95
|
_CALL_RE = re.compile(r"\bCALL\s+([A-Za-z0-9_.`]+)\s*\(", re.IGNORECASE)
|
|
96
96
|
|
|
97
97
|
|
|
98
|
+
# sqlglot's BigQuery dialect rejects parameter modes (OUT / INOUT / IN) in a
|
|
99
|
+
# CREATE PROCEDURE signature, which fails the entire file — so a real procedure
|
|
100
|
+
# lost all of its lineage and showed up as a syntax error. Stripping the modes
|
|
101
|
+
# from the parameter list is enough for sqlglot to accept it; the body, name and
|
|
102
|
+
# dependencies are then read normally.
|
|
103
|
+
_PROC_SIGNATURE_RE = re.compile(
|
|
104
|
+
r"(\bPROCEDURE\b\s+(?:`[^`]+`|[\w.\-]+)\s*\()([^)]*)(\))",
|
|
105
|
+
re.IGNORECASE | re.DOTALL,
|
|
106
|
+
)
|
|
107
|
+
_PARAM_MODE_RE = re.compile(r"\b(?:INOUT|OUT|IN)\s+", re.IGNORECASE)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _normalize_procedure_signature(sql_text):
|
|
111
|
+
"""Drop OUT/INOUT/IN modifiers from a procedure's parameter list."""
|
|
112
|
+
def strip_modes(match):
|
|
113
|
+
return match.group(1) + _PARAM_MODE_RE.sub("", match.group(2)) + match.group(3)
|
|
114
|
+
|
|
115
|
+
return _PROC_SIGNATURE_RE.sub(strip_modes, sql_text or "", count=1)
|
|
116
|
+
|
|
117
|
+
|
|
98
118
|
def _qualified_table_name(table_exp):
|
|
99
119
|
"""'catalog.db.name' for a sqlglot Table, as far as it is qualified."""
|
|
100
120
|
if not isinstance(table_exp, exp.Table):
|
|
@@ -492,8 +512,20 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
|
|
|
492
512
|
sql_content = f.read()
|
|
493
513
|
|
|
494
514
|
try:
|
|
495
|
-
# Parse with BigQuery dialect to support CREATE OR REPLACE TABLE/VIEW
|
|
496
|
-
|
|
515
|
+
# Parse with BigQuery dialect to support CREATE OR REPLACE TABLE/VIEW.
|
|
516
|
+
# `sql_for_parsing` may differ from the on-disk text: a stored
|
|
517
|
+
# procedure whose signature sqlglot rejects is retried with the
|
|
518
|
+
# parameter modes stripped. The original text is still what we
|
|
519
|
+
# store and display.
|
|
520
|
+
sql_for_parsing = sql_content
|
|
521
|
+
try:
|
|
522
|
+
parsed = sqlglot.parse_one(sql_for_parsing, read=dialect)
|
|
523
|
+
except sqlglot.errors.ParseError:
|
|
524
|
+
normalized = _normalize_procedure_signature(sql_content)
|
|
525
|
+
if normalized == sql_content:
|
|
526
|
+
raise
|
|
527
|
+
parsed = sqlglot.parse_one(normalized, read=dialect)
|
|
528
|
+
sql_for_parsing = normalized
|
|
497
529
|
|
|
498
530
|
# Detect Node Type (table, view or stored procedure)
|
|
499
531
|
node_type = "table" # default
|
|
@@ -633,7 +665,7 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
|
|
|
633
665
|
# lineage: what it reads, and — unlike a plain model — the
|
|
634
666
|
# tables it writes to. Writes become outgoing edges later.
|
|
635
667
|
writes = {}
|
|
636
|
-
extra_statements = _collect_statements(
|
|
668
|
+
extra_statements = _collect_statements(sql_for_parsing, dialect)
|
|
637
669
|
if len(extra_statements) > 1 or node_type == "procedure":
|
|
638
670
|
own_names = {target_table_name}
|
|
639
671
|
own_fqn = _build_fqn(project, dataset, target_table_name)
|
|
@@ -660,7 +692,7 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery", visi
|
|
|
660
692
|
dependencies[full] = "FROM"
|
|
661
693
|
|
|
662
694
|
# CALL <proc>() — sqlglot leaves these as opaque Commands
|
|
663
|
-
for called in _extract_calls(
|
|
695
|
+
for called in _extract_calls(sql_for_parsing):
|
|
664
696
|
if called and called not in own_names:
|
|
665
697
|
dependencies[called] = "CALL"
|
|
666
698
|
|