sql-dag-flow 0.7.1__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {sql_dag_flow-0.7.1/src/sql_dag_flow.egg-info → sql_dag_flow-0.9.0}/PKG-INFO +7 -2
  2. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/README.md +4 -1
  3. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/pyproject.toml +48 -41
  4. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/main.py +75 -68
  5. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/parser.py +490 -184
  6. sql_dag_flow-0.7.1/src/sql_dag_flow/static/assets/index-BqmvUL2G.js → sql_dag_flow-0.9.0/src/sql_dag_flow/static/assets/index-CcBSO8Tg.js +52 -52
  7. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/static/index.html +1 -1
  8. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0/src/sql_dag_flow.egg-info}/PKG-INFO +7 -2
  9. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow.egg-info/SOURCES.txt +7 -5
  10. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow.egg-info/requires.txt +3 -0
  11. sql_dag_flow-0.9.0/tests/test_consistency.py +59 -0
  12. sql_dag_flow-0.9.0/tests/test_graph.py +151 -0
  13. sql_dag_flow-0.9.0/tests/test_identity.py +171 -0
  14. sql_dag_flow-0.9.0/tests/test_procedures.py +140 -0
  15. sql_dag_flow-0.9.0/tests/test_schema_scope.py +54 -0
  16. sql_dag_flow-0.7.1/src/sql_dag_flow/test_api_endpoints.py +0 -44
  17. sql_dag_flow-0.7.1/src/sql_dag_flow/test_parser.py +0 -29
  18. sql_dag_flow-0.7.1/src/sql_dag_flow/verify_counts.py +0 -21
  19. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/LICENSE +0 -0
  20. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/MANIFEST.in +0 -0
  21. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/setup.cfg +0 -0
  22. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/__init__.py +0 -0
  23. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/static/assets/index-HIS38g9M.css +0 -0
  24. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow/static/vite.svg +0 -0
  25. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
  26. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
  27. {sql_dag_flow-0.7.1 → sql_dag_flow-0.9.0}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sql-dag-flow
3
- Version: 0.7.1
3
+ Version: 0.9.0
4
4
  Summary: A sophisticated SQL lineage visualization tool for Medallion Architectures.
5
5
  Author-email: Flavio Sandoval <dsandovalflavio@gmail.com>
6
6
  License: MIT
@@ -21,6 +21,8 @@ Requires-Dist: uvicorn
21
21
  Requires-Dist: sqlglot
22
22
  Requires-Dist: networkx
23
23
  Requires-Dist: pydantic
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=7; extra == "dev"
24
26
  Dynamic: license-file
25
27
 
26
28
  # SQL DAG Flow
@@ -58,10 +60,13 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
58
60
 
59
61
  ### 🔍 Visualization & Analysis
60
62
  * **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
63
+ * **Trustworthy Lineage (New in v0.8.0 🎯)**: Every `.sql` file appears exactly once, even when two folders hold the same filename (`staging/customers.sql` and `marts/customers.sql` no longer overwrite each other — they're addressed by path). Dependency resolution is strict: a reference naming a dataset will never be matched to a model in a *different* dataset, and an ambiguous name draws no edge at all. Anything unresolved, ambiguous or duplicated is reported in a warnings banner instead of failing silently — a missing edge you can see beats a wrong edge you can't.
61
64
  * **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
62
65
  * **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
63
66
  * **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
64
67
  * **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
68
+ * **Stored Procedures (New in v0.9.0 ⚙️)**: Full support for `CREATE PROCEDURE` files. The body between `BEGIN ... END` is parsed statement by statement, so a procedure shows both what it **reads** (`FROM`/`JOIN`) and what it **writes** (`INSERT`, `MERGE`, `UPDATE`, `DELETE`) — writes become outgoing edges, placing the procedure between its inputs and the tables it produces. `CALL` between procedures is tracked too. Multi-statement scripts benefit as well: every statement is inspected, not just the first.
69
+ * **20x Faster Column Analysis (New in v0.9.0 ⚡)**: The `qualify_columns` and column-lineage passes used to receive the *entire* project's schema for every model, and re-parse a model once per column. They now get only the tables a model actually reads, and reuse a single parsed AST. On a 300-model project the visible-node analysis went from **457 ms to 23 ms per model** — a full-project run that never finished within two minutes now takes under 7 seconds. The same fix repaired a silent bug: computed columns stored an expression in the type slot, which made `qualify_columns` raise and discard its own work on **every** model. It now runs successfully across the board.
65
70
  * **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
66
71
  * **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
67
72
  * **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
@@ -153,7 +158,7 @@ Install easily via `pip`:
153
158
  pip install sql-dag-flow
154
159
  ```
155
160
 
156
- To update to the latest version (**v0.7.1**):
161
+ To update to the latest version (**v0.9.0**):
157
162
 
158
163
  ```bash
159
164
  pip install --upgrade sql-dag-flow
@@ -33,10 +33,13 @@ Specially optimized for the **Medallion Architecture** (Bronze, Silver, Gold) an
33
33
 
34
34
  ### 🔍 Visualization & Analysis
35
35
  * **Automatic Parsing**: Recursively scans `.sql` files to detect dependencies (`FROM`, `JOIN`, `CTE`s) using `sqlglot`.
36
+ * **Trustworthy Lineage (New in v0.8.0 🎯)**: Every `.sql` file appears exactly once, even when two folders hold the same filename (`staging/customers.sql` and `marts/customers.sql` no longer overwrite each other — they're addressed by path). Dependency resolution is strict: a reference naming a dataset will never be matched to a model in a *different* dataset, and an ambiguous name draws no edge at all. Anything unresolved, ambiguous or duplicated is reported in a warnings banner instead of failing silently — a missing edge you can see beats a wrong edge you can't.
36
37
  * **Persistent File Cache (New in v0.6.0 ⚡)**: Avoids re-parsing unchanged SQL files across application restarts. Instantly loads your DAG on consecutive days.
37
38
  * **Selective Processing (New in v0.6.0 🚀)**: Dramatically improves performance on large projects (10x-20x) by only analyzing visible nodes for complex features like Qualify Columns and Column-Level Lineage.
38
39
  * **Scoped Views (New in v0.7.0 🎯)**: A saved diagram is now a *scope*. Reopening or refreshing it only re-parses the models actually on your canvas, so parsing stays proportional to your view instead of your whole project — and newly added `.sql` files never flood a curated architecture.
39
40
  * **Scan New Models (New in v0.7.0 🔎)**: A pure filesystem diff (zero SQL parsing, instant on huge projects) that surfaces `.sql` files not yet on your canvas, so you pull them in on demand instead of re-indexing everything.
41
+ * **Stored Procedures (New in v0.9.0 ⚙️)**: Full support for `CREATE PROCEDURE` files. The body between `BEGIN ... END` is parsed statement by statement, so a procedure shows both what it **reads** (`FROM`/`JOIN`) and what it **writes** (`INSERT`, `MERGE`, `UPDATE`, `DELETE`) — writes become outgoing edges, placing the procedure between its inputs and the tables it produces. `CALL` between procedures is tracked too. Multi-statement scripts benefit as well: every statement is inspected, not just the first.
42
+ * **20x Faster Column Analysis (New in v0.9.0 ⚡)**: The `qualify_columns` and column-lineage passes used to receive the *entire* project's schema for every model, and re-parse a model once per column. They now get only the tables a model actually reads, and reuse a single parsed AST. On a 300-model project the visible-node analysis went from **457 ms to 23 ms per model** — a full-project run that never finished within two minutes now takes under 7 seconds. The same fix repaired a silent bug: computed columns stored an expression in the type slot, which made `qualify_columns` raise and discard its own work on **every** model. It now runs successfully across the board.
40
43
  * **Medallion Architecture Support**: Automatically categorizes and colors nodes based on folder structure (Bronze, Silver, Gold).
41
44
  * **Advanced Discovery Mode (Improved in v0.6.0 👻)**: Visualize "Ghost Nodes" (missing files or external tables). Includes specific filters to show **Both**, **Only External**, or **Only CTEs**.
42
45
  * **CTE Visualization**: Detects internal Common Table Expressions and displays them as distinct Pink nodes.
@@ -128,7 +131,7 @@ Install easily via `pip`:
128
131
  pip install sql-dag-flow
129
132
  ```
130
133
 
131
- To update to the latest version (**v0.7.1**):
134
+ To update to the latest version (**v0.9.0**):
132
135
 
133
136
  ```bash
134
137
  pip install --upgrade sql-dag-flow
@@ -1,42 +1,49 @@
1
- [build-system]
2
- requires = ["setuptools>=42", "wheel"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [project]
6
- name = "sql-dag-flow"
7
- version = "0.7.1"
8
- description = "A sophisticated SQL lineage visualization tool for Medallion Architectures."
9
- readme = "README.md"
10
- requires-python = ">=3.8"
11
- license = {text = "MIT"}
12
- authors = [
13
- {name = "Flavio Sandoval", email = "dsandovalflavio@gmail.com"}
14
- ]
15
- keywords = ["sql", "lineage", "dag", "visualization", "medallion-architecture", "data-engineering"]
16
- classifiers = [
17
- "Development Status :: 4 - Beta",
18
- "Intended Audience :: Developers",
19
- "License :: OSI Approved :: MIT License",
20
- "Programming Language :: Python :: 3",
21
- "Programming Language :: Python :: 3.8",
22
- "Programming Language :: Python :: 3.9",
23
- "Programming Language :: Python :: 3.10",
24
- "Programming Language :: Python :: 3.11",
25
- ]
26
- dependencies = [
27
- "fastapi",
28
- "uvicorn",
29
- "sqlglot",
30
- "networkx",
31
- "pydantic"
32
- ]
33
-
34
- [project.scripts]
35
- sql-dag-flow = "sql_dag_flow.main:start"
36
-
37
- [tool.setuptools.packages.find]
38
- where = ["src"]
39
- include = ["sql_dag_flow*"]
40
-
41
- [tool.setuptools.package-data]
1
+ [build-system]
2
+ requires = ["setuptools>=42", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "sql-dag-flow"
7
+ version = "0.9.0"
8
+ description = "A sophisticated SQL lineage visualization tool for Medallion Architectures."
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ license = {text = "MIT"}
12
+ authors = [
13
+ {name = "Flavio Sandoval", email = "dsandovalflavio@gmail.com"}
14
+ ]
15
+ keywords = ["sql", "lineage", "dag", "visualization", "medallion-architecture", "data-engineering"]
16
+ classifiers = [
17
+ "Development Status :: 4 - Beta",
18
+ "Intended Audience :: Developers",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.8",
22
+ "Programming Language :: Python :: 3.9",
23
+ "Programming Language :: Python :: 3.10",
24
+ "Programming Language :: Python :: 3.11",
25
+ ]
26
+ dependencies = [
27
+ "fastapi",
28
+ "uvicorn",
29
+ "sqlglot",
30
+ "networkx",
31
+ "pydantic"
32
+ ]
33
+
34
+ [project.scripts]
35
+ sql-dag-flow = "sql_dag_flow.main:start"
36
+
37
+ [project.optional-dependencies]
38
+ dev = ["pytest>=7"]
39
+
40
+ [tool.pytest.ini_options]
41
+ testpaths = ["tests"]
42
+ pythonpath = ["src", "tests"]
43
+
44
+ [tool.setuptools.packages.find]
45
+ where = ["src"]
46
+ include = ["sql_dag_flow*"]
47
+
48
+ [tool.setuptools.package-data]
42
49
  sql_dag_flow = ["static/**/*"]
@@ -56,33 +56,66 @@ def _cached_parse(directory, subfolders_tuple, dialect, visible_node_ids=None, t
56
56
  _parse_cache["time"] = now
57
57
  return tables
58
58
 
59
- @app.get("/graph")
60
- def get_graph(dialect: str = "bigquery", discovery: bool = False, expanded_nodes: str = "", visible_node_ids: str = "", discovery_filter: str = "all"):
61
- """Parses SQL files in the current directory and returns graph data."""
59
+ def _normalize_expanded(expanded_nodes):
60
+ """Accept the dict form, the legacy list form, or the query-string form."""
61
+ if isinstance(expanded_nodes, dict):
62
+ return expanded_nodes
63
+ if isinstance(expanded_nodes, list):
64
+ return {n: "all" for n in expanded_nodes}
65
+ result = {}
66
+ for item in (expanded_nodes or "").split(","):
67
+ item = item.strip()
68
+ if not item:
69
+ continue
70
+ parts = item.rsplit(":", 1)
71
+ if len(parts) == 2 and parts[1] in ("all", "external", "cte"):
72
+ result[parts[0]] = parts[1]
73
+ else:
74
+ result[item] = "all"
75
+ return result
76
+
77
+
78
+ def _graph_response(dialect="bigquery", discovery=False, expanded_nodes=None,
79
+ discovery_filter="all", subfolders=None,
80
+ visible_node_ids=None, target_ids=None):
81
+ """Single implementation behind every graph route.
82
+
83
+ The three routes below differ only in how the request arrives (query
84
+ string, subfolder filter, explicit scope) — the parse → build → serialize
85
+ pipeline is identical, so it lives here once.
86
+ """
62
87
  if not os.path.exists(CURRENT_DIRECTORY):
63
- return {"nodes": [], "edges": [], "cycles": [], "error": "Directory not found"}
64
-
65
- # Parse expanded_nodes as dict {node_id: mode}
66
- exp_nodes_dict = {}
67
- if expanded_nodes:
68
- for item in expanded_nodes.split(","):
69
- item = item.strip()
70
- if not item:
71
- continue
72
- parts = item.rsplit(":", 1)
73
- if len(parts) == 2 and parts[1] in ("all", "external", "cte"):
74
- exp_nodes_dict[parts[0]] = parts[1]
75
- else:
76
- exp_nodes_dict[item] = "all"
77
- visible_list = [n.strip() for n in visible_node_ids.split(",")] if visible_node_ids else None
88
+ return {"nodes": [], "edges": [], "cycles": [], "warnings": [], "error": "Directory not found"}
89
+
78
90
  try:
79
- tables = _cached_parse(CURRENT_DIRECTORY, None, dialect, visible_node_ids=visible_list)
80
- nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=exp_nodes_dict, discovery_filter=discovery_filter)
81
- return {"nodes": nodes, "edges": edges, "cycles": cycles}
91
+ tables = _cached_parse(
92
+ CURRENT_DIRECTORY,
93
+ tuple(subfolders) if subfolders else None,
94
+ dialect,
95
+ visible_node_ids=visible_node_ids,
96
+ target_ids=target_ids,
97
+ )
98
+ nodes, edges, cycles, warnings = build_graph(
99
+ tables,
100
+ discovery_mode=discovery,
101
+ expanded_nodes=_normalize_expanded(expanded_nodes),
102
+ discovery_filter=discovery_filter,
103
+ )
104
+ return {"nodes": nodes, "edges": edges, "cycles": cycles, "warnings": warnings}
82
105
  except Exception as e:
83
106
  import traceback
84
107
  traceback.print_exc()
85
- return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
108
+ return {"nodes": [], "edges": [], "cycles": [], "warnings": [], "error": f"Backend Error: {str(e)}"}
109
+
110
+
111
+ @app.get("/graph")
112
+ def get_graph(dialect: str = "bigquery", discovery: bool = False, expanded_nodes: str = "", visible_node_ids: str = "", discovery_filter: str = "all"):
113
+ """Full-project graph."""
114
+ visible_list = [n.strip() for n in visible_node_ids.split(",")] if visible_node_ids else None
115
+ return _graph_response(
116
+ dialect=dialect, discovery=discovery, expanded_nodes=expanded_nodes,
117
+ discovery_filter=discovery_filter, visible_node_ids=visible_list,
118
+ )
86
119
 
87
120
  @app.post("/config/path")
88
121
  def set_path(path_data: dict = Body(...)):
@@ -127,28 +160,15 @@ def scan_folders(path_data: dict = Body(...)):
127
160
 
128
161
  @app.post("/graph/filtered")
129
162
  def get_filtered_graph(data: dict = Body(...)):
130
- """Parses SQL files with subfolder filtering."""
131
- if not os.path.exists(CURRENT_DIRECTORY):
132
- return {"nodes": [], "edges": [], "error": "Directory not found"}
133
-
134
- subfolders = data.get("subfolders")
135
- dialect = data.get("dialect", "bigquery")
136
- discovery = data.get("discovery", False)
137
- expanded_nodes = data.get("expanded_nodes", {})
138
- # Normalize: if frontend sends a list (legacy), convert to dict
139
- if isinstance(expanded_nodes, list):
140
- expanded_nodes = {n: 'all' for n in expanded_nodes}
141
- visible_node_ids = data.get("visible_node_ids", None)
142
- discovery_filter = data.get("discovery_filter", "all")
143
-
144
- try:
145
- tables = _cached_parse(CURRENT_DIRECTORY, tuple(subfolders) if subfolders else None, dialect, visible_node_ids=visible_node_ids)
146
- nodes, edges, cycles = build_graph(tables, discovery_mode=discovery, expanded_nodes=expanded_nodes, discovery_filter=discovery_filter)
147
- return {"nodes": nodes, "edges": edges, "cycles": cycles}
148
- except Exception as e:
149
- import traceback
150
- traceback.print_exc()
151
- return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
163
+ """Graph restricted to a set of subfolders."""
164
+ return _graph_response(
165
+ dialect=data.get("dialect", "bigquery"),
166
+ discovery=data.get("discovery", False),
167
+ expanded_nodes=data.get("expanded_nodes", {}),
168
+ discovery_filter=data.get("discovery_filter", "all"),
169
+ subfolders=data.get("subfolders"),
170
+ visible_node_ids=data.get("visible_node_ids"),
171
+ )
152
172
 
153
173
  @app.post("/graph/scoped")
154
174
  def get_scoped_graph(data: dict = Body(...)):
@@ -165,29 +185,16 @@ def get_scoped_graph(data: dict = Body(...)):
165
185
 
166
186
  node_ids = data.get("node_ids") or []
167
187
  if not node_ids:
168
- return {"nodes": [], "edges": [], "cycles": []}
169
-
170
- dialect = data.get("dialect", "bigquery")
171
- discovery = data.get("discovery", False)
172
- expanded_nodes = data.get("expanded_nodes", {})
173
- if isinstance(expanded_nodes, list):
174
- expanded_nodes = {n: 'all' for n in expanded_nodes}
175
- discovery_filter = data.get("discovery_filter", "all")
176
-
177
- try:
178
- tables = _cached_parse(
179
- CURRENT_DIRECTORY, None, dialect,
180
- visible_node_ids=node_ids, target_ids=node_ids,
181
- )
182
- nodes, edges, cycles = build_graph(
183
- tables, discovery_mode=discovery,
184
- expanded_nodes=expanded_nodes, discovery_filter=discovery_filter,
185
- )
186
- return {"nodes": nodes, "edges": edges, "cycles": cycles}
187
- except Exception as e:
188
- import traceback
189
- traceback.print_exc()
190
- return {"nodes": [], "edges": [], "error": f"Backend Error: {str(e)}"}
188
+ return {"nodes": [], "edges": [], "cycles": [], "warnings": []}
189
+
190
+ return _graph_response(
191
+ dialect=data.get("dialect", "bigquery"),
192
+ discovery=data.get("discovery", False),
193
+ expanded_nodes=data.get("expanded_nodes", {}),
194
+ discovery_filter=data.get("discovery_filter", "all"),
195
+ visible_node_ids=node_ids,
196
+ target_ids=node_ids,
197
+ )
191
198
 
192
199
 
193
200
  @app.post("/scan/new")
@@ -309,7 +316,7 @@ def export_data_dictionary(data: dict = Body(...)):
309
316
  visible_node_ids=visible_node_ids,
310
317
  target_ids=visible_node_ids,
311
318
  )
312
- nodes, edges, cycles = build_graph(tables, discovery_mode=False)
319
+ nodes, edges, cycles, _warnings = build_graph(tables, discovery_mode=False)
313
320
 
314
321
  # Filter to only visible nodes if list provided
315
322
  if visible_node_ids is not None: