sql-dag-flow 0.5.1__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sql_dag_flow-0.5.1/src/sql_dag_flow.egg-info → sql_dag_flow-0.5.2}/PKG-INFO +1 -1
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/pyproject.toml +1 -1
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/parser.py +86 -48
- sql_dag_flow-0.5.1/src/sql_dag_flow/static/assets/index-ZylJfG6u.js → sql_dag_flow-0.5.2/src/sql_dag_flow/static/assets/index-BPI03jTK.js +21 -21
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/static/index.html +1 -1
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2/src/sql_dag_flow.egg-info}/PKG-INFO +1 -1
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow.egg-info/SOURCES.txt +1 -1
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/LICENSE +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/MANIFEST.in +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/README.md +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/setup.cfg +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/__init__.py +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/main.py +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/static/assets/index-BM6GPh_j.css +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/static/vite.svg +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/test_api_endpoints.py +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/test_parser.py +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow/verify_counts.py +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow.egg-info/dependency_links.txt +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow.egg-info/entry_points.txt +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow.egg-info/requires.txt +0 -0
- {sql_dag_flow-0.5.1 → sql_dag_flow-0.5.2}/src/sql_dag_flow.egg-info/top_level.txt +0 -0
|
@@ -566,18 +566,28 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
566
566
|
global_schema["default"][label] = col_dict
|
|
567
567
|
|
|
568
568
|
# Re-extract column_references using qualify_columns for precision
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
569
|
+
# Protected with per-table time budget to prevent hanging on large projects
|
|
570
|
+
qualify_tables = [(tid, tdata) for tid, tdata in tables.items()
|
|
571
|
+
if not tdata.get("error") and tdata.get("content")]
|
|
572
|
+
total_qualify = len(qualify_tables)
|
|
573
|
+
|
|
574
|
+
for idx, (tid, tdata) in enumerate(qualify_tables):
|
|
575
|
+
table_start = time.time()
|
|
576
|
+
if idx % 10 == 0 and total_qualify > 10:
|
|
577
|
+
print(f" Qualifying columns: {idx}/{total_qualify} tables...")
|
|
572
578
|
try:
|
|
573
579
|
parsed = sqlglot.parse_one(tdata["content"], read=dialect)
|
|
574
580
|
|
|
575
|
-
# Try to qualify columns using the global schema
|
|
581
|
+
# Try to qualify columns using the global schema (with 2s budget)
|
|
576
582
|
try:
|
|
577
583
|
qualified_ast = sqlglot_qualify_columns(parsed.copy(), schema=global_schema, dialect=dialect)
|
|
578
584
|
except Exception:
|
|
579
585
|
qualified_ast = parsed # Fallback to unqualified
|
|
580
586
|
|
|
587
|
+
# Check time budget
|
|
588
|
+
if time.time() - table_start > 2.0:
|
|
589
|
+
continue # Skip extraction if qualification took too long
|
|
590
|
+
|
|
581
591
|
# Re-extract column references from qualified AST
|
|
582
592
|
alias_map = {}
|
|
583
593
|
defined_ctes = set()
|
|
@@ -634,55 +644,83 @@ def parse_sql_files(directory, allowed_subfolders=None, dialect="bigquery"):
|
|
|
634
644
|
except Exception:
|
|
635
645
|
pass # Keep original column_references
|
|
636
646
|
|
|
647
|
+
if total_qualify > 10:
|
|
648
|
+
print(f" Qualifying columns: {total_qualify}/{total_qualify} done.")
|
|
649
|
+
|
|
637
650
|
# ===== Column-Level Lineage =====
|
|
638
651
|
# For each table, trace how each output column derives from source columns
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
+
# Protected: skip if project is very large (>100 tables) and cap columns per table
|
|
653
|
+
MAX_LINEAGE_TABLES = 100
|
|
654
|
+
MAX_COLS_PER_TABLE = 30
|
|
655
|
+
LINEAGE_TIME_BUDGET = 1.5 # seconds per table
|
|
656
|
+
|
|
657
|
+
lineage_tables = [(tid, tdata) for tid, tdata in tables.items()
|
|
658
|
+
if not tdata.get("error") and tdata.get("content") and tdata.get("schema")]
|
|
659
|
+
total_lineage = len(lineage_tables)
|
|
660
|
+
|
|
661
|
+
if total_lineage > MAX_LINEAGE_TABLES:
|
|
662
|
+
print(f" Skipping column lineage: {total_lineage} tables exceeds limit of {MAX_LINEAGE_TABLES}")
|
|
663
|
+
else:
|
|
664
|
+
for idx, (tid, tdata) in enumerate(lineage_tables):
|
|
665
|
+
table_start = time.time()
|
|
666
|
+
if idx % 10 == 0 and total_lineage > 10:
|
|
667
|
+
print(f" Column lineage: {idx}/{total_lineage} tables...")
|
|
668
|
+
|
|
669
|
+
schema_cols = tdata.get("schema", [])
|
|
670
|
+
if not schema_cols:
|
|
652
671
|
continue
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
source_table = ""
|
|
665
|
-
if child.source and isinstance(child.source, exp.Table):
|
|
666
|
-
source_table = child.source.name
|
|
667
|
-
if child.source.db:
|
|
668
|
-
source_table = f"{child.source.db}.{child.source.name}"
|
|
669
|
-
|
|
670
|
-
transform = ""
|
|
671
|
-
if child.expression and str(child.expression) != source_col:
|
|
672
|
-
transform = str(child.expression)
|
|
672
|
+
|
|
673
|
+
column_lineage = {} # col_name -> [{source_table, source_column, transform}]
|
|
674
|
+
sql_content = tdata["content"]
|
|
675
|
+
|
|
676
|
+
cols_processed = 0
|
|
677
|
+
for col_info in schema_cols:
|
|
678
|
+
# Check time budget
|
|
679
|
+
if time.time() - table_start > LINEAGE_TIME_BUDGET:
|
|
680
|
+
break
|
|
681
|
+
if cols_processed >= MAX_COLS_PER_TABLE:
|
|
682
|
+
break
|
|
673
683
|
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
684
|
+
col_name = col_info["name"]
|
|
685
|
+
if col_name == "*":
|
|
686
|
+
continue
|
|
687
|
+
cols_processed += 1
|
|
688
|
+
try:
|
|
689
|
+
from sqlglot.lineage import lineage as sqlglot_lineage
|
|
690
|
+
node = sqlglot_lineage(
|
|
691
|
+
col_name, sql_content,
|
|
692
|
+
schema=global_schema,
|
|
693
|
+
dialect=dialect
|
|
694
|
+
)
|
|
695
|
+
sources = []
|
|
696
|
+
for child in node.downstream:
|
|
697
|
+
source_col = child.name
|
|
698
|
+
# Extract the source table from the expression
|
|
699
|
+
source_table = ""
|
|
700
|
+
if child.source and isinstance(child.source, exp.Table):
|
|
701
|
+
source_table = child.source.name
|
|
702
|
+
if child.source.db:
|
|
703
|
+
source_table = f"{child.source.db}.{child.source.name}"
|
|
704
|
+
|
|
705
|
+
transform = ""
|
|
706
|
+
if child.expression and str(child.expression) != source_col:
|
|
707
|
+
transform = str(child.expression)
|
|
708
|
+
|
|
709
|
+
sources.append({
|
|
710
|
+
"source_table": source_table,
|
|
711
|
+
"source_column": source_col,
|
|
712
|
+
"transform": transform
|
|
713
|
+
})
|
|
714
|
+
if sources:
|
|
715
|
+
column_lineage[col_name] = sources
|
|
716
|
+
except Exception:
|
|
717
|
+
pass # Lineage not available for this column
|
|
718
|
+
|
|
719
|
+
if column_lineage:
|
|
720
|
+
tdata["column_lineage"] = column_lineage
|
|
683
721
|
|
|
684
|
-
if
|
|
685
|
-
|
|
722
|
+
if total_lineage > 10:
|
|
723
|
+
print(f" Column lineage: {total_lineage}/{total_lineage} done.")
|
|
686
724
|
|
|
687
725
|
return tables
|
|
688
726
|
|