scope-lineage 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. scope_lineage/__init__.py +120 -0
  2. scope_lineage/cli.py +305 -0
  3. scope_lineage/contract/__init__.py +17 -0
  4. scope_lineage/contract/lineage.py +525 -0
  5. scope_lineage/contract/validation.py +67 -0
  6. scope_lineage/metadata/__init__.py +0 -0
  7. scope_lineage/metadata/related_metadata.py +152 -0
  8. scope_lineage/metadata/schema_metadata.py +568 -0
  9. scope_lineage/metadata/target_table_metadata.py +411 -0
  10. scope_lineage/schemas/__init__.py +1 -0
  11. scope_lineage/schemas/diagnostics.schema.json +35 -0
  12. scope_lineage/schemas/lineage.schema.json +1694 -0
  13. scope_lineage/scope/__init__.py +0 -0
  14. scope_lineage/scope/_shared.py +1435 -0
  15. scope_lineage/scope/column_expression_resolution.py +202 -0
  16. scope_lineage/scope/column_ref_resolver.py +580 -0
  17. scope_lineage/scope/end_to_end.py +590 -0
  18. scope_lineage/scope/lineage_fact_gaps.py +196 -0
  19. scope_lineage/scope/logic_block.py +800 -0
  20. scope_lineage/scope/parser.py +198 -0
  21. scope_lineage/scope/passthrough_resolution.py +267 -0
  22. scope_lineage/scope/scope_builder.py +1280 -0
  23. scope_lineage/scope/scope_facts.py +2456 -0
  24. scope_lineage/scope/scope_resolver.py +997 -0
  25. scope_lineage/scope/scope_role_inferrer.py +49 -0
  26. scope_lineage/scope/scope_types.py +295 -0
  27. scope_lineage/scope/scope_warnings.py +223 -0
  28. scope_lineage/scope/select_scope.py +840 -0
  29. scope_lineage/scope/sqlglot_config.py +18 -0
  30. scope_lineage/scope/target_field_binding.py +237 -0
  31. scope_lineage/scope/types.py +112 -0
  32. scope_lineage/serialize/__init__.py +0 -0
  33. scope_lineage/serialize/scope_profile.py +323 -0
  34. scope_lineage-0.1.0.dist-info/METADATA +430 -0
  35. scope_lineage-0.1.0.dist-info/RECORD +39 -0
  36. scope_lineage-0.1.0.dist-info/WHEEL +5 -0
  37. scope_lineage-0.1.0.dist-info/entry_points.txt +2 -0
  38. scope_lineage-0.1.0.dist-info/licenses/LICENSE +202 -0
  39. scope_lineage-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,120 @@
1
+ """Stable public API for SQL parsing and versioned Lineage Core artifacts."""
2
+
3
+ # ruff: noqa: F401 -- imports below are the intentionally declared public facade.
4
+
5
+ from .contract import (
6
+ to_lineage_dict,
7
+ to_lineage_json,
8
+ validate_cross_references,
9
+ validate_diagnostics_document,
10
+ validate_lineage_document,
11
+ write_lineage,
12
+ )
13
+ from .contract.lineage import to_dict, to_json
14
+ from .metadata.schema_metadata import (
15
+ DictSchemaProvider,
16
+ MetadataFileError,
17
+ SchemaMap,
18
+ SchemaProvider,
19
+ column_details_for_table,
20
+ check_metadata_file,
21
+ catalog_prefixes,
22
+ load_schema,
23
+ materialize_schema,
24
+ metadata_dict_reader,
25
+ normalize_schema_map,
26
+ normalize_table_name,
27
+ table_details_for_table,
28
+ )
29
+ from .metadata.target_table_metadata import (
30
+ TargetColumnMetadata,
31
+ TargetMetadataMap,
32
+ TargetTableMetadata,
33
+ load_target_table_metadata,
34
+ lookup_target_table_metadata,
35
+ )
36
+ from .scope._shared import extract_qualified_field_refs
37
+ from .scope.end_to_end import build_end_to_end_lineage
38
+ from .scope.parser import resolve_display_expression
39
+ from .scope.scope_builder import (
40
+ NoSupportedWriteStatementError,
41
+ parse_all_scope_lineage,
42
+ parse_scope_lineage,
43
+ )
44
+ from .scope.scope_types import (
45
+ CONSTANT_SCOPE_ID,
46
+ NON_PHYSICAL_SOURCE_SCOPES,
47
+ SYSTEM_SCOPE_ID,
48
+ ScopeColumn,
49
+ ScopeData,
50
+ ScopeFieldUsage,
51
+ ScopeGraph,
52
+ ScopeGraphEdge,
53
+ ScopeInputEdge,
54
+ ScopeLineageResult,
55
+ ScopeLogicBlock,
56
+ ScopeOutputField,
57
+ SourceRef,
58
+ )
59
+ from .scope.sqlglot_config import suppress_invalid_json_path_warnings
60
+ from .scope.types import Column, ColumnRef, JoinKey, LineageResult, Unresolved
61
+ from .serialize.scope_profile import build_scope_profile
62
+
63
+
64
+ PUBLIC_CORE_API = frozenset({
65
+ "PUBLIC_CORE_API",
66
+ "Column",
67
+ "ColumnRef",
68
+ "CONSTANT_SCOPE_ID",
69
+ "DictSchemaProvider",
70
+ "JoinKey",
71
+ "LineageResult",
72
+ "MetadataFileError",
73
+ "NoSupportedWriteStatementError",
74
+ "NON_PHYSICAL_SOURCE_SCOPES",
75
+ "SchemaMap",
76
+ "SchemaProvider",
77
+ "ScopeColumn",
78
+ "ScopeData",
79
+ "ScopeFieldUsage",
80
+ "ScopeGraph",
81
+ "ScopeGraphEdge",
82
+ "ScopeInputEdge",
83
+ "ScopeLineageResult",
84
+ "ScopeLogicBlock",
85
+ "ScopeOutputField",
86
+ "SourceRef",
87
+ "SYSTEM_SCOPE_ID",
88
+ "TargetColumnMetadata",
89
+ "TargetMetadataMap",
90
+ "TargetTableMetadata",
91
+ "Unresolved",
92
+ "build_end_to_end_lineage",
93
+ "build_scope_profile",
94
+ "catalog_prefixes",
95
+ "column_details_for_table",
96
+ "check_metadata_file",
97
+ "extract_qualified_field_refs",
98
+ "load_schema",
99
+ "load_target_table_metadata",
100
+ "lookup_target_table_metadata",
101
+ "materialize_schema",
102
+ "metadata_dict_reader",
103
+ "normalize_schema_map",
104
+ "normalize_table_name",
105
+ "parse_all_scope_lineage",
106
+ "parse_scope_lineage",
107
+ "resolve_display_expression",
108
+ "suppress_invalid_json_path_warnings",
109
+ "table_details_for_table",
110
+ "to_dict",
111
+ "to_json",
112
+ "to_lineage_dict",
113
+ "to_lineage_json",
114
+ "validate_diagnostics_document",
115
+ "validate_cross_references",
116
+ "validate_lineage_document",
117
+ "write_lineage",
118
+ })
119
+
120
+ __all__ = sorted(PUBLIC_CORE_API)
scope_lineage/cli.py ADDED
@@ -0,0 +1,305 @@
1
+ """Minimal command line interface for the public Lineage Core."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import os
8
+ import sys
9
+ from contextlib import contextmanager
10
+ from dataclasses import dataclass
11
+ from pathlib import Path
12
+
13
+ from .contract import write_lineage
14
+ from .metadata.schema_metadata import load_schema
15
+ from .metadata.target_table_metadata import load_target_table_metadata
16
+ from .scope.scope_builder import parse_all_scope_lineage
17
+
18
+
19
+ def main(argv: list[str] | None = None) -> int:
20
+ parser = argparse.ArgumentParser(prog="scope-lineage")
21
+ subcommands = parser.add_subparsers(dest="command", required=True)
22
+ parse_cmd = subcommands.add_parser(
23
+ "parse",
24
+ help="Parse SQL or exported task JSON into Core artifacts",
25
+ )
26
+ input_group = parse_cmd.add_mutually_exclusive_group(required=True)
27
+ input_group.add_argument("--sql-file", help="Path to one SQL file")
28
+ input_group.add_argument(
29
+ "--task-file",
30
+ help="Path to one task JSON (meta/sql wrapper or legacy task_name/sql object)",
31
+ )
32
+ input_group.add_argument(
33
+ "--input-dir",
34
+ help="Directory of task JSON files; files are discovered recursively",
35
+ )
36
+ parse_cmd.add_argument(
37
+ "--task-name",
38
+ help="Override the task name for --sql-file or --task-file",
39
+ )
40
+ parse_cmd.add_argument("--out", required=True, help="Output directory")
41
+ parse_cmd.add_argument("--schema", help="Optional CSV/JSON schema metadata")
42
+ parse_cmd.add_argument(
43
+ "--target-ddl-metadata",
44
+ help="Optional target-table DDL/Schema metadata JSON file or directory",
45
+ )
46
+ parse_cmd.add_argument(
47
+ "--catalog-prefixes",
48
+ help=(
49
+ "Comma-separated leading catalog names to remove from table identities. "
50
+ "Overrides SCOPE_LINEAGE_CATALOG_PREFIXES; by default catalogs are preserved."
51
+ ),
52
+ )
53
+ parse_cmd.add_argument(
54
+ "--sanitize-metadata-nul",
55
+ action="store_true",
56
+ help="Remove NUL bytes from metadata inputs and report provenance",
57
+ )
58
+ parse_cmd.add_argument(
59
+ "--allow-partial",
60
+ action="store_true",
61
+ help="Return zero even when a statement produced parse_status=failed",
62
+ )
63
+
64
+ args = parser.parse_args(argv)
65
+ if args.command == "parse":
66
+ if args.input_dir and args.task_name:
67
+ parser.error("--task-name cannot be used with --input-dir")
68
+ with _catalog_prefix_override(args.catalog_prefixes):
69
+ return _parse_inputs(args)
70
+ parser.error(f"unknown command: {args.command}")
71
+ return 2
72
+
73
+
74
+ @dataclass(frozen=True)
75
+ class _TaskInput:
76
+ source_path: Path
77
+ relative_parent: Path
78
+ task_name: str
79
+ sql: str
80
+ task_dependencies: dict
81
+
82
+
83
+ @contextmanager
84
+ def _catalog_prefix_override(value: str | None):
85
+ """Apply a CLI-only catalog policy without leaking it to later in-process calls."""
86
+ if value is None:
87
+ yield
88
+ return
89
+ key = "SCOPE_LINEAGE_CATALOG_PREFIXES"
90
+ previous = os.environ.get(key)
91
+ os.environ[key] = value
92
+ try:
93
+ yield
94
+ finally:
95
+ if previous is None:
96
+ os.environ.pop(key, None)
97
+ else:
98
+ os.environ[key] = previous
99
+
100
+
101
+ def _parse_inputs(args: argparse.Namespace) -> int:
102
+ schema = (
103
+ load_schema(args.schema, sanitize_nul=args.sanitize_metadata_nul)
104
+ if args.schema
105
+ else None
106
+ )
107
+ target_metadata = (
108
+ load_target_table_metadata(
109
+ args.target_ddl_metadata,
110
+ sanitize_nul=args.sanitize_metadata_nul,
111
+ )
112
+ if args.target_ddl_metadata
113
+ else None
114
+ )
115
+ out_root = Path(args.out)
116
+ source_paths, input_root = _source_paths(args)
117
+ result_count = 0
118
+ failed_count = 0
119
+ input_failed_count = 0
120
+ claimed_output_dirs: dict[Path, Path] = {}
121
+
122
+ for source_path in source_paths:
123
+ try:
124
+ task = _load_task_input(source_path, input_root, args.task_name)
125
+ results = parse_all_scope_lineage(
126
+ task.sql,
127
+ task_name=task.task_name,
128
+ schema=schema,
129
+ target_metadata=target_metadata,
130
+ )
131
+ for result in results:
132
+ result.task_dependencies = task.task_dependencies
133
+ task_out = (
134
+ out_root
135
+ / task.relative_parent
136
+ / result.task_id.replace("#", "_")
137
+ )
138
+ claimed_by = claimed_output_dirs.get(task_out)
139
+ if claimed_by is not None and claimed_by != source_path:
140
+ raise ValueError(
141
+ f"output directory collision: {task_out} is already used by "
142
+ f"{claimed_by}"
143
+ )
144
+ claimed_output_dirs[task_out] = source_path
145
+ write_lineage(result, task_out)
146
+ result_count += 1
147
+ if result.parse_status == "failed":
148
+ failed_count += 1
149
+ _print_parse_failure(result)
150
+ except Exception as exc:
151
+ input_failed_count += 1
152
+ print(f" FAILED {source_path}: {type(exc).__name__}: {exc}", file=sys.stderr)
153
+
154
+ print(
155
+ f"Parsed {result_count} statement(s) from {len(source_paths)} input(s) "
156
+ f"into {out_root} "
157
+ f"(ok={result_count - failed_count}, failed={failed_count}, "
158
+ f"input_failed={input_failed_count})"
159
+ )
160
+ if not failed_count and not input_failed_count:
161
+ return 0
162
+ return 0 if args.allow_partial else 1
163
+
164
+
165
+ def _source_paths(args: argparse.Namespace) -> tuple[list[Path], Path | None]:
166
+ if args.sql_file:
167
+ return [Path(args.sql_file)], None
168
+ if args.task_file:
169
+ return [Path(args.task_file)], None
170
+ input_root = Path(args.input_dir)
171
+ if not input_root.is_dir():
172
+ raise ValueError(f"task input directory does not exist: {input_root}")
173
+ paths = sorted(input_root.rglob("*.json"))
174
+ if not paths:
175
+ raise ValueError(f"task input directory contains no JSON files: {input_root}")
176
+ return paths, input_root
177
+
178
+
179
+ def _load_task_input(
180
+ source_path: Path,
181
+ input_root: Path | None,
182
+ task_name_override: str | None,
183
+ ) -> _TaskInput:
184
+ relative_parent = (
185
+ source_path.parent.relative_to(input_root)
186
+ if input_root is not None
187
+ else Path()
188
+ )
189
+ if source_path.suffix.lower() == ".sql":
190
+ return _TaskInput(
191
+ source_path=source_path,
192
+ relative_parent=relative_parent,
193
+ task_name=task_name_override or source_path.stem,
194
+ sql=source_path.read_text(encoding="utf-8"),
195
+ task_dependencies=_empty_task_dependencies("sql_file"),
196
+ )
197
+
198
+ document = json.loads(source_path.read_text(encoding="utf-8"))
199
+ if not isinstance(document, dict):
200
+ raise ValueError("task JSON top level must be an object")
201
+ meta = document.get("meta")
202
+ payload = meta if isinstance(meta, dict) else document
203
+ sql = payload.get("sql")
204
+ if not isinstance(sql, str) or not sql.strip():
205
+ raise ValueError("task JSON must contain a non-empty string at meta.sql or sql")
206
+ task_name = (
207
+ task_name_override
208
+ or _clean_value(payload.get("task_name"))
209
+ or _clean_value(payload.get("task_id"))
210
+ or source_path.stem
211
+ )
212
+ return _TaskInput(
213
+ source_path=source_path,
214
+ relative_parent=relative_parent,
215
+ task_name=task_name,
216
+ sql=sql,
217
+ task_dependencies=_task_dependencies(document, source_path),
218
+ )
219
+
220
+
221
+ def _task_dependencies(document: dict, source_path: Path) -> dict:
222
+ meta = document.get("meta")
223
+ if not isinstance(meta, dict):
224
+ return _empty_task_dependencies("task_json_legacy")
225
+ upstream = _dependency_items(
226
+ meta.get("upstream_tasks"), "upstream", source_path
227
+ )
228
+ downstream = _dependency_items(
229
+ meta.get("downstream_tasks"), "downstream", source_path
230
+ )
231
+ return {
232
+ "upstream_tasks": upstream,
233
+ "downstream_tasks": downstream,
234
+ "source_summary": {
235
+ "source_format": "task_info_meta",
236
+ "upstream_count": len(upstream),
237
+ "downstream_count": len(downstream),
238
+ "has_declared_task_dependencies": bool(upstream or downstream),
239
+ },
240
+ }
241
+
242
+
243
+ def _dependency_items(records, direction: str, source_path: Path) -> list[dict]:
244
+ items = []
245
+ for record in records if isinstance(records, list) else []:
246
+ if not isinstance(record, dict):
247
+ continue
248
+ task_name = _clean_value(record.get("task_name") or record.get("task_id"))
249
+ if not task_name:
250
+ continue
251
+ items.append(
252
+ {
253
+ "dependency_id": f"taskdep:{direction}:{len(items) + 1:03d}",
254
+ "direction": direction,
255
+ "task_id": _clean_value(record.get("task_id")),
256
+ "task_name": task_name,
257
+ "task_group": _clean_value(record.get("task_group")),
258
+ "project_name": _clean_value(record.get("project_name")),
259
+ "dependency_type": "declared",
260
+ "dependency_table": _clean_value(
261
+ record.get("dependency_table") or record.get("table")
262
+ ),
263
+ "source": f"task_info.meta.{direction}_tasks",
264
+ "source_file": source_path.as_posix(),
265
+ "raw_record": record,
266
+ }
267
+ )
268
+ return items
269
+
270
+
271
+ def _empty_task_dependencies(source_format: str) -> dict:
272
+ return {
273
+ "upstream_tasks": [],
274
+ "downstream_tasks": [],
275
+ "source_summary": {
276
+ "source_format": source_format,
277
+ "upstream_count": 0,
278
+ "downstream_count": 0,
279
+ "has_declared_task_dependencies": False,
280
+ },
281
+ }
282
+
283
+
284
+ def _clean_value(value) -> str | None:
285
+ if value is None:
286
+ return None
287
+ cleaned = str(value).strip()
288
+ return cleaned or None
289
+
290
+
291
+ def _print_parse_failure(result) -> None:
292
+ reasons = [
293
+ warning.msg
294
+ for warning in result.diagnostics.warnings
295
+ if warning.type == "LINEAGE_ERROR"
296
+ ]
297
+ print(
298
+ f" FAILED {result.task_id}: "
299
+ f"{reasons[0] if reasons else 'scope build failed'}",
300
+ file=sys.stderr,
301
+ )
302
+
303
+
304
+ if __name__ == "__main__":
305
+ raise SystemExit(main())
@@ -0,0 +1,17 @@
1
+ """Public Lineage contract conversion, validation, and writing APIs."""
2
+
3
+ from .lineage import to_lineage_dict, to_lineage_json, write_lineage
4
+ from .validation import (
5
+ validate_cross_references,
6
+ validate_diagnostics_document,
7
+ validate_lineage_document,
8
+ )
9
+
10
+ __all__ = [
11
+ "to_lineage_dict",
12
+ "to_lineage_json",
13
+ "validate_cross_references",
14
+ "validate_diagnostics_document",
15
+ "validate_lineage_document",
16
+ "write_lineage",
17
+ ]