etlpipe 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- etlpipe/__init__.py +53 -0
- etlpipe/_config.py +70 -0
- etlpipe/_contracts.py +29 -0
- etlpipe/_pii.py +27 -0
- etlpipe/_validators.py +85 -0
- etlpipe/_version.py +3 -0
- etlpipe/convert.py +1088 -0
- etlpipe/developer.py +287 -0
- etlpipe/engines/__init__.py +4 -0
- etlpipe/engines/base.py +336 -0
- etlpipe/engines/pandas_engine.py +1654 -0
- etlpipe/engines/spark_engine.py +1359 -0
- etlpipe/in_out.py +228 -0
- etlpipe/join.py +225 -0
- etlpipe/parse.py +230 -0
- etlpipe/pipeline.py +326 -0
- etlpipe/preparation.py +570 -0
- etlpipe/transform.py +255 -0
- etlpipe-2.0.0.dist-info/METADATA +526 -0
- etlpipe-2.0.0.dist-info/RECORD +32 -0
- etlpipe-2.0.0.dist-info/WHEEL +4 -0
- etlpipe-2.0.0.dist-info/entry_points.txt +3 -0
- etlpipe-2.0.0.dist-info/licenses/LICENSE +21 -0
- etlpipe_governance/README_GOVERNANCE.md +99 -0
- etlpipe_governance/__init__.py +61 -0
- etlpipe_governance/_version.py +1 -0
- etlpipe_governance/contracts.py +545 -0
- etlpipe_governance/pii.py +398 -0
- etlpipe_governance/pyproject.toml +50 -0
- etlpipe_governance/tests/__init__.py +1 -0
- etlpipe_governance/tests/test_contracts.py +415 -0
- etlpipe_governance/tests/test_pii.py +314 -0
etlpipe/__init__.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""etlpipe — The fastest path from proprietary visual ETL to open-source Python.
|
|
2
|
+
|
|
3
|
+
Accelerates migration from legacy visual ETL tools to
|
|
4
|
+
open-source Python by providing 1:1 tool palette mappings (Preparation, Join,
|
|
5
|
+
Transform, Parse, InOut, Developer). Uses **pandas** (local) or **PySpark** (cluster)
|
|
6
|
+
as the underlying data engine. Includes automated ``.yxmd`` workflow conversion tools.
|
|
7
|
+
|
|
8
|
+
Quick start::
|
|
9
|
+
|
|
10
|
+
from etlpipe import InOut, Preparation, Join, Transform, Parse, Developer
|
|
11
|
+
|
|
12
|
+
df = InOut.input_data("sales.csv")
|
|
13
|
+
high, low = Preparation.filter(df, "Revenue > 1000")
|
|
14
|
+
summary = Transform.summarize(high, group_by="Region",
|
|
15
|
+
aggregations={"Revenue": "sum"})
|
|
16
|
+
InOut.output_data(summary, "summary.parquet")
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from etlpipe._config import backend, get_backend, set_backend
|
|
20
|
+
from etlpipe._contracts import SchemaViolationError, expect_schema, infer_schema
|
|
21
|
+
from etlpipe._pii import scan_pii
|
|
22
|
+
from etlpipe._version import __version__
|
|
23
|
+
from etlpipe.convert import YxmdConverter
|
|
24
|
+
from etlpipe.developer import Developer
|
|
25
|
+
from etlpipe.in_out import InOut
|
|
26
|
+
from etlpipe.join import Join
|
|
27
|
+
from etlpipe.parse import Parse
|
|
28
|
+
from etlpipe.pipeline import Pipeline
|
|
29
|
+
from etlpipe.preparation import Preparation
|
|
30
|
+
from etlpipe.transform import Transform
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"__version__",
|
|
34
|
+
# Converter
|
|
35
|
+
"YxmdConverter",
|
|
36
|
+
# Tool palettes
|
|
37
|
+
"Developer",
|
|
38
|
+
"InOut",
|
|
39
|
+
"Join",
|
|
40
|
+
"Parse",
|
|
41
|
+
"Pipeline",
|
|
42
|
+
"Preparation",
|
|
43
|
+
"Transform",
|
|
44
|
+
"set_backend",
|
|
45
|
+
"get_backend",
|
|
46
|
+
"backend",
|
|
47
|
+
# Data contracts
|
|
48
|
+
"expect_schema",
|
|
49
|
+
"infer_schema",
|
|
50
|
+
"SchemaViolationError",
|
|
51
|
+
# PII scanning
|
|
52
|
+
"scan_pii",
|
|
53
|
+
]
|
etlpipe/_config.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import threading
|
|
2
|
+
from contextlib import contextmanager
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from etlpipe.engines.pandas_engine import PandasEngine
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
from etlpipe.engines.spark_engine import SparkEngine
|
|
9
|
+
except ImportError:
|
|
10
|
+
SparkEngine = None
|
|
11
|
+
|
|
12
|
+
# Thread-local storage for configuration so multiple pipelines can run safely
|
|
13
|
+
_local_state = threading.local()
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _init_state():
|
|
17
|
+
if not hasattr(_local_state, "backend"):
|
|
18
|
+
_local_state.backend = "pandas"
|
|
19
|
+
if not hasattr(_local_state, "engine"):
|
|
20
|
+
_local_state.engine = PandasEngine()
|
|
21
|
+
if not hasattr(_local_state, "spark_config"):
|
|
22
|
+
_local_state.spark_config = {}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def set_backend(backend_name: str, **kwargs: Any) -> None:
|
|
26
|
+
_init_state()
|
|
27
|
+
backend_name = backend_name.lower().strip()
|
|
28
|
+
if backend_name == "pandas":
|
|
29
|
+
_local_state.backend = "pandas"
|
|
30
|
+
_local_state.engine = PandasEngine()
|
|
31
|
+
elif backend_name == "spark":
|
|
32
|
+
if SparkEngine is None:
|
|
33
|
+
raise ImportError("PySpark is required for the Spark backend. Install it with: pip install etlpipe[spark]")
|
|
34
|
+
_local_state.backend = "spark"
|
|
35
|
+
_local_state.engine = SparkEngine(**kwargs)
|
|
36
|
+
_local_state.spark_config = kwargs
|
|
37
|
+
else:
|
|
38
|
+
raise ValueError(f"Unknown backend: '{backend_name}'. Supported backends are: 'pandas', 'spark'.")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def get_backend() -> str:
|
|
42
|
+
_init_state()
|
|
43
|
+
return _local_state.backend
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def get_engine() -> Any:
|
|
47
|
+
_init_state()
|
|
48
|
+
return _local_state.engine
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def reset_backend() -> None:
|
|
52
|
+
_local_state.backend = "pandas"
|
|
53
|
+
_local_state.engine = PandasEngine()
|
|
54
|
+
_local_state.spark_config = {}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@contextmanager
|
|
58
|
+
def backend(backend_name: str, **kwargs: Any):
|
|
59
|
+
_init_state()
|
|
60
|
+
prev_backend = _local_state.backend
|
|
61
|
+
prev_engine = _local_state.engine
|
|
62
|
+
prev_spark_config = getattr(_local_state, "spark_config", {})
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
set_backend(backend_name, **kwargs)
|
|
66
|
+
yield
|
|
67
|
+
finally:
|
|
68
|
+
_local_state.backend = prev_backend
|
|
69
|
+
_local_state.engine = prev_engine
|
|
70
|
+
_local_state.spark_config = prev_spark_config
|
etlpipe/_contracts.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Data contract validation — backward-compatibility shim.
|
|
2
|
+
|
|
3
|
+
The schema contract implementation has moved to the ``etlpipe-governance``
|
|
4
|
+
sub-package. This module re-exports everything from there so that all
|
|
5
|
+
existing code continues to work unchanged:
|
|
6
|
+
|
|
7
|
+
from etlpipe._contracts import expect_schema, SchemaViolationError # still works
|
|
8
|
+
from etlpipe import expect_schema, infer_schema # still works
|
|
9
|
+
from etlpipe_governance import expect_schema, profile, ContractSuite # new canonical home
|
|
10
|
+
|
|
11
|
+
.. deprecated::
|
|
12
|
+
Import from ``etlpipe_governance`` directly for access to the full
|
|
13
|
+
feature set, including :func:`etlpipe_governance.contracts.profile`
|
|
14
|
+
and :class:`etlpipe_governance.contracts.ContractSuite`.
|
|
15
|
+
The shim re-exports will be removed in etlpipe 3.0.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
# Re-export everything the original module exposed, sourced from the
|
|
21
|
+
# governance sub-package (single source of truth).
|
|
22
|
+
from etlpipe_governance.contracts import (
|
|
23
|
+
SchemaViolationError,
|
|
24
|
+
_dtype_matches,
|
|
25
|
+
expect_schema,
|
|
26
|
+
infer_schema,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
__all__ = ["SchemaViolationError", "expect_schema", "infer_schema", "_dtype_matches"]
|
etlpipe/_pii.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""PII scanner — backward-compatibility shim.
|
|
2
|
+
|
|
3
|
+
The PII detection implementation has moved to the ``etlpipe-governance``
|
|
4
|
+
sub-package. This module re-exports everything from there so that all
|
|
5
|
+
existing code continues to work unchanged:
|
|
6
|
+
|
|
7
|
+
from etlpipe._pii import scan_pii, PIIWarning # still works
|
|
8
|
+
from etlpipe import scan_pii # still works
|
|
9
|
+
from etlpipe_governance import scan_pii, mask_pii # new canonical home
|
|
10
|
+
|
|
11
|
+
.. deprecated::
|
|
12
|
+
Import from ``etlpipe_governance`` directly for access to the full
|
|
13
|
+
feature set, including :func:`etlpipe_governance.pii.mask_pii`.
|
|
14
|
+
The shim re-exports will be removed in etlpipe 3.0.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
# Re-export everything the original module exposed, sourced from the
|
|
20
|
+
# governance sub-package (single source of truth).
|
|
21
|
+
from etlpipe_governance.pii import (
|
|
22
|
+
_DEFAULT_PATTERNS,
|
|
23
|
+
PIIWarning,
|
|
24
|
+
scan_pii,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
__all__ = ["scan_pii", "PIIWarning", "_DEFAULT_PATTERNS"]
|
etlpipe/_validators.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Shared input validation helpers for etlpipe tool functions.
|
|
2
|
+
|
|
3
|
+
Every public tool function validates its inputs through these helpers
|
|
4
|
+
to provide clear, consistent error messages across the library.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def validate_dataframe(df: Any, param_name: str = "df") -> None:
|
|
16
|
+
"""Ensure the given object is a pandas DataFrame.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
df: The object to validate.
|
|
20
|
+
param_name: Name of the parameter (for error messages).
|
|
21
|
+
|
|
22
|
+
Raises:
|
|
23
|
+
TypeError: If *df* is not a ``pandas.DataFrame``.
|
|
24
|
+
"""
|
|
25
|
+
valid_types = (pd.DataFrame,)
|
|
26
|
+
try:
|
|
27
|
+
from pyspark.sql import DataFrame as SparkDF
|
|
28
|
+
|
|
29
|
+
valid_types = (pd.DataFrame, SparkDF)
|
|
30
|
+
except ImportError:
|
|
31
|
+
pass
|
|
32
|
+
if not isinstance(df, valid_types):
|
|
33
|
+
raise TypeError(f"'{param_name}' must be a pandas or PySpark DataFrame, got {type(df).__name__}.")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def validate_columns(
|
|
37
|
+
df: Any,
|
|
38
|
+
columns: str | Sequence[str],
|
|
39
|
+
param_name: str = "columns",
|
|
40
|
+
) -> list[str]:
|
|
41
|
+
"""Ensure the specified columns exist in the DataFrame.
|
|
42
|
+
|
|
43
|
+
Accepts a single column name (``str``) or a sequence of names and
|
|
44
|
+
always returns a ``list[str]`` for uniform downstream handling.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
df: The DataFrame to check against.
|
|
48
|
+
columns: Column name(s) to validate.
|
|
49
|
+
param_name: Name of the parameter (for error messages).
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
A list of validated column names.
|
|
53
|
+
|
|
54
|
+
Raises:
|
|
55
|
+
TypeError: If *columns* is not a string or sequence of strings.
|
|
56
|
+
KeyError: If any column is missing from *df*.
|
|
57
|
+
"""
|
|
58
|
+
if isinstance(columns, str):
|
|
59
|
+
columns = [columns]
|
|
60
|
+
elif not isinstance(columns, (list, tuple)):
|
|
61
|
+
raise TypeError(f"'{param_name}' must be a string or list of strings, got {type(columns).__name__}.")
|
|
62
|
+
|
|
63
|
+
missing = [c for c in columns if c not in df.columns]
|
|
64
|
+
if missing:
|
|
65
|
+
raise KeyError(f"Column(s) not found in DataFrame: {missing}. Available columns: {list(df.columns)}")
|
|
66
|
+
return list(columns)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def validate_not_empty(df: Any, param_name: str = "df") -> None:
|
|
70
|
+
"""Ensure the DataFrame is not empty.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
df: The DataFrame to check.
|
|
74
|
+
param_name: Name of the parameter (for error messages).
|
|
75
|
+
|
|
76
|
+
Raises:
|
|
77
|
+
ValueError: If *df* has zero rows.
|
|
78
|
+
"""
|
|
79
|
+
if isinstance(df, pd.DataFrame):
|
|
80
|
+
if df.empty:
|
|
81
|
+
raise ValueError(f"'{param_name}' must not be an empty DataFrame.")
|
|
82
|
+
else:
|
|
83
|
+
# For PySpark, check if at least 1 row exists without counting all
|
|
84
|
+
if len(df.head(1)) == 0:
|
|
85
|
+
raise ValueError(f"'{param_name}' must not be an empty DataFrame.")
|
etlpipe/_version.py
ADDED