etlpipe 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
etlpipe/__init__.py ADDED
@@ -0,0 +1,53 @@
1
+ """etlpipe — The fastest path from proprietary visual ETL to open-source Python.
2
+
3
+ Accelerates migration from legacy visual ETL tools to
4
+ open-source Python by providing 1:1 tool palette mappings (Preparation, Join,
5
+ Transform, Parse, InOut, Developer). Uses **pandas** (local) or **PySpark** (cluster)
6
+ as the underlying data engine. Includes automated ``.yxmd`` workflow conversion tools.
7
+
8
+ Quick start::
9
+
10
+ from etlpipe import InOut, Preparation, Join, Transform, Parse, Developer
11
+
12
+ df = InOut.input_data("sales.csv")
13
+ high, low = Preparation.filter(df, "Revenue > 1000")
14
+ summary = Transform.summarize(high, group_by="Region",
15
+ aggregations={"Revenue": "sum"})
16
+ InOut.output_data(summary, "summary.parquet")
17
+ """
18
+
19
+ from etlpipe._config import backend, get_backend, set_backend
20
+ from etlpipe._contracts import SchemaViolationError, expect_schema, infer_schema
21
+ from etlpipe._pii import scan_pii
22
+ from etlpipe._version import __version__
23
+ from etlpipe.convert import YxmdConverter
24
+ from etlpipe.developer import Developer
25
+ from etlpipe.in_out import InOut
26
+ from etlpipe.join import Join
27
+ from etlpipe.parse import Parse
28
+ from etlpipe.pipeline import Pipeline
29
+ from etlpipe.preparation import Preparation
30
+ from etlpipe.transform import Transform
31
+
32
+ __all__ = [
33
+ "__version__",
34
+ # Converter
35
+ "YxmdConverter",
36
+ # Tool palettes
37
+ "Developer",
38
+ "InOut",
39
+ "Join",
40
+ "Parse",
41
+ "Pipeline",
42
+ "Preparation",
43
+ "Transform",
44
+ "set_backend",
45
+ "get_backend",
46
+ "backend",
47
+ # Data contracts
48
+ "expect_schema",
49
+ "infer_schema",
50
+ "SchemaViolationError",
51
+ # PII scanning
52
+ "scan_pii",
53
+ ]
etlpipe/_config.py ADDED
@@ -0,0 +1,70 @@
1
+ import threading
2
+ from contextlib import contextmanager
3
+ from typing import Any
4
+
5
+ from etlpipe.engines.pandas_engine import PandasEngine
6
+
7
+ try:
8
+ from etlpipe.engines.spark_engine import SparkEngine
9
+ except ImportError:
10
+ SparkEngine = None
11
+
12
+ # Thread-local storage for configuration so multiple pipelines can run safely
13
+ _local_state = threading.local()
14
+
15
+
16
+ def _init_state():
17
+ if not hasattr(_local_state, "backend"):
18
+ _local_state.backend = "pandas"
19
+ if not hasattr(_local_state, "engine"):
20
+ _local_state.engine = PandasEngine()
21
+ if not hasattr(_local_state, "spark_config"):
22
+ _local_state.spark_config = {}
23
+
24
+
25
+ def set_backend(backend_name: str, **kwargs: Any) -> None:
26
+ _init_state()
27
+ backend_name = backend_name.lower().strip()
28
+ if backend_name == "pandas":
29
+ _local_state.backend = "pandas"
30
+ _local_state.engine = PandasEngine()
31
+ elif backend_name == "spark":
32
+ if SparkEngine is None:
33
+ raise ImportError("PySpark is required for the Spark backend. Install it with: pip install etlpipe[spark]")
34
+ _local_state.backend = "spark"
35
+ _local_state.engine = SparkEngine(**kwargs)
36
+ _local_state.spark_config = kwargs
37
+ else:
38
+ raise ValueError(f"Unknown backend: '{backend_name}'. Supported backends are: 'pandas', 'spark'.")
39
+
40
+
41
+ def get_backend() -> str:
42
+ _init_state()
43
+ return _local_state.backend
44
+
45
+
46
+ def get_engine() -> Any:
47
+ _init_state()
48
+ return _local_state.engine
49
+
50
+
51
+ def reset_backend() -> None:
52
+ _local_state.backend = "pandas"
53
+ _local_state.engine = PandasEngine()
54
+ _local_state.spark_config = {}
55
+
56
+
57
+ @contextmanager
58
+ def backend(backend_name: str, **kwargs: Any):
59
+ _init_state()
60
+ prev_backend = _local_state.backend
61
+ prev_engine = _local_state.engine
62
+ prev_spark_config = getattr(_local_state, "spark_config", {})
63
+
64
+ try:
65
+ set_backend(backend_name, **kwargs)
66
+ yield
67
+ finally:
68
+ _local_state.backend = prev_backend
69
+ _local_state.engine = prev_engine
70
+ _local_state.spark_config = prev_spark_config
etlpipe/_contracts.py ADDED
@@ -0,0 +1,29 @@
1
+ """Data contract validation — backward-compatibility shim.
2
+
3
+ The schema contract implementation has moved to the ``etlpipe-governance``
4
+ sub-package. This module re-exports everything from there so that all
5
+ existing code continues to work unchanged:
6
+
7
+ from etlpipe._contracts import expect_schema, SchemaViolationError # still works
8
+ from etlpipe import expect_schema, infer_schema # still works
9
+ from etlpipe_governance import expect_schema, profile, ContractSuite # new canonical home
10
+
11
+ .. deprecated::
12
+ Import from ``etlpipe_governance`` directly for access to the full
13
+ feature set, including :func:`etlpipe_governance.contracts.profile`
14
+ and :class:`etlpipe_governance.contracts.ContractSuite`.
15
+ The shim re-exports will be removed in etlpipe 3.0.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ # Re-export everything the original module exposed, sourced from the
21
+ # governance sub-package (single source of truth).
22
+ from etlpipe_governance.contracts import (
23
+ SchemaViolationError,
24
+ _dtype_matches,
25
+ expect_schema,
26
+ infer_schema,
27
+ )
28
+
29
+ __all__ = ["SchemaViolationError", "expect_schema", "infer_schema", "_dtype_matches"]
etlpipe/_pii.py ADDED
@@ -0,0 +1,27 @@
1
+ """PII scanner — backward-compatibility shim.
2
+
3
+ The PII detection implementation has moved to the ``etlpipe-governance``
4
+ sub-package. This module re-exports everything from there so that all
5
+ existing code continues to work unchanged:
6
+
7
+ from etlpipe._pii import scan_pii, PIIWarning # still works
8
+ from etlpipe import scan_pii # still works
9
+ from etlpipe_governance import scan_pii, mask_pii # new canonical home
10
+
11
+ .. deprecated::
12
+ Import from ``etlpipe_governance`` directly for access to the full
13
+ feature set, including :func:`etlpipe_governance.pii.mask_pii`.
14
+ The shim re-exports will be removed in etlpipe 3.0.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ # Re-export everything the original module exposed, sourced from the
20
+ # governance sub-package (single source of truth).
21
+ from etlpipe_governance.pii import (
22
+ _DEFAULT_PATTERNS,
23
+ PIIWarning,
24
+ scan_pii,
25
+ )
26
+
27
+ __all__ = ["scan_pii", "PIIWarning", "_DEFAULT_PATTERNS"]
etlpipe/_validators.py ADDED
@@ -0,0 +1,85 @@
1
+ """Shared input validation helpers for etlpipe tool functions.
2
+
3
+ Every public tool function validates its inputs through these helpers
4
+ to provide clear, consistent error messages across the library.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Sequence
10
+ from typing import Any
11
+
12
+ import pandas as pd
13
+
14
+
15
+ def validate_dataframe(df: Any, param_name: str = "df") -> None:
16
+ """Ensure the given object is a pandas DataFrame.
17
+
18
+ Args:
19
+ df: The object to validate.
20
+ param_name: Name of the parameter (for error messages).
21
+
22
+ Raises:
23
+ TypeError: If *df* is not a ``pandas.DataFrame``.
24
+ """
25
+ valid_types = (pd.DataFrame,)
26
+ try:
27
+ from pyspark.sql import DataFrame as SparkDF
28
+
29
+ valid_types = (pd.DataFrame, SparkDF)
30
+ except ImportError:
31
+ pass
32
+ if not isinstance(df, valid_types):
33
+ raise TypeError(f"'{param_name}' must be a pandas or PySpark DataFrame, got {type(df).__name__}.")
34
+
35
+
36
+ def validate_columns(
37
+ df: Any,
38
+ columns: str | Sequence[str],
39
+ param_name: str = "columns",
40
+ ) -> list[str]:
41
+ """Ensure the specified columns exist in the DataFrame.
42
+
43
+ Accepts a single column name (``str``) or a sequence of names and
44
+ always returns a ``list[str]`` for uniform downstream handling.
45
+
46
+ Args:
47
+ df: The DataFrame to check against.
48
+ columns: Column name(s) to validate.
49
+ param_name: Name of the parameter (for error messages).
50
+
51
+ Returns:
52
+ A list of validated column names.
53
+
54
+ Raises:
55
+ TypeError: If *columns* is not a string or sequence of strings.
56
+ KeyError: If any column is missing from *df*.
57
+ """
58
+ if isinstance(columns, str):
59
+ columns = [columns]
60
+ elif not isinstance(columns, (list, tuple)):
61
+ raise TypeError(f"'{param_name}' must be a string or list of strings, got {type(columns).__name__}.")
62
+
63
+ missing = [c for c in columns if c not in df.columns]
64
+ if missing:
65
+ raise KeyError(f"Column(s) not found in DataFrame: {missing}. Available columns: {list(df.columns)}")
66
+ return list(columns)
67
+
68
+
69
+ def validate_not_empty(df: Any, param_name: str = "df") -> None:
70
+ """Ensure the DataFrame is not empty.
71
+
72
+ Args:
73
+ df: The DataFrame to check.
74
+ param_name: Name of the parameter (for error messages).
75
+
76
+ Raises:
77
+ ValueError: If *df* has zero rows.
78
+ """
79
+ if isinstance(df, pd.DataFrame):
80
+ if df.empty:
81
+ raise ValueError(f"'{param_name}' must not be an empty DataFrame.")
82
+ else:
83
+ # For PySpark, check if at least 1 row exists without counting all
84
+ if len(df.head(1)) == 0:
85
+ raise ValueError(f"'{param_name}' must not be an empty DataFrame.")
etlpipe/_version.py ADDED
@@ -0,0 +1,3 @@
1
+ """Single source of truth for the etlpipe package version."""
2
+
3
+ __version__ = "2.0.0"