xplainable-preprocessing 0.3.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/PKG-INFO +1 -1
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/pyproject.toml +1 -1
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/expression.py +16 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_transformers/test_expression.py +23 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/.github/workflows/publish-pypi.yml +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/.gitignore +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/README.md +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/docs/dag-pipeline-proposal.md +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/docs/feature-pipeline-architectures.md +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/docs/feature-store-proposal.md +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/__init__.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/compiler.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/guard.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/pipeline.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/preview.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/registry.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/sandbox.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/schema.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/serialization.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/__init__.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/clip.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/text_clean.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/__init__.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_compiler.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_guard.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_pipeline.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_preview.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_sandbox.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_schema.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_serialization.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_transformers/__init__.py +0 -0
- {xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_transformers/test_all_transformers.py +0 -0
|
@@ -5,6 +5,17 @@ from __future__ import annotations
|
|
|
5
5
|
import pandas as pd
|
|
6
6
|
from sklearn.base import BaseEstimator, TransformerMixin
|
|
7
7
|
|
|
8
|
+
try:
|
|
9
|
+
# Semi-private pandas API (stable since 1.0, present through 2.x). It is
|
|
10
|
+
# the canonical way to reproduce how pandas mangles backtick-quoted names
|
|
11
|
+
# (`Monthly Charges` -> BACKTICK_QUOTED_STRING_Monthly_Charges) before
|
|
12
|
+
# resolving them; the mangling table is version-specific, so we do not
|
|
13
|
+
# re-implement it. If a future pandas removes it, the package still
|
|
14
|
+
# imports and only backtick-quoted column references lose support.
|
|
15
|
+
from pandas.core.computation.parsing import clean_column_name
|
|
16
|
+
except ImportError: # pragma: no cover
|
|
17
|
+
clean_column_name = None
|
|
18
|
+
|
|
8
19
|
|
|
9
20
|
class ExpressionTransformer(BaseEstimator, TransformerMixin):
|
|
10
21
|
"""Create new columns using pandas.eval() expressions.
|
|
@@ -30,6 +41,11 @@ class ExpressionTransformer(BaseEstimator, TransformerMixin):
|
|
|
30
41
|
Xt = X.copy()
|
|
31
42
|
# Support both bare column names ("age * salary") and df reference ("df['age'] * df['salary']")
|
|
32
43
|
local_dict = {col: Xt[col] for col in Xt.columns}
|
|
44
|
+
# Backtick-quoted names ("`Monthly Charges` * Tenure") are rewritten by
|
|
45
|
+
# pandas' parser into mangled identifiers regardless of engine, so the
|
|
46
|
+
# same Series must also be reachable under the mangled alias.
|
|
47
|
+
if clean_column_name is not None:
|
|
48
|
+
local_dict.update({clean_column_name(col): Xt[col] for col in Xt.columns})
|
|
33
49
|
local_dict["df"] = Xt
|
|
34
50
|
Xt[self.output_column] = pd.eval(self.expression, local_dict=local_dict, engine="python")
|
|
35
51
|
return Xt
|
|
@@ -28,3 +28,26 @@ class TestExpressionTransformer:
|
|
|
28
28
|
df = pd.DataFrame({"a": [100, 200], "b": [50, 60]})
|
|
29
29
|
result = t.fit_transform(df)
|
|
30
30
|
assert list(result["result"]) == [5.0, 12.0]
|
|
31
|
+
|
|
32
|
+
def test_backtick_quoted_column_with_spaces(self):
|
|
33
|
+
# pandas' expression parser mangles `Monthly Charges` into the
|
|
34
|
+
# identifier BACKTICK_QUOTED_STRING_Monthly_Charges before lookup, so
|
|
35
|
+
# local_dict must also be reachable under that mangled name.
|
|
36
|
+
t = ExpressionTransformer(
|
|
37
|
+
expression="`Monthly Charges` * Tenure",
|
|
38
|
+
output_column="Total Charges",
|
|
39
|
+
)
|
|
40
|
+
df = pd.DataFrame({"Monthly Charges": [10.0, 20.0, 30.0], "Tenure": [1, 2, 3]})
|
|
41
|
+
result = t.fit_transform(df)
|
|
42
|
+
assert list(result["Total Charges"]) == [10.0, 40.0, 90.0]
|
|
43
|
+
assert "Monthly Charges" in result.columns
|
|
44
|
+
assert "Tenure" in result.columns
|
|
45
|
+
|
|
46
|
+
def test_df_indexing_style_with_spaced_column_still_works(self):
|
|
47
|
+
t = ExpressionTransformer(
|
|
48
|
+
expression="df['Monthly Charges'] * 2",
|
|
49
|
+
output_column="Doubled",
|
|
50
|
+
)
|
|
51
|
+
df = pd.DataFrame({"Monthly Charges": [10.0, 20.0, 30.0]})
|
|
52
|
+
result = t.fit_transform(df)
|
|
53
|
+
assert list(result["Doubled"]) == [20.0, 40.0, 60.0]
|
{xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/.github/workflows/publish-pypi.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/docs/dag-pipeline-proposal.md
RENAMED
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/docs/feature-store-proposal.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.3.0 → xplainable_preprocessing-0.3.1}/tests/test_serialization.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|