xplainable-preprocessing 0.2.3__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xplainable_preprocessing-0.3.0/.gitignore +5 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/PKG-INFO +3 -2
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/pyproject.toml +2 -1
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/__init__.py +14 -0
- xplainable_preprocessing-0.3.0/src/xplainable_preprocessing/guard.py +282 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/pipeline.py +6 -0
- xplainable_preprocessing-0.3.0/tests/test_guard.py +119 -0
- xplainable_preprocessing-0.3.0/tests/test_pipeline.py +73 -0
- xplainable_preprocessing-0.2.3/.gitignore +0 -1
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/.github/workflows/publish-pypi.yml +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/README.md +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/dag-pipeline-proposal.md +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/feature-pipeline-architectures.md +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/feature-store-proposal.md +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/compiler.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/preview.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/registry.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/sandbox.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/schema.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/serialization.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/__init__.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/clip.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/expression.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/text_clean.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/__init__.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_compiler.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_preview.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_sandbox.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_schema.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_serialization.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/__init__.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/test_all_transformers.py +0 -0
- {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/test_expression.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: xplainable-preprocessing
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Shared preprocessing pipeline package for xplainable
|
|
5
5
|
Requires-Python: >=3.9
|
|
6
6
|
Requires-Dist: cloudpickle>=3.0
|
|
@@ -8,6 +8,7 @@ Requires-Dist: numpy>=1.24
|
|
|
8
8
|
Requires-Dist: pandas>=2.0
|
|
9
9
|
Requires-Dist: pydantic>=2.0
|
|
10
10
|
Requires-Dist: scikit-learn>=1.3
|
|
11
|
+
Requires-Dist: scipy>=1.8
|
|
11
12
|
Provides-Extra: dev
|
|
12
13
|
Requires-Dist: pytest-cov; extra == 'dev'
|
|
13
14
|
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "xplainable-preprocessing"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Shared preprocessing pipeline package for xplainable"
|
|
9
9
|
requires-python = ">=3.9"
|
|
10
10
|
dependencies = [
|
|
@@ -12,6 +12,7 @@ dependencies = [
|
|
|
12
12
|
"scikit-learn>=1.3",
|
|
13
13
|
"pandas>=2.0",
|
|
14
14
|
"numpy>=1.24",
|
|
15
|
+
"scipy>=1.8",
|
|
15
16
|
"cloudpickle>=3.0",
|
|
16
17
|
]
|
|
17
18
|
|
|
@@ -4,6 +4,14 @@ from xplainable_preprocessing.schema import validate_spec
|
|
|
4
4
|
from xplainable_preprocessing.serialization import save_pipeline, load_pipeline
|
|
5
5
|
from xplainable_preprocessing.pipeline import DataFramePipeline, DataFrameColumnTransformer
|
|
6
6
|
from xplainable_preprocessing.registry import REGISTRY, register, generate_catalog
|
|
7
|
+
from xplainable_preprocessing.guard import (
|
|
8
|
+
ROW_COLLAPSING,
|
|
9
|
+
preview_spec,
|
|
10
|
+
preview_step,
|
|
11
|
+
safe_catalog,
|
|
12
|
+
spec_safety_errors,
|
|
13
|
+
step_safety_error,
|
|
14
|
+
)
|
|
7
15
|
|
|
8
16
|
__all__ = [
|
|
9
17
|
"PipelineSpec",
|
|
@@ -17,4 +25,10 @@ __all__ = [
|
|
|
17
25
|
"REGISTRY",
|
|
18
26
|
"register",
|
|
19
27
|
"generate_catalog",
|
|
28
|
+
"ROW_COLLAPSING",
|
|
29
|
+
"safe_catalog",
|
|
30
|
+
"step_safety_error",
|
|
31
|
+
"spec_safety_errors",
|
|
32
|
+
"preview_step",
|
|
33
|
+
"preview_spec",
|
|
20
34
|
]
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
"""Step safety guard and step preview.
|
|
2
|
+
|
|
3
|
+
Shared by the platform API (preprocessor create/add), the training service
|
|
4
|
+
(agent apply/preview) and the client, so every surface rejects the same
|
|
5
|
+
unsafe steps with the same message.
|
|
6
|
+
|
|
7
|
+
A step is unsafe when it:
|
|
8
|
+
- uses a row-collapsing transformer (one row per group; the column wrapper
|
|
9
|
+
concat-misaligns the result back onto the frame and corrupts other columns),
|
|
10
|
+
- changes the row count,
|
|
11
|
+
- drops the label column, or
|
|
12
|
+
- mutates the label values.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import math
|
|
18
|
+
from typing import Any, Optional, Union
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
from xplainable_preprocessing.compiler import compile_spec
|
|
24
|
+
from xplainable_preprocessing.pipeline import DataFrameColumnTransformer
|
|
25
|
+
from xplainable_preprocessing.preview import compute_delta
|
|
26
|
+
from xplainable_preprocessing.registry import generate_catalog
|
|
27
|
+
from xplainable_preprocessing.schema import PipelineSpec, StepSpec
|
|
28
|
+
|
|
29
|
+
ROW_COLLAPSING = frozenset({"GroupByAggTransformer"})
|
|
30
|
+
"""Transformer types that change the row count by design. Never offer them to an LLM."""
|
|
31
|
+
|
|
32
|
+
DEFAULT_MAX_ROWS = 10_000
|
|
33
|
+
|
|
34
|
+
StepLike = Union[StepSpec, dict]
|
|
35
|
+
SpecLike = Union[PipelineSpec, dict]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _as_step(step: StepLike) -> StepSpec:
|
|
39
|
+
return step if isinstance(step, StepSpec) else StepSpec(**step)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _as_spec(spec: SpecLike) -> PipelineSpec:
|
|
43
|
+
return spec if isinstance(spec, PipelineSpec) else PipelineSpec(**spec)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _sample(df: pd.DataFrame, max_rows: int) -> pd.DataFrame:
|
|
47
|
+
if max_rows and len(df) > max_rows:
|
|
48
|
+
return df.sample(max_rows, random_state=42)
|
|
49
|
+
return df.copy()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def safe_catalog() -> str:
|
|
53
|
+
"""The transformer catalog with row-collapsing entries removed."""
|
|
54
|
+
lines = generate_catalog().splitlines()
|
|
55
|
+
return "\n".join(
|
|
56
|
+
line for line in lines
|
|
57
|
+
if not any(line.lstrip().startswith(f"- {name}(") for name in ROW_COLLAPSING)
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def step_safety_error(
|
|
62
|
+
step: StepLike,
|
|
63
|
+
df: pd.DataFrame,
|
|
64
|
+
label: Optional[str] = None,
|
|
65
|
+
max_rows: int = DEFAULT_MAX_ROWS,
|
|
66
|
+
) -> Optional[str]:
|
|
67
|
+
"""Return why `step` is unsafe on `df`, or None when it is safe.
|
|
68
|
+
|
|
69
|
+
The step is compiled alone and fit on a sample. A step that cannot be
|
|
70
|
+
evaluated in isolation (for example it reads a column an earlier step
|
|
71
|
+
creates) is NOT rejected; use `spec_safety_errors` to validate a whole
|
|
72
|
+
spec in order.
|
|
73
|
+
"""
|
|
74
|
+
step_spec = _as_step(step)
|
|
75
|
+
if step_spec.type in ROW_COLLAPSING:
|
|
76
|
+
return f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}"
|
|
77
|
+
try:
|
|
78
|
+
sample = _sample(df, max_rows)
|
|
79
|
+
pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
|
|
80
|
+
_, transformer = pipeline.steps[0]
|
|
81
|
+
|
|
82
|
+
# The column wrapper masks row collapses by concat-aligning onto the
|
|
83
|
+
# original index; check the inner transformer's raw output.
|
|
84
|
+
inner, inner_input = transformer, sample
|
|
85
|
+
if isinstance(transformer, DataFrameColumnTransformer):
|
|
86
|
+
inner = transformer.transformer
|
|
87
|
+
inner_input = sample[list(transformer.columns)]
|
|
88
|
+
raw = inner.fit_transform(inner_input.copy())
|
|
89
|
+
if hasattr(raw, "__len__") and len(raw) != len(sample):
|
|
90
|
+
return (
|
|
91
|
+
f"step '{step_spec.id}' changes the row count "
|
|
92
|
+
f"({len(sample)} -> {len(raw)}); row-collapsing steps corrupt the training frame"
|
|
93
|
+
)
|
|
94
|
+
transformed = pipeline.fit_transform(sample)
|
|
95
|
+
except Exception:
|
|
96
|
+
return None
|
|
97
|
+
return _label_error(step_spec.id, sample, transformed, label)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _label_error(step_id: str, before: pd.DataFrame, after: pd.DataFrame, label: Optional[str]) -> Optional[str]:
|
|
101
|
+
if not label:
|
|
102
|
+
return None
|
|
103
|
+
if label not in after.columns:
|
|
104
|
+
return f"step '{step_id}' drops the label column '{label}'"
|
|
105
|
+
b = before[label].reset_index(drop=True)
|
|
106
|
+
a = after[label].reset_index(drop=True)
|
|
107
|
+
if not isinstance(a, pd.Series) or not a.equals(b):
|
|
108
|
+
return f"step '{step_id}' mutates the label column '{label}'"
|
|
109
|
+
return None
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def spec_safety_errors(
|
|
113
|
+
spec: SpecLike,
|
|
114
|
+
df: pd.DataFrame,
|
|
115
|
+
label: Optional[str] = None,
|
|
116
|
+
max_rows: int = DEFAULT_MAX_ROWS,
|
|
117
|
+
report_failures: bool = True,
|
|
118
|
+
) -> list:
|
|
119
|
+
"""Validate every step of a spec in order on a sample of `df`.
|
|
120
|
+
|
|
121
|
+
Steps are applied cumulatively so later steps see the columns earlier
|
|
122
|
+
steps created. Returns a list of error strings (empty when the spec is
|
|
123
|
+
safe). A step that fails to fit is reported as well unless
|
|
124
|
+
`report_failures=False` (for callers that already surface fit failures
|
|
125
|
+
through their own compile path); such a step is skipped so later steps
|
|
126
|
+
are still checked.
|
|
127
|
+
"""
|
|
128
|
+
pipeline_spec = _as_spec(spec)
|
|
129
|
+
errors = []
|
|
130
|
+
frame = _sample(df, max_rows)
|
|
131
|
+
for step_spec in pipeline_spec.steps:
|
|
132
|
+
if step_spec.type in ROW_COLLAPSING:
|
|
133
|
+
errors.append(f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}")
|
|
134
|
+
continue
|
|
135
|
+
try:
|
|
136
|
+
single = compile_spec(PipelineSpec(steps=[step_spec]))
|
|
137
|
+
after = single.fit_transform(frame.copy())
|
|
138
|
+
except Exception as e: # noqa: BLE001
|
|
139
|
+
if report_failures:
|
|
140
|
+
errors.append(f"step '{step_spec.id}' failed to apply: {e}")
|
|
141
|
+
continue
|
|
142
|
+
if len(after) != len(frame):
|
|
143
|
+
errors.append(
|
|
144
|
+
f"step '{step_spec.id}' changes the row count ({len(frame)} -> {len(after)})"
|
|
145
|
+
)
|
|
146
|
+
continue
|
|
147
|
+
label_err = _label_error(step_spec.id, frame, after, label)
|
|
148
|
+
if label_err:
|
|
149
|
+
errors.append(label_err)
|
|
150
|
+
continue
|
|
151
|
+
frame = after
|
|
152
|
+
return errors
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _num(v: Any) -> Optional[float]:
|
|
156
|
+
try:
|
|
157
|
+
f = float(v)
|
|
158
|
+
except (TypeError, ValueError):
|
|
159
|
+
return None
|
|
160
|
+
return None if (math.isnan(f) or math.isinf(f)) else f
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _column_stats(s: pd.Series) -> dict:
|
|
164
|
+
numeric = pd.api.types.is_numeric_dtype(s) and not pd.api.types.is_bool_dtype(s)
|
|
165
|
+
return {
|
|
166
|
+
"mean": _num(s.mean()) if numeric else None,
|
|
167
|
+
"nulls": int(s.isnull().sum()),
|
|
168
|
+
"min": _num(s.min()) if numeric else None,
|
|
169
|
+
"max": _num(s.max()) if numeric else None,
|
|
170
|
+
"unique": int(s.nunique()),
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def preview_step(
|
|
175
|
+
step: StepLike,
|
|
176
|
+
df: pd.DataFrame,
|
|
177
|
+
max_rows: int = DEFAULT_MAX_ROWS,
|
|
178
|
+
) -> dict:
|
|
179
|
+
"""Dry-run one step on a sample of `df`.
|
|
180
|
+
|
|
181
|
+
Returns
|
|
182
|
+
-------
|
|
183
|
+
dict with keys:
|
|
184
|
+
- delta: {dropped, added, updated} column lists
|
|
185
|
+
- stats: {column: {"before": stats, "after": stats}} for changed columns
|
|
186
|
+
- warnings: human-readable warnings (new nulls, infinities, row-count change, ...)
|
|
187
|
+
- rows_before / rows_after
|
|
188
|
+
- error: str when the step failed to apply, else None
|
|
189
|
+
"""
|
|
190
|
+
step_spec = _as_step(step)
|
|
191
|
+
before = _sample(df, max_rows)
|
|
192
|
+
rows_before = len(before)
|
|
193
|
+
try:
|
|
194
|
+
pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
|
|
195
|
+
after = pipeline.fit_transform(before.copy())
|
|
196
|
+
except Exception as e: # noqa: BLE001
|
|
197
|
+
return {
|
|
198
|
+
"delta": {"dropped": [], "added": [], "updated": []},
|
|
199
|
+
"stats": {},
|
|
200
|
+
"warnings": [f"fit_transform failed: {e}"],
|
|
201
|
+
"rows_before": rows_before,
|
|
202
|
+
"rows_after": rows_before,
|
|
203
|
+
"error": str(e),
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
raw = compute_delta(before, after)
|
|
207
|
+
delta = {k: list(raw.get(k, [])) for k in ("dropped", "added", "updated")}
|
|
208
|
+
|
|
209
|
+
stats = {}
|
|
210
|
+
for col in delta["added"] + delta["updated"]:
|
|
211
|
+
entry = {}
|
|
212
|
+
if col in before.columns:
|
|
213
|
+
entry["before"] = _column_stats(before[col])
|
|
214
|
+
if col in after.columns:
|
|
215
|
+
entry["after"] = _column_stats(after[col])
|
|
216
|
+
if entry:
|
|
217
|
+
stats[col] = entry
|
|
218
|
+
|
|
219
|
+
warnings = []
|
|
220
|
+
for col in delta["updated"]:
|
|
221
|
+
if col in before.columns and col in after.columns:
|
|
222
|
+
new_nulls = int(after[col].isnull().sum()) - int(before[col].isnull().sum())
|
|
223
|
+
if new_nulls > 0:
|
|
224
|
+
warnings.append(f"Column '{col}': {new_nulls} new null values introduced")
|
|
225
|
+
for col in after.select_dtypes(include=["number"]).columns:
|
|
226
|
+
try:
|
|
227
|
+
inf_count = int(np.isinf(after[col].astype(float)).sum())
|
|
228
|
+
except (TypeError, ValueError):
|
|
229
|
+
inf_count = 0
|
|
230
|
+
if inf_count:
|
|
231
|
+
warnings.append(f"Column '{col}': {inf_count} infinite values detected")
|
|
232
|
+
if len(delta["added"]) > 100:
|
|
233
|
+
warnings.append(f"Massive column expansion: {len(delta['added'])} new columns added")
|
|
234
|
+
rows_after = len(after)
|
|
235
|
+
if rows_after != rows_before:
|
|
236
|
+
warnings.append(f"Row count changed from {rows_before} to {rows_after}")
|
|
237
|
+
|
|
238
|
+
return {
|
|
239
|
+
"delta": delta,
|
|
240
|
+
"stats": stats,
|
|
241
|
+
"warnings": warnings,
|
|
242
|
+
"rows_before": rows_before,
|
|
243
|
+
"rows_after": rows_after,
|
|
244
|
+
"error": None,
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def preview_spec(
|
|
249
|
+
spec: SpecLike,
|
|
250
|
+
df: pd.DataFrame,
|
|
251
|
+
label: Optional[str] = None,
|
|
252
|
+
max_rows: int = DEFAULT_MAX_ROWS,
|
|
253
|
+
) -> dict:
|
|
254
|
+
"""Preview every step of a spec in order, plus the safety verdict.
|
|
255
|
+
|
|
256
|
+
Returns
|
|
257
|
+
-------
|
|
258
|
+
dict with keys:
|
|
259
|
+
- steps: [{"step_id", "step_type", **preview_step(...)}] applied cumulatively
|
|
260
|
+
- safety_errors: from `spec_safety_errors`
|
|
261
|
+
- rows_before / rows_after
|
|
262
|
+
- output_columns: columns after the last step that applied cleanly
|
|
263
|
+
"""
|
|
264
|
+
pipeline_spec = _as_spec(spec)
|
|
265
|
+
frame = _sample(df, max_rows)
|
|
266
|
+
rows_before = len(frame)
|
|
267
|
+
steps = []
|
|
268
|
+
for step_spec in pipeline_spec.steps:
|
|
269
|
+
result = preview_step(step_spec, frame, max_rows=0)
|
|
270
|
+
steps.append({"step_id": step_spec.id, "step_type": step_spec.type, **result})
|
|
271
|
+
if result["error"] is None:
|
|
272
|
+
try:
|
|
273
|
+
frame = compile_spec(PipelineSpec(steps=[step_spec])).fit_transform(frame.copy())
|
|
274
|
+
except Exception: # noqa: BLE001
|
|
275
|
+
pass
|
|
276
|
+
return {
|
|
277
|
+
"steps": steps,
|
|
278
|
+
"safety_errors": spec_safety_errors(pipeline_spec, df, label=label, max_rows=max_rows),
|
|
279
|
+
"rows_before": rows_before,
|
|
280
|
+
"rows_after": len(frame),
|
|
281
|
+
"output_columns": list(frame.columns),
|
|
282
|
+
}
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import pandas as pd
|
|
6
|
+
from scipy import sparse
|
|
6
7
|
from sklearn.base import BaseEstimator, TransformerMixin
|
|
7
8
|
|
|
8
9
|
|
|
@@ -20,6 +21,11 @@ class DataFrameColumnTransformer(BaseEstimator, TransformerMixin):
|
|
|
20
21
|
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
21
22
|
Xt = X.copy()
|
|
22
23
|
result = self.transformer.transform(Xt[self.columns])
|
|
24
|
+
if sparse.issparse(result):
|
|
25
|
+
# e.g. OneHotEncoder defaults to sparse_output=True; pd.DataFrame
|
|
26
|
+
# would treat the matrix as a single object column and raise
|
|
27
|
+
# "Shape of passed values is (N, 1), indices imply (N, K)".
|
|
28
|
+
result = result.toarray()
|
|
23
29
|
if isinstance(result, pd.DataFrame):
|
|
24
30
|
Xt = Xt.drop(columns=self.columns)
|
|
25
31
|
Xt = pd.concat([Xt, result], axis=1)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Tests for the step safety guard and step preview."""
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from xplainable_preprocessing import (
|
|
8
|
+
ROW_COLLAPSING,
|
|
9
|
+
preview_spec,
|
|
10
|
+
preview_step,
|
|
11
|
+
safe_catalog,
|
|
12
|
+
spec_safety_errors,
|
|
13
|
+
step_safety_error,
|
|
14
|
+
)
|
|
15
|
+
from xplainable_preprocessing.schema import PipelineSpec, StepSpec
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@pytest.fixture
|
|
19
|
+
def frame():
|
|
20
|
+
rng = np.random.default_rng(0)
|
|
21
|
+
n = 120
|
|
22
|
+
return pd.DataFrame({
|
|
23
|
+
"id": [f"C{i}" for i in range(n)],
|
|
24
|
+
"tenure": rng.integers(1, 72, n),
|
|
25
|
+
"charges": rng.uniform(20, 120, n).round(2),
|
|
26
|
+
"total": [str(v) for v in rng.uniform(100, 5000, n).round(1)],
|
|
27
|
+
"plan": rng.choice(["a", "b", "c", "d", "e"], n),
|
|
28
|
+
"churn": rng.integers(0, 2, n),
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
DROP_ID = {"id": "drop_id", "type": "DropColumnsTransformer", "columns": ["id"]}
|
|
33
|
+
CAST_TOTAL = {"id": "cast_total", "type": "TypeCastTransformer", "columns": ["total"], "params": {"dtypes": {"total": "float"}}}
|
|
34
|
+
DROP_LABEL = {"id": "drop_label", "type": "DropColumnsTransformer", "columns": ["churn"]}
|
|
35
|
+
COLLAPSE = {"id": "grp", "type": "GroupByAggTransformer", "columns": ["plan", "charges"], "params": {}}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class TestSafeCatalog:
|
|
39
|
+
def test_excludes_row_collapsing_transformers(self):
|
|
40
|
+
catalog = safe_catalog()
|
|
41
|
+
for name in ROW_COLLAPSING:
|
|
42
|
+
assert f"- {name}(" not in catalog
|
|
43
|
+
assert "- DropColumnsTransformer(" in catalog
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class TestStepSafetyError:
|
|
47
|
+
def test_safe_step_is_none(self, frame):
|
|
48
|
+
assert step_safety_error(DROP_ID, frame, label="churn") is None
|
|
49
|
+
assert step_safety_error(StepSpec(**CAST_TOTAL), frame, label="churn") is None
|
|
50
|
+
|
|
51
|
+
def test_row_collapsing_type_is_rejected_without_running(self, frame):
|
|
52
|
+
assert "row-collapsing" in step_safety_error(COLLAPSE, frame)
|
|
53
|
+
|
|
54
|
+
def test_dropping_the_label_is_rejected(self, frame):
|
|
55
|
+
assert "drops the label" in step_safety_error(DROP_LABEL, frame, label="churn")
|
|
56
|
+
|
|
57
|
+
def test_mutating_the_label_is_rejected(self, frame):
|
|
58
|
+
step = {"id": "cast_label", "type": "TypeCastTransformer", "columns": ["churn"], "params": {"dtypes": {"churn": "str"}}}
|
|
59
|
+
assert "mutates the label" in step_safety_error(step, frame, label="churn")
|
|
60
|
+
|
|
61
|
+
def test_step_that_cannot_run_in_isolation_is_not_rejected(self, frame):
|
|
62
|
+
# References a column that does not exist yet (created by an earlier step).
|
|
63
|
+
step = {"id": "later", "type": "DropColumnsTransformer", "columns": ["derived_col"]}
|
|
64
|
+
assert step_safety_error(step, frame, label="churn") is None
|
|
65
|
+
|
|
66
|
+
def test_without_label_only_structural_checks_apply(self, frame):
|
|
67
|
+
assert step_safety_error(DROP_LABEL, frame) is None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class TestSpecSafetyErrors:
|
|
71
|
+
def test_clean_spec_has_no_errors(self, frame):
|
|
72
|
+
spec = {"version": "2.0", "steps": [DROP_ID, CAST_TOTAL]}
|
|
73
|
+
assert spec_safety_errors(spec, frame, label="churn") == []
|
|
74
|
+
|
|
75
|
+
def test_reports_every_unsafe_step(self, frame):
|
|
76
|
+
spec = PipelineSpec(steps=[StepSpec(**DROP_ID), StepSpec(**DROP_LABEL), StepSpec(**COLLAPSE)])
|
|
77
|
+
errors = spec_safety_errors(spec, frame, label="churn")
|
|
78
|
+
assert len(errors) == 2
|
|
79
|
+
assert "drop_label" in errors[0] and "grp" in errors[1]
|
|
80
|
+
|
|
81
|
+
def test_steps_are_validated_cumulatively(self, frame):
|
|
82
|
+
# The second step reads a column the first step drops: in-order validation catches it.
|
|
83
|
+
spec = {"steps": [DROP_ID, {"id": "cast_id", "type": "TypeCastTransformer", "columns": ["id"], "params": {"dtypes": {"id": "str"}}}]}
|
|
84
|
+
errors = spec_safety_errors(spec, frame)
|
|
85
|
+
assert len(errors) == 1 and "cast_id" in errors[0] and "failed to apply" in errors[0]
|
|
86
|
+
assert spec_safety_errors(spec, frame, report_failures=False) == []
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class TestPreviewStep:
|
|
90
|
+
def test_reports_delta_stats_and_row_counts(self, frame):
|
|
91
|
+
out = preview_step(CAST_TOTAL, frame)
|
|
92
|
+
assert out["error"] is None
|
|
93
|
+
assert out["delta"]["updated"] == ["total"]
|
|
94
|
+
assert out["stats"]["total"]["after"]["mean"] is not None
|
|
95
|
+
assert out["stats"]["total"]["before"]["mean"] is None # was text
|
|
96
|
+
assert out["rows_before"] == out["rows_after"] == len(frame)
|
|
97
|
+
assert out["warnings"] == []
|
|
98
|
+
|
|
99
|
+
def test_failure_is_reported_not_raised(self, frame):
|
|
100
|
+
bad = {"id": "bad", "type": "TypeCastTransformer", "columns": ["plan"], "params": {"dtypes": {"plan": "float"}, "errors": "raise"}}
|
|
101
|
+
out = preview_step(bad, frame)
|
|
102
|
+
assert out["error"] is not None or out["delta"]["updated"] == ["plan"]
|
|
103
|
+
|
|
104
|
+
def test_samples_large_frames(self, frame):
|
|
105
|
+
big = pd.concat([frame] * 100, ignore_index=True)
|
|
106
|
+
out = preview_step(DROP_ID, big, max_rows=500)
|
|
107
|
+
assert out["rows_before"] == 500
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class TestPreviewSpec:
|
|
111
|
+
def test_steps_apply_cumulatively_and_report_safety(self, frame):
|
|
112
|
+
spec = {"steps": [DROP_ID, CAST_TOTAL, DROP_LABEL]}
|
|
113
|
+
out = preview_spec(spec, frame, label="churn")
|
|
114
|
+
assert [s["step_id"] for s in out["steps"]] == ["drop_id", "cast_total", "drop_label"]
|
|
115
|
+
assert out["steps"][0]["delta"]["dropped"] == ["id"]
|
|
116
|
+
assert out["steps"][1]["delta"]["updated"] == ["total"]
|
|
117
|
+
assert "id" not in out["output_columns"]
|
|
118
|
+
assert out["safety_errors"] and "drop_label" in out["safety_errors"][0]
|
|
119
|
+
assert out["rows_before"] == out["rows_after"] == len(frame)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Tests for DataFrame-preserving pipeline pieces.
|
|
2
|
+
|
|
3
|
+
Regression coverage for sparse transformer output: sklearn's OneHotEncoder
|
|
4
|
+
defaults to sparse_output=True (csr_matrix). pd.DataFrame(csr_matrix, ...)
|
|
5
|
+
treats the matrix as a single object column, raising
|
|
6
|
+
"Shape of passed values is (N, 1), indices imply (N, K)".
|
|
7
|
+
Observed live as HTTP 500s from /v1/preprocessors/create.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import pandas as pd
|
|
11
|
+
import pytest
|
|
12
|
+
from sklearn.preprocessing import OneHotEncoder
|
|
13
|
+
|
|
14
|
+
from xplainable_preprocessing.compiler import compile_spec
|
|
15
|
+
from xplainable_preprocessing.pipeline import DataFrameColumnTransformer
|
|
16
|
+
from xplainable_preprocessing.schema import PipelineSpec
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@pytest.fixture
|
|
20
|
+
def df():
|
|
21
|
+
return pd.DataFrame({
|
|
22
|
+
"PhoneService": ["Yes", "No"] * 25,
|
|
23
|
+
"Contract": ["Month-to-month", "One year", "Two year", "One year",
|
|
24
|
+
"Month-to-month"] * 10,
|
|
25
|
+
"tenure": range(50),
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TestSparseTransformerOutput:
|
|
30
|
+
def test_column_transformer_densifies_sparse_output(self, df):
|
|
31
|
+
t = DataFrameColumnTransformer(OneHotEncoder(), ["PhoneService", "Contract"])
|
|
32
|
+
out = t.fit_transform(df)
|
|
33
|
+
|
|
34
|
+
assert isinstance(out, pd.DataFrame)
|
|
35
|
+
assert "tenure" in out.columns
|
|
36
|
+
assert "PhoneService_Yes" in out.columns
|
|
37
|
+
assert "Contract_Two year" in out.columns
|
|
38
|
+
# Row 0: PhoneService == "Yes"
|
|
39
|
+
assert out.loc[0, "PhoneService_Yes"] == 1.0
|
|
40
|
+
assert out.loc[0, "PhoneService_No"] == 0.0
|
|
41
|
+
|
|
42
|
+
def test_compile_spec_onehot_multi_column_step(self, df):
|
|
43
|
+
spec = PipelineSpec(version="2.0", steps=[
|
|
44
|
+
{"id": "oh", "type": "OneHotEncoder",
|
|
45
|
+
"columns": ["PhoneService", "Contract"], "params": {}},
|
|
46
|
+
])
|
|
47
|
+
pipeline = compile_spec(spec)
|
|
48
|
+
out = pipeline.fit_transform(df)
|
|
49
|
+
|
|
50
|
+
assert "PhoneService" not in out.columns
|
|
51
|
+
assert "PhoneService_No" in out.columns
|
|
52
|
+
assert "Contract_One year" in out.columns
|
|
53
|
+
assert len(out) == 50
|
|
54
|
+
|
|
55
|
+
def test_compile_spec_onehot_one_column_per_step(self, df):
|
|
56
|
+
spec = PipelineSpec(version="2.0", steps=[
|
|
57
|
+
{"id": "oh1", "type": "OneHotEncoder",
|
|
58
|
+
"columns": ["PhoneService"], "params": {}},
|
|
59
|
+
{"id": "oh2", "type": "OneHotEncoder",
|
|
60
|
+
"columns": ["Contract"], "params": {}},
|
|
61
|
+
])
|
|
62
|
+
pipeline = compile_spec(spec)
|
|
63
|
+
out = pipeline.fit_transform(df)
|
|
64
|
+
|
|
65
|
+
assert "PhoneService_Yes" in out.columns
|
|
66
|
+
assert "Contract_Month-to-month" in out.columns
|
|
67
|
+
assert "tenure" in out.columns
|
|
68
|
+
|
|
69
|
+
def test_transform_after_fit_matches_fit_transform(self, df):
|
|
70
|
+
t = DataFrameColumnTransformer(OneHotEncoder(), ["PhoneService"])
|
|
71
|
+
fitted_out = t.fit_transform(df)
|
|
72
|
+
transform_out = t.transform(df)
|
|
73
|
+
pd.testing.assert_frame_equal(fitted_out, transform_out)
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
.env
|
{xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/.github/workflows/publish-pypi.yml
RENAMED
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/dag-pipeline-proposal.md
RENAMED
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/feature-store-proposal.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_serialization.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|