xplainable-preprocessing 0.2.3__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. xplainable_preprocessing-0.3.0/.gitignore +5 -0
  2. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/PKG-INFO +3 -2
  3. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/pyproject.toml +2 -1
  4. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/__init__.py +14 -0
  5. xplainable_preprocessing-0.3.0/src/xplainable_preprocessing/guard.py +282 -0
  6. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/pipeline.py +6 -0
  7. xplainable_preprocessing-0.3.0/tests/test_guard.py +119 -0
  8. xplainable_preprocessing-0.3.0/tests/test_pipeline.py +73 -0
  9. xplainable_preprocessing-0.2.3/.gitignore +0 -1
  10. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/.github/workflows/publish-pypi.yml +0 -0
  11. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/README.md +0 -0
  12. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/dag-pipeline-proposal.md +0 -0
  13. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/feature-pipeline-architectures.md +0 -0
  14. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/docs/feature-store-proposal.md +0 -0
  15. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/compiler.py +0 -0
  16. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/preview.py +0 -0
  17. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/registry.py +0 -0
  18. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/sandbox.py +0 -0
  19. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/schema.py +0 -0
  20. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/serialization.py +0 -0
  21. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/__init__.py +0 -0
  22. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
  23. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/clip.py +0 -0
  24. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -0
  25. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
  26. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/expression.py +0 -0
  27. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
  28. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
  29. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -0
  30. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
  31. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
  32. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -0
  33. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/text_clean.py +0 -0
  34. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
  35. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/__init__.py +0 -0
  36. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_compiler.py +0 -0
  37. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_preview.py +0 -0
  38. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_sandbox.py +0 -0
  39. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_schema.py +0 -0
  40. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_serialization.py +0 -0
  41. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/__init__.py +0 -0
  42. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/test_all_transformers.py +0 -0
  43. {xplainable_preprocessing-0.2.3 → xplainable_preprocessing-0.3.0}/tests/test_transformers/test_expression.py +0 -0
@@ -0,0 +1,5 @@
1
+ .env
2
+ .worktrees/
3
+ .venv/
4
+ __pycache__/
5
+ *.pyc
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: xplainable-preprocessing
3
- Version: 0.2.3
3
+ Version: 0.3.0
4
4
  Summary: Shared preprocessing pipeline package for xplainable
5
5
  Requires-Python: >=3.9
6
6
  Requires-Dist: cloudpickle>=3.0
@@ -8,6 +8,7 @@ Requires-Dist: numpy>=1.24
8
8
  Requires-Dist: pandas>=2.0
9
9
  Requires-Dist: pydantic>=2.0
10
10
  Requires-Dist: scikit-learn>=1.3
11
+ Requires-Dist: scipy>=1.8
11
12
  Provides-Extra: dev
12
13
  Requires-Dist: pytest-cov; extra == 'dev'
13
14
  Requires-Dist: pytest>=7.0; extra == 'dev'
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "xplainable-preprocessing"
7
- version = "0.2.3"
7
+ version = "0.3.0"
8
8
  description = "Shared preprocessing pipeline package for xplainable"
9
9
  requires-python = ">=3.9"
10
10
  dependencies = [
@@ -12,6 +12,7 @@ dependencies = [
12
12
  "scikit-learn>=1.3",
13
13
  "pandas>=2.0",
14
14
  "numpy>=1.24",
15
+ "scipy>=1.8",
15
16
  "cloudpickle>=3.0",
16
17
  ]
17
18
 
@@ -4,6 +4,14 @@ from xplainable_preprocessing.schema import validate_spec
4
4
  from xplainable_preprocessing.serialization import save_pipeline, load_pipeline
5
5
  from xplainable_preprocessing.pipeline import DataFramePipeline, DataFrameColumnTransformer
6
6
  from xplainable_preprocessing.registry import REGISTRY, register, generate_catalog
7
+ from xplainable_preprocessing.guard import (
8
+ ROW_COLLAPSING,
9
+ preview_spec,
10
+ preview_step,
11
+ safe_catalog,
12
+ spec_safety_errors,
13
+ step_safety_error,
14
+ )
7
15
 
8
16
  __all__ = [
9
17
  "PipelineSpec",
@@ -17,4 +25,10 @@ __all__ = [
17
25
  "REGISTRY",
18
26
  "register",
19
27
  "generate_catalog",
28
+ "ROW_COLLAPSING",
29
+ "safe_catalog",
30
+ "step_safety_error",
31
+ "spec_safety_errors",
32
+ "preview_step",
33
+ "preview_spec",
20
34
  ]
@@ -0,0 +1,282 @@
1
+ """Step safety guard and step preview.
2
+
3
+ Shared by the platform API (preprocessor create/add), the training service
4
+ (agent apply/preview) and the client, so every surface rejects the same
5
+ unsafe steps with the same message.
6
+
7
+ A step is unsafe when it:
8
+ - uses a row-collapsing transformer (one row per group; the column wrapper
9
+ concat-misaligns the result back onto the frame and corrupts other columns),
10
+ - changes the row count,
11
+ - drops the label column, or
12
+ - mutates the label values.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import math
18
+ from typing import Any, Optional, Union
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+ from xplainable_preprocessing.compiler import compile_spec
24
+ from xplainable_preprocessing.pipeline import DataFrameColumnTransformer
25
+ from xplainable_preprocessing.preview import compute_delta
26
+ from xplainable_preprocessing.registry import generate_catalog
27
+ from xplainable_preprocessing.schema import PipelineSpec, StepSpec
28
+
29
+ ROW_COLLAPSING = frozenset({"GroupByAggTransformer"})
30
+ """Transformer types that change the row count by design. Never offer them to an LLM."""
31
+
32
+ DEFAULT_MAX_ROWS = 10_000
33
+
34
+ StepLike = Union[StepSpec, dict]
35
+ SpecLike = Union[PipelineSpec, dict]
36
+
37
+
38
+ def _as_step(step: StepLike) -> StepSpec:
39
+ return step if isinstance(step, StepSpec) else StepSpec(**step)
40
+
41
+
42
+ def _as_spec(spec: SpecLike) -> PipelineSpec:
43
+ return spec if isinstance(spec, PipelineSpec) else PipelineSpec(**spec)
44
+
45
+
46
+ def _sample(df: pd.DataFrame, max_rows: int) -> pd.DataFrame:
47
+ if max_rows and len(df) > max_rows:
48
+ return df.sample(max_rows, random_state=42)
49
+ return df.copy()
50
+
51
+
52
+ def safe_catalog() -> str:
53
+ """The transformer catalog with row-collapsing entries removed."""
54
+ lines = generate_catalog().splitlines()
55
+ return "\n".join(
56
+ line for line in lines
57
+ if not any(line.lstrip().startswith(f"- {name}(") for name in ROW_COLLAPSING)
58
+ )
59
+
60
+
61
+ def step_safety_error(
62
+ step: StepLike,
63
+ df: pd.DataFrame,
64
+ label: Optional[str] = None,
65
+ max_rows: int = DEFAULT_MAX_ROWS,
66
+ ) -> Optional[str]:
67
+ """Return why `step` is unsafe on `df`, or None when it is safe.
68
+
69
+ The step is compiled alone and fit on a sample. A step that cannot be
70
+ evaluated in isolation (for example it reads a column an earlier step
71
+ creates) is NOT rejected; use `spec_safety_errors` to validate a whole
72
+ spec in order.
73
+ """
74
+ step_spec = _as_step(step)
75
+ if step_spec.type in ROW_COLLAPSING:
76
+ return f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}"
77
+ try:
78
+ sample = _sample(df, max_rows)
79
+ pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
80
+ _, transformer = pipeline.steps[0]
81
+
82
+ # The column wrapper masks row collapses by concat-aligning onto the
83
+ # original index; check the inner transformer's raw output.
84
+ inner, inner_input = transformer, sample
85
+ if isinstance(transformer, DataFrameColumnTransformer):
86
+ inner = transformer.transformer
87
+ inner_input = sample[list(transformer.columns)]
88
+ raw = inner.fit_transform(inner_input.copy())
89
+ if hasattr(raw, "__len__") and len(raw) != len(sample):
90
+ return (
91
+ f"step '{step_spec.id}' changes the row count "
92
+ f"({len(sample)} -> {len(raw)}); row-collapsing steps corrupt the training frame"
93
+ )
94
+ transformed = pipeline.fit_transform(sample)
95
+ except Exception:
96
+ return None
97
+ return _label_error(step_spec.id, sample, transformed, label)
98
+
99
+
100
+ def _label_error(step_id: str, before: pd.DataFrame, after: pd.DataFrame, label: Optional[str]) -> Optional[str]:
101
+ if not label:
102
+ return None
103
+ if label not in after.columns:
104
+ return f"step '{step_id}' drops the label column '{label}'"
105
+ b = before[label].reset_index(drop=True)
106
+ a = after[label].reset_index(drop=True)
107
+ if not isinstance(a, pd.Series) or not a.equals(b):
108
+ return f"step '{step_id}' mutates the label column '{label}'"
109
+ return None
110
+
111
+
112
+ def spec_safety_errors(
113
+ spec: SpecLike,
114
+ df: pd.DataFrame,
115
+ label: Optional[str] = None,
116
+ max_rows: int = DEFAULT_MAX_ROWS,
117
+ report_failures: bool = True,
118
+ ) -> list:
119
+ """Validate every step of a spec in order on a sample of `df`.
120
+
121
+ Steps are applied cumulatively so later steps see the columns earlier
122
+ steps created. Returns a list of error strings (empty when the spec is
123
+ safe). A step that fails to fit is reported as well unless
124
+ `report_failures=False` (for callers that already surface fit failures
125
+ through their own compile path); such a step is skipped so later steps
126
+ are still checked.
127
+ """
128
+ pipeline_spec = _as_spec(spec)
129
+ errors = []
130
+ frame = _sample(df, max_rows)
131
+ for step_spec in pipeline_spec.steps:
132
+ if step_spec.type in ROW_COLLAPSING:
133
+ errors.append(f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}")
134
+ continue
135
+ try:
136
+ single = compile_spec(PipelineSpec(steps=[step_spec]))
137
+ after = single.fit_transform(frame.copy())
138
+ except Exception as e: # noqa: BLE001
139
+ if report_failures:
140
+ errors.append(f"step '{step_spec.id}' failed to apply: {e}")
141
+ continue
142
+ if len(after) != len(frame):
143
+ errors.append(
144
+ f"step '{step_spec.id}' changes the row count ({len(frame)} -> {len(after)})"
145
+ )
146
+ continue
147
+ label_err = _label_error(step_spec.id, frame, after, label)
148
+ if label_err:
149
+ errors.append(label_err)
150
+ continue
151
+ frame = after
152
+ return errors
153
+
154
+
155
+ def _num(v: Any) -> Optional[float]:
156
+ try:
157
+ f = float(v)
158
+ except (TypeError, ValueError):
159
+ return None
160
+ return None if (math.isnan(f) or math.isinf(f)) else f
161
+
162
+
163
+ def _column_stats(s: pd.Series) -> dict:
164
+ numeric = pd.api.types.is_numeric_dtype(s) and not pd.api.types.is_bool_dtype(s)
165
+ return {
166
+ "mean": _num(s.mean()) if numeric else None,
167
+ "nulls": int(s.isnull().sum()),
168
+ "min": _num(s.min()) if numeric else None,
169
+ "max": _num(s.max()) if numeric else None,
170
+ "unique": int(s.nunique()),
171
+ }
172
+
173
+
174
+ def preview_step(
175
+ step: StepLike,
176
+ df: pd.DataFrame,
177
+ max_rows: int = DEFAULT_MAX_ROWS,
178
+ ) -> dict:
179
+ """Dry-run one step on a sample of `df`.
180
+
181
+ Returns
182
+ -------
183
+ dict with keys:
184
+ - delta: {dropped, added, updated} column lists
185
+ - stats: {column: {"before": stats, "after": stats}} for changed columns
186
+ - warnings: human-readable warnings (new nulls, infinities, row-count change, ...)
187
+ - rows_before / rows_after
188
+ - error: str when the step failed to apply, else None
189
+ """
190
+ step_spec = _as_step(step)
191
+ before = _sample(df, max_rows)
192
+ rows_before = len(before)
193
+ try:
194
+ pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
195
+ after = pipeline.fit_transform(before.copy())
196
+ except Exception as e: # noqa: BLE001
197
+ return {
198
+ "delta": {"dropped": [], "added": [], "updated": []},
199
+ "stats": {},
200
+ "warnings": [f"fit_transform failed: {e}"],
201
+ "rows_before": rows_before,
202
+ "rows_after": rows_before,
203
+ "error": str(e),
204
+ }
205
+
206
+ raw = compute_delta(before, after)
207
+ delta = {k: list(raw.get(k, [])) for k in ("dropped", "added", "updated")}
208
+
209
+ stats = {}
210
+ for col in delta["added"] + delta["updated"]:
211
+ entry = {}
212
+ if col in before.columns:
213
+ entry["before"] = _column_stats(before[col])
214
+ if col in after.columns:
215
+ entry["after"] = _column_stats(after[col])
216
+ if entry:
217
+ stats[col] = entry
218
+
219
+ warnings = []
220
+ for col in delta["updated"]:
221
+ if col in before.columns and col in after.columns:
222
+ new_nulls = int(after[col].isnull().sum()) - int(before[col].isnull().sum())
223
+ if new_nulls > 0:
224
+ warnings.append(f"Column '{col}': {new_nulls} new null values introduced")
225
+ for col in after.select_dtypes(include=["number"]).columns:
226
+ try:
227
+ inf_count = int(np.isinf(after[col].astype(float)).sum())
228
+ except (TypeError, ValueError):
229
+ inf_count = 0
230
+ if inf_count:
231
+ warnings.append(f"Column '{col}': {inf_count} infinite values detected")
232
+ if len(delta["added"]) > 100:
233
+ warnings.append(f"Massive column expansion: {len(delta['added'])} new columns added")
234
+ rows_after = len(after)
235
+ if rows_after != rows_before:
236
+ warnings.append(f"Row count changed from {rows_before} to {rows_after}")
237
+
238
+ return {
239
+ "delta": delta,
240
+ "stats": stats,
241
+ "warnings": warnings,
242
+ "rows_before": rows_before,
243
+ "rows_after": rows_after,
244
+ "error": None,
245
+ }
246
+
247
+
248
+ def preview_spec(
249
+ spec: SpecLike,
250
+ df: pd.DataFrame,
251
+ label: Optional[str] = None,
252
+ max_rows: int = DEFAULT_MAX_ROWS,
253
+ ) -> dict:
254
+ """Preview every step of a spec in order, plus the safety verdict.
255
+
256
+ Returns
257
+ -------
258
+ dict with keys:
259
+ - steps: [{"step_id", "step_type", **preview_step(...)}] applied cumulatively
260
+ - safety_errors: from `spec_safety_errors`
261
+ - rows_before / rows_after
262
+ - output_columns: columns after the last step that applied cleanly
263
+ """
264
+ pipeline_spec = _as_spec(spec)
265
+ frame = _sample(df, max_rows)
266
+ rows_before = len(frame)
267
+ steps = []
268
+ for step_spec in pipeline_spec.steps:
269
+ result = preview_step(step_spec, frame, max_rows=0)
270
+ steps.append({"step_id": step_spec.id, "step_type": step_spec.type, **result})
271
+ if result["error"] is None:
272
+ try:
273
+ frame = compile_spec(PipelineSpec(steps=[step_spec])).fit_transform(frame.copy())
274
+ except Exception: # noqa: BLE001
275
+ pass
276
+ return {
277
+ "steps": steps,
278
+ "safety_errors": spec_safety_errors(pipeline_spec, df, label=label, max_rows=max_rows),
279
+ "rows_before": rows_before,
280
+ "rows_after": len(frame),
281
+ "output_columns": list(frame.columns),
282
+ }
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import pandas as pd
6
+ from scipy import sparse
6
7
  from sklearn.base import BaseEstimator, TransformerMixin
7
8
 
8
9
 
@@ -20,6 +21,11 @@ class DataFrameColumnTransformer(BaseEstimator, TransformerMixin):
20
21
  def transform(self, X: pd.DataFrame) -> pd.DataFrame:
21
22
  Xt = X.copy()
22
23
  result = self.transformer.transform(Xt[self.columns])
24
+ if sparse.issparse(result):
25
+ # e.g. OneHotEncoder defaults to sparse_output=True; pd.DataFrame
26
+ # would treat the matrix as a single object column and raise
27
+ # "Shape of passed values is (N, 1), indices imply (N, K)".
28
+ result = result.toarray()
23
29
  if isinstance(result, pd.DataFrame):
24
30
  Xt = Xt.drop(columns=self.columns)
25
31
  Xt = pd.concat([Xt, result], axis=1)
@@ -0,0 +1,119 @@
1
+ """Tests for the step safety guard and step preview."""
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+ import pytest
6
+
7
+ from xplainable_preprocessing import (
8
+ ROW_COLLAPSING,
9
+ preview_spec,
10
+ preview_step,
11
+ safe_catalog,
12
+ spec_safety_errors,
13
+ step_safety_error,
14
+ )
15
+ from xplainable_preprocessing.schema import PipelineSpec, StepSpec
16
+
17
+
18
+ @pytest.fixture
19
+ def frame():
20
+ rng = np.random.default_rng(0)
21
+ n = 120
22
+ return pd.DataFrame({
23
+ "id": [f"C{i}" for i in range(n)],
24
+ "tenure": rng.integers(1, 72, n),
25
+ "charges": rng.uniform(20, 120, n).round(2),
26
+ "total": [str(v) for v in rng.uniform(100, 5000, n).round(1)],
27
+ "plan": rng.choice(["a", "b", "c", "d", "e"], n),
28
+ "churn": rng.integers(0, 2, n),
29
+ })
30
+
31
+
32
+ DROP_ID = {"id": "drop_id", "type": "DropColumnsTransformer", "columns": ["id"]}
33
+ CAST_TOTAL = {"id": "cast_total", "type": "TypeCastTransformer", "columns": ["total"], "params": {"dtypes": {"total": "float"}}}
34
+ DROP_LABEL = {"id": "drop_label", "type": "DropColumnsTransformer", "columns": ["churn"]}
35
+ COLLAPSE = {"id": "grp", "type": "GroupByAggTransformer", "columns": ["plan", "charges"], "params": {}}
36
+
37
+
38
+ class TestSafeCatalog:
39
+ def test_excludes_row_collapsing_transformers(self):
40
+ catalog = safe_catalog()
41
+ for name in ROW_COLLAPSING:
42
+ assert f"- {name}(" not in catalog
43
+ assert "- DropColumnsTransformer(" in catalog
44
+
45
+
46
+ class TestStepSafetyError:
47
+ def test_safe_step_is_none(self, frame):
48
+ assert step_safety_error(DROP_ID, frame, label="churn") is None
49
+ assert step_safety_error(StepSpec(**CAST_TOTAL), frame, label="churn") is None
50
+
51
+ def test_row_collapsing_type_is_rejected_without_running(self, frame):
52
+ assert "row-collapsing" in step_safety_error(COLLAPSE, frame)
53
+
54
+ def test_dropping_the_label_is_rejected(self, frame):
55
+ assert "drops the label" in step_safety_error(DROP_LABEL, frame, label="churn")
56
+
57
+ def test_mutating_the_label_is_rejected(self, frame):
58
+ step = {"id": "cast_label", "type": "TypeCastTransformer", "columns": ["churn"], "params": {"dtypes": {"churn": "str"}}}
59
+ assert "mutates the label" in step_safety_error(step, frame, label="churn")
60
+
61
+ def test_step_that_cannot_run_in_isolation_is_not_rejected(self, frame):
62
+ # References a column that does not exist yet (created by an earlier step).
63
+ step = {"id": "later", "type": "DropColumnsTransformer", "columns": ["derived_col"]}
64
+ assert step_safety_error(step, frame, label="churn") is None
65
+
66
+ def test_without_label_only_structural_checks_apply(self, frame):
67
+ assert step_safety_error(DROP_LABEL, frame) is None
68
+
69
+
70
+ class TestSpecSafetyErrors:
71
+ def test_clean_spec_has_no_errors(self, frame):
72
+ spec = {"version": "2.0", "steps": [DROP_ID, CAST_TOTAL]}
73
+ assert spec_safety_errors(spec, frame, label="churn") == []
74
+
75
+ def test_reports_every_unsafe_step(self, frame):
76
+ spec = PipelineSpec(steps=[StepSpec(**DROP_ID), StepSpec(**DROP_LABEL), StepSpec(**COLLAPSE)])
77
+ errors = spec_safety_errors(spec, frame, label="churn")
78
+ assert len(errors) == 2
79
+ assert "drop_label" in errors[0] and "grp" in errors[1]
80
+
81
+ def test_steps_are_validated_cumulatively(self, frame):
82
+ # The second step reads a column the first step drops: in-order validation catches it.
83
+ spec = {"steps": [DROP_ID, {"id": "cast_id", "type": "TypeCastTransformer", "columns": ["id"], "params": {"dtypes": {"id": "str"}}}]}
84
+ errors = spec_safety_errors(spec, frame)
85
+ assert len(errors) == 1 and "cast_id" in errors[0] and "failed to apply" in errors[0]
86
+ assert spec_safety_errors(spec, frame, report_failures=False) == []
87
+
88
+
89
+ class TestPreviewStep:
90
+ def test_reports_delta_stats_and_row_counts(self, frame):
91
+ out = preview_step(CAST_TOTAL, frame)
92
+ assert out["error"] is None
93
+ assert out["delta"]["updated"] == ["total"]
94
+ assert out["stats"]["total"]["after"]["mean"] is not None
95
+ assert out["stats"]["total"]["before"]["mean"] is None # was text
96
+ assert out["rows_before"] == out["rows_after"] == len(frame)
97
+ assert out["warnings"] == []
98
+
99
+ def test_failure_is_reported_not_raised(self, frame):
100
+ bad = {"id": "bad", "type": "TypeCastTransformer", "columns": ["plan"], "params": {"dtypes": {"plan": "float"}, "errors": "raise"}}
101
+ out = preview_step(bad, frame)
102
+ assert out["error"] is not None or out["delta"]["updated"] == ["plan"]
103
+
104
+ def test_samples_large_frames(self, frame):
105
+ big = pd.concat([frame] * 100, ignore_index=True)
106
+ out = preview_step(DROP_ID, big, max_rows=500)
107
+ assert out["rows_before"] == 500
108
+
109
+
110
+ class TestPreviewSpec:
111
+ def test_steps_apply_cumulatively_and_report_safety(self, frame):
112
+ spec = {"steps": [DROP_ID, CAST_TOTAL, DROP_LABEL]}
113
+ out = preview_spec(spec, frame, label="churn")
114
+ assert [s["step_id"] for s in out["steps"]] == ["drop_id", "cast_total", "drop_label"]
115
+ assert out["steps"][0]["delta"]["dropped"] == ["id"]
116
+ assert out["steps"][1]["delta"]["updated"] == ["total"]
117
+ assert "id" not in out["output_columns"]
118
+ assert out["safety_errors"] and "drop_label" in out["safety_errors"][0]
119
+ assert out["rows_before"] == out["rows_after"] == len(frame)
@@ -0,0 +1,73 @@
1
+ """Tests for DataFrame-preserving pipeline pieces.
2
+
3
+ Regression coverage for sparse transformer output: sklearn's OneHotEncoder
4
+ defaults to sparse_output=True (csr_matrix). pd.DataFrame(csr_matrix, ...)
5
+ treats the matrix as a single object column, raising
6
+ "Shape of passed values is (N, 1), indices imply (N, K)".
7
+ Observed live as HTTP 500s from /v1/preprocessors/create.
8
+ """
9
+
10
+ import pandas as pd
11
+ import pytest
12
+ from sklearn.preprocessing import OneHotEncoder
13
+
14
+ from xplainable_preprocessing.compiler import compile_spec
15
+ from xplainable_preprocessing.pipeline import DataFrameColumnTransformer
16
+ from xplainable_preprocessing.schema import PipelineSpec
17
+
18
+
19
+ @pytest.fixture
20
+ def df():
21
+ return pd.DataFrame({
22
+ "PhoneService": ["Yes", "No"] * 25,
23
+ "Contract": ["Month-to-month", "One year", "Two year", "One year",
24
+ "Month-to-month"] * 10,
25
+ "tenure": range(50),
26
+ })
27
+
28
+
29
+ class TestSparseTransformerOutput:
30
+ def test_column_transformer_densifies_sparse_output(self, df):
31
+ t = DataFrameColumnTransformer(OneHotEncoder(), ["PhoneService", "Contract"])
32
+ out = t.fit_transform(df)
33
+
34
+ assert isinstance(out, pd.DataFrame)
35
+ assert "tenure" in out.columns
36
+ assert "PhoneService_Yes" in out.columns
37
+ assert "Contract_Two year" in out.columns
38
+ # Row 0: PhoneService == "Yes"
39
+ assert out.loc[0, "PhoneService_Yes"] == 1.0
40
+ assert out.loc[0, "PhoneService_No"] == 0.0
41
+
42
+ def test_compile_spec_onehot_multi_column_step(self, df):
43
+ spec = PipelineSpec(version="2.0", steps=[
44
+ {"id": "oh", "type": "OneHotEncoder",
45
+ "columns": ["PhoneService", "Contract"], "params": {}},
46
+ ])
47
+ pipeline = compile_spec(spec)
48
+ out = pipeline.fit_transform(df)
49
+
50
+ assert "PhoneService" not in out.columns
51
+ assert "PhoneService_No" in out.columns
52
+ assert "Contract_One year" in out.columns
53
+ assert len(out) == 50
54
+
55
+ def test_compile_spec_onehot_one_column_per_step(self, df):
56
+ spec = PipelineSpec(version="2.0", steps=[
57
+ {"id": "oh1", "type": "OneHotEncoder",
58
+ "columns": ["PhoneService"], "params": {}},
59
+ {"id": "oh2", "type": "OneHotEncoder",
60
+ "columns": ["Contract"], "params": {}},
61
+ ])
62
+ pipeline = compile_spec(spec)
63
+ out = pipeline.fit_transform(df)
64
+
65
+ assert "PhoneService_Yes" in out.columns
66
+ assert "Contract_Month-to-month" in out.columns
67
+ assert "tenure" in out.columns
68
+
69
+ def test_transform_after_fit_matches_fit_transform(self, df):
70
+ t = DataFrameColumnTransformer(OneHotEncoder(), ["PhoneService"])
71
+ fitted_out = t.fit_transform(df)
72
+ transform_out = t.transform(df)
73
+ pd.testing.assert_frame_equal(fitted_out, transform_out)
@@ -1 +0,0 @@
1
- .env