xplainable-preprocessing 0.2.4__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. xplainable_preprocessing-0.3.1/.gitignore +5 -0
  2. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/PKG-INFO +1 -1
  3. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/pyproject.toml +1 -1
  4. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/__init__.py +14 -0
  5. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/guard.py +282 -0
  6. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/expression.py +16 -0
  7. xplainable_preprocessing-0.3.1/tests/test_guard.py +119 -0
  8. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_transformers/test_expression.py +23 -0
  9. xplainable_preprocessing-0.2.4/.gitignore +0 -1
  10. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/.github/workflows/publish-pypi.yml +0 -0
  11. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/README.md +0 -0
  12. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/docs/dag-pipeline-proposal.md +0 -0
  13. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/docs/feature-pipeline-architectures.md +0 -0
  14. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/docs/feature-store-proposal.md +0 -0
  15. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/compiler.py +0 -0
  16. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/pipeline.py +0 -0
  17. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/preview.py +0 -0
  18. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/registry.py +0 -0
  19. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/sandbox.py +0 -0
  20. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/schema.py +0 -0
  21. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/serialization.py +0 -0
  22. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/__init__.py +0 -0
  23. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
  24. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/clip.py +0 -0
  25. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -0
  26. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
  27. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
  28. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
  29. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -0
  30. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
  31. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
  32. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -0
  33. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/text_clean.py +0 -0
  34. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
  35. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/__init__.py +0 -0
  36. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_compiler.py +0 -0
  37. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_pipeline.py +0 -0
  38. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_preview.py +0 -0
  39. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_sandbox.py +0 -0
  40. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_schema.py +0 -0
  41. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_serialization.py +0 -0
  42. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_transformers/__init__.py +0 -0
  43. {xplainable_preprocessing-0.2.4 → xplainable_preprocessing-0.3.1}/tests/test_transformers/test_all_transformers.py +0 -0
@@ -0,0 +1,5 @@
1
+ .env
2
+ .worktrees/
3
+ .venv/
4
+ __pycache__/
5
+ *.pyc
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: xplainable-preprocessing
3
- Version: 0.2.4
3
+ Version: 0.3.1
4
4
  Summary: Shared preprocessing pipeline package for xplainable
5
5
  Requires-Python: >=3.9
6
6
  Requires-Dist: cloudpickle>=3.0
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "xplainable-preprocessing"
7
- version = "0.2.4"
7
+ version = "0.3.1"
8
8
  description = "Shared preprocessing pipeline package for xplainable"
9
9
  requires-python = ">=3.9"
10
10
  dependencies = [
@@ -4,6 +4,14 @@ from xplainable_preprocessing.schema import validate_spec
4
4
  from xplainable_preprocessing.serialization import save_pipeline, load_pipeline
5
5
  from xplainable_preprocessing.pipeline import DataFramePipeline, DataFrameColumnTransformer
6
6
  from xplainable_preprocessing.registry import REGISTRY, register, generate_catalog
7
+ from xplainable_preprocessing.guard import (
8
+ ROW_COLLAPSING,
9
+ preview_spec,
10
+ preview_step,
11
+ safe_catalog,
12
+ spec_safety_errors,
13
+ step_safety_error,
14
+ )
7
15
 
8
16
  __all__ = [
9
17
  "PipelineSpec",
@@ -17,4 +25,10 @@ __all__ = [
17
25
  "REGISTRY",
18
26
  "register",
19
27
  "generate_catalog",
28
+ "ROW_COLLAPSING",
29
+ "safe_catalog",
30
+ "step_safety_error",
31
+ "spec_safety_errors",
32
+ "preview_step",
33
+ "preview_spec",
20
34
  ]
@@ -0,0 +1,282 @@
1
+ """Step safety guard and step preview.
2
+
3
+ Shared by the platform API (preprocessor create/add), the training service
4
+ (agent apply/preview) and the client, so every surface rejects the same
5
+ unsafe steps with the same message.
6
+
7
+ A step is unsafe when it:
8
+ - uses a row-collapsing transformer (one row per group; the column wrapper
9
+ concat-misaligns the result back onto the frame and corrupts other columns),
10
+ - changes the row count,
11
+ - drops the label column, or
12
+ - mutates the label values.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import math
18
+ from typing import Any, Optional, Union
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+ from xplainable_preprocessing.compiler import compile_spec
24
+ from xplainable_preprocessing.pipeline import DataFrameColumnTransformer
25
+ from xplainable_preprocessing.preview import compute_delta
26
+ from xplainable_preprocessing.registry import generate_catalog
27
+ from xplainable_preprocessing.schema import PipelineSpec, StepSpec
28
+
29
+ ROW_COLLAPSING = frozenset({"GroupByAggTransformer"})
30
+ """Transformer types that change the row count by design. Never offer them to an LLM."""
31
+
32
+ DEFAULT_MAX_ROWS = 10_000
33
+
34
+ StepLike = Union[StepSpec, dict]
35
+ SpecLike = Union[PipelineSpec, dict]
36
+
37
+
38
+ def _as_step(step: StepLike) -> StepSpec:
39
+ return step if isinstance(step, StepSpec) else StepSpec(**step)
40
+
41
+
42
+ def _as_spec(spec: SpecLike) -> PipelineSpec:
43
+ return spec if isinstance(spec, PipelineSpec) else PipelineSpec(**spec)
44
+
45
+
46
+ def _sample(df: pd.DataFrame, max_rows: int) -> pd.DataFrame:
47
+ if max_rows and len(df) > max_rows:
48
+ return df.sample(max_rows, random_state=42)
49
+ return df.copy()
50
+
51
+
52
+ def safe_catalog() -> str:
53
+ """The transformer catalog with row-collapsing entries removed."""
54
+ lines = generate_catalog().splitlines()
55
+ return "\n".join(
56
+ line for line in lines
57
+ if not any(line.lstrip().startswith(f"- {name}(") for name in ROW_COLLAPSING)
58
+ )
59
+
60
+
61
+ def step_safety_error(
62
+ step: StepLike,
63
+ df: pd.DataFrame,
64
+ label: Optional[str] = None,
65
+ max_rows: int = DEFAULT_MAX_ROWS,
66
+ ) -> Optional[str]:
67
+ """Return why `step` is unsafe on `df`, or None when it is safe.
68
+
69
+ The step is compiled alone and fit on a sample. A step that cannot be
70
+ evaluated in isolation (for example it reads a column an earlier step
71
+ creates) is NOT rejected; use `spec_safety_errors` to validate a whole
72
+ spec in order.
73
+ """
74
+ step_spec = _as_step(step)
75
+ if step_spec.type in ROW_COLLAPSING:
76
+ return f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}"
77
+ try:
78
+ sample = _sample(df, max_rows)
79
+ pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
80
+ _, transformer = pipeline.steps[0]
81
+
82
+ # The column wrapper masks row collapses by concat-aligning onto the
83
+ # original index; check the inner transformer's raw output.
84
+ inner, inner_input = transformer, sample
85
+ if isinstance(transformer, DataFrameColumnTransformer):
86
+ inner = transformer.transformer
87
+ inner_input = sample[list(transformer.columns)]
88
+ raw = inner.fit_transform(inner_input.copy())
89
+ if hasattr(raw, "__len__") and len(raw) != len(sample):
90
+ return (
91
+ f"step '{step_spec.id}' changes the row count "
92
+ f"({len(sample)} -> {len(raw)}); row-collapsing steps corrupt the training frame"
93
+ )
94
+ transformed = pipeline.fit_transform(sample)
95
+ except Exception:
96
+ return None
97
+ return _label_error(step_spec.id, sample, transformed, label)
98
+
99
+
100
+ def _label_error(step_id: str, before: pd.DataFrame, after: pd.DataFrame, label: Optional[str]) -> Optional[str]:
101
+ if not label:
102
+ return None
103
+ if label not in after.columns:
104
+ return f"step '{step_id}' drops the label column '{label}'"
105
+ b = before[label].reset_index(drop=True)
106
+ a = after[label].reset_index(drop=True)
107
+ if not isinstance(a, pd.Series) or not a.equals(b):
108
+ return f"step '{step_id}' mutates the label column '{label}'"
109
+ return None
110
+
111
+
112
+ def spec_safety_errors(
113
+ spec: SpecLike,
114
+ df: pd.DataFrame,
115
+ label: Optional[str] = None,
116
+ max_rows: int = DEFAULT_MAX_ROWS,
117
+ report_failures: bool = True,
118
+ ) -> list:
119
+ """Validate every step of a spec in order on a sample of `df`.
120
+
121
+ Steps are applied cumulatively so later steps see the columns earlier
122
+ steps created. Returns a list of error strings (empty when the spec is
123
+ safe). A step that fails to fit is reported as well unless
124
+ `report_failures=False` (for callers that already surface fit failures
125
+ through their own compile path); such a step is skipped so later steps
126
+ are still checked.
127
+ """
128
+ pipeline_spec = _as_spec(spec)
129
+ errors = []
130
+ frame = _sample(df, max_rows)
131
+ for step_spec in pipeline_spec.steps:
132
+ if step_spec.type in ROW_COLLAPSING:
133
+ errors.append(f"step '{step_spec.id}' uses row-collapsing transformer {step_spec.type}")
134
+ continue
135
+ try:
136
+ single = compile_spec(PipelineSpec(steps=[step_spec]))
137
+ after = single.fit_transform(frame.copy())
138
+ except Exception as e: # noqa: BLE001
139
+ if report_failures:
140
+ errors.append(f"step '{step_spec.id}' failed to apply: {e}")
141
+ continue
142
+ if len(after) != len(frame):
143
+ errors.append(
144
+ f"step '{step_spec.id}' changes the row count ({len(frame)} -> {len(after)})"
145
+ )
146
+ continue
147
+ label_err = _label_error(step_spec.id, frame, after, label)
148
+ if label_err:
149
+ errors.append(label_err)
150
+ continue
151
+ frame = after
152
+ return errors
153
+
154
+
155
+ def _num(v: Any) -> Optional[float]:
156
+ try:
157
+ f = float(v)
158
+ except (TypeError, ValueError):
159
+ return None
160
+ return None if (math.isnan(f) or math.isinf(f)) else f
161
+
162
+
163
+ def _column_stats(s: pd.Series) -> dict:
164
+ numeric = pd.api.types.is_numeric_dtype(s) and not pd.api.types.is_bool_dtype(s)
165
+ return {
166
+ "mean": _num(s.mean()) if numeric else None,
167
+ "nulls": int(s.isnull().sum()),
168
+ "min": _num(s.min()) if numeric else None,
169
+ "max": _num(s.max()) if numeric else None,
170
+ "unique": int(s.nunique()),
171
+ }
172
+
173
+
174
+ def preview_step(
175
+ step: StepLike,
176
+ df: pd.DataFrame,
177
+ max_rows: int = DEFAULT_MAX_ROWS,
178
+ ) -> dict:
179
+ """Dry-run one step on a sample of `df`.
180
+
181
+ Returns
182
+ -------
183
+ dict with keys:
184
+ - delta: {dropped, added, updated} column lists
185
+ - stats: {column: {"before": stats, "after": stats}} for changed columns
186
+ - warnings: human-readable warnings (new nulls, infinities, row-count change, ...)
187
+ - rows_before / rows_after
188
+ - error: str when the step failed to apply, else None
189
+ """
190
+ step_spec = _as_step(step)
191
+ before = _sample(df, max_rows)
192
+ rows_before = len(before)
193
+ try:
194
+ pipeline = compile_spec(PipelineSpec(steps=[step_spec]))
195
+ after = pipeline.fit_transform(before.copy())
196
+ except Exception as e: # noqa: BLE001
197
+ return {
198
+ "delta": {"dropped": [], "added": [], "updated": []},
199
+ "stats": {},
200
+ "warnings": [f"fit_transform failed: {e}"],
201
+ "rows_before": rows_before,
202
+ "rows_after": rows_before,
203
+ "error": str(e),
204
+ }
205
+
206
+ raw = compute_delta(before, after)
207
+ delta = {k: list(raw.get(k, [])) for k in ("dropped", "added", "updated")}
208
+
209
+ stats = {}
210
+ for col in delta["added"] + delta["updated"]:
211
+ entry = {}
212
+ if col in before.columns:
213
+ entry["before"] = _column_stats(before[col])
214
+ if col in after.columns:
215
+ entry["after"] = _column_stats(after[col])
216
+ if entry:
217
+ stats[col] = entry
218
+
219
+ warnings = []
220
+ for col in delta["updated"]:
221
+ if col in before.columns and col in after.columns:
222
+ new_nulls = int(after[col].isnull().sum()) - int(before[col].isnull().sum())
223
+ if new_nulls > 0:
224
+ warnings.append(f"Column '{col}': {new_nulls} new null values introduced")
225
+ for col in after.select_dtypes(include=["number"]).columns:
226
+ try:
227
+ inf_count = int(np.isinf(after[col].astype(float)).sum())
228
+ except (TypeError, ValueError):
229
+ inf_count = 0
230
+ if inf_count:
231
+ warnings.append(f"Column '{col}': {inf_count} infinite values detected")
232
+ if len(delta["added"]) > 100:
233
+ warnings.append(f"Massive column expansion: {len(delta['added'])} new columns added")
234
+ rows_after = len(after)
235
+ if rows_after != rows_before:
236
+ warnings.append(f"Row count changed from {rows_before} to {rows_after}")
237
+
238
+ return {
239
+ "delta": delta,
240
+ "stats": stats,
241
+ "warnings": warnings,
242
+ "rows_before": rows_before,
243
+ "rows_after": rows_after,
244
+ "error": None,
245
+ }
246
+
247
+
248
+ def preview_spec(
249
+ spec: SpecLike,
250
+ df: pd.DataFrame,
251
+ label: Optional[str] = None,
252
+ max_rows: int = DEFAULT_MAX_ROWS,
253
+ ) -> dict:
254
+ """Preview every step of a spec in order, plus the safety verdict.
255
+
256
+ Returns
257
+ -------
258
+ dict with keys:
259
+ - steps: [{"step_id", "step_type", **preview_step(...)}] applied cumulatively
260
+ - safety_errors: from `spec_safety_errors`
261
+ - rows_before / rows_after
262
+ - output_columns: columns after the last step that applied cleanly
263
+ """
264
+ pipeline_spec = _as_spec(spec)
265
+ frame = _sample(df, max_rows)
266
+ rows_before = len(frame)
267
+ steps = []
268
+ for step_spec in pipeline_spec.steps:
269
+ result = preview_step(step_spec, frame, max_rows=0)
270
+ steps.append({"step_id": step_spec.id, "step_type": step_spec.type, **result})
271
+ if result["error"] is None:
272
+ try:
273
+ frame = compile_spec(PipelineSpec(steps=[step_spec])).fit_transform(frame.copy())
274
+ except Exception: # noqa: BLE001
275
+ pass
276
+ return {
277
+ "steps": steps,
278
+ "safety_errors": spec_safety_errors(pipeline_spec, df, label=label, max_rows=max_rows),
279
+ "rows_before": rows_before,
280
+ "rows_after": len(frame),
281
+ "output_columns": list(frame.columns),
282
+ }
@@ -5,6 +5,17 @@ from __future__ import annotations
5
5
  import pandas as pd
6
6
  from sklearn.base import BaseEstimator, TransformerMixin
7
7
 
8
+ try:
9
+ # Semi-private pandas API (stable since 1.0, present through 2.x). It is
10
+ # the canonical way to reproduce how pandas mangles backtick-quoted names
11
+ # (`Monthly Charges` -> BACKTICK_QUOTED_STRING_Monthly_Charges) before
12
+ # resolving them; the mangling table is version-specific, so we do not
13
+ # re-implement it. If a future pandas removes it, the package still
14
+ # imports and only backtick-quoted column references lose support.
15
+ from pandas.core.computation.parsing import clean_column_name
16
+ except ImportError: # pragma: no cover
17
+ clean_column_name = None
18
+
8
19
 
9
20
  class ExpressionTransformer(BaseEstimator, TransformerMixin):
10
21
  """Create new columns using pandas.eval() expressions.
@@ -30,6 +41,11 @@ class ExpressionTransformer(BaseEstimator, TransformerMixin):
30
41
  Xt = X.copy()
31
42
  # Support both bare column names ("age * salary") and df reference ("df['age'] * df['salary']")
32
43
  local_dict = {col: Xt[col] for col in Xt.columns}
44
+ # Backtick-quoted names ("`Monthly Charges` * Tenure") are rewritten by
45
+ # pandas' parser into mangled identifiers regardless of engine, so the
46
+ # same Series must also be reachable under the mangled alias.
47
+ if clean_column_name is not None:
48
+ local_dict.update({clean_column_name(col): Xt[col] for col in Xt.columns})
33
49
  local_dict["df"] = Xt
34
50
  Xt[self.output_column] = pd.eval(self.expression, local_dict=local_dict, engine="python")
35
51
  return Xt
@@ -0,0 +1,119 @@
1
+ """Tests for the step safety guard and step preview."""
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+ import pytest
6
+
7
+ from xplainable_preprocessing import (
8
+ ROW_COLLAPSING,
9
+ preview_spec,
10
+ preview_step,
11
+ safe_catalog,
12
+ spec_safety_errors,
13
+ step_safety_error,
14
+ )
15
+ from xplainable_preprocessing.schema import PipelineSpec, StepSpec
16
+
17
+
18
+ @pytest.fixture
19
+ def frame():
20
+ rng = np.random.default_rng(0)
21
+ n = 120
22
+ return pd.DataFrame({
23
+ "id": [f"C{i}" for i in range(n)],
24
+ "tenure": rng.integers(1, 72, n),
25
+ "charges": rng.uniform(20, 120, n).round(2),
26
+ "total": [str(v) for v in rng.uniform(100, 5000, n).round(1)],
27
+ "plan": rng.choice(["a", "b", "c", "d", "e"], n),
28
+ "churn": rng.integers(0, 2, n),
29
+ })
30
+
31
+
32
+ DROP_ID = {"id": "drop_id", "type": "DropColumnsTransformer", "columns": ["id"]}
33
+ CAST_TOTAL = {"id": "cast_total", "type": "TypeCastTransformer", "columns": ["total"], "params": {"dtypes": {"total": "float"}}}
34
+ DROP_LABEL = {"id": "drop_label", "type": "DropColumnsTransformer", "columns": ["churn"]}
35
+ COLLAPSE = {"id": "grp", "type": "GroupByAggTransformer", "columns": ["plan", "charges"], "params": {}}
36
+
37
+
38
+ class TestSafeCatalog:
39
+ def test_excludes_row_collapsing_transformers(self):
40
+ catalog = safe_catalog()
41
+ for name in ROW_COLLAPSING:
42
+ assert f"- {name}(" not in catalog
43
+ assert "- DropColumnsTransformer(" in catalog
44
+
45
+
46
+ class TestStepSafetyError:
47
+ def test_safe_step_is_none(self, frame):
48
+ assert step_safety_error(DROP_ID, frame, label="churn") is None
49
+ assert step_safety_error(StepSpec(**CAST_TOTAL), frame, label="churn") is None
50
+
51
+ def test_row_collapsing_type_is_rejected_without_running(self, frame):
52
+ assert "row-collapsing" in step_safety_error(COLLAPSE, frame)
53
+
54
+ def test_dropping_the_label_is_rejected(self, frame):
55
+ assert "drops the label" in step_safety_error(DROP_LABEL, frame, label="churn")
56
+
57
+ def test_mutating_the_label_is_rejected(self, frame):
58
+ step = {"id": "cast_label", "type": "TypeCastTransformer", "columns": ["churn"], "params": {"dtypes": {"churn": "str"}}}
59
+ assert "mutates the label" in step_safety_error(step, frame, label="churn")
60
+
61
+ def test_step_that_cannot_run_in_isolation_is_not_rejected(self, frame):
62
+ # References a column that does not exist yet (created by an earlier step).
63
+ step = {"id": "later", "type": "DropColumnsTransformer", "columns": ["derived_col"]}
64
+ assert step_safety_error(step, frame, label="churn") is None
65
+
66
+ def test_without_label_only_structural_checks_apply(self, frame):
67
+ assert step_safety_error(DROP_LABEL, frame) is None
68
+
69
+
70
+ class TestSpecSafetyErrors:
71
+ def test_clean_spec_has_no_errors(self, frame):
72
+ spec = {"version": "2.0", "steps": [DROP_ID, CAST_TOTAL]}
73
+ assert spec_safety_errors(spec, frame, label="churn") == []
74
+
75
+ def test_reports_every_unsafe_step(self, frame):
76
+ spec = PipelineSpec(steps=[StepSpec(**DROP_ID), StepSpec(**DROP_LABEL), StepSpec(**COLLAPSE)])
77
+ errors = spec_safety_errors(spec, frame, label="churn")
78
+ assert len(errors) == 2
79
+ assert "drop_label" in errors[0] and "grp" in errors[1]
80
+
81
+ def test_steps_are_validated_cumulatively(self, frame):
82
+ # The second step reads a column the first step drops: in-order validation catches it.
83
+ spec = {"steps": [DROP_ID, {"id": "cast_id", "type": "TypeCastTransformer", "columns": ["id"], "params": {"dtypes": {"id": "str"}}}]}
84
+ errors = spec_safety_errors(spec, frame)
85
+ assert len(errors) == 1 and "cast_id" in errors[0] and "failed to apply" in errors[0]
86
+ assert spec_safety_errors(spec, frame, report_failures=False) == []
87
+
88
+
89
+ class TestPreviewStep:
90
+ def test_reports_delta_stats_and_row_counts(self, frame):
91
+ out = preview_step(CAST_TOTAL, frame)
92
+ assert out["error"] is None
93
+ assert out["delta"]["updated"] == ["total"]
94
+ assert out["stats"]["total"]["after"]["mean"] is not None
95
+ assert out["stats"]["total"]["before"]["mean"] is None # was text
96
+ assert out["rows_before"] == out["rows_after"] == len(frame)
97
+ assert out["warnings"] == []
98
+
99
+ def test_failure_is_reported_not_raised(self, frame):
100
+ bad = {"id": "bad", "type": "TypeCastTransformer", "columns": ["plan"], "params": {"dtypes": {"plan": "float"}, "errors": "raise"}}
101
+ out = preview_step(bad, frame)
102
+ assert out["error"] is not None or out["delta"]["updated"] == ["plan"]
103
+
104
+ def test_samples_large_frames(self, frame):
105
+ big = pd.concat([frame] * 100, ignore_index=True)
106
+ out = preview_step(DROP_ID, big, max_rows=500)
107
+ assert out["rows_before"] == 500
108
+
109
+
110
+ class TestPreviewSpec:
111
+ def test_steps_apply_cumulatively_and_report_safety(self, frame):
112
+ spec = {"steps": [DROP_ID, CAST_TOTAL, DROP_LABEL]}
113
+ out = preview_spec(spec, frame, label="churn")
114
+ assert [s["step_id"] for s in out["steps"]] == ["drop_id", "cast_total", "drop_label"]
115
+ assert out["steps"][0]["delta"]["dropped"] == ["id"]
116
+ assert out["steps"][1]["delta"]["updated"] == ["total"]
117
+ assert "id" not in out["output_columns"]
118
+ assert out["safety_errors"] and "drop_label" in out["safety_errors"][0]
119
+ assert out["rows_before"] == out["rows_after"] == len(frame)
@@ -28,3 +28,26 @@ class TestExpressionTransformer:
28
28
  df = pd.DataFrame({"a": [100, 200], "b": [50, 60]})
29
29
  result = t.fit_transform(df)
30
30
  assert list(result["result"]) == [5.0, 12.0]
31
+
32
+ def test_backtick_quoted_column_with_spaces(self):
33
+ # pandas' expression parser mangles `Monthly Charges` into the
34
+ # identifier BACKTICK_QUOTED_STRING_Monthly_Charges before lookup, so
35
+ # local_dict must also be reachable under that mangled name.
36
+ t = ExpressionTransformer(
37
+ expression="`Monthly Charges` * Tenure",
38
+ output_column="Total Charges",
39
+ )
40
+ df = pd.DataFrame({"Monthly Charges": [10.0, 20.0, 30.0], "Tenure": [1, 2, 3]})
41
+ result = t.fit_transform(df)
42
+ assert list(result["Total Charges"]) == [10.0, 40.0, 90.0]
43
+ assert "Monthly Charges" in result.columns
44
+ assert "Tenure" in result.columns
45
+
46
+ def test_df_indexing_style_with_spaced_column_still_works(self):
47
+ t = ExpressionTransformer(
48
+ expression="df['Monthly Charges'] * 2",
49
+ output_column="Doubled",
50
+ )
51
+ df = pd.DataFrame({"Monthly Charges": [10.0, 20.0, 30.0]})
52
+ result = t.fit_transform(df)
53
+ assert list(result["Doubled"]) == [20.0, 40.0, 60.0]
@@ -1 +0,0 @@
1
- .env