xplainable-preprocessing 0.3.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xplainable_preprocessing-0.4.0/.github/workflows/ci.yml +90 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/.gitignore +2 -0
- xplainable_preprocessing-0.4.0/CHANGELOG.md +231 -0
- xplainable_preprocessing-0.4.0/PKG-INFO +200 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/README.md +21 -6
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/pyproject.toml +13 -2
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/__init__.py +77 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/compiler.py +20 -1
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/contracts.py +1286 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/__init__.py +181 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/content.py +1896 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/structure.py +1307 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/target.py +1093 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/types.py +242 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/guard.py +1099 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/io.py +1292 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/numeric_text.py +909 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/pipeline.py +180 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/preview.py +254 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/registry.py +265 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/sandbox.py +345 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/serialization.py +139 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/__init__.py +14 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/_util.py +74 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/clip.py +101 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/cross_categories.py +177 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/datetime_extract.py +303 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/expression.py +625 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/group_broadcast.py +335 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/grouped_lag.py +88 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/null_tokens.py +182 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/ordinal_map.py +337 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/parse_numeric.py +174 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/rolling_agg.py +155 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/text_clean.py +177 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/text_features.py +349 -0
- xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/time_lag.py +627 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/README.md +38 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/adult_income.csv +1201 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/ames_housing.csv +580 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/bank_marketing.csv +1003 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/bank_marketing_raw.csv +201 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/bike_sharing_hour.csv +1001 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/credit_default.csv +1002 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/diamonds.csv +1004 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/heart_cleveland.csv +303 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/insurance.csv +1339 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/iris.csv +151 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/lending_club.csv +217 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/metro_traffic.csv +1525 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/sms_spam.csv +601 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/student_math.csv +396 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/titanic.csv +892 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/eval/winequality-red.csv +1600 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/adult_placeholders.csv +41 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/ames_literal_none.csv +39 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/bank_semicolon.csv +31 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/cp1252_semicolon.csv +7 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/credit_two_header_rows.xls +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/credit_two_header_rows.xlsx +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/expected.json +221 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/heart_headerless.csv +34 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/metro_holiday.csv +24 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/mixed_types.xlsx +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/pipe.txt +16 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/title_rows.csv +12 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/io/utf16_tab.txt +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/builtin_pipeline.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_missing_flag.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_rename.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_wrapped_expression.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/custom_step_by_value.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/datetime_unwrapped.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/expression_outside_allow_list.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/grouped_lag_order_by.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/make_fixtures.py +258 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/manifest.json +21 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/rolling_order_by.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/sklearn_steps.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/text_clean_mixed_types.pkl +0 -0
- xplainable_preprocessing-0.4.0/tests/portability_roundtrip.py +312 -0
- xplainable_preprocessing-0.4.0/tests/test_contracts.py +701 -0
- xplainable_preprocessing-0.4.0/tests/test_detect_content.py +990 -0
- xplainable_preprocessing-0.4.0/tests/test_detect_structure.py +764 -0
- xplainable_preprocessing-0.4.0/tests/test_detect_target.py +671 -0
- xplainable_preprocessing-0.4.0/tests/test_detect_types.py +133 -0
- xplainable_preprocessing-0.4.0/tests/test_guard.py +591 -0
- xplainable_preprocessing-0.4.0/tests/test_io.py +783 -0
- xplainable_preprocessing-0.4.0/tests/test_legacy_pickles.py +201 -0
- xplainable_preprocessing-0.4.0/tests/test_numeric_text.py +481 -0
- xplainable_preprocessing-0.4.0/tests/test_pipeline.py +189 -0
- xplainable_preprocessing-0.4.0/tests/test_portable_custom.py +478 -0
- xplainable_preprocessing-0.4.0/tests/test_preview.py +219 -0
- xplainable_preprocessing-0.4.0/tests/test_registry.py +118 -0
- xplainable_preprocessing-0.4.0/tests/test_release.py +95 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_clip.py +91 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_cross_categories.py +147 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_datetime_extract.py +210 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_expression.py +334 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_group_broadcast.py +221 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_null_tokens.py +171 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_ordinal_map.py +264 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_parse_numeric.py +195 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_text_clean.py +107 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_text_features.py +229 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_time_lag.py +418 -0
- xplainable_preprocessing-0.4.0/tests/test_transformers/test_time_order.py +211 -0
- xplainable_preprocessing-0.3.1/PKG-INFO +0 -14
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/__init__.py +0 -34
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/guard.py +0 -282
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/pipeline.py +0 -93
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/preview.py +0 -137
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/registry.py +0 -114
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/sandbox.py +0 -113
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/serialization.py +0 -20
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/clip.py +0 -45
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -72
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/expression.py +0 -51
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -55
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -79
- xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/text_clean.py +0 -70
- xplainable_preprocessing-0.3.1/tests/test_guard.py +0 -119
- xplainable_preprocessing-0.3.1/tests/test_pipeline.py +0 -73
- xplainable_preprocessing-0.3.1/tests/test_preview.py +0 -90
- xplainable_preprocessing-0.3.1/tests/test_transformers/test_expression.py +0 -53
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/.github/workflows/publish-pypi.yml +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/dag-pipeline-proposal.md +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/feature-pipeline-architectures.md +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/feature-store-proposal.md +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/schema.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/__init__.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_compiler.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_sandbox.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_schema.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_serialization.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_transformers/__init__.py +0 -0
- {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_transformers/test_all_transformers.py +0 -0
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
# Tests on both Pythons the platform runs (core-api and the inference server
|
|
4
|
+
# on 3.10, the train service on 3.12), plus a cross-version check that a
|
|
5
|
+
# pipeline with custom steps fitted and saved under one Python loads and
|
|
6
|
+
# transforms identically under the other.
|
|
7
|
+
|
|
8
|
+
on:
|
|
9
|
+
push:
|
|
10
|
+
branches: [main, 'release/**']
|
|
11
|
+
pull_request:
|
|
12
|
+
|
|
13
|
+
permissions:
|
|
14
|
+
contents: read
|
|
15
|
+
|
|
16
|
+
jobs:
|
|
17
|
+
test:
|
|
18
|
+
name: pytest (Python ${{ matrix.python-version }})
|
|
19
|
+
runs-on: ubuntu-latest
|
|
20
|
+
strategy:
|
|
21
|
+
fail-fast: false
|
|
22
|
+
matrix:
|
|
23
|
+
include:
|
|
24
|
+
# core-api / inference-server stack
|
|
25
|
+
- python-version: '3.10'
|
|
26
|
+
pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
|
|
27
|
+
# train-service stack
|
|
28
|
+
- python-version: '3.12'
|
|
29
|
+
pins: 'pandas==2.3.3'
|
|
30
|
+
steps:
|
|
31
|
+
- uses: actions/checkout@v4
|
|
32
|
+
- uses: actions/setup-python@v5
|
|
33
|
+
with:
|
|
34
|
+
python-version: ${{ matrix.python-version }}
|
|
35
|
+
- name: Install
|
|
36
|
+
run: pip install -e '.[dev]' ${{ matrix.pins }}
|
|
37
|
+
- name: Test
|
|
38
|
+
run: python -m pytest -q
|
|
39
|
+
|
|
40
|
+
portability-dump:
|
|
41
|
+
name: dump under Python ${{ matrix.python-version }}
|
|
42
|
+
runs-on: ubuntu-latest
|
|
43
|
+
strategy:
|
|
44
|
+
matrix:
|
|
45
|
+
include:
|
|
46
|
+
- python-version: '3.10'
|
|
47
|
+
pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
|
|
48
|
+
- python-version: '3.12'
|
|
49
|
+
pins: 'pandas==2.3.3'
|
|
50
|
+
steps:
|
|
51
|
+
- uses: actions/checkout@v4
|
|
52
|
+
- uses: actions/setup-python@v5
|
|
53
|
+
with:
|
|
54
|
+
python-version: ${{ matrix.python-version }}
|
|
55
|
+
- name: Install
|
|
56
|
+
run: pip install -e . ${{ matrix.pins }}
|
|
57
|
+
- name: Fit and save the bike_sharing custom-step pipeline
|
|
58
|
+
run: python tests/portability_roundtrip.py dump dump
|
|
59
|
+
- uses: actions/upload-artifact@v4
|
|
60
|
+
with:
|
|
61
|
+
name: pipeline-py${{ matrix.python-version }}
|
|
62
|
+
path: dump/
|
|
63
|
+
|
|
64
|
+
portability-load:
|
|
65
|
+
name: load a Python ${{ matrix.written-by }} binary under Python ${{ matrix.python-version }}
|
|
66
|
+
needs: portability-dump
|
|
67
|
+
runs-on: ubuntu-latest
|
|
68
|
+
strategy:
|
|
69
|
+
fail-fast: false
|
|
70
|
+
matrix:
|
|
71
|
+
include:
|
|
72
|
+
- python-version: '3.12'
|
|
73
|
+
written-by: '3.10'
|
|
74
|
+
pins: 'pandas==2.3.3'
|
|
75
|
+
- python-version: '3.10'
|
|
76
|
+
written-by: '3.12'
|
|
77
|
+
pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
|
|
78
|
+
steps:
|
|
79
|
+
- uses: actions/checkout@v4
|
|
80
|
+
- uses: actions/setup-python@v5
|
|
81
|
+
with:
|
|
82
|
+
python-version: ${{ matrix.python-version }}
|
|
83
|
+
- name: Install
|
|
84
|
+
run: pip install -e . ${{ matrix.pins }}
|
|
85
|
+
- uses: actions/download-artifact@v4
|
|
86
|
+
with:
|
|
87
|
+
name: pipeline-py${{ matrix.written-by }}
|
|
88
|
+
path: dump/
|
|
89
|
+
- name: Load, transform and compare
|
|
90
|
+
run: python tests/portability_roundtrip.py load dump
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to xplainable-preprocessing. Versions follow semantic
|
|
4
|
+
versioning; before 1.0 a minor release may break compatibility, and every
|
|
5
|
+
break is listed under **Breaking changes**.
|
|
6
|
+
|
|
7
|
+
## [0.4.0] - 2026-09-28
|
|
8
|
+
|
|
9
|
+
A breaking release. Pipelines now refuse to produce two columns with the same
|
|
10
|
+
name, expressions run on an allow-listed evaluator instead of `pandas.eval`,
|
|
11
|
+
several transformers keep data they used to destroy, and custom steps pickle
|
|
12
|
+
so a binary loads on another Python version. It also adds seven transformers
|
|
13
|
+
and four modules: `detect`, `io`, `numeric_text` and `contracts`.
|
|
14
|
+
|
|
15
|
+
Supported: Python 3.9 and later (tested on 3.10 and 3.12), pandas 2.x
|
|
16
|
+
(`pandas>=2.0,<3`), numpy 1.24 or later, scikit-learn 1.3 or later.
|
|
17
|
+
|
|
18
|
+
### Breaking changes
|
|
19
|
+
|
|
20
|
+
| Area | Change |
|
|
21
|
+
|---|---|
|
|
22
|
+
| Column collisions | `DataFrameColumnTransformer` raises `ColumnCollisionError` when a wrapped step emits a column that is also a pass-through column. `DataFramePipeline` raises it, with the step id, when any step's output repeats a column name (for example `RenameColumnsTransformer` renaming onto an existing column). 0.3.x returned a frame with two columns of that name. The error carries `columns`, `step` and `inputs`, and pickles intact. |
|
|
23
|
+
| Expressions | `ExpressionTransformer` parses the expression with `ast` and evaluates it with an allow-list: arithmetic, comparisons, element-wise logic, the `pandas.eval` math functions, column references (bare, backticked, `df['x']`, `df.x`), and the methods `.where .abs .round .isna .notna .isnull .notnull .fillna .clip .between .isin .astype` (restricted dtypes) plus `.str.len/count/contains/startswith/endswith/lower/upper/strip`. Anything else (other methods such as `.cumsum()`, `.shift()`, `.rank()` or `.pow()`, dunders, I/O) raises `ExpressionError` at fit and at transform. Nothing reaches `eval` or `pandas.eval`. A column named `df` is read as `` `df` `` or `df['df']`. |
|
|
24
|
+
| `save_pipeline` | Defaults to `portable=True` and raises `NonPortablePipelineError`, naming the steps, when any fitted state would be pickled by value (a lambda, or a function or class from custom code kept on `self`). `portable=False` saves as 0.3.x did. |
|
|
25
|
+
| TextCleanTransformer | Operations apply to `str` cells only. Numbers, None and other values in a mixed column are kept; 0.3.x turned them into NaN. `string` and `category` columns are cleaned too (a category column stays categorical). An unknown operation raises at fit; 0.3.x ignored it. |
|
|
26
|
+
| GroupedLagTransformer, RollingAggTransformer | Compute on a stably sorted copy and return rows in input order with the input index. With `order_by`, 0.3.x returned the rows sorted by group and `order_by`. Each row's values are unchanged, but row order is not. Negative lag periods and missing `group_by`/`order_by` columns raise at fit. Categorical group keys use `observed=True`. |
|
|
27
|
+
| DateTimeExtractTransformer without step columns | At fit it picks datetime columns, and text columns where at least 80% of values parse as dates. Only those are parsed and dropped. Numeric, bool and other text columns are left untouched; 0.3.x converted every column to dates, extracted parts and dropped it. Fit raises when no column holds dates. |
|
|
28
|
+
| ClipTransformer | Bounds are validated at fit: a non-number, `min > max`, or a `bounds` column that is missing or not numeric raises. |
|
|
29
|
+
| `generate_catalog()` | Lines show typed params (`name: type = default`). A required param is shown without a default, and a choice between params is spelled out ("Requires order or mapping."). Code that parses catalog lines must follow the new format. |
|
|
30
|
+
| Guard and previews | `step_safety_error` and `spec_safety_errors` also reject steps that reorder rows, read the label, or copy the label (see Added). Checks and previews run on the whole frame up to `full_rows_limit` (100k) rows, where 0.3.x used a random 10k-row sample. They see every row, and take longer on large frames. |
|
|
31
|
+
|
|
32
|
+
#### Stored 0.3.x pipelines
|
|
33
|
+
|
|
34
|
+
Built-in steps pickle by reference, so a 0.3.x binary loads into the 0.4.0
|
|
35
|
+
classes. The 0.3.x pickles of `RollingAggTransformer`, `ClipTransformer` and
|
|
36
|
+
`DateTimeExtractTransformer` are filled in with the new params' defaults. A
|
|
37
|
+
0.3.x pipeline of built-in steps serves the same output under 0.4.0
|
|
38
|
+
(`tests/test_legacy_pickles.py` checks this on binaries written by 0.3.1)
|
|
39
|
+
with these exceptions:
|
|
40
|
+
|
|
41
|
+
| Stored pipeline | Under 0.4.0 | What to do |
|
|
42
|
+
|---|---|---|
|
|
43
|
+
| A wrapped step whose output name already exists, or a rename onto an existing column | `transform` raises `ColumnCollisionError` (0.3.x served duplicate columns) | Re-fit with a new output name, or drop the existing column first |
|
|
44
|
+
| An expression outside the allow-list | `transform` raises `ExpressionError` | Rewrite it with allowed constructs (`a ** 2` for `a.pow(2)`) and re-fit, or ask for the method to be allow-listed |
|
|
45
|
+
| TextCleanTransformer on a column mixing text and other values | Non-text cells are kept, so the output changes | Re-fit, or accept the change |
|
|
46
|
+
| GroupedLagTransformer or RollingAggTransformer with `order_by` | Rows come back in input order | Re-fit, and check anything that relied on the sorted order |
|
|
47
|
+
| DateTimeExtractTransformer without step columns | Non-date columns are kept, so the output columns change. A 0.3.x binary records no date columns, so each served batch decides which columns are dates: a single row with an empty date keeps that column and gets no date parts | Re-fit |
|
|
48
|
+
| Custom steps saved by 0.3.x (pickled by value) | Load and serve on the Python version that wrote them, as before; on another Python version `load_pipeline` still fails (typically a `TypeError` from the pickled code). `save_pipeline` refuses them unless `portable=False` | Re-fit to get a portable binary |
|
|
49
|
+
|
|
50
|
+
Steps from scikit-learn (`OneHotEncoder`, `StandardScaler`, `SimpleImputer`,
|
|
51
|
+
...) pickle as scikit-learn's own objects. They load only under the
|
|
52
|
+
scikit-learn release that fitted them (a 1.6.1 `SimpleImputer` fails under
|
|
53
|
+
1.9), whichever xplainable-preprocessing release is installed.
|
|
54
|
+
|
|
55
|
+
### Added
|
|
56
|
+
|
|
57
|
+
**Transformers** (registered, in the catalog, with required params in
|
|
58
|
+
`REQUIRED_PARAMS`):
|
|
59
|
+
- `ParseNumericTransformer`: numbers stored as text into floats. Handles `%`,
|
|
60
|
+
currency, `1,234`, `(1,234)`, units such as `36 months`, ranges, bounds
|
|
61
|
+
such as `10+` and NA tokens. Idempotent on numbers.
|
|
62
|
+
- `NullTokensTransformer`: placeholder text (`?`, `N/A`, blanks) and sentinel
|
|
63
|
+
numbers (`999`) to NaN, with optional 0/1 flags.
|
|
64
|
+
- `OrdinalMapTransformer`: an ordered scale to numbers in place, by `order`
|
|
65
|
+
or `mapping`. NaN- and unknown-safe, with JSON-safe params. It rejects a
|
|
66
|
+
mapping that gives one level two codes.
|
|
67
|
+
- `CrossCategoriesTransformer`: 2 to 3 low-cardinality columns crossed into
|
|
68
|
+
one readable category, with zero padding and a `max_levels` cap.
|
|
69
|
+
- `TimeLagTransformer`: lags and trailing-window aggregates at exact time
|
|
70
|
+
offsets over strictly earlier timestamps, per group. Supports `gap`,
|
|
71
|
+
deduplication of repeated timestamps, a bounded history tail and
|
|
72
|
+
`serving_requirements()`. A `gap` that covers every lag is the one safe way
|
|
73
|
+
to use past target values.
|
|
74
|
+
- `GroupBroadcastTransformer`: a group aggregate written to every row of its
|
|
75
|
+
group. A group can be a calendar date, hour, week or month of a time
|
|
76
|
+
column, and the mapping is kept from fit so single-row serving works.
|
|
77
|
+
- `TextFeaturesTransformer`: free text into numeric features (length, words,
|
|
78
|
+
digits, capitals, punctuation, URLs, emails, phone-like numbers, whole-word
|
|
79
|
+
keyword flags). The feature set is frozen as `feature_set_version=1`.
|
|
80
|
+
|
|
81
|
+
**Transformer options:**
|
|
82
|
+
- `RollingAggTransformer`: `closed` (`"left"` excludes the current row) and
|
|
83
|
+
`shift`. List params accept a single name.
|
|
84
|
+
- `GroupedLagTransformer`: `order_by` takes a list, and scalar params are
|
|
85
|
+
coerced.
|
|
86
|
+
- `DateTimeExtractTransformer`: components `elapsed_days`, `elapsed_hours`
|
|
87
|
+
(from an `origin` fixed at fit), `hour_of_week`, `date` and `month_index`,
|
|
88
|
+
plus `as_categorical` (zero-padded strings). The text format is fixed at
|
|
89
|
+
fit, and ISO 8601 values it misses are still read.
|
|
90
|
+
- `ClipTransformer`: per-column `bounds` (`{"x": [3, 11], "z": [1.5, None]}`).
|
|
91
|
+
- `TextCleanTransformer`: operations `html_unescape`, `fix_encoding` (cp1252
|
|
92
|
+
mojibake), `nfkc`, `mask_digits`, `mask_urls` and `mask_emails`.
|
|
93
|
+
|
|
94
|
+
**Modules:**
|
|
95
|
+
- `numeric_text`: `parse_numeric` and `numeric_text_report`, which says
|
|
96
|
+
whether a text column holds numbers. The report rejects dates, codes with
|
|
97
|
+
leading zeros and identifiers, and reports collisions and mixed units. The
|
|
98
|
+
module also holds the NA-token vocabulary (`NA_TOKENS`, `MAYBE_NA_TOKENS`).
|
|
99
|
+
- `io.read_table(content, filename, options) -> (DataFrame, ReadReport)`:
|
|
100
|
+
one reader for uploaded CSV and Excel files.
|
|
101
|
+
- Encoding: UTF-8 with cp1252 and latin-1 fallbacks.
|
|
102
|
+
- Delimiter sniffing, with a guard against single-column reads.
|
|
103
|
+
- Header rows below the top and headerless files are detected from the
|
|
104
|
+
raw bytes and read again losslessly.
|
|
105
|
+
- NA policies: `literal_v1` (the default; only empty cells are missing,
|
|
106
|
+
and text tokens stay literal) and `legacy` (pandas' defaults).
|
|
107
|
+
- Typed Excel with a safe 99% numeric coercion.
|
|
108
|
+
- `ReadError` subclasses carry messages meant for users.
|
|
109
|
+
- `coerce_numeric_columns` applies the same coercion to frames already
|
|
110
|
+
stored.
|
|
111
|
+
- Excel needs the new `excel` extra: `pip install
|
|
112
|
+
"xplainable-preprocessing[excel]"`.
|
|
113
|
+
- `detect`: pure functions that describe a raw frame and return JSON-safe
|
|
114
|
+
`Issue`s. Each issue has a severity, a confidence and, where a library
|
|
115
|
+
step fixes it, `suggested_steps`.
|
|
116
|
+
- `scan_frame`: column content. Placeholder and maybe-NA tokens, numbers
|
|
117
|
+
stored as text, sentinels and top or bottom codes, structural and
|
|
118
|
+
co-missing gaps, whitespace and case variants, impossible years and
|
|
119
|
+
broken year orderings, ordinal scales, free text, sparse markers,
|
|
120
|
+
impossible zeros, and a regression target's point mass.
|
|
121
|
+
- `scan_structure`: frame shape. A delimited file read as one column,
|
|
122
|
+
headerless files (pandas' `.N` suffixes undone), header rows inside the
|
|
123
|
+
data, repeated headers, junk rows, and identifiers measured against
|
|
124
|
+
distinct rows.
|
|
125
|
+
- `scan_relations`: needs a target. Target proxies, columns that rebuild
|
|
126
|
+
the target, time-order drift, duplicate keys whose target is constant,
|
|
127
|
+
and 1:1 or functional redundancy.
|
|
128
|
+
- `scan_all` runs all three.
|
|
129
|
+
- `repair_frame` applies only lossless repairs to frames already stored.
|
|
130
|
+
- `contracts`:
|
|
131
|
+
- `dependency_map(fitted, spec, data)`: for each engineered output column,
|
|
132
|
+
the step that computed it, its inputs and raw sources, and whether it
|
|
133
|
+
can be recomputed row by row. For ExpressionTransformer outputs it also
|
|
134
|
+
gives a `DataFrame.eval` formula, verified on the rows, that
|
|
135
|
+
xplainable-gm's optimiser accepts as a `derived` declaration.
|
|
136
|
+
`DependencyMap.is_engineered(column)` checks a single column.
|
|
137
|
+
- `partial_transform(fitted, base_prepared, raw_rows, changed_columns)`:
|
|
138
|
+
recomputes only the outputs that a changed raw column feeds.
|
|
139
|
+
- Both work on 0.3.x pipelines when the spec is passed.
|
|
140
|
+
|
|
141
|
+
**Pipelines and serialisation:**
|
|
142
|
+
- `compile_spec` attaches the spec to the pipeline as `pipeline.spec`, as a
|
|
143
|
+
plain dict.
|
|
144
|
+
- Custom steps pickle as their spec plus fitted state. On load, the
|
|
145
|
+
module-level `sandbox.rebuild_custom` validates and runs the source again,
|
|
146
|
+
so a pipeline saved on Python 3.10 loads on 3.12 and the reverse. Copies
|
|
147
|
+
keep the original class.
|
|
148
|
+
- A custom step's constructor `TypeError` names the params its `__init__`
|
|
149
|
+
does not accept.
|
|
150
|
+
|
|
151
|
+
**Guard and previews** (`guard`, `preview`):
|
|
152
|
+
- `step_safety_error` and `spec_safety_errors` reject four more kinds of
|
|
153
|
+
step:
|
|
154
|
+
- a step that reorders rows;
|
|
155
|
+
- a step that reads the label. This is checked statically (columns,
|
|
156
|
+
params, expression names, custom-code AST) and behaviourally (transform
|
|
157
|
+
with the label dropped or shuffled). A `TimeLagTransformer` whose `gap`
|
|
158
|
+
covers every lag is exempt (`LABEL_READ_EXEMPTIONS`).
|
|
159
|
+
- a step that adds a copy of the label: the new column equals the label
|
|
160
|
+
on at least 50% of rows, or has |Spearman rho| >= 0.995. A preview
|
|
161
|
+
warns from 5% up.
|
|
162
|
+
- a step whose output collides with an existing column.
|
|
163
|
+
- `preview_step` and `preview_spec` add:
|
|
164
|
+
- a `label` parameter;
|
|
165
|
+
- `serving_warnings`, from a single-row batch-dependence probe;
|
|
166
|
+
- `changes`, with `changed_cells` and `top_changes` for each updated
|
|
167
|
+
column;
|
|
168
|
+
- `top_values` for non-numeric columns;
|
|
169
|
+
- a warning when a step changes 0 cells;
|
|
170
|
+
- `rows_total` and `sampled`.
|
|
171
|
+
- Past 100k rows, checks and previews use an order-preserving sample. Lag,
|
|
172
|
+
rolling, group-statistic and custom steps get a contiguous block; any
|
|
173
|
+
other step gets rows spread over the frame that always include the
|
|
174
|
+
extreme rows of its columns.
|
|
175
|
+
- `registry.missing_required_params` and `REQUIRED_PARAMS` list what each
|
|
176
|
+
step type must set. `registry.catalog_entry` builds one catalog line.
|
|
177
|
+
|
|
178
|
+
**Package:**
|
|
179
|
+
- `__version__`. The top level now exports `ColumnCollisionError`,
|
|
180
|
+
`NonPortablePipelineError`, `ExpressionError`, `read_table`, `ReadReport`,
|
|
181
|
+
`ReadError`, `Issue`, `scan_all`, `parse_numeric` and
|
|
182
|
+
`numeric_text_report`. The `detect`, `io` and `numeric_text` submodules are
|
|
183
|
+
reachable as attributes.
|
|
184
|
+
- CI runs the suite on Python 3.10 (pandas 2.3.3, numpy 2.0.2,
|
|
185
|
+
scikit-learn 1.6.1) and 3.12, plus a matrix that dumps a pipeline on 3.10
|
|
186
|
+
and loads it on 3.12, and the reverse.
|
|
187
|
+
|
|
188
|
+
### Fixed
|
|
189
|
+
|
|
190
|
+
- `infer_schema`, `compute_preview` and `compute_step_deltas` returned samples
|
|
191
|
+
that `json.dumps` rejects when a column held timestamps, timedeltas, periods,
|
|
192
|
+
numpy scalars or pandas NA values. Storing such a schema failed, e.g. after a
|
|
193
|
+
`DateTimeExtractTransformer` with `drop_original=False`. Samples are now
|
|
194
|
+
JSON-safe: missing values become null, timestamps and dates ISO-8601 text,
|
|
195
|
+
timedeltas ISO-8601 durations, numpy scalars Python values.
|
|
196
|
+
- A step whose output repeated a column name produced a frame with two
|
|
197
|
+
columns of that name, and anything downstream then failed or read the
|
|
198
|
+
wrong one. It now raises (see Breaking changes).
|
|
199
|
+
- `GroupedLagTransformer` and `RollingAggTransformer` misaligned rows when
|
|
200
|
+
`order_by` sorted them. `RollingAggTransformer` crashed on categorical
|
|
201
|
+
group keys with unused categories.
|
|
202
|
+
- Unwrapped `DateTimeExtractTransformer` destroyed every non-date column.
|
|
203
|
+
- `TextCleanTransformer` erased numbers in mixed columns.
|
|
204
|
+
- Custom-step pipelines saved on one Python version failed to load on
|
|
205
|
+
another (`code() argument 13 must be str, not int`).
|
|
206
|
+
- `preview_step` no longer raises when statistics or deltas fail, and
|
|
207
|
+
repeated input column names come back as an error.
|
|
208
|
+
|
|
209
|
+
### Known limitations
|
|
210
|
+
|
|
211
|
+
- pandas 3 is not supported, and the dependency is capped at `<3`. Under
|
|
212
|
+
pandas 3.0 the suite fails in four places:
|
|
213
|
+
- `detect.target` writes into arrays that copy-on-write makes read-only,
|
|
214
|
+
so most relation scans raise.
|
|
215
|
+
- `detect.structure`'s headerless restore handles missing values in the
|
|
216
|
+
default `str` dtype differently.
|
|
217
|
+
- `ExpressionTransformer`'s `.str` methods also handle missing values in
|
|
218
|
+
the `str` dtype differently.
|
|
219
|
+
- `CategoryCondenseTransformer` fits only `object` and `category`
|
|
220
|
+
columns, so it leaves `str` columns uncondensed.
|
|
221
|
+
|
|
222
|
+
## [0.3.1]
|
|
223
|
+
|
|
224
|
+
- `ExpressionTransformer` resolves backtick-quoted column names.
|
|
225
|
+
|
|
226
|
+
## [0.3.0]
|
|
227
|
+
|
|
228
|
+
- Step safety guard (`step_safety_error`, `spec_safety_errors`), step and
|
|
229
|
+
spec previews (`preview_step`, `preview_spec`), and `safe_catalog`.
|
|
230
|
+
|
|
231
|
+
Earlier releases are described in the git history.
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: xplainable-preprocessing
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Shared preprocessing pipeline package for xplainable
|
|
5
|
+
Requires-Python: >=3.9
|
|
6
|
+
Requires-Dist: cloudpickle>=3.0
|
|
7
|
+
Requires-Dist: numpy>=1.24
|
|
8
|
+
Requires-Dist: pandas<3,>=2.0
|
|
9
|
+
Requires-Dist: pydantic>=2.0
|
|
10
|
+
Requires-Dist: scikit-learn>=1.3
|
|
11
|
+
Requires-Dist: scipy>=1.8
|
|
12
|
+
Provides-Extra: dev
|
|
13
|
+
Requires-Dist: openpyxl>=3.1; extra == 'dev'
|
|
14
|
+
Requires-Dist: pytest-cov; extra == 'dev'
|
|
15
|
+
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
16
|
+
Requires-Dist: xlrd>=2.0.1; extra == 'dev'
|
|
17
|
+
Provides-Extra: excel
|
|
18
|
+
Requires-Dist: openpyxl>=3.1; extra == 'excel'
|
|
19
|
+
Requires-Dist: xlrd>=2.0.1; extra == 'excel'
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# xplainable-preprocessing
|
|
23
|
+
|
|
24
|
+
Shared preprocessing pipeline library for [xplainable](https://xplainable.io). Define, compile, fit, serialise, and apply data transformation pipelines using a declarative JSON spec.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install xplainable-preprocessing
|
|
30
|
+
pip install "xplainable-preprocessing[excel]" # io.read_table for .xlsx/.xls
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Upgrading from 0.3.x? 0.4.0 has breaking changes; see [CHANGELOG.md](CHANGELOG.md).
|
|
34
|
+
|
|
35
|
+
## Quick Start
|
|
36
|
+
|
|
37
|
+
### Define a pipeline as JSON
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from xplainable_preprocessing.schema import PipelineSpec
|
|
41
|
+
from xplainable_preprocessing.compiler import compile_spec
|
|
42
|
+
|
|
43
|
+
spec = PipelineSpec(
|
|
44
|
+
version="2.0",
|
|
45
|
+
steps=[
|
|
46
|
+
{"id": "drop_ids", "type": "DropColumnsTransformer", "params": {"columns": ["customer_id", "row_id"]}},
|
|
47
|
+
{"id": "fill_numeric", "type": "FillMissingTransformer", "columns": ["tenure", "charges"], "params": {"strategies": {"tenure": "median", "charges": "median"}}},
|
|
48
|
+
{"id": "clip_outliers", "type": "ClipTransformer", "columns": ["charges"], "params": {"min_val": 0, "max_val": 500}},
|
|
49
|
+
{"id": "missing_flags", "type": "MissingFlagTransformer", "columns": ["tenure", "charges"]},
|
|
50
|
+
{"id": "condense_plan", "type": "CategoryCondenseTransformer", "columns": ["plan_type"], "params": {"max_categories": 10}},
|
|
51
|
+
{"id": "extract_dates", "type": "DateTimeExtractTransformer", "columns": ["signup_date"], "params": {"components": ["month", "dayofweek"], "drop_original": true}},
|
|
52
|
+
]
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
pipeline = compile_spec(spec)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Fit and transform
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
import pandas as pd
|
|
62
|
+
|
|
63
|
+
df = pd.read_csv("data.csv")
|
|
64
|
+
pipeline.fit(df)
|
|
65
|
+
transformed = pipeline.transform(df)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
### Serialise and load
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from xplainable_preprocessing.serialization import save_pipeline, load_pipeline
|
|
72
|
+
|
|
73
|
+
# Save
|
|
74
|
+
binary = save_pipeline(pipeline)
|
|
75
|
+
|
|
76
|
+
# Load
|
|
77
|
+
loaded = load_pipeline(binary)
|
|
78
|
+
result = loaded.transform(new_data)
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Available Transformers
|
|
82
|
+
|
|
83
|
+
### Data Cleaning
|
|
84
|
+
| Transformer | Description |
|
|
85
|
+
|-------------|-------------|
|
|
86
|
+
| `DropColumnsTransformer` | Drop specified columns |
|
|
87
|
+
| `RenameColumnsTransformer` | Rename columns |
|
|
88
|
+
| `TypeCastTransformer` | Cast column dtypes (e.g. str to float64, with error coercion) |
|
|
89
|
+
| `ClipTransformer` | Set numeric values outside (min, max), or per-column `bounds`, to NaN |
|
|
90
|
+
| `ParseNumericTransformer` | Parse numbers stored as text (%, currency, `1,234`, `36 months`, ranges, `10+`) |
|
|
91
|
+
| `NullTokensTransformer` | Turn placeholder text (`?`, `N/A`) and sentinel numbers (`999`) into NaN |
|
|
92
|
+
|
|
93
|
+
### Missing Values
|
|
94
|
+
| Transformer | Description |
|
|
95
|
+
|-------------|-------------|
|
|
96
|
+
| `FillMissingTransformer` | Fill nulls with median, mode, constant, or other strategies |
|
|
97
|
+
| `MissingFlagTransformer` | Create binary 0/1 indicator columns for nulls |
|
|
98
|
+
|
|
99
|
+
### Feature Engineering
|
|
100
|
+
| Transformer | Description |
|
|
101
|
+
|-------------|-------------|
|
|
102
|
+
| `ExpressionTransformer` | Create derived columns from expressions (allow-listed evaluator) |
|
|
103
|
+
| `DateTimeExtractTransformer` | Extract year, month, dayofweek, elapsed time, etc. from datetime columns |
|
|
104
|
+
| `GroupByAggTransformer` | Aggregate features by group (count, sum, mean) |
|
|
105
|
+
| `GroupedLagTransformer` | Create row-shift lags within groups |
|
|
106
|
+
| `RollingAggTransformer` | Rolling window aggregations |
|
|
107
|
+
| `TimeLagTransformer` | Lags and trailing windows at exact time offsets over strictly earlier rows |
|
|
108
|
+
| `GroupBroadcastTransformer` | Write a group aggregate (e.g. a holiday flag per date) to every row of the group |
|
|
109
|
+
| `TextFeaturesTransformer` | Numeric features from free text (length, digits, URLs, keyword flags) |
|
|
110
|
+
|
|
111
|
+
### Categorical
|
|
112
|
+
| Transformer | Description |
|
|
113
|
+
|-------------|-------------|
|
|
114
|
+
| `CategoryCondenseTransformer` | Condense high-cardinality categoricals to top N + "Other" |
|
|
115
|
+
| `TextCleanTransformer` | Clean text cells: lowercase, strip, remove HTML/extra whitespace, fix encoding, mask digits/URLs/emails |
|
|
116
|
+
| `OrdinalMapTransformer` | Map an ordered scale (grades, ratings, bands) to numbers |
|
|
117
|
+
| `CrossCategoriesTransformer` | Cross 2-3 low-cardinality columns into one category |
|
|
118
|
+
|
|
119
|
+
### Scaling (use with caution -- breaks xplainable model explainability)
|
|
120
|
+
| Transformer | Description |
|
|
121
|
+
|-------------|-------------|
|
|
122
|
+
| `StandardScaler` | Standardise to zero mean, unit variance |
|
|
123
|
+
| `MinMaxScaler` | Scale to [0, 1] range |
|
|
124
|
+
| `RobustScaler` | Scale using median and IQR |
|
|
125
|
+
| `PowerTransformer` | Apply power transform for normality |
|
|
126
|
+
| `QuantileTransformer` | Transform to uniform or normal distribution |
|
|
127
|
+
|
|
128
|
+
> **Note:** Scaling transformers are available but should NOT be used with xplainable models. xplainable models are inherently explainable and require raw feature values. Scaling destroys interpretability.
|
|
129
|
+
|
|
130
|
+
### Other sklearn
|
|
131
|
+
| Transformer | Description |
|
|
132
|
+
|-------------|-------------|
|
|
133
|
+
| `SimpleImputer` | sklearn's imputer |
|
|
134
|
+
| `OneHotEncoder` | One-hot encode categoricals |
|
|
135
|
+
| `OrdinalEncoder` | Ordinal encode categoricals |
|
|
136
|
+
| `KBinsDiscretizer` | Discretise continuous features into bins |
|
|
137
|
+
| `Binarizer` | Threshold features to binary |
|
|
138
|
+
|
|
139
|
+
## Pipeline Spec Format
|
|
140
|
+
|
|
141
|
+
Pipelines are defined as JSON for portability across the xplainable platform, MCP tools, and API:
|
|
142
|
+
|
|
143
|
+
```json
|
|
144
|
+
{
|
|
145
|
+
"version": "2.0",
|
|
146
|
+
"steps": [
|
|
147
|
+
{
|
|
148
|
+
"id": "step_name",
|
|
149
|
+
"type": "TransformerName",
|
|
150
|
+
"columns": ["col1", "col2"],
|
|
151
|
+
"params": {"param1": "value1"},
|
|
152
|
+
"description": "Optional description"
|
|
153
|
+
}
|
|
154
|
+
]
|
|
155
|
+
}
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
- `id`: unique step identifier
|
|
159
|
+
- `type`: transformer class name (from the registry)
|
|
160
|
+
- `columns`: optional list of columns to apply to (wraps in `DataFrameColumnTransformer`)
|
|
161
|
+
- `params`: constructor arguments for the transformer
|
|
162
|
+
- `description`: optional human-readable description
|
|
163
|
+
|
|
164
|
+
## Architecture
|
|
165
|
+
|
|
166
|
+
```
|
|
167
|
+
PipelineSpec (JSON)
|
|
168
|
+
↓ compile_spec()
|
|
169
|
+
DataFramePipeline (sklearn Pipeline of transformers)
|
|
170
|
+
↓ fit() / transform()
|
|
171
|
+
Transformed DataFrame
|
|
172
|
+
↓ save_pipeline()
|
|
173
|
+
Binary (cloudpickle)
|
|
174
|
+
↓ load_pipeline()
|
|
175
|
+
DataFramePipeline (restored)
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Key modules:
|
|
179
|
+
- `schema.py` -- Pydantic models for `PipelineSpec` and `StepSpec`
|
|
180
|
+
- `compiler.py` -- compiles JSON spec into sklearn pipeline
|
|
181
|
+
- `pipeline.py` -- `DataFramePipeline` wrapper around sklearn `Pipeline`
|
|
182
|
+
- `registry.py` -- maps transformer names to classes, generates LLM-ready catalog
|
|
183
|
+
- `serialization.py` -- cloudpickle save/load; portable across Python versions
|
|
184
|
+
- `guard.py` -- step safety checks and step/spec previews
|
|
185
|
+
- `preview.py` -- before/after preview with schema deltas
|
|
186
|
+
- `contracts.py` -- which engineered columns each step computes (`dependency_map`, `partial_transform`)
|
|
187
|
+
- `detect/` -- detectors that describe a raw frame (`scan_all`, `Issue`)
|
|
188
|
+
- `io.py` -- `read_table`, the shared CSV/Excel reader
|
|
189
|
+
- `numeric_text.py` -- parse numbers stored as text
|
|
190
|
+
|
|
191
|
+
## Development
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
pip install -e ".[dev]"
|
|
195
|
+
pytest
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## License
|
|
199
|
+
|
|
200
|
+
MIT
|
|
@@ -6,8 +6,11 @@ Shared preprocessing pipeline library for [xplainable](https://xplainable.io). D
|
|
|
6
6
|
|
|
7
7
|
```bash
|
|
8
8
|
pip install xplainable-preprocessing
|
|
9
|
+
pip install "xplainable-preprocessing[excel]" # io.read_table for .xlsx/.xls
|
|
9
10
|
```
|
|
10
11
|
|
|
12
|
+
Upgrading from 0.3.x? 0.4.0 has breaking changes; see [CHANGELOG.md](CHANGELOG.md).
|
|
13
|
+
|
|
11
14
|
## Quick Start
|
|
12
15
|
|
|
13
16
|
### Define a pipeline as JSON
|
|
@@ -62,7 +65,9 @@ result = loaded.transform(new_data)
|
|
|
62
65
|
| `DropColumnsTransformer` | Drop specified columns |
|
|
63
66
|
| `RenameColumnsTransformer` | Rename columns |
|
|
64
67
|
| `TypeCastTransformer` | Cast column dtypes (e.g. str to float64, with error coercion) |
|
|
65
|
-
| `ClipTransformer` |
|
|
68
|
+
| `ClipTransformer` | Set numeric values outside (min, max), or per-column `bounds`, to NaN |
|
|
69
|
+
| `ParseNumericTransformer` | Parse numbers stored as text (%, currency, `1,234`, `36 months`, ranges, `10+`) |
|
|
70
|
+
| `NullTokensTransformer` | Turn placeholder text (`?`, `N/A`) and sentinel numbers (`999`) into NaN |
|
|
66
71
|
|
|
67
72
|
### Missing Values
|
|
68
73
|
| Transformer | Description |
|
|
@@ -73,17 +78,22 @@ result = loaded.transform(new_data)
|
|
|
73
78
|
### Feature Engineering
|
|
74
79
|
| Transformer | Description |
|
|
75
80
|
|-------------|-------------|
|
|
76
|
-
| `ExpressionTransformer` | Create derived columns
|
|
77
|
-
| `DateTimeExtractTransformer` | Extract year, month, dayofweek, etc. from datetime columns |
|
|
81
|
+
| `ExpressionTransformer` | Create derived columns from expressions (allow-listed evaluator) |
|
|
82
|
+
| `DateTimeExtractTransformer` | Extract year, month, dayofweek, elapsed time, etc. from datetime columns |
|
|
78
83
|
| `GroupByAggTransformer` | Aggregate features by group (count, sum, mean) |
|
|
79
|
-
| `GroupedLagTransformer` | Create
|
|
84
|
+
| `GroupedLagTransformer` | Create row-shift lags within groups |
|
|
80
85
|
| `RollingAggTransformer` | Rolling window aggregations |
|
|
86
|
+
| `TimeLagTransformer` | Lags and trailing windows at exact time offsets over strictly earlier rows |
|
|
87
|
+
| `GroupBroadcastTransformer` | Write a group aggregate (e.g. a holiday flag per date) to every row of the group |
|
|
88
|
+
| `TextFeaturesTransformer` | Numeric features from free text (length, digits, URLs, keyword flags) |
|
|
81
89
|
|
|
82
90
|
### Categorical
|
|
83
91
|
| Transformer | Description |
|
|
84
92
|
|-------------|-------------|
|
|
85
93
|
| `CategoryCondenseTransformer` | Condense high-cardinality categoricals to top N + "Other" |
|
|
86
|
-
| `TextCleanTransformer` | Clean text: lowercase, strip, remove HTML/extra whitespace |
|
|
94
|
+
| `TextCleanTransformer` | Clean text cells: lowercase, strip, remove HTML/extra whitespace, fix encoding, mask digits/URLs/emails |
|
|
95
|
+
| `OrdinalMapTransformer` | Map an ordered scale (grades, ratings, bands) to numbers |
|
|
96
|
+
| `CrossCategoriesTransformer` | Cross 2-3 low-cardinality columns into one category |
|
|
87
97
|
|
|
88
98
|
### Scaling (use with caution -- breaks xplainable model explainability)
|
|
89
99
|
| Transformer | Description |
|
|
@@ -149,8 +159,13 @@ Key modules:
|
|
|
149
159
|
- `compiler.py` -- compiles JSON spec into sklearn pipeline
|
|
150
160
|
- `pipeline.py` -- `DataFramePipeline` wrapper around sklearn `Pipeline`
|
|
151
161
|
- `registry.py` -- maps transformer names to classes, generates LLM-ready catalog
|
|
152
|
-
- `serialization.py` -- cloudpickle save/load
|
|
162
|
+
- `serialization.py` -- cloudpickle save/load; portable across Python versions
|
|
163
|
+
- `guard.py` -- step safety checks and step/spec previews
|
|
153
164
|
- `preview.py` -- before/after preview with schema deltas
|
|
165
|
+
- `contracts.py` -- which engineered columns each step computes (`dependency_map`, `partial_transform`)
|
|
166
|
+
- `detect/` -- detectors that describe a raw frame (`scan_all`, `Issue`)
|
|
167
|
+
- `io.py` -- `read_table`, the shared CSV/Excel reader
|
|
168
|
+
- `numeric_text.py` -- parse numbers stored as text
|
|
154
169
|
|
|
155
170
|
## Development
|
|
156
171
|
|
|
@@ -4,22 +4,33 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "xplainable-preprocessing"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.4.0"
|
|
8
8
|
description = "Shared preprocessing pipeline package for xplainable"
|
|
9
|
+
readme = "README.md"
|
|
9
10
|
requires-python = ">=3.9"
|
|
10
11
|
dependencies = [
|
|
11
12
|
"pydantic>=2.0",
|
|
12
13
|
"scikit-learn>=1.3",
|
|
13
|
-
|
|
14
|
+
# pandas 3 is not supported yet: its copy-on-write read-only arrays and
|
|
15
|
+
# default string dtype break detect.target, detect.structure and a few
|
|
16
|
+
# transformers (see CHANGELOG 0.4.0).
|
|
17
|
+
"pandas>=2.0,<3",
|
|
14
18
|
"numpy>=1.24",
|
|
15
19
|
"scipy>=1.8",
|
|
16
20
|
"cloudpickle>=3.0",
|
|
17
21
|
]
|
|
18
22
|
|
|
19
23
|
[project.optional-dependencies]
|
|
24
|
+
# Excel uploads in xplainable_preprocessing.io.read_table: .xlsx and .xls.
|
|
25
|
+
excel = [
|
|
26
|
+
"openpyxl>=3.1",
|
|
27
|
+
"xlrd>=2.0.1",
|
|
28
|
+
]
|
|
20
29
|
dev = [
|
|
21
30
|
"pytest>=7.0",
|
|
22
31
|
"pytest-cov",
|
|
32
|
+
"openpyxl>=3.1",
|
|
33
|
+
"xlrd>=2.0.1",
|
|
23
34
|
]
|
|
24
35
|
|
|
25
36
|
[tool.hatch.build.targets.wheel]
|