xplainable-preprocessing 0.3.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. xplainable_preprocessing-0.4.0/.github/workflows/ci.yml +90 -0
  2. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/.gitignore +2 -0
  3. xplainable_preprocessing-0.4.0/CHANGELOG.md +231 -0
  4. xplainable_preprocessing-0.4.0/PKG-INFO +200 -0
  5. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/README.md +21 -6
  6. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/pyproject.toml +13 -2
  7. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/__init__.py +77 -0
  8. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/compiler.py +20 -1
  9. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/contracts.py +1286 -0
  10. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/__init__.py +181 -0
  11. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/content.py +1896 -0
  12. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/structure.py +1307 -0
  13. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/target.py +1093 -0
  14. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/detect/types.py +242 -0
  15. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/guard.py +1099 -0
  16. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/io.py +1292 -0
  17. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/numeric_text.py +909 -0
  18. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/pipeline.py +180 -0
  19. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/preview.py +254 -0
  20. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/registry.py +265 -0
  21. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/sandbox.py +345 -0
  22. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/serialization.py +139 -0
  23. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/__init__.py +14 -0
  24. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/_util.py +74 -0
  25. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/clip.py +101 -0
  26. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/cross_categories.py +177 -0
  27. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/datetime_extract.py +303 -0
  28. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/expression.py +625 -0
  29. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/group_broadcast.py +335 -0
  30. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/grouped_lag.py +88 -0
  31. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/null_tokens.py +182 -0
  32. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/ordinal_map.py +337 -0
  33. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/parse_numeric.py +174 -0
  34. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/rolling_agg.py +155 -0
  35. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/text_clean.py +177 -0
  36. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/text_features.py +349 -0
  37. xplainable_preprocessing-0.4.0/src/xplainable_preprocessing/transformers/time_lag.py +627 -0
  38. xplainable_preprocessing-0.4.0/tests/fixtures/eval/README.md +38 -0
  39. xplainable_preprocessing-0.4.0/tests/fixtures/eval/adult_income.csv +1201 -0
  40. xplainable_preprocessing-0.4.0/tests/fixtures/eval/ames_housing.csv +580 -0
  41. xplainable_preprocessing-0.4.0/tests/fixtures/eval/bank_marketing.csv +1003 -0
  42. xplainable_preprocessing-0.4.0/tests/fixtures/eval/bank_marketing_raw.csv +201 -0
  43. xplainable_preprocessing-0.4.0/tests/fixtures/eval/bike_sharing_hour.csv +1001 -0
  44. xplainable_preprocessing-0.4.0/tests/fixtures/eval/credit_default.csv +1002 -0
  45. xplainable_preprocessing-0.4.0/tests/fixtures/eval/diamonds.csv +1004 -0
  46. xplainable_preprocessing-0.4.0/tests/fixtures/eval/heart_cleveland.csv +303 -0
  47. xplainable_preprocessing-0.4.0/tests/fixtures/eval/insurance.csv +1339 -0
  48. xplainable_preprocessing-0.4.0/tests/fixtures/eval/iris.csv +151 -0
  49. xplainable_preprocessing-0.4.0/tests/fixtures/eval/lending_club.csv +217 -0
  50. xplainable_preprocessing-0.4.0/tests/fixtures/eval/metro_traffic.csv +1525 -0
  51. xplainable_preprocessing-0.4.0/tests/fixtures/eval/sms_spam.csv +601 -0
  52. xplainable_preprocessing-0.4.0/tests/fixtures/eval/student_math.csv +396 -0
  53. xplainable_preprocessing-0.4.0/tests/fixtures/eval/titanic.csv +892 -0
  54. xplainable_preprocessing-0.4.0/tests/fixtures/eval/winequality-red.csv +1600 -0
  55. xplainable_preprocessing-0.4.0/tests/fixtures/io/adult_placeholders.csv +41 -0
  56. xplainable_preprocessing-0.4.0/tests/fixtures/io/ames_literal_none.csv +39 -0
  57. xplainable_preprocessing-0.4.0/tests/fixtures/io/bank_semicolon.csv +31 -0
  58. xplainable_preprocessing-0.4.0/tests/fixtures/io/cp1252_semicolon.csv +7 -0
  59. xplainable_preprocessing-0.4.0/tests/fixtures/io/credit_two_header_rows.xls +0 -0
  60. xplainable_preprocessing-0.4.0/tests/fixtures/io/credit_two_header_rows.xlsx +0 -0
  61. xplainable_preprocessing-0.4.0/tests/fixtures/io/expected.json +221 -0
  62. xplainable_preprocessing-0.4.0/tests/fixtures/io/heart_headerless.csv +34 -0
  63. xplainable_preprocessing-0.4.0/tests/fixtures/io/metro_holiday.csv +24 -0
  64. xplainable_preprocessing-0.4.0/tests/fixtures/io/mixed_types.xlsx +0 -0
  65. xplainable_preprocessing-0.4.0/tests/fixtures/io/pipe.txt +16 -0
  66. xplainable_preprocessing-0.4.0/tests/fixtures/io/title_rows.csv +12 -0
  67. xplainable_preprocessing-0.4.0/tests/fixtures/io/utf16_tab.txt +0 -0
  68. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/builtin_pipeline.pkl +0 -0
  69. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_missing_flag.pkl +0 -0
  70. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_rename.pkl +0 -0
  71. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/collision_wrapped_expression.pkl +0 -0
  72. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/custom_step_by_value.pkl +0 -0
  73. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/datetime_unwrapped.pkl +0 -0
  74. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/expression_outside_allow_list.pkl +0 -0
  75. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/grouped_lag_order_by.pkl +0 -0
  76. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/make_fixtures.py +258 -0
  77. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/manifest.json +21 -0
  78. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/rolling_order_by.pkl +0 -0
  79. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/sklearn_steps.pkl +0 -0
  80. xplainable_preprocessing-0.4.0/tests/fixtures/legacy_0_3_1/text_clean_mixed_types.pkl +0 -0
  81. xplainable_preprocessing-0.4.0/tests/portability_roundtrip.py +312 -0
  82. xplainable_preprocessing-0.4.0/tests/test_contracts.py +701 -0
  83. xplainable_preprocessing-0.4.0/tests/test_detect_content.py +990 -0
  84. xplainable_preprocessing-0.4.0/tests/test_detect_structure.py +764 -0
  85. xplainable_preprocessing-0.4.0/tests/test_detect_target.py +671 -0
  86. xplainable_preprocessing-0.4.0/tests/test_detect_types.py +133 -0
  87. xplainable_preprocessing-0.4.0/tests/test_guard.py +591 -0
  88. xplainable_preprocessing-0.4.0/tests/test_io.py +783 -0
  89. xplainable_preprocessing-0.4.0/tests/test_legacy_pickles.py +201 -0
  90. xplainable_preprocessing-0.4.0/tests/test_numeric_text.py +481 -0
  91. xplainable_preprocessing-0.4.0/tests/test_pipeline.py +189 -0
  92. xplainable_preprocessing-0.4.0/tests/test_portable_custom.py +478 -0
  93. xplainable_preprocessing-0.4.0/tests/test_preview.py +219 -0
  94. xplainable_preprocessing-0.4.0/tests/test_registry.py +118 -0
  95. xplainable_preprocessing-0.4.0/tests/test_release.py +95 -0
  96. xplainable_preprocessing-0.4.0/tests/test_transformers/test_clip.py +91 -0
  97. xplainable_preprocessing-0.4.0/tests/test_transformers/test_cross_categories.py +147 -0
  98. xplainable_preprocessing-0.4.0/tests/test_transformers/test_datetime_extract.py +210 -0
  99. xplainable_preprocessing-0.4.0/tests/test_transformers/test_expression.py +334 -0
  100. xplainable_preprocessing-0.4.0/tests/test_transformers/test_group_broadcast.py +221 -0
  101. xplainable_preprocessing-0.4.0/tests/test_transformers/test_null_tokens.py +171 -0
  102. xplainable_preprocessing-0.4.0/tests/test_transformers/test_ordinal_map.py +264 -0
  103. xplainable_preprocessing-0.4.0/tests/test_transformers/test_parse_numeric.py +195 -0
  104. xplainable_preprocessing-0.4.0/tests/test_transformers/test_text_clean.py +107 -0
  105. xplainable_preprocessing-0.4.0/tests/test_transformers/test_text_features.py +229 -0
  106. xplainable_preprocessing-0.4.0/tests/test_transformers/test_time_lag.py +418 -0
  107. xplainable_preprocessing-0.4.0/tests/test_transformers/test_time_order.py +211 -0
  108. xplainable_preprocessing-0.3.1/PKG-INFO +0 -14
  109. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/__init__.py +0 -34
  110. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/guard.py +0 -282
  111. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/pipeline.py +0 -93
  112. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/preview.py +0 -137
  113. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/registry.py +0 -114
  114. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/sandbox.py +0 -113
  115. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/serialization.py +0 -20
  116. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/clip.py +0 -45
  117. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/datetime_extract.py +0 -72
  118. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/expression.py +0 -51
  119. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/grouped_lag.py +0 -55
  120. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/rolling_agg.py +0 -79
  121. xplainable_preprocessing-0.3.1/src/xplainable_preprocessing/transformers/text_clean.py +0 -70
  122. xplainable_preprocessing-0.3.1/tests/test_guard.py +0 -119
  123. xplainable_preprocessing-0.3.1/tests/test_pipeline.py +0 -73
  124. xplainable_preprocessing-0.3.1/tests/test_preview.py +0 -90
  125. xplainable_preprocessing-0.3.1/tests/test_transformers/test_expression.py +0 -53
  126. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/.github/workflows/publish-pypi.yml +0 -0
  127. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/dag-pipeline-proposal.md +0 -0
  128. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/feature-pipeline-architectures.md +0 -0
  129. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/docs/feature-store-proposal.md +0 -0
  130. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/schema.py +0 -0
  131. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/category_condense.py +0 -0
  132. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/drop_columns.py +0 -0
  133. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/fill_missing.py +0 -0
  134. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/groupby_agg.py +0 -0
  135. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/missing_flag.py +0 -0
  136. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/rename_columns.py +0 -0
  137. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/src/xplainable_preprocessing/transformers/type_cast.py +0 -0
  138. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/__init__.py +0 -0
  139. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_compiler.py +0 -0
  140. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_sandbox.py +0 -0
  141. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_schema.py +0 -0
  142. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_serialization.py +0 -0
  143. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_transformers/__init__.py +0 -0
  144. {xplainable_preprocessing-0.3.1 → xplainable_preprocessing-0.4.0}/tests/test_transformers/test_all_transformers.py +0 -0
@@ -0,0 +1,90 @@
1
+ name: CI
2
+
3
+ # Tests on both Pythons the platform runs (core-api and the inference server
4
+ # on 3.10, the train service on 3.12), plus a cross-version check that a
5
+ # pipeline with custom steps fitted and saved under one Python loads and
6
+ # transforms identically under the other.
7
+
8
+ on:
9
+ push:
10
+ branches: [main, 'release/**']
11
+ pull_request:
12
+
13
+ permissions:
14
+ contents: read
15
+
16
+ jobs:
17
+ test:
18
+ name: pytest (Python ${{ matrix.python-version }})
19
+ runs-on: ubuntu-latest
20
+ strategy:
21
+ fail-fast: false
22
+ matrix:
23
+ include:
24
+ # core-api / inference-server stack
25
+ - python-version: '3.10'
26
+ pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
27
+ # train-service stack
28
+ - python-version: '3.12'
29
+ pins: 'pandas==2.3.3'
30
+ steps:
31
+ - uses: actions/checkout@v4
32
+ - uses: actions/setup-python@v5
33
+ with:
34
+ python-version: ${{ matrix.python-version }}
35
+ - name: Install
36
+ run: pip install -e '.[dev]' ${{ matrix.pins }}
37
+ - name: Test
38
+ run: python -m pytest -q
39
+
40
+ portability-dump:
41
+ name: dump under Python ${{ matrix.python-version }}
42
+ runs-on: ubuntu-latest
43
+ strategy:
44
+ matrix:
45
+ include:
46
+ - python-version: '3.10'
47
+ pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
48
+ - python-version: '3.12'
49
+ pins: 'pandas==2.3.3'
50
+ steps:
51
+ - uses: actions/checkout@v4
52
+ - uses: actions/setup-python@v5
53
+ with:
54
+ python-version: ${{ matrix.python-version }}
55
+ - name: Install
56
+ run: pip install -e . ${{ matrix.pins }}
57
+ - name: Fit and save the bike_sharing custom-step pipeline
58
+ run: python tests/portability_roundtrip.py dump dump
59
+ - uses: actions/upload-artifact@v4
60
+ with:
61
+ name: pipeline-py${{ matrix.python-version }}
62
+ path: dump/
63
+
64
+ portability-load:
65
+ name: load a Python ${{ matrix.written-by }} binary under Python ${{ matrix.python-version }}
66
+ needs: portability-dump
67
+ runs-on: ubuntu-latest
68
+ strategy:
69
+ fail-fast: false
70
+ matrix:
71
+ include:
72
+ - python-version: '3.12'
73
+ written-by: '3.10'
74
+ pins: 'pandas==2.3.3'
75
+ - python-version: '3.10'
76
+ written-by: '3.12'
77
+ pins: 'pandas==2.3.3 numpy==2.0.2 scikit-learn==1.6.1'
78
+ steps:
79
+ - uses: actions/checkout@v4
80
+ - uses: actions/setup-python@v5
81
+ with:
82
+ python-version: ${{ matrix.python-version }}
83
+ - name: Install
84
+ run: pip install -e . ${{ matrix.pins }}
85
+ - uses: actions/download-artifact@v4
86
+ with:
87
+ name: pipeline-py${{ matrix.written-by }}
88
+ path: dump/
89
+ - name: Load, transform and compare
90
+ run: python tests/portability_roundtrip.py load dump
@@ -3,3 +3,5 @@
3
3
  .venv/
4
4
  __pycache__/
5
5
  *.pyc
6
+ dist/
7
+ build/
@@ -0,0 +1,231 @@
1
+ # Changelog
2
+
3
+ All notable changes to xplainable-preprocessing. Versions follow semantic
4
+ versioning; before 1.0 a minor release may break compatibility, and every
5
+ break is listed under **Breaking changes**.
6
+
7
+ ## [0.4.0] - 2026-09-28
8
+
9
+ A breaking release. Pipelines now refuse to produce two columns with the same
10
+ name, expressions run on an allow-listed evaluator instead of `pandas.eval`,
11
+ several transformers keep data they used to destroy, and custom steps pickle
12
+ so a binary loads on another Python version. It also adds seven transformers
13
+ and four modules: `detect`, `io`, `numeric_text` and `contracts`.
14
+
15
+ Supported: Python 3.9 and later (tested on 3.10 and 3.12), pandas 2.x
16
+ (`pandas>=2.0,<3`), numpy 1.24 or later, scikit-learn 1.3 or later.
17
+
18
+ ### Breaking changes
19
+
20
+ | Area | Change |
21
+ |---|---|
22
+ | Column collisions | `DataFrameColumnTransformer` raises `ColumnCollisionError` when a wrapped step emits a column that is also a pass-through column. `DataFramePipeline` raises it, with the step id, when any step's output repeats a column name (for example `RenameColumnsTransformer` renaming onto an existing column). 0.3.x returned a frame with two columns of that name. The error carries `columns`, `step` and `inputs`, and pickles intact. |
23
+ | Expressions | `ExpressionTransformer` parses the expression with `ast` and evaluates it with an allow-list: arithmetic, comparisons, element-wise logic, the `pandas.eval` math functions, column references (bare, backticked, `df['x']`, `df.x`), and the methods `.where .abs .round .isna .notna .isnull .notnull .fillna .clip .between .isin .astype` (restricted dtypes) plus `.str.len/count/contains/startswith/endswith/lower/upper/strip`. Anything else (other methods such as `.cumsum()`, `.shift()`, `.rank()` or `.pow()`, dunders, I/O) raises `ExpressionError` at fit and at transform. Nothing reaches `eval` or `pandas.eval`. A column named `df` is read as `` `df` `` or `df['df']`. |
24
+ | `save_pipeline` | Defaults to `portable=True` and raises `NonPortablePipelineError`, naming the steps, when any fitted state would be pickled by value (a lambda, or a function or class from custom code kept on `self`). `portable=False` saves as 0.3.x did. |
25
+ | TextCleanTransformer | Operations apply to `str` cells only. Numbers, None and other values in a mixed column are kept; 0.3.x turned them into NaN. `string` and `category` columns are cleaned too (a category column stays categorical). An unknown operation raises at fit; 0.3.x ignored it. |
26
+ | GroupedLagTransformer, RollingAggTransformer | Compute on a stably sorted copy and return rows in input order with the input index. With `order_by`, 0.3.x returned the rows sorted by group and `order_by`. Each row's values are unchanged, but row order is not. Negative lag periods and missing `group_by`/`order_by` columns raise at fit. Categorical group keys use `observed=True`. |
27
+ | DateTimeExtractTransformer without step columns | At fit it picks datetime columns, and text columns where at least 80% of values parse as dates. Only those are parsed and dropped. Numeric, bool and other text columns are left untouched; 0.3.x converted every column to dates, extracted parts and dropped it. Fit raises when no column holds dates. |
28
+ | ClipTransformer | Bounds are validated at fit: a non-number, `min > max`, or a `bounds` column that is missing or not numeric raises. |
29
+ | `generate_catalog()` | Lines show typed params (`name: type = default`). A required param is shown without a default, and a choice between params is spelled out ("Requires order or mapping."). Code that parses catalog lines must follow the new format. |
30
+ | Guard and previews | `step_safety_error` and `spec_safety_errors` also reject steps that reorder rows, read the label, or copy the label (see Added). Checks and previews run on the whole frame up to `full_rows_limit` (100k) rows, where 0.3.x used a random 10k-row sample. They see every row, and take longer on large frames. |
31
+
32
+ #### Stored 0.3.x pipelines
33
+
34
+ Built-in steps pickle by reference, so a 0.3.x binary loads into the 0.4.0
35
+ classes. The 0.3.x pickles of `RollingAggTransformer`, `ClipTransformer` and
36
+ `DateTimeExtractTransformer` are filled in with the new params' defaults. A
37
+ 0.3.x pipeline of built-in steps serves the same output under 0.4.0
38
+ (`tests/test_legacy_pickles.py` checks this on binaries written by 0.3.1)
39
+ with these exceptions:
40
+
41
+ | Stored pipeline | Under 0.4.0 | What to do |
42
+ |---|---|---|
43
+ | A wrapped step whose output name already exists, or a rename onto an existing column | `transform` raises `ColumnCollisionError` (0.3.x served duplicate columns) | Re-fit with a new output name, or drop the existing column first |
44
+ | An expression outside the allow-list | `transform` raises `ExpressionError` | Rewrite it with allowed constructs (`a ** 2` for `a.pow(2)`) and re-fit, or ask for the method to be allow-listed |
45
+ | TextCleanTransformer on a column mixing text and other values | Non-text cells are kept, so the output changes | Re-fit, or accept the change |
46
+ | GroupedLagTransformer or RollingAggTransformer with `order_by` | Rows come back in input order | Re-fit, and check anything that relied on the sorted order |
47
+ | DateTimeExtractTransformer without step columns | Non-date columns are kept, so the output columns change. A 0.3.x binary records no date columns, so each served batch decides which columns are dates: a single row with an empty date keeps that column and gets no date parts | Re-fit |
48
+ | Custom steps saved by 0.3.x (pickled by value) | Load and serve on the Python version that wrote them, as before; on another Python version `load_pipeline` still fails (typically a `TypeError` from the pickled code). `save_pipeline` refuses them unless `portable=False` | Re-fit to get a portable binary |
49
+
50
+ Steps from scikit-learn (`OneHotEncoder`, `StandardScaler`, `SimpleImputer`,
51
+ ...) pickle as scikit-learn's own objects. They load only under the
52
+ scikit-learn release that fitted them (a 1.6.1 `SimpleImputer` fails under
53
+ 1.9), whichever xplainable-preprocessing release is installed.
54
+
55
+ ### Added
56
+
57
+ **Transformers** (registered, in the catalog, with required params in
58
+ `REQUIRED_PARAMS`):
59
+ - `ParseNumericTransformer`: numbers stored as text into floats. Handles `%`,
60
+ currency, `1,234`, `(1,234)`, units such as `36 months`, ranges, bounds
61
+ such as `10+` and NA tokens. Idempotent on numbers.
62
+ - `NullTokensTransformer`: placeholder text (`?`, `N/A`, blanks) and sentinel
63
+ numbers (`999`) to NaN, with optional 0/1 flags.
64
+ - `OrdinalMapTransformer`: an ordered scale to numbers in place, by `order`
65
+ or `mapping`. NaN- and unknown-safe, with JSON-safe params. It rejects a
66
+ mapping that gives one level two codes.
67
+ - `CrossCategoriesTransformer`: 2 to 3 low-cardinality columns crossed into
68
+ one readable category, with zero padding and a `max_levels` cap.
69
+ - `TimeLagTransformer`: lags and trailing-window aggregates at exact time
70
+ offsets over strictly earlier timestamps, per group. Supports `gap`,
71
+ deduplication of repeated timestamps, a bounded history tail and
72
+ `serving_requirements()`. A `gap` that covers every lag is the one safe way
73
+ to use past target values.
74
+ - `GroupBroadcastTransformer`: a group aggregate written to every row of its
75
+ group. A group can be a calendar date, hour, week or month of a time
76
+ column, and the mapping is kept from fit so single-row serving works.
77
+ - `TextFeaturesTransformer`: free text into numeric features (length, words,
78
+ digits, capitals, punctuation, URLs, emails, phone-like numbers, whole-word
79
+ keyword flags). The feature set is frozen as `feature_set_version=1`.
80
+
81
+ **Transformer options:**
82
+ - `RollingAggTransformer`: `closed` (`"left"` excludes the current row) and
83
+ `shift`. List params accept a single name.
84
+ - `GroupedLagTransformer`: `order_by` takes a list, and scalar params are
85
+ coerced.
86
+ - `DateTimeExtractTransformer`: components `elapsed_days`, `elapsed_hours`
87
+ (from an `origin` fixed at fit), `hour_of_week`, `date` and `month_index`,
88
+ plus `as_categorical` (zero-padded strings). The text format is fixed at
89
+ fit, and ISO 8601 values it misses are still read.
90
+ - `ClipTransformer`: per-column `bounds` (`{"x": [3, 11], "z": [1.5, None]}`).
91
+ - `TextCleanTransformer`: operations `html_unescape`, `fix_encoding` (cp1252
92
+ mojibake), `nfkc`, `mask_digits`, `mask_urls` and `mask_emails`.
93
+
94
+ **Modules:**
95
+ - `numeric_text`: `parse_numeric` and `numeric_text_report`, which says
96
+ whether a text column holds numbers. The report rejects dates, codes with
97
+ leading zeros and identifiers, and reports collisions and mixed units. The
98
+ module also holds the NA-token vocabulary (`NA_TOKENS`, `MAYBE_NA_TOKENS`).
99
+ - `io.read_table(content, filename, options) -> (DataFrame, ReadReport)`:
100
+ one reader for uploaded CSV and Excel files.
101
+ - Encoding: UTF-8 with cp1252 and latin-1 fallbacks.
102
+ - Delimiter sniffing, with a guard against single-column reads.
103
+ - Header rows below the top and headerless files are detected from the
104
+ raw bytes and read again losslessly.
105
+ - NA policies: `literal_v1` (the default; only empty cells are missing,
106
+ and text tokens stay literal) and `legacy` (pandas' defaults).
107
+ - Typed Excel with a safe 99% numeric coercion.
108
+ - `ReadError` subclasses carry messages meant for users.
109
+ - `coerce_numeric_columns` applies the same coercion to frames already
110
+ stored.
111
+ - Excel needs the new `excel` extra: `pip install
112
+ "xplainable-preprocessing[excel]"`.
113
+ - `detect`: pure functions that describe a raw frame and return JSON-safe
114
+ `Issue`s. Each issue has a severity, a confidence and, where a library
115
+ step fixes it, `suggested_steps`.
116
+ - `scan_frame`: column content. Placeholder and maybe-NA tokens, numbers
117
+ stored as text, sentinels and top or bottom codes, structural and
118
+ co-missing gaps, whitespace and case variants, impossible years and
119
+ broken year orderings, ordinal scales, free text, sparse markers,
120
+ impossible zeros, and a regression target's point mass.
121
+ - `scan_structure`: frame shape. A delimited file read as one column,
122
+ headerless files (pandas' `.N` suffixes undone), header rows inside the
123
+ data, repeated headers, junk rows, and identifiers measured against
124
+ distinct rows.
125
+ - `scan_relations`: needs a target. Target proxies, columns that rebuild
126
+ the target, time-order drift, duplicate keys whose target is constant,
127
+ and 1:1 or functional redundancy.
128
+ - `scan_all` runs all three.
129
+ - `repair_frame` applies only lossless repairs to frames already stored.
130
+ - `contracts`:
131
+ - `dependency_map(fitted, spec, data)`: for each engineered output column,
132
+ the step that computed it, its inputs and raw sources, and whether it
133
+ can be recomputed row by row. For ExpressionTransformer outputs it also
134
+ gives a `DataFrame.eval` formula, verified on the rows, that
135
+ xplainable-gm's optimiser accepts as a `derived` declaration.
136
+ `DependencyMap.is_engineered(column)` checks a single column.
137
+ - `partial_transform(fitted, base_prepared, raw_rows, changed_columns)`:
138
+ recomputes only the outputs that a changed raw column feeds.
139
+ - Both work on 0.3.x pipelines when the spec is passed.
140
+
141
+ **Pipelines and serialisation:**
142
+ - `compile_spec` attaches the spec to the pipeline as `pipeline.spec`, as a
143
+ plain dict.
144
+ - Custom steps pickle as their spec plus fitted state. On load, the
145
+ module-level `sandbox.rebuild_custom` validates and runs the source again,
146
+ so a pipeline saved on Python 3.10 loads on 3.12 and the reverse. Copies
147
+ keep the original class.
148
+ - A custom step's constructor `TypeError` names the params its `__init__`
149
+ does not accept.
150
+
151
+ **Guard and previews** (`guard`, `preview`):
152
+ - `step_safety_error` and `spec_safety_errors` reject four more kinds of
153
+ step:
154
+ - a step that reorders rows;
155
+ - a step that reads the label. This is checked statically (columns,
156
+ params, expression names, custom-code AST) and behaviourally (transform
157
+ with the label dropped or shuffled). A `TimeLagTransformer` whose `gap`
158
+ covers every lag is exempt (`LABEL_READ_EXEMPTIONS`).
159
+ - a step that adds a copy of the label: the new column equals the label
160
+ on at least 50% of rows, or has |Spearman rho| >= 0.995. A preview
161
+ warns from 5% up.
162
+ - a step whose output collides with an existing column.
163
+ - `preview_step` and `preview_spec` add:
164
+ - a `label` parameter;
165
+ - `serving_warnings`, from a single-row batch-dependence probe;
166
+ - `changes`, with `changed_cells` and `top_changes` for each updated
167
+ column;
168
+ - `top_values` for non-numeric columns;
169
+ - a warning when a step changes 0 cells;
170
+ - `rows_total` and `sampled`.
171
+ - Past 100k rows, checks and previews use an order-preserving sample. Lag,
172
+ rolling, group-statistic and custom steps get a contiguous block; any
173
+ other step gets rows spread over the frame that always include the
174
+ extreme rows of its columns.
175
+ - `registry.missing_required_params` and `REQUIRED_PARAMS` list what each
176
+ step type must set. `registry.catalog_entry` builds one catalog line.
177
+
178
+ **Package:**
179
+ - `__version__`. The top level now exports `ColumnCollisionError`,
180
+ `NonPortablePipelineError`, `ExpressionError`, `read_table`, `ReadReport`,
181
+ `ReadError`, `Issue`, `scan_all`, `parse_numeric` and
182
+ `numeric_text_report`. The `detect`, `io` and `numeric_text` submodules are
183
+ reachable as attributes.
184
+ - CI runs the suite on Python 3.10 (pandas 2.3.3, numpy 2.0.2,
185
+ scikit-learn 1.6.1) and 3.12, plus a matrix that dumps a pipeline on 3.10
186
+ and loads it on 3.12, and the reverse.
187
+
188
+ ### Fixed
189
+
190
+ - `infer_schema`, `compute_preview` and `compute_step_deltas` returned samples
191
+ that `json.dumps` rejects when a column held timestamps, timedeltas, periods,
192
+ numpy scalars or pandas NA values. Storing such a schema failed, e.g. after a
193
+ `DateTimeExtractTransformer` with `drop_original=False`. Samples are now
194
+ JSON-safe: missing values become null, timestamps and dates ISO-8601 text,
195
+ timedeltas ISO-8601 durations, numpy scalars Python values.
196
+ - A step whose output repeated a column name produced a frame with two
197
+ columns of that name, and anything downstream then failed or read the
198
+ wrong one. It now raises (see Breaking changes).
199
+ - `GroupedLagTransformer` and `RollingAggTransformer` misaligned rows when
200
+ `order_by` sorted them. `RollingAggTransformer` crashed on categorical
201
+ group keys with unused categories.
202
+ - Unwrapped `DateTimeExtractTransformer` destroyed every non-date column.
203
+ - `TextCleanTransformer` erased numbers in mixed columns.
204
+ - Custom-step pipelines saved on one Python version failed to load on
205
+ another (`code() argument 13 must be str, not int`).
206
+ - `preview_step` no longer raises when statistics or deltas fail, and
207
+ repeated input column names come back as an error.
208
+
209
+ ### Known limitations
210
+
211
+ - pandas 3 is not supported, and the dependency is capped at `<3`. Under
212
+ pandas 3.0 the suite fails in four places:
213
+ - `detect.target` writes into arrays that copy-on-write makes read-only,
214
+ so most relation scans raise.
215
+ - `detect.structure`'s headerless restore handles missing values in the
216
+ default `str` dtype differently.
217
+ - `ExpressionTransformer`'s `.str` methods also handle missing values in
218
+ the `str` dtype differently.
219
+ - `CategoryCondenseTransformer` fits only `object` and `category`
220
+ columns, so it leaves `str` columns uncondensed.
221
+
222
+ ## [0.3.1]
223
+
224
+ - `ExpressionTransformer` resolves backtick-quoted column names.
225
+
226
+ ## [0.3.0]
227
+
228
+ - Step safety guard (`step_safety_error`, `spec_safety_errors`), step and
229
+ spec previews (`preview_step`, `preview_spec`), and `safe_catalog`.
230
+
231
+ Earlier releases are described in the git history.
@@ -0,0 +1,200 @@
1
+ Metadata-Version: 2.5
2
+ Name: xplainable-preprocessing
3
+ Version: 0.4.0
4
+ Summary: Shared preprocessing pipeline package for xplainable
5
+ Requires-Python: >=3.9
6
+ Requires-Dist: cloudpickle>=3.0
7
+ Requires-Dist: numpy>=1.24
8
+ Requires-Dist: pandas<3,>=2.0
9
+ Requires-Dist: pydantic>=2.0
10
+ Requires-Dist: scikit-learn>=1.3
11
+ Requires-Dist: scipy>=1.8
12
+ Provides-Extra: dev
13
+ Requires-Dist: openpyxl>=3.1; extra == 'dev'
14
+ Requires-Dist: pytest-cov; extra == 'dev'
15
+ Requires-Dist: pytest>=7.0; extra == 'dev'
16
+ Requires-Dist: xlrd>=2.0.1; extra == 'dev'
17
+ Provides-Extra: excel
18
+ Requires-Dist: openpyxl>=3.1; extra == 'excel'
19
+ Requires-Dist: xlrd>=2.0.1; extra == 'excel'
20
+ Description-Content-Type: text/markdown
21
+
22
+ # xplainable-preprocessing
23
+
24
+ Shared preprocessing pipeline library for [xplainable](https://xplainable.io). Define, compile, fit, serialise, and apply data transformation pipelines using a declarative JSON spec.
25
+
26
+ ## Install
27
+
28
+ ```bash
29
+ pip install xplainable-preprocessing
30
+ pip install "xplainable-preprocessing[excel]" # io.read_table for .xlsx/.xls
31
+ ```
32
+
33
+ Upgrading from 0.3.x? 0.4.0 has breaking changes; see [CHANGELOG.md](CHANGELOG.md).
34
+
35
+ ## Quick Start
36
+
37
+ ### Define a pipeline as JSON
38
+
39
+ ```python
40
+ from xplainable_preprocessing.schema import PipelineSpec
41
+ from xplainable_preprocessing.compiler import compile_spec
42
+
43
+ spec = PipelineSpec(
44
+ version="2.0",
45
+ steps=[
46
+ {"id": "drop_ids", "type": "DropColumnsTransformer", "params": {"columns": ["customer_id", "row_id"]}},
47
+ {"id": "fill_numeric", "type": "FillMissingTransformer", "columns": ["tenure", "charges"], "params": {"strategies": {"tenure": "median", "charges": "median"}}},
48
+ {"id": "clip_outliers", "type": "ClipTransformer", "columns": ["charges"], "params": {"min_val": 0, "max_val": 500}},
49
+ {"id": "missing_flags", "type": "MissingFlagTransformer", "columns": ["tenure", "charges"]},
50
+ {"id": "condense_plan", "type": "CategoryCondenseTransformer", "columns": ["plan_type"], "params": {"max_categories": 10}},
51
+ {"id": "extract_dates", "type": "DateTimeExtractTransformer", "columns": ["signup_date"], "params": {"components": ["month", "dayofweek"], "drop_original": true}},
52
+ ]
53
+ )
54
+
55
+ pipeline = compile_spec(spec)
56
+ ```
57
+
58
+ ### Fit and transform
59
+
60
+ ```python
61
+ import pandas as pd
62
+
63
+ df = pd.read_csv("data.csv")
64
+ pipeline.fit(df)
65
+ transformed = pipeline.transform(df)
66
+ ```
67
+
68
+ ### Serialise and load
69
+
70
+ ```python
71
+ from xplainable_preprocessing.serialization import save_pipeline, load_pipeline
72
+
73
+ # Save
74
+ binary = save_pipeline(pipeline)
75
+
76
+ # Load
77
+ loaded = load_pipeline(binary)
78
+ result = loaded.transform(new_data)
79
+ ```
80
+
81
+ ## Available Transformers
82
+
83
+ ### Data Cleaning
84
+ | Transformer | Description |
85
+ |-------------|-------------|
86
+ | `DropColumnsTransformer` | Drop specified columns |
87
+ | `RenameColumnsTransformer` | Rename columns |
88
+ | `TypeCastTransformer` | Cast column dtypes (e.g. str to float64, with error coercion) |
89
+ | `ClipTransformer` | Set numeric values outside (min, max), or per-column `bounds`, to NaN |
90
+ | `ParseNumericTransformer` | Parse numbers stored as text (%, currency, `1,234`, `36 months`, ranges, `10+`) |
91
+ | `NullTokensTransformer` | Turn placeholder text (`?`, `N/A`) and sentinel numbers (`999`) into NaN |
92
+
93
+ ### Missing Values
94
+ | Transformer | Description |
95
+ |-------------|-------------|
96
+ | `FillMissingTransformer` | Fill nulls with median, mode, constant, or other strategies |
97
+ | `MissingFlagTransformer` | Create binary 0/1 indicator columns for nulls |
98
+
99
+ ### Feature Engineering
100
+ | Transformer | Description |
101
+ |-------------|-------------|
102
+ | `ExpressionTransformer` | Create derived columns from expressions (allow-listed evaluator) |
103
+ | `DateTimeExtractTransformer` | Extract year, month, dayofweek, elapsed time, etc. from datetime columns |
104
+ | `GroupByAggTransformer` | Aggregate features by group (count, sum, mean) |
105
+ | `GroupedLagTransformer` | Create row-shift lags within groups |
106
+ | `RollingAggTransformer` | Rolling window aggregations |
107
+ | `TimeLagTransformer` | Lags and trailing windows at exact time offsets over strictly earlier rows |
108
+ | `GroupBroadcastTransformer` | Write a group aggregate (e.g. a holiday flag per date) to every row of the group |
109
+ | `TextFeaturesTransformer` | Numeric features from free text (length, digits, URLs, keyword flags) |
110
+
111
+ ### Categorical
112
+ | Transformer | Description |
113
+ |-------------|-------------|
114
+ | `CategoryCondenseTransformer` | Condense high-cardinality categoricals to top N + "Other" |
115
+ | `TextCleanTransformer` | Clean text cells: lowercase, strip, remove HTML/extra whitespace, fix encoding, mask digits/URLs/emails |
116
+ | `OrdinalMapTransformer` | Map an ordered scale (grades, ratings, bands) to numbers |
117
+ | `CrossCategoriesTransformer` | Cross 2-3 low-cardinality columns into one category |
118
+
119
+ ### Scaling (use with caution -- breaks xplainable model explainability)
120
+ | Transformer | Description |
121
+ |-------------|-------------|
122
+ | `StandardScaler` | Standardise to zero mean, unit variance |
123
+ | `MinMaxScaler` | Scale to [0, 1] range |
124
+ | `RobustScaler` | Scale using median and IQR |
125
+ | `PowerTransformer` | Apply power transform for normality |
126
+ | `QuantileTransformer` | Transform to uniform or normal distribution |
127
+
128
+ > **Note:** Scaling transformers are available but should NOT be used with xplainable models. xplainable models are inherently explainable and require raw feature values. Scaling destroys interpretability.
129
+
130
+ ### Other sklearn
131
+ | Transformer | Description |
132
+ |-------------|-------------|
133
+ | `SimpleImputer` | sklearn's imputer |
134
+ | `OneHotEncoder` | One-hot encode categoricals |
135
+ | `OrdinalEncoder` | Ordinal encode categoricals |
136
+ | `KBinsDiscretizer` | Discretise continuous features into bins |
137
+ | `Binarizer` | Threshold features to binary |
138
+
139
+ ## Pipeline Spec Format
140
+
141
+ Pipelines are defined as JSON for portability across the xplainable platform, MCP tools, and API:
142
+
143
+ ```json
144
+ {
145
+ "version": "2.0",
146
+ "steps": [
147
+ {
148
+ "id": "step_name",
149
+ "type": "TransformerName",
150
+ "columns": ["col1", "col2"],
151
+ "params": {"param1": "value1"},
152
+ "description": "Optional description"
153
+ }
154
+ ]
155
+ }
156
+ ```
157
+
158
+ - `id`: unique step identifier
159
+ - `type`: transformer class name (from the registry)
160
+ - `columns`: optional list of columns to apply to (wraps in `DataFrameColumnTransformer`)
161
+ - `params`: constructor arguments for the transformer
162
+ - `description`: optional human-readable description
163
+
164
+ ## Architecture
165
+
166
+ ```
167
+ PipelineSpec (JSON)
168
+ ↓ compile_spec()
169
+ DataFramePipeline (sklearn Pipeline of transformers)
170
+ ↓ fit() / transform()
171
+ Transformed DataFrame
172
+ ↓ save_pipeline()
173
+ Binary (cloudpickle)
174
+ ↓ load_pipeline()
175
+ DataFramePipeline (restored)
176
+ ```
177
+
178
+ Key modules:
179
+ - `schema.py` -- Pydantic models for `PipelineSpec` and `StepSpec`
180
+ - `compiler.py` -- compiles JSON spec into sklearn pipeline
181
+ - `pipeline.py` -- `DataFramePipeline` wrapper around sklearn `Pipeline`
182
+ - `registry.py` -- maps transformer names to classes, generates LLM-ready catalog
183
+ - `serialization.py` -- cloudpickle save/load; portable across Python versions
184
+ - `guard.py` -- step safety checks and step/spec previews
185
+ - `preview.py` -- before/after preview with schema deltas
186
+ - `contracts.py` -- which engineered columns each step computes (`dependency_map`, `partial_transform`)
187
+ - `detect/` -- detectors that describe a raw frame (`scan_all`, `Issue`)
188
+ - `io.py` -- `read_table`, the shared CSV/Excel reader
189
+ - `numeric_text.py` -- parse numbers stored as text
190
+
191
+ ## Development
192
+
193
+ ```bash
194
+ pip install -e ".[dev]"
195
+ pytest
196
+ ```
197
+
198
+ ## License
199
+
200
+ MIT
@@ -6,8 +6,11 @@ Shared preprocessing pipeline library for [xplainable](https://xplainable.io). D
6
6
 
7
7
  ```bash
8
8
  pip install xplainable-preprocessing
9
+ pip install "xplainable-preprocessing[excel]" # io.read_table for .xlsx/.xls
9
10
  ```
10
11
 
12
+ Upgrading from 0.3.x? 0.4.0 has breaking changes; see [CHANGELOG.md](CHANGELOG.md).
13
+
11
14
  ## Quick Start
12
15
 
13
16
  ### Define a pipeline as JSON
@@ -62,7 +65,9 @@ result = loaded.transform(new_data)
62
65
  | `DropColumnsTransformer` | Drop specified columns |
63
66
  | `RenameColumnsTransformer` | Rename columns |
64
67
  | `TypeCastTransformer` | Cast column dtypes (e.g. str to float64, with error coercion) |
65
- | `ClipTransformer` | Clip numeric values outside (min, max) to NaN |
68
+ | `ClipTransformer` | Set numeric values outside (min, max), or per-column `bounds`, to NaN |
69
+ | `ParseNumericTransformer` | Parse numbers stored as text (%, currency, `1,234`, `36 months`, ranges, `10+`) |
70
+ | `NullTokensTransformer` | Turn placeholder text (`?`, `N/A`) and sentinel numbers (`999`) into NaN |
66
71
 
67
72
  ### Missing Values
68
73
  | Transformer | Description |
@@ -73,17 +78,22 @@ result = loaded.transform(new_data)
73
78
  ### Feature Engineering
74
79
  | Transformer | Description |
75
80
  |-------------|-------------|
76
- | `ExpressionTransformer` | Create derived columns via pandas expressions |
77
- | `DateTimeExtractTransformer` | Extract year, month, dayofweek, etc. from datetime columns |
81
+ | `ExpressionTransformer` | Create derived columns from expressions (allow-listed evaluator) |
82
+ | `DateTimeExtractTransformer` | Extract year, month, dayofweek, elapsed time, etc. from datetime columns |
78
83
  | `GroupByAggTransformer` | Aggregate features by group (count, sum, mean) |
79
- | `GroupedLagTransformer` | Create lagged features within groups |
84
+ | `GroupedLagTransformer` | Create row-shift lags within groups |
80
85
  | `RollingAggTransformer` | Rolling window aggregations |
86
+ | `TimeLagTransformer` | Lags and trailing windows at exact time offsets over strictly earlier rows |
87
+ | `GroupBroadcastTransformer` | Write a group aggregate (e.g. a holiday flag per date) to every row of the group |
88
+ | `TextFeaturesTransformer` | Numeric features from free text (length, digits, URLs, keyword flags) |
81
89
 
82
90
  ### Categorical
83
91
  | Transformer | Description |
84
92
  |-------------|-------------|
85
93
  | `CategoryCondenseTransformer` | Condense high-cardinality categoricals to top N + "Other" |
86
- | `TextCleanTransformer` | Clean text: lowercase, strip, remove HTML/extra whitespace |
94
+ | `TextCleanTransformer` | Clean text cells: lowercase, strip, remove HTML/extra whitespace, fix encoding, mask digits/URLs/emails |
95
+ | `OrdinalMapTransformer` | Map an ordered scale (grades, ratings, bands) to numbers |
96
+ | `CrossCategoriesTransformer` | Cross 2-3 low-cardinality columns into one category |
87
97
 
88
98
  ### Scaling (use with caution -- breaks xplainable model explainability)
89
99
  | Transformer | Description |
@@ -149,8 +159,13 @@ Key modules:
149
159
  - `compiler.py` -- compiles JSON spec into sklearn pipeline
150
160
  - `pipeline.py` -- `DataFramePipeline` wrapper around sklearn `Pipeline`
151
161
  - `registry.py` -- maps transformer names to classes, generates LLM-ready catalog
152
- - `serialization.py` -- cloudpickle save/load
162
+ - `serialization.py` -- cloudpickle save/load; portable across Python versions
163
+ - `guard.py` -- step safety checks and step/spec previews
153
164
  - `preview.py` -- before/after preview with schema deltas
165
+ - `contracts.py` -- which engineered columns each step computes (`dependency_map`, `partial_transform`)
166
+ - `detect/` -- detectors that describe a raw frame (`scan_all`, `Issue`)
167
+ - `io.py` -- `read_table`, the shared CSV/Excel reader
168
+ - `numeric_text.py` -- parse numbers stored as text
154
169
 
155
170
  ## Development
156
171
 
@@ -4,22 +4,33 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "xplainable-preprocessing"
7
- version = "0.3.1"
7
+ version = "0.4.0"
8
8
  description = "Shared preprocessing pipeline package for xplainable"
9
+ readme = "README.md"
9
10
  requires-python = ">=3.9"
10
11
  dependencies = [
11
12
  "pydantic>=2.0",
12
13
  "scikit-learn>=1.3",
13
- "pandas>=2.0",
14
+ # pandas 3 is not supported yet: its copy-on-write read-only arrays and
15
+ # default string dtype break detect.target, detect.structure and a few
16
+ # transformers (see CHANGELOG 0.4.0).
17
+ "pandas>=2.0,<3",
14
18
  "numpy>=1.24",
15
19
  "scipy>=1.8",
16
20
  "cloudpickle>=3.0",
17
21
  ]
18
22
 
19
23
  [project.optional-dependencies]
24
+ # Excel uploads in xplainable_preprocessing.io.read_table: .xlsx and .xls.
25
+ excel = [
26
+ "openpyxl>=3.1",
27
+ "xlrd>=2.0.1",
28
+ ]
20
29
  dev = [
21
30
  "pytest>=7.0",
22
31
  "pytest-cov",
32
+ "openpyxl>=3.1",
33
+ "xlrd>=2.0.1",
23
34
  ]
24
35
 
25
36
  [tool.hatch.build.targets.wheel]