cleanflow-kit 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.1.0.dist-info/METADATA +119 -0
- cleanflow_kit-1.1.0.dist-info/RECORD +17 -0
- cleanflow_kit-1.1.0.dist-info/WHEEL +5 -0
- cleanflow_kit-1.1.0.dist-info/licenses/LICENSE +21 -0
- cleanflow_kit-1.1.0.dist-info/top_level.txt +1 -0
- dataclean/__init__.py +44 -0
- dataclean/_compat.py +24 -0
- dataclean/data_cleaner.py +988 -0
- dataclean/data_loader.py +194 -0
- dataclean/drift_detector.py +111 -0
- dataclean/eda.py +400 -0
- dataclean/feature_engineer.py +608 -0
- dataclean/model_trainer.py +874 -0
- dataclean/pipeline.py +548 -0
- dataclean/py.typed +0 -0
- dataclean/report_generator.py +365 -0
- dataclean/synthetic_generator.py +97 -0
|
@@ -0,0 +1,608 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Feature Engineering Module
|
|
3
|
+
==========================
|
|
4
|
+
Handles feature transformation, encoding, scaling, and creation.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import numpy as np
|
|
11
|
+
from typing import Optional, List, Dict, Any, Tuple, Union
|
|
12
|
+
from sklearn.preprocessing import (
|
|
13
|
+
StandardScaler, MinMaxScaler, LabelEncoder,
|
|
14
|
+
OneHotEncoder, PolynomialFeatures
|
|
15
|
+
)
|
|
16
|
+
from sklearn.feature_selection import VarianceThreshold, mutual_info_classif, mutual_info_regression
|
|
17
|
+
import warnings
|
|
18
|
+
from ._compat import normalize_string_columns
|
|
19
|
+
|
|
20
|
+
warnings.filterwarnings('ignore')
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class FeatureEngineer:
|
|
24
|
+
"""Feature engineering and transformation toolkit."""
|
|
25
|
+
|
|
26
|
+
def __init__(
|
|
27
|
+
self,
|
|
28
|
+
df: pd.DataFrame,
|
|
29
|
+
target_col: Optional[str] = None,
|
|
30
|
+
problem_type: Optional[str] = None
|
|
31
|
+
):
|
|
32
|
+
"""
|
|
33
|
+
Initialize feature engineer.
|
|
34
|
+
|
|
35
|
+
Parameters:
|
|
36
|
+
-----------
|
|
37
|
+
df : pd.DataFrame
|
|
38
|
+
Input DataFrame
|
|
39
|
+
target_col : str, optional
|
|
40
|
+
Target column name
|
|
41
|
+
problem_type : str, optional
|
|
42
|
+
'classification', 'regression', or None
|
|
43
|
+
"""
|
|
44
|
+
self.df = normalize_string_columns(df.copy())
|
|
45
|
+
self.target_col = target_col
|
|
46
|
+
self.problem_type = problem_type
|
|
47
|
+
self.transformations: List[str] = []
|
|
48
|
+
self.encoders: Dict[str, Any] = {}
|
|
49
|
+
self.scalers: Dict[str, Any] = {}
|
|
50
|
+
self.feature_importance: Dict[str, float] = {}
|
|
51
|
+
|
|
52
|
+
def encode_categorical(
|
|
53
|
+
self,
|
|
54
|
+
method: str = 'auto',
|
|
55
|
+
columns: Optional[List[str]] = None,
|
|
56
|
+
max_categories: int = 10,
|
|
57
|
+
drop_first: bool = True
|
|
58
|
+
) -> 'FeatureEngineer':
|
|
59
|
+
"""
|
|
60
|
+
Encode categorical variables.
|
|
61
|
+
|
|
62
|
+
Parameters:
|
|
63
|
+
-----------
|
|
64
|
+
method : str
|
|
65
|
+
'auto', 'onehot', 'label', or 'ordinal'
|
|
66
|
+
columns : list, optional
|
|
67
|
+
Specific columns to encode
|
|
68
|
+
max_categories : int
|
|
69
|
+
Max categories for one-hot encoding
|
|
70
|
+
drop_first : bool
|
|
71
|
+
Drop first category in one-hot encoding
|
|
72
|
+
"""
|
|
73
|
+
if columns is None:
|
|
74
|
+
columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
75
|
+
# Exclude target column
|
|
76
|
+
if self.target_col in columns:
|
|
77
|
+
columns.remove(self.target_col)
|
|
78
|
+
|
|
79
|
+
for col in columns:
|
|
80
|
+
if col not in self.df.columns:
|
|
81
|
+
continue
|
|
82
|
+
|
|
83
|
+
n_unique = self.df[col].nunique()
|
|
84
|
+
|
|
85
|
+
# Determine encoding method
|
|
86
|
+
if method == 'auto':
|
|
87
|
+
if n_unique == 2:
|
|
88
|
+
use_method = 'label'
|
|
89
|
+
elif n_unique <= max_categories:
|
|
90
|
+
use_method = 'onehot'
|
|
91
|
+
else:
|
|
92
|
+
use_method = 'label'
|
|
93
|
+
else:
|
|
94
|
+
use_method = method
|
|
95
|
+
|
|
96
|
+
if use_method == 'onehot':
|
|
97
|
+
# One-hot encoding
|
|
98
|
+
dummies = pd.get_dummies(
|
|
99
|
+
self.df[col],
|
|
100
|
+
prefix=col,
|
|
101
|
+
drop_first=drop_first,
|
|
102
|
+
dtype=int
|
|
103
|
+
)
|
|
104
|
+
self.df = pd.concat([self.df.drop(columns=[col]), dummies], axis=1)
|
|
105
|
+
self.transformations.append(f"One-hot encoded '{col}' ā {len(dummies.columns)} columns")
|
|
106
|
+
|
|
107
|
+
elif use_method == 'label':
|
|
108
|
+
# Label encoding
|
|
109
|
+
le = LabelEncoder()
|
|
110
|
+
# Handle NaN values
|
|
111
|
+
mask = self.df[col].notna()
|
|
112
|
+
self.df.loc[mask, col] = le.fit_transform(self.df.loc[mask, col].astype(str))
|
|
113
|
+
self.df[col] = self.df[col].astype(float)
|
|
114
|
+
self.encoders[col] = le
|
|
115
|
+
self.transformations.append(f"Label encoded '{col}'")
|
|
116
|
+
|
|
117
|
+
return self
|
|
118
|
+
|
|
119
|
+
def scale_features(
|
|
120
|
+
self,
|
|
121
|
+
method: str = 'standard',
|
|
122
|
+
columns: Optional[List[str]] = None
|
|
123
|
+
) -> 'FeatureEngineer':
|
|
124
|
+
"""
|
|
125
|
+
Scale numeric features.
|
|
126
|
+
|
|
127
|
+
Parameters:
|
|
128
|
+
-----------
|
|
129
|
+
method : str
|
|
130
|
+
'standard' (z-score) or 'minmax' (0-1 range)
|
|
131
|
+
columns : list, optional
|
|
132
|
+
Specific columns to scale
|
|
133
|
+
"""
|
|
134
|
+
if columns is None:
|
|
135
|
+
columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
136
|
+
# Exclude target column
|
|
137
|
+
if self.target_col in columns:
|
|
138
|
+
columns.remove(self.target_col)
|
|
139
|
+
|
|
140
|
+
if len(columns) == 0:
|
|
141
|
+
return self
|
|
142
|
+
|
|
143
|
+
if method == 'standard':
|
|
144
|
+
scaler = StandardScaler()
|
|
145
|
+
elif method == 'minmax':
|
|
146
|
+
scaler = MinMaxScaler()
|
|
147
|
+
else:
|
|
148
|
+
raise ValueError(f"Unknown scaling method: {method}")
|
|
149
|
+
|
|
150
|
+
# Handle missing values for scaling
|
|
151
|
+
cols_to_scale = [c for c in columns if c in self.df.columns]
|
|
152
|
+
|
|
153
|
+
if cols_to_scale:
|
|
154
|
+
self.df[cols_to_scale] = scaler.fit_transform(self.df[cols_to_scale])
|
|
155
|
+
self.scalers['main'] = scaler
|
|
156
|
+
self.transformations.append(f"{method.capitalize()} scaled {len(cols_to_scale)} numeric features")
|
|
157
|
+
|
|
158
|
+
return self
|
|
159
|
+
|
|
160
|
+
def create_datetime_features(
|
|
161
|
+
self,
|
|
162
|
+
columns: Optional[List[str]] = None,
|
|
163
|
+
features: List[str] = None
|
|
164
|
+
) -> 'FeatureEngineer':
|
|
165
|
+
"""
|
|
166
|
+
Extract features from datetime columns.
|
|
167
|
+
|
|
168
|
+
Parameters:
|
|
169
|
+
-----------
|
|
170
|
+
columns : list, optional
|
|
171
|
+
Datetime columns to process
|
|
172
|
+
features : list
|
|
173
|
+
Features to extract: 'year', 'month', 'day', 'dayofweek',
|
|
174
|
+
'hour', 'minute', 'quarter', 'is_weekend'
|
|
175
|
+
"""
|
|
176
|
+
if features is None:
|
|
177
|
+
features = ['year', 'month', 'day', 'dayofweek', 'is_weekend']
|
|
178
|
+
|
|
179
|
+
if columns is None:
|
|
180
|
+
columns = self.df.select_dtypes(include=['datetime64']).columns.tolist()
|
|
181
|
+
|
|
182
|
+
for col in columns:
|
|
183
|
+
if col not in self.df.columns:
|
|
184
|
+
continue
|
|
185
|
+
|
|
186
|
+
dt = self.df[col]
|
|
187
|
+
|
|
188
|
+
if 'year' in features:
|
|
189
|
+
self.df[f'{col}_year'] = dt.dt.year
|
|
190
|
+
if 'month' in features:
|
|
191
|
+
self.df[f'{col}_month'] = dt.dt.month
|
|
192
|
+
if 'day' in features:
|
|
193
|
+
self.df[f'{col}_day'] = dt.dt.day
|
|
194
|
+
if 'dayofweek' in features:
|
|
195
|
+
self.df[f'{col}_dayofweek'] = dt.dt.dayofweek
|
|
196
|
+
if 'quarter' in features:
|
|
197
|
+
self.df[f'{col}_quarter'] = dt.dt.quarter
|
|
198
|
+
if 'hour' in features and hasattr(dt.dt, 'hour'):
|
|
199
|
+
self.df[f'{col}_hour'] = dt.dt.hour
|
|
200
|
+
if 'minute' in features and hasattr(dt.dt, 'minute'):
|
|
201
|
+
self.df[f'{col}_minute'] = dt.dt.minute
|
|
202
|
+
if 'is_weekend' in features:
|
|
203
|
+
self.df[f'{col}_is_weekend'] = (dt.dt.dayofweek >= 5).astype(int)
|
|
204
|
+
|
|
205
|
+
# Drop original datetime column
|
|
206
|
+
self.df = self.df.drop(columns=[col])
|
|
207
|
+
self.transformations.append(f"Extracted {len(features)} features from '{col}'")
|
|
208
|
+
|
|
209
|
+
return self
|
|
210
|
+
|
|
211
|
+
def create_polynomial_features(
|
|
212
|
+
self,
|
|
213
|
+
columns: Optional[List[str]] = None,
|
|
214
|
+
degree: int = 2,
|
|
215
|
+
interaction_only: bool = False,
|
|
216
|
+
include_bias: bool = False
|
|
217
|
+
) -> 'FeatureEngineer':
|
|
218
|
+
"""
|
|
219
|
+
Create polynomial and interaction features.
|
|
220
|
+
|
|
221
|
+
Parameters:
|
|
222
|
+
-----------
|
|
223
|
+
columns : list, optional
|
|
224
|
+
Columns to use for polynomial features
|
|
225
|
+
degree : int
|
|
226
|
+
Polynomial degree
|
|
227
|
+
interaction_only : bool
|
|
228
|
+
If True, only interaction features
|
|
229
|
+
include_bias : bool
|
|
230
|
+
Include bias column
|
|
231
|
+
"""
|
|
232
|
+
if columns is None:
|
|
233
|
+
# Use top numeric columns (limit to avoid explosion)
|
|
234
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
235
|
+
if self.target_col in numeric_cols:
|
|
236
|
+
numeric_cols.remove(self.target_col)
|
|
237
|
+
columns = numeric_cols[:5] # Limit to 5 columns
|
|
238
|
+
|
|
239
|
+
if len(columns) < 2:
|
|
240
|
+
return self
|
|
241
|
+
|
|
242
|
+
poly = PolynomialFeatures(
|
|
243
|
+
degree=degree,
|
|
244
|
+
interaction_only=interaction_only,
|
|
245
|
+
include_bias=include_bias
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
# Create polynomial features
|
|
249
|
+
poly_data = poly.fit_transform(self.df[columns])
|
|
250
|
+
poly_features = poly.get_feature_names_out(columns)
|
|
251
|
+
|
|
252
|
+
# Add new features (excluding original columns)
|
|
253
|
+
new_features = poly_features[len(columns):]
|
|
254
|
+
new_data = poly_data[:, len(columns):]
|
|
255
|
+
|
|
256
|
+
for i, feat_name in enumerate(new_features):
|
|
257
|
+
self.df[feat_name] = new_data[:, i]
|
|
258
|
+
|
|
259
|
+
self.transformations.append(f"Created {len(new_features)} polynomial features (degree={degree})")
|
|
260
|
+
|
|
261
|
+
return self
|
|
262
|
+
|
|
263
|
+
def create_binned_features(
|
|
264
|
+
self,
|
|
265
|
+
columns: Optional[List[str]] = None,
|
|
266
|
+
n_bins: int = 5,
|
|
267
|
+
strategy: str = 'quantile'
|
|
268
|
+
) -> 'FeatureEngineer':
|
|
269
|
+
"""
|
|
270
|
+
Create binned versions of numeric features.
|
|
271
|
+
|
|
272
|
+
Parameters:
|
|
273
|
+
-----------
|
|
274
|
+
columns : list, optional
|
|
275
|
+
Columns to bin
|
|
276
|
+
n_bins : int
|
|
277
|
+
Number of bins
|
|
278
|
+
strategy : str
|
|
279
|
+
'quantile' or 'uniform'
|
|
280
|
+
"""
|
|
281
|
+
if columns is None:
|
|
282
|
+
columns = self.df.select_dtypes(include=[np.number]).columns.tolist()[:5]
|
|
283
|
+
if self.target_col in columns:
|
|
284
|
+
columns.remove(self.target_col)
|
|
285
|
+
|
|
286
|
+
for col in columns:
|
|
287
|
+
if col not in self.df.columns:
|
|
288
|
+
continue
|
|
289
|
+
|
|
290
|
+
if strategy == 'quantile':
|
|
291
|
+
self.df[f'{col}_binned'] = pd.qcut(
|
|
292
|
+
self.df[col], q=n_bins, labels=False, duplicates='drop'
|
|
293
|
+
)
|
|
294
|
+
else:
|
|
295
|
+
self.df[f'{col}_binned'] = pd.cut(
|
|
296
|
+
self.df[col], bins=n_bins, labels=False
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
self.transformations.append(f"Created binned features for {len(columns)} columns")
|
|
300
|
+
|
|
301
|
+
return self
|
|
302
|
+
|
|
303
|
+
def compute_feature_importance(
|
|
304
|
+
self,
|
|
305
|
+
n_features: int = 20
|
|
306
|
+
) -> Dict[str, float]:
|
|
307
|
+
"""
|
|
308
|
+
Compute feature importance using mutual information.
|
|
309
|
+
|
|
310
|
+
Parameters:
|
|
311
|
+
-----------
|
|
312
|
+
n_features : int
|
|
313
|
+
Number of top features to return
|
|
314
|
+
"""
|
|
315
|
+
if self.target_col is None or self.target_col not in self.df.columns:
|
|
316
|
+
return {}
|
|
317
|
+
|
|
318
|
+
# Get numeric features
|
|
319
|
+
feature_cols = [c for c in self.df.select_dtypes(include=[np.number]).columns
|
|
320
|
+
if c != self.target_col]
|
|
321
|
+
|
|
322
|
+
if len(feature_cols) == 0:
|
|
323
|
+
return {}
|
|
324
|
+
|
|
325
|
+
X = self.df[feature_cols].fillna(0)
|
|
326
|
+
y = self.df[self.target_col]
|
|
327
|
+
|
|
328
|
+
# Compute mutual information
|
|
329
|
+
if self.problem_type == 'classification' or y.dtype == 'object':
|
|
330
|
+
mi = mutual_info_classif(X, y, random_state=42)
|
|
331
|
+
else:
|
|
332
|
+
mi = mutual_info_regression(X, y, random_state=42)
|
|
333
|
+
|
|
334
|
+
# Create importance dictionary
|
|
335
|
+
importance = dict(zip(feature_cols, mi))
|
|
336
|
+
importance = dict(sorted(importance.items(), key=lambda x: x[1], reverse=True))
|
|
337
|
+
|
|
338
|
+
# Keep top n
|
|
339
|
+
self.feature_importance = dict(list(importance.items())[:n_features])
|
|
340
|
+
|
|
341
|
+
return self.feature_importance
|
|
342
|
+
|
|
343
|
+
def drop_low_importance_features(
|
|
344
|
+
self,
|
|
345
|
+
threshold: float = 0.01,
|
|
346
|
+
keep_n: Optional[int] = None
|
|
347
|
+
) -> 'FeatureEngineer':
|
|
348
|
+
"""
|
|
349
|
+
Drop features with low importance.
|
|
350
|
+
|
|
351
|
+
Parameters:
|
|
352
|
+
-----------
|
|
353
|
+
threshold : float
|
|
354
|
+
Minimum importance threshold
|
|
355
|
+
keep_n : int, optional
|
|
356
|
+
Keep top n features regardless of threshold
|
|
357
|
+
"""
|
|
358
|
+
if not self.feature_importance:
|
|
359
|
+
self.compute_feature_importance()
|
|
360
|
+
|
|
361
|
+
if not self.feature_importance:
|
|
362
|
+
return self
|
|
363
|
+
|
|
364
|
+
# Determine features to keep
|
|
365
|
+
if keep_n:
|
|
366
|
+
features_to_keep = list(self.feature_importance.keys())[:keep_n]
|
|
367
|
+
else:
|
|
368
|
+
features_to_keep = [f for f, imp in self.feature_importance.items()
|
|
369
|
+
if imp >= threshold]
|
|
370
|
+
|
|
371
|
+
# Add target column
|
|
372
|
+
if self.target_col:
|
|
373
|
+
features_to_keep.append(self.target_col)
|
|
374
|
+
|
|
375
|
+
# Get current columns
|
|
376
|
+
cols_to_drop = [c for c in self.df.columns if c not in features_to_keep]
|
|
377
|
+
|
|
378
|
+
if cols_to_drop:
|
|
379
|
+
self.df = self.df.drop(columns=cols_to_drop)
|
|
380
|
+
self.transformations.append(f"Dropped {len(cols_to_drop)} low-importance features")
|
|
381
|
+
|
|
382
|
+
return self
|
|
383
|
+
|
|
384
|
+
def drop_low_variance_features(
|
|
385
|
+
self,
|
|
386
|
+
threshold: float = 0.01
|
|
387
|
+
) -> 'FeatureEngineer':
|
|
388
|
+
"""
|
|
389
|
+
Drop features with low variance.
|
|
390
|
+
|
|
391
|
+
Parameters:
|
|
392
|
+
-----------
|
|
393
|
+
threshold : float
|
|
394
|
+
Variance threshold
|
|
395
|
+
"""
|
|
396
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
397
|
+
if self.target_col in numeric_cols:
|
|
398
|
+
numeric_cols.remove(self.target_col)
|
|
399
|
+
|
|
400
|
+
if len(numeric_cols) == 0:
|
|
401
|
+
return self
|
|
402
|
+
|
|
403
|
+
selector = VarianceThreshold(threshold=threshold)
|
|
404
|
+
|
|
405
|
+
try:
|
|
406
|
+
selector.fit(self.df[numeric_cols])
|
|
407
|
+
mask = selector.get_support()
|
|
408
|
+
cols_to_drop = [c for c, keep in zip(numeric_cols, mask) if not keep]
|
|
409
|
+
|
|
410
|
+
if cols_to_drop:
|
|
411
|
+
self.df = self.df.drop(columns=cols_to_drop)
|
|
412
|
+
self.transformations.append(f"Dropped {len(cols_to_drop)} low-variance features")
|
|
413
|
+
except:
|
|
414
|
+
pass
|
|
415
|
+
|
|
416
|
+
return self
|
|
417
|
+
|
|
418
|
+
def drop_highly_correlated(
|
|
419
|
+
self,
|
|
420
|
+
threshold: float = 0.95
|
|
421
|
+
) -> 'FeatureEngineer':
|
|
422
|
+
"""
|
|
423
|
+
Drop one of each pair of highly correlated features.
|
|
424
|
+
|
|
425
|
+
Parameters:
|
|
426
|
+
-----------
|
|
427
|
+
threshold : float
|
|
428
|
+
Correlation threshold
|
|
429
|
+
"""
|
|
430
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
431
|
+
if self.target_col in numeric_cols:
|
|
432
|
+
numeric_cols.remove(self.target_col)
|
|
433
|
+
|
|
434
|
+
if len(numeric_cols) < 2:
|
|
435
|
+
return self
|
|
436
|
+
|
|
437
|
+
corr_matrix = self.df[numeric_cols].corr().abs()
|
|
438
|
+
upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))
|
|
439
|
+
|
|
440
|
+
cols_to_drop = [col for col in upper.columns if any(upper[col] > threshold)]
|
|
441
|
+
|
|
442
|
+
if cols_to_drop:
|
|
443
|
+
self.df = self.df.drop(columns=cols_to_drop)
|
|
444
|
+
self.transformations.append(f"Dropped {len(cols_to_drop)} highly correlated features (r > {threshold})")
|
|
445
|
+
|
|
446
|
+
return self
|
|
447
|
+
|
|
448
|
+
def handle_class_imbalance(
|
|
449
|
+
self,
|
|
450
|
+
method: str = 'smote',
|
|
451
|
+
sampling_strategy: Union[str, float] = 'auto'
|
|
452
|
+
) -> 'FeatureEngineer':
|
|
453
|
+
"""
|
|
454
|
+
Handle class imbalance for classification problems.
|
|
455
|
+
|
|
456
|
+
Parameters:
|
|
457
|
+
-----------
|
|
458
|
+
method : str
|
|
459
|
+
'smote', 'oversample', or 'undersample'
|
|
460
|
+
sampling_strategy : str or float
|
|
461
|
+
Sampling strategy
|
|
462
|
+
"""
|
|
463
|
+
if self.target_col is None or self.problem_type != 'classification':
|
|
464
|
+
return self
|
|
465
|
+
|
|
466
|
+
try:
|
|
467
|
+
if method == 'smote':
|
|
468
|
+
from imblearn.over_sampling import SMOTE
|
|
469
|
+
|
|
470
|
+
X = self.df.drop(columns=[self.target_col])
|
|
471
|
+
y = self.df[self.target_col]
|
|
472
|
+
|
|
473
|
+
smote = SMOTE(sampling_strategy=sampling_strategy, random_state=42)
|
|
474
|
+
X_resampled, y_resampled = smote.fit_resample(X, y)
|
|
475
|
+
|
|
476
|
+
self.df = pd.concat([X_resampled, y_resampled], axis=1)
|
|
477
|
+
self.transformations.append(f"Applied SMOTE: {len(y)} ā {len(y_resampled)} samples")
|
|
478
|
+
|
|
479
|
+
elif method == 'oversample':
|
|
480
|
+
# Simple oversampling
|
|
481
|
+
max_count = self.df[self.target_col].value_counts().max()
|
|
482
|
+
dfs = []
|
|
483
|
+
for val in self.df[self.target_col].unique():
|
|
484
|
+
df_class = self.df[self.df[self.target_col] == val]
|
|
485
|
+
df_upsampled = df_class.sample(max_count, replace=True, random_state=42)
|
|
486
|
+
dfs.append(df_upsampled)
|
|
487
|
+
self.df = pd.concat(dfs).reset_index(drop=True)
|
|
488
|
+
self.transformations.append(f"Oversampled minority classes")
|
|
489
|
+
|
|
490
|
+
elif method == 'undersample':
|
|
491
|
+
# Simple undersampling
|
|
492
|
+
min_count = self.df[self.target_col].value_counts().min()
|
|
493
|
+
dfs = []
|
|
494
|
+
for val in self.df[self.target_col].unique():
|
|
495
|
+
df_class = self.df[self.df[self.target_col] == val]
|
|
496
|
+
df_downsampled = df_class.sample(min_count, random_state=42)
|
|
497
|
+
dfs.append(df_downsampled)
|
|
498
|
+
self.df = pd.concat(dfs).reset_index(drop=True)
|
|
499
|
+
self.transformations.append(f"Undersampled majority classes")
|
|
500
|
+
|
|
501
|
+
except ImportError:
|
|
502
|
+
self.transformations.append("SMOTE not available (install imbalanced-learn)")
|
|
503
|
+
except Exception as e:
|
|
504
|
+
self.transformations.append(f"Could not handle imbalance: {e}")
|
|
505
|
+
|
|
506
|
+
return self
|
|
507
|
+
|
|
508
|
+
def auto_evolve(self, max_new_features: int = 10) -> 'FeatureEngineer':
|
|
509
|
+
"""
|
|
510
|
+
Automatically generate and select best interaction features ("Super Features").
|
|
511
|
+
"""
|
|
512
|
+
if self.target_col is None:
|
|
513
|
+
return self
|
|
514
|
+
|
|
515
|
+
# 1. Identify top numeric features
|
|
516
|
+
if not self.feature_importance:
|
|
517
|
+
self.compute_feature_importance()
|
|
518
|
+
|
|
519
|
+
# Filter for numeric only
|
|
520
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
521
|
+
if self.target_col in numeric_cols:
|
|
522
|
+
numeric_cols.remove(self.target_col)
|
|
523
|
+
|
|
524
|
+
# Get top features from importance
|
|
525
|
+
top_features = [f for f in self.feature_importance.keys() if f in numeric_cols][:5]
|
|
526
|
+
|
|
527
|
+
if len(top_features) < 2:
|
|
528
|
+
return self
|
|
529
|
+
|
|
530
|
+
initial_features = set(self.df.columns)
|
|
531
|
+
|
|
532
|
+
# 2. Generate interactions
|
|
533
|
+
# We use PolynomialFeatures with interaction_only=True to get A*B terms
|
|
534
|
+
poly = PolynomialFeatures(degree=2, interaction_only=True, include_bias=False)
|
|
535
|
+
try:
|
|
536
|
+
poly_data = poly.fit_transform(self.df[top_features])
|
|
537
|
+
new_feature_names = poly.get_feature_names_out(top_features)
|
|
538
|
+
|
|
539
|
+
# Add new features to df
|
|
540
|
+
# Skip the first len(top_features) as they are the original ones
|
|
541
|
+
new_feats = new_feature_names[len(top_features):]
|
|
542
|
+
new_data = poly_data[:, len(top_features):]
|
|
543
|
+
|
|
544
|
+
created_features = []
|
|
545
|
+
for i, feat_name in enumerate(new_feats):
|
|
546
|
+
# Clean name (replace spaces with _)
|
|
547
|
+
clean_name = feat_name.replace(' ', '_x_')
|
|
548
|
+
self.df[clean_name] = new_data[:, i]
|
|
549
|
+
created_features.append(clean_name)
|
|
550
|
+
|
|
551
|
+
# 3. Re-evaluate importance
|
|
552
|
+
self.compute_feature_importance()
|
|
553
|
+
|
|
554
|
+
# 4. Filter: Keep only features that have decent importance
|
|
555
|
+
# Threshold: median importance of original features? or just > 0.01?
|
|
556
|
+
# Let's simple keep top N new features
|
|
557
|
+
new_feat_importance = {f: self.feature_importance.get(f, 0) for f in created_features}
|
|
558
|
+
|
|
559
|
+
# Sort by importance
|
|
560
|
+
sorted_new = sorted(new_feat_importance.items(), key=lambda x: x[1], reverse=True)
|
|
561
|
+
|
|
562
|
+
# Keep top max_new_features
|
|
563
|
+
keep_features = [f for f, imp in sorted_new[:max_new_features] if imp > 0.001]
|
|
564
|
+
drop_features = [f for f in created_features if f not in keep_features]
|
|
565
|
+
|
|
566
|
+
if drop_features:
|
|
567
|
+
self.df = self.df.drop(columns=drop_features)
|
|
568
|
+
|
|
569
|
+
if keep_features:
|
|
570
|
+
self.transformations.append(f"Auto-evolved {len(keep_features)} new interaction features: {', '.join(keep_features[:3])}...")
|
|
571
|
+
|
|
572
|
+
except Exception as e:
|
|
573
|
+
self.transformations.append(f"Auto-evolution failed: {e}")
|
|
574
|
+
|
|
575
|
+
return self
|
|
576
|
+
|
|
577
|
+
def get_transformed_data(self) -> pd.DataFrame:
|
|
578
|
+
"""Return the transformed DataFrame."""
|
|
579
|
+
return self.df
|
|
580
|
+
|
|
581
|
+
def get_summary(self) -> Dict[str, Any]:
|
|
582
|
+
"""Return summary of all transformations."""
|
|
583
|
+
return {
|
|
584
|
+
"final_shape": self.df.shape,
|
|
585
|
+
"transformations": self.transformations,
|
|
586
|
+
"encoders": list(self.encoders.keys()),
|
|
587
|
+
"feature_importance": self.feature_importance
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
def print_summary(self) -> None:
|
|
591
|
+
"""Print transformation summary."""
|
|
592
|
+
summary = self.get_summary()
|
|
593
|
+
|
|
594
|
+
print("=" * 60)
|
|
595
|
+
logger.info("FEATURE ENGINEERING SUMMARY")
|
|
596
|
+
print("=" * 60)
|
|
597
|
+
logger.info(f"\nš Final shape: {summary['final_shape']}")
|
|
598
|
+
|
|
599
|
+
logger.info("\nš§ Transformations applied:")
|
|
600
|
+
for t in summary['transformations']:
|
|
601
|
+
logger.info(f" ⢠{t}")
|
|
602
|
+
|
|
603
|
+
if summary['feature_importance']:
|
|
604
|
+
logger.info("\nš Top Feature Importances:")
|
|
605
|
+
for feat, imp in list(summary['feature_importance'].items())[:10]:
|
|
606
|
+
logger.info(f" ⢠{feat}: {imp:.4f}")
|
|
607
|
+
|
|
608
|
+
print("\n" + "=" * 60)
|