cleanflow-kit 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,608 @@
1
+ """
2
+ Feature Engineering Module
3
+ ==========================
4
+ Handles feature transformation, encoding, scaling, and creation.
5
+ """
6
+
7
+ import logging
8
+ logger = logging.getLogger(__name__)
9
+ import pandas as pd
10
+ import numpy as np
11
+ from typing import Optional, List, Dict, Any, Tuple, Union
12
+ from sklearn.preprocessing import (
13
+ StandardScaler, MinMaxScaler, LabelEncoder,
14
+ OneHotEncoder, PolynomialFeatures
15
+ )
16
+ from sklearn.feature_selection import VarianceThreshold, mutual_info_classif, mutual_info_regression
17
+ import warnings
18
+ from ._compat import normalize_string_columns
19
+
20
+ warnings.filterwarnings('ignore')
21
+
22
+
23
+ class FeatureEngineer:
24
+ """Feature engineering and transformation toolkit."""
25
+
26
+ def __init__(
27
+ self,
28
+ df: pd.DataFrame,
29
+ target_col: Optional[str] = None,
30
+ problem_type: Optional[str] = None
31
+ ):
32
+ """
33
+ Initialize feature engineer.
34
+
35
+ Parameters:
36
+ -----------
37
+ df : pd.DataFrame
38
+ Input DataFrame
39
+ target_col : str, optional
40
+ Target column name
41
+ problem_type : str, optional
42
+ 'classification', 'regression', or None
43
+ """
44
+ self.df = normalize_string_columns(df.copy())
45
+ self.target_col = target_col
46
+ self.problem_type = problem_type
47
+ self.transformations: List[str] = []
48
+ self.encoders: Dict[str, Any] = {}
49
+ self.scalers: Dict[str, Any] = {}
50
+ self.feature_importance: Dict[str, float] = {}
51
+
52
+ def encode_categorical(
53
+ self,
54
+ method: str = 'auto',
55
+ columns: Optional[List[str]] = None,
56
+ max_categories: int = 10,
57
+ drop_first: bool = True
58
+ ) -> 'FeatureEngineer':
59
+ """
60
+ Encode categorical variables.
61
+
62
+ Parameters:
63
+ -----------
64
+ method : str
65
+ 'auto', 'onehot', 'label', or 'ordinal'
66
+ columns : list, optional
67
+ Specific columns to encode
68
+ max_categories : int
69
+ Max categories for one-hot encoding
70
+ drop_first : bool
71
+ Drop first category in one-hot encoding
72
+ """
73
+ if columns is None:
74
+ columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
75
+ # Exclude target column
76
+ if self.target_col in columns:
77
+ columns.remove(self.target_col)
78
+
79
+ for col in columns:
80
+ if col not in self.df.columns:
81
+ continue
82
+
83
+ n_unique = self.df[col].nunique()
84
+
85
+ # Determine encoding method
86
+ if method == 'auto':
87
+ if n_unique == 2:
88
+ use_method = 'label'
89
+ elif n_unique <= max_categories:
90
+ use_method = 'onehot'
91
+ else:
92
+ use_method = 'label'
93
+ else:
94
+ use_method = method
95
+
96
+ if use_method == 'onehot':
97
+ # One-hot encoding
98
+ dummies = pd.get_dummies(
99
+ self.df[col],
100
+ prefix=col,
101
+ drop_first=drop_first,
102
+ dtype=int
103
+ )
104
+ self.df = pd.concat([self.df.drop(columns=[col]), dummies], axis=1)
105
+ self.transformations.append(f"One-hot encoded '{col}' → {len(dummies.columns)} columns")
106
+
107
+ elif use_method == 'label':
108
+ # Label encoding
109
+ le = LabelEncoder()
110
+ # Handle NaN values
111
+ mask = self.df[col].notna()
112
+ self.df.loc[mask, col] = le.fit_transform(self.df.loc[mask, col].astype(str))
113
+ self.df[col] = self.df[col].astype(float)
114
+ self.encoders[col] = le
115
+ self.transformations.append(f"Label encoded '{col}'")
116
+
117
+ return self
118
+
119
+ def scale_features(
120
+ self,
121
+ method: str = 'standard',
122
+ columns: Optional[List[str]] = None
123
+ ) -> 'FeatureEngineer':
124
+ """
125
+ Scale numeric features.
126
+
127
+ Parameters:
128
+ -----------
129
+ method : str
130
+ 'standard' (z-score) or 'minmax' (0-1 range)
131
+ columns : list, optional
132
+ Specific columns to scale
133
+ """
134
+ if columns is None:
135
+ columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
136
+ # Exclude target column
137
+ if self.target_col in columns:
138
+ columns.remove(self.target_col)
139
+
140
+ if len(columns) == 0:
141
+ return self
142
+
143
+ if method == 'standard':
144
+ scaler = StandardScaler()
145
+ elif method == 'minmax':
146
+ scaler = MinMaxScaler()
147
+ else:
148
+ raise ValueError(f"Unknown scaling method: {method}")
149
+
150
+ # Handle missing values for scaling
151
+ cols_to_scale = [c for c in columns if c in self.df.columns]
152
+
153
+ if cols_to_scale:
154
+ self.df[cols_to_scale] = scaler.fit_transform(self.df[cols_to_scale])
155
+ self.scalers['main'] = scaler
156
+ self.transformations.append(f"{method.capitalize()} scaled {len(cols_to_scale)} numeric features")
157
+
158
+ return self
159
+
160
+ def create_datetime_features(
161
+ self,
162
+ columns: Optional[List[str]] = None,
163
+ features: List[str] = None
164
+ ) -> 'FeatureEngineer':
165
+ """
166
+ Extract features from datetime columns.
167
+
168
+ Parameters:
169
+ -----------
170
+ columns : list, optional
171
+ Datetime columns to process
172
+ features : list
173
+ Features to extract: 'year', 'month', 'day', 'dayofweek',
174
+ 'hour', 'minute', 'quarter', 'is_weekend'
175
+ """
176
+ if features is None:
177
+ features = ['year', 'month', 'day', 'dayofweek', 'is_weekend']
178
+
179
+ if columns is None:
180
+ columns = self.df.select_dtypes(include=['datetime64']).columns.tolist()
181
+
182
+ for col in columns:
183
+ if col not in self.df.columns:
184
+ continue
185
+
186
+ dt = self.df[col]
187
+
188
+ if 'year' in features:
189
+ self.df[f'{col}_year'] = dt.dt.year
190
+ if 'month' in features:
191
+ self.df[f'{col}_month'] = dt.dt.month
192
+ if 'day' in features:
193
+ self.df[f'{col}_day'] = dt.dt.day
194
+ if 'dayofweek' in features:
195
+ self.df[f'{col}_dayofweek'] = dt.dt.dayofweek
196
+ if 'quarter' in features:
197
+ self.df[f'{col}_quarter'] = dt.dt.quarter
198
+ if 'hour' in features and hasattr(dt.dt, 'hour'):
199
+ self.df[f'{col}_hour'] = dt.dt.hour
200
+ if 'minute' in features and hasattr(dt.dt, 'minute'):
201
+ self.df[f'{col}_minute'] = dt.dt.minute
202
+ if 'is_weekend' in features:
203
+ self.df[f'{col}_is_weekend'] = (dt.dt.dayofweek >= 5).astype(int)
204
+
205
+ # Drop original datetime column
206
+ self.df = self.df.drop(columns=[col])
207
+ self.transformations.append(f"Extracted {len(features)} features from '{col}'")
208
+
209
+ return self
210
+
211
+ def create_polynomial_features(
212
+ self,
213
+ columns: Optional[List[str]] = None,
214
+ degree: int = 2,
215
+ interaction_only: bool = False,
216
+ include_bias: bool = False
217
+ ) -> 'FeatureEngineer':
218
+ """
219
+ Create polynomial and interaction features.
220
+
221
+ Parameters:
222
+ -----------
223
+ columns : list, optional
224
+ Columns to use for polynomial features
225
+ degree : int
226
+ Polynomial degree
227
+ interaction_only : bool
228
+ If True, only interaction features
229
+ include_bias : bool
230
+ Include bias column
231
+ """
232
+ if columns is None:
233
+ # Use top numeric columns (limit to avoid explosion)
234
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
235
+ if self.target_col in numeric_cols:
236
+ numeric_cols.remove(self.target_col)
237
+ columns = numeric_cols[:5] # Limit to 5 columns
238
+
239
+ if len(columns) < 2:
240
+ return self
241
+
242
+ poly = PolynomialFeatures(
243
+ degree=degree,
244
+ interaction_only=interaction_only,
245
+ include_bias=include_bias
246
+ )
247
+
248
+ # Create polynomial features
249
+ poly_data = poly.fit_transform(self.df[columns])
250
+ poly_features = poly.get_feature_names_out(columns)
251
+
252
+ # Add new features (excluding original columns)
253
+ new_features = poly_features[len(columns):]
254
+ new_data = poly_data[:, len(columns):]
255
+
256
+ for i, feat_name in enumerate(new_features):
257
+ self.df[feat_name] = new_data[:, i]
258
+
259
+ self.transformations.append(f"Created {len(new_features)} polynomial features (degree={degree})")
260
+
261
+ return self
262
+
263
+ def create_binned_features(
264
+ self,
265
+ columns: Optional[List[str]] = None,
266
+ n_bins: int = 5,
267
+ strategy: str = 'quantile'
268
+ ) -> 'FeatureEngineer':
269
+ """
270
+ Create binned versions of numeric features.
271
+
272
+ Parameters:
273
+ -----------
274
+ columns : list, optional
275
+ Columns to bin
276
+ n_bins : int
277
+ Number of bins
278
+ strategy : str
279
+ 'quantile' or 'uniform'
280
+ """
281
+ if columns is None:
282
+ columns = self.df.select_dtypes(include=[np.number]).columns.tolist()[:5]
283
+ if self.target_col in columns:
284
+ columns.remove(self.target_col)
285
+
286
+ for col in columns:
287
+ if col not in self.df.columns:
288
+ continue
289
+
290
+ if strategy == 'quantile':
291
+ self.df[f'{col}_binned'] = pd.qcut(
292
+ self.df[col], q=n_bins, labels=False, duplicates='drop'
293
+ )
294
+ else:
295
+ self.df[f'{col}_binned'] = pd.cut(
296
+ self.df[col], bins=n_bins, labels=False
297
+ )
298
+
299
+ self.transformations.append(f"Created binned features for {len(columns)} columns")
300
+
301
+ return self
302
+
303
+ def compute_feature_importance(
304
+ self,
305
+ n_features: int = 20
306
+ ) -> Dict[str, float]:
307
+ """
308
+ Compute feature importance using mutual information.
309
+
310
+ Parameters:
311
+ -----------
312
+ n_features : int
313
+ Number of top features to return
314
+ """
315
+ if self.target_col is None or self.target_col not in self.df.columns:
316
+ return {}
317
+
318
+ # Get numeric features
319
+ feature_cols = [c for c in self.df.select_dtypes(include=[np.number]).columns
320
+ if c != self.target_col]
321
+
322
+ if len(feature_cols) == 0:
323
+ return {}
324
+
325
+ X = self.df[feature_cols].fillna(0)
326
+ y = self.df[self.target_col]
327
+
328
+ # Compute mutual information
329
+ if self.problem_type == 'classification' or y.dtype == 'object':
330
+ mi = mutual_info_classif(X, y, random_state=42)
331
+ else:
332
+ mi = mutual_info_regression(X, y, random_state=42)
333
+
334
+ # Create importance dictionary
335
+ importance = dict(zip(feature_cols, mi))
336
+ importance = dict(sorted(importance.items(), key=lambda x: x[1], reverse=True))
337
+
338
+ # Keep top n
339
+ self.feature_importance = dict(list(importance.items())[:n_features])
340
+
341
+ return self.feature_importance
342
+
343
+ def drop_low_importance_features(
344
+ self,
345
+ threshold: float = 0.01,
346
+ keep_n: Optional[int] = None
347
+ ) -> 'FeatureEngineer':
348
+ """
349
+ Drop features with low importance.
350
+
351
+ Parameters:
352
+ -----------
353
+ threshold : float
354
+ Minimum importance threshold
355
+ keep_n : int, optional
356
+ Keep top n features regardless of threshold
357
+ """
358
+ if not self.feature_importance:
359
+ self.compute_feature_importance()
360
+
361
+ if not self.feature_importance:
362
+ return self
363
+
364
+ # Determine features to keep
365
+ if keep_n:
366
+ features_to_keep = list(self.feature_importance.keys())[:keep_n]
367
+ else:
368
+ features_to_keep = [f for f, imp in self.feature_importance.items()
369
+ if imp >= threshold]
370
+
371
+ # Add target column
372
+ if self.target_col:
373
+ features_to_keep.append(self.target_col)
374
+
375
+ # Get current columns
376
+ cols_to_drop = [c for c in self.df.columns if c not in features_to_keep]
377
+
378
+ if cols_to_drop:
379
+ self.df = self.df.drop(columns=cols_to_drop)
380
+ self.transformations.append(f"Dropped {len(cols_to_drop)} low-importance features")
381
+
382
+ return self
383
+
384
+ def drop_low_variance_features(
385
+ self,
386
+ threshold: float = 0.01
387
+ ) -> 'FeatureEngineer':
388
+ """
389
+ Drop features with low variance.
390
+
391
+ Parameters:
392
+ -----------
393
+ threshold : float
394
+ Variance threshold
395
+ """
396
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
397
+ if self.target_col in numeric_cols:
398
+ numeric_cols.remove(self.target_col)
399
+
400
+ if len(numeric_cols) == 0:
401
+ return self
402
+
403
+ selector = VarianceThreshold(threshold=threshold)
404
+
405
+ try:
406
+ selector.fit(self.df[numeric_cols])
407
+ mask = selector.get_support()
408
+ cols_to_drop = [c for c, keep in zip(numeric_cols, mask) if not keep]
409
+
410
+ if cols_to_drop:
411
+ self.df = self.df.drop(columns=cols_to_drop)
412
+ self.transformations.append(f"Dropped {len(cols_to_drop)} low-variance features")
413
+ except:
414
+ pass
415
+
416
+ return self
417
+
418
+ def drop_highly_correlated(
419
+ self,
420
+ threshold: float = 0.95
421
+ ) -> 'FeatureEngineer':
422
+ """
423
+ Drop one of each pair of highly correlated features.
424
+
425
+ Parameters:
426
+ -----------
427
+ threshold : float
428
+ Correlation threshold
429
+ """
430
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
431
+ if self.target_col in numeric_cols:
432
+ numeric_cols.remove(self.target_col)
433
+
434
+ if len(numeric_cols) < 2:
435
+ return self
436
+
437
+ corr_matrix = self.df[numeric_cols].corr().abs()
438
+ upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))
439
+
440
+ cols_to_drop = [col for col in upper.columns if any(upper[col] > threshold)]
441
+
442
+ if cols_to_drop:
443
+ self.df = self.df.drop(columns=cols_to_drop)
444
+ self.transformations.append(f"Dropped {len(cols_to_drop)} highly correlated features (r > {threshold})")
445
+
446
+ return self
447
+
448
+ def handle_class_imbalance(
449
+ self,
450
+ method: str = 'smote',
451
+ sampling_strategy: Union[str, float] = 'auto'
452
+ ) -> 'FeatureEngineer':
453
+ """
454
+ Handle class imbalance for classification problems.
455
+
456
+ Parameters:
457
+ -----------
458
+ method : str
459
+ 'smote', 'oversample', or 'undersample'
460
+ sampling_strategy : str or float
461
+ Sampling strategy
462
+ """
463
+ if self.target_col is None or self.problem_type != 'classification':
464
+ return self
465
+
466
+ try:
467
+ if method == 'smote':
468
+ from imblearn.over_sampling import SMOTE
469
+
470
+ X = self.df.drop(columns=[self.target_col])
471
+ y = self.df[self.target_col]
472
+
473
+ smote = SMOTE(sampling_strategy=sampling_strategy, random_state=42)
474
+ X_resampled, y_resampled = smote.fit_resample(X, y)
475
+
476
+ self.df = pd.concat([X_resampled, y_resampled], axis=1)
477
+ self.transformations.append(f"Applied SMOTE: {len(y)} → {len(y_resampled)} samples")
478
+
479
+ elif method == 'oversample':
480
+ # Simple oversampling
481
+ max_count = self.df[self.target_col].value_counts().max()
482
+ dfs = []
483
+ for val in self.df[self.target_col].unique():
484
+ df_class = self.df[self.df[self.target_col] == val]
485
+ df_upsampled = df_class.sample(max_count, replace=True, random_state=42)
486
+ dfs.append(df_upsampled)
487
+ self.df = pd.concat(dfs).reset_index(drop=True)
488
+ self.transformations.append(f"Oversampled minority classes")
489
+
490
+ elif method == 'undersample':
491
+ # Simple undersampling
492
+ min_count = self.df[self.target_col].value_counts().min()
493
+ dfs = []
494
+ for val in self.df[self.target_col].unique():
495
+ df_class = self.df[self.df[self.target_col] == val]
496
+ df_downsampled = df_class.sample(min_count, random_state=42)
497
+ dfs.append(df_downsampled)
498
+ self.df = pd.concat(dfs).reset_index(drop=True)
499
+ self.transformations.append(f"Undersampled majority classes")
500
+
501
+ except ImportError:
502
+ self.transformations.append("SMOTE not available (install imbalanced-learn)")
503
+ except Exception as e:
504
+ self.transformations.append(f"Could not handle imbalance: {e}")
505
+
506
+ return self
507
+
508
+ def auto_evolve(self, max_new_features: int = 10) -> 'FeatureEngineer':
509
+ """
510
+ Automatically generate and select best interaction features ("Super Features").
511
+ """
512
+ if self.target_col is None:
513
+ return self
514
+
515
+ # 1. Identify top numeric features
516
+ if not self.feature_importance:
517
+ self.compute_feature_importance()
518
+
519
+ # Filter for numeric only
520
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns.tolist()
521
+ if self.target_col in numeric_cols:
522
+ numeric_cols.remove(self.target_col)
523
+
524
+ # Get top features from importance
525
+ top_features = [f for f in self.feature_importance.keys() if f in numeric_cols][:5]
526
+
527
+ if len(top_features) < 2:
528
+ return self
529
+
530
+ initial_features = set(self.df.columns)
531
+
532
+ # 2. Generate interactions
533
+ # We use PolynomialFeatures with interaction_only=True to get A*B terms
534
+ poly = PolynomialFeatures(degree=2, interaction_only=True, include_bias=False)
535
+ try:
536
+ poly_data = poly.fit_transform(self.df[top_features])
537
+ new_feature_names = poly.get_feature_names_out(top_features)
538
+
539
+ # Add new features to df
540
+ # Skip the first len(top_features) as they are the original ones
541
+ new_feats = new_feature_names[len(top_features):]
542
+ new_data = poly_data[:, len(top_features):]
543
+
544
+ created_features = []
545
+ for i, feat_name in enumerate(new_feats):
546
+ # Clean name (replace spaces with _)
547
+ clean_name = feat_name.replace(' ', '_x_')
548
+ self.df[clean_name] = new_data[:, i]
549
+ created_features.append(clean_name)
550
+
551
+ # 3. Re-evaluate importance
552
+ self.compute_feature_importance()
553
+
554
+ # 4. Filter: Keep only features that have decent importance
555
+ # Threshold: median importance of original features? or just > 0.01?
556
+ # Let's simple keep top N new features
557
+ new_feat_importance = {f: self.feature_importance.get(f, 0) for f in created_features}
558
+
559
+ # Sort by importance
560
+ sorted_new = sorted(new_feat_importance.items(), key=lambda x: x[1], reverse=True)
561
+
562
+ # Keep top max_new_features
563
+ keep_features = [f for f, imp in sorted_new[:max_new_features] if imp > 0.001]
564
+ drop_features = [f for f in created_features if f not in keep_features]
565
+
566
+ if drop_features:
567
+ self.df = self.df.drop(columns=drop_features)
568
+
569
+ if keep_features:
570
+ self.transformations.append(f"Auto-evolved {len(keep_features)} new interaction features: {', '.join(keep_features[:3])}...")
571
+
572
+ except Exception as e:
573
+ self.transformations.append(f"Auto-evolution failed: {e}")
574
+
575
+ return self
576
+
577
+ def get_transformed_data(self) -> pd.DataFrame:
578
+ """Return the transformed DataFrame."""
579
+ return self.df
580
+
581
+ def get_summary(self) -> Dict[str, Any]:
582
+ """Return summary of all transformations."""
583
+ return {
584
+ "final_shape": self.df.shape,
585
+ "transformations": self.transformations,
586
+ "encoders": list(self.encoders.keys()),
587
+ "feature_importance": self.feature_importance
588
+ }
589
+
590
+ def print_summary(self) -> None:
591
+ """Print transformation summary."""
592
+ summary = self.get_summary()
593
+
594
+ print("=" * 60)
595
+ logger.info("FEATURE ENGINEERING SUMMARY")
596
+ print("=" * 60)
597
+ logger.info(f"\nšŸ“Š Final shape: {summary['final_shape']}")
598
+
599
+ logger.info("\nšŸ”§ Transformations applied:")
600
+ for t in summary['transformations']:
601
+ logger.info(f" • {t}")
602
+
603
+ if summary['feature_importance']:
604
+ logger.info("\nšŸ“ˆ Top Feature Importances:")
605
+ for feat, imp in list(summary['feature_importance'].items())[:10]:
606
+ logger.info(f" • {feat}: {imp:.4f}")
607
+
608
+ print("\n" + "=" * 60)