cleanflow-kit 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,988 @@
1
+ """
2
+ Data Cleaner Module
3
+ ===================
4
+ Handles data cleaning operations including missing values,
5
+ duplicates, outliers, and data type corrections.
6
+ """
7
+
8
+ import logging
9
+ logger = logging.getLogger(__name__)
10
+ import pandas as pd
11
+ import numpy as np
12
+ from typing import Optional, Dict, List, Any, Union
13
+ from sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler, LabelEncoder, OneHotEncoder
14
+ from sklearn.feature_selection import VarianceThreshold
15
+ import re
16
+ from ._compat import normalize_string_columns
17
+
18
+
19
+ class DataCleaner:
20
+ """Clean and preprocess datasets."""
21
+
22
+ def __init__(self, df: pd.DataFrame):
23
+ """
24
+ Initialize cleaner with a DataFrame.
25
+
26
+ Parameters:
27
+ -----------
28
+ df : pd.DataFrame
29
+ Input DataFrame to clean
30
+ """
31
+ self.df = normalize_string_columns(df.copy())
32
+ # Convert nullable integer/float extension types to standard NumPy
33
+ # dtypes for compatibility with NumPy/scikit-learn.
34
+ for col in self.df.columns:
35
+ if pd.api.types.is_integer_dtype(self.df[col]) and hasattr(self.df[col].dtype, 'numpy_dtype'):
36
+ self.df[col] = self.df[col].astype(self.df[col].dtype.numpy_dtype)
37
+ elif pd.api.types.is_float_dtype(self.df[col]) and hasattr(self.df[col].dtype, 'numpy_dtype'):
38
+ self.df[col] = self.df[col].astype(self.df[col].dtype.numpy_dtype)
39
+ self.original_shape = df.shape
40
+ self.cleaning_log: List[str] = []
41
+ self.transformations: List[Dict[str, Any]] = []
42
+ self.row_changes: List[Dict[str, Any]] = []
43
+
44
+ def remove_duplicates(
45
+ self,
46
+ subset: Optional[List[str]] = None,
47
+ keep: str = 'first'
48
+ ) -> 'DataCleaner':
49
+ """
50
+ Remove duplicate rows.
51
+
52
+ Parameters:
53
+ -----------
54
+ subset : list, optional
55
+ Columns to consider for duplicates
56
+ keep : str
57
+ Which duplicate to keep ('first', 'last', False)
58
+ """
59
+ before = len(self.df)
60
+
61
+ # Identify duplicates for logging before removing
62
+ dupes = self.df[self.df.duplicated(subset=subset, keep=keep)]
63
+ if not dupes.empty:
64
+ for idx in dupes.index[:2]:
65
+ # Create a string representation of the row (first 3 cols)
66
+ row_str = ", ".join([str(x) for x in self.df.loc[idx].values[:3]]) + "..."
67
+ self.row_changes.append({
68
+ "index": int(idx),
69
+ "column": "Row",
70
+ "old_value": row_str,
71
+ "new_value": "Deleted",
72
+ "operation": "Row Removed",
73
+ "reason": "Duplicate"
74
+ })
75
+
76
+ self.df = self.df.drop_duplicates(subset=subset, keep=keep)
77
+ removed = before - len(self.df)
78
+
79
+ if removed > 0:
80
+ msg = f"Removed {removed:,} duplicate rows"
81
+ self.cleaning_log.append(msg)
82
+ self.transformations.append({
83
+ "operation": "remove_duplicates",
84
+ "rows_removed": removed
85
+ })
86
+
87
+ return self
88
+
89
+ def handle_missing_values(
90
+ self,
91
+ strategy: str = 'auto',
92
+ numeric_strategy: str = 'median',
93
+ categorical_strategy: str = 'mode',
94
+ drop_threshold: float = 0.4,
95
+ fill_value: Optional[Any] = None
96
+ ) -> 'DataCleaner':
97
+ """
98
+ Handle missing values in the dataset.
99
+
100
+ Parameters:
101
+ -----------
102
+ strategy : str
103
+ 'auto', 'drop_rows', 'drop_cols', 'fill'
104
+ numeric_strategy : str
105
+ Strategy for numeric columns: 'mean', 'median', 'zero'
106
+ categorical_strategy : str
107
+ Strategy for categorical: 'mode', 'unknown'
108
+ drop_threshold : float
109
+ Drop columns with missing > threshold (0-1)
110
+ fill_value : any, optional
111
+ Custom fill value when strategy='fill'
112
+ """
113
+ missing_before = self.df.isnull().sum().sum()
114
+
115
+ if missing_before == 0:
116
+ self.cleaning_log.append("No missing values to handle")
117
+ return self
118
+
119
+ # Drop columns with too many missing values
120
+ missing_pct = self.df.isnull().sum() / len(self.df)
121
+ cols_to_drop = missing_pct[missing_pct > drop_threshold].index.tolist()
122
+
123
+ if cols_to_drop:
124
+ self.df = self.df.drop(columns=cols_to_drop)
125
+ msg = f"Dropped {len(cols_to_drop)} columns with >{drop_threshold*100}% missing: {cols_to_drop}"
126
+ self.cleaning_log.append(msg)
127
+ self.transformations.append({
128
+ "operation": "drop_high_missing_columns",
129
+ "columns": cols_to_drop,
130
+ "threshold": drop_threshold
131
+ })
132
+
133
+ if strategy == 'drop_rows':
134
+ before = len(self.df)
135
+ self.df = self.df.dropna()
136
+ msg = f"Dropped {before - len(self.df):,} rows with missing values"
137
+ self.cleaning_log.append(msg)
138
+ return self
139
+
140
+ # Fill numeric columns
141
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns
142
+ for col in numeric_cols:
143
+ if self.df[col].isnull().any():
144
+ # Log usage
145
+ missing_indices = self.df[self.df[col].isnull()].index.tolist()
146
+
147
+ # Determine strategy name and value
148
+ strat_name = "Imputed (Custom)"
149
+ val_used = fill_value
150
+
151
+ if fill_value is not None:
152
+ self.df[col] = self.df[col].fillna(fill_value)
153
+ strat_name = f"Filled {fill_value}"
154
+ elif numeric_strategy == 'mean':
155
+ val_used = self.df[col].mean()
156
+ self.df[col] = self.df[col].fillna(val_used)
157
+ strat_name = "Imputed Mean"
158
+ elif numeric_strategy == 'median':
159
+ val_used = self.df[col].median()
160
+ self.df[col] = self.df[col].fillna(val_used)
161
+ strat_name = "Imputed Median"
162
+ elif numeric_strategy == 'zero':
163
+ val_used = 0
164
+ self.df[col] = self.df[col].fillna(0)
165
+ strat_name = "Filled Zero"
166
+
167
+ # Capture 2 examples
168
+ for idx in missing_indices[:2]:
169
+ self.row_changes.append({
170
+ "index": int(idx),
171
+ "column": col,
172
+ "old_value": "NaN",
173
+ "new_value": round(float(val_used), 4) if isinstance(val_used, (float, int)) else str(val_used),
174
+ "operation": strat_name,
175
+ "reason": "Missing Value"
176
+ })
177
+
178
+ # Fill categorical columns
179
+ categorical_cols = self.df.select_dtypes(include=['object', 'category']).columns
180
+ for col in categorical_cols:
181
+ if self.df[col].isnull().any():
182
+ missing_indices = self.df[self.df[col].isnull()].index.tolist()
183
+ strat_name = "Imputed (Custom)"
184
+ val_used = fill_value
185
+
186
+ if fill_value is not None:
187
+ self.df[col] = self.df[col].fillna(fill_value)
188
+ strat_name = f"Filled {fill_value}"
189
+ elif categorical_strategy == 'mode':
190
+ mode_val = self.df[col].mode()
191
+ if len(mode_val) > 0:
192
+ val_used = mode_val[0]
193
+ self.df[col] = self.df[col].fillna(val_used)
194
+ strat_name = "Imputed Mode"
195
+ elif categorical_strategy == 'unknown':
196
+ val_used = "Unknown"
197
+ self.df[col] = self.df[col].fillna('Unknown')
198
+ strat_name = "Filled 'Unknown'"
199
+
200
+ # Capture 2 examples
201
+ for idx in missing_indices[:2]:
202
+ self.row_changes.append({
203
+ "index": int(idx),
204
+ "column": col,
205
+ "old_value": "NaN",
206
+ "new_value": str(val_used),
207
+ "operation": strat_name,
208
+ "reason": "Missing Value"
209
+ })
210
+
211
+ missing_after = self.df.isnull().sum().sum()
212
+ msg = f"Handled missing values: {missing_before:,} → {missing_after:,}"
213
+ self.cleaning_log.append(msg)
214
+ self.transformations.append({
215
+ "operation": "handle_missing",
216
+ "numeric_strategy": numeric_strategy,
217
+ "categorical_strategy": categorical_strategy,
218
+ "values_filled": missing_before - missing_after
219
+ })
220
+
221
+ return self
222
+
223
+ def fix_data_types(
224
+ self,
225
+ type_mapping: Optional[Dict[str, str]] = None,
226
+ infer_types: bool = True
227
+ ) -> 'DataCleaner':
228
+ """
229
+ Fix and optimize data types.
230
+
231
+ Parameters:
232
+ -----------
233
+ type_mapping : dict, optional
234
+ Manual mapping of column names to types
235
+ infer_types : bool
236
+ Whether to automatically infer types
237
+ """
238
+ changes = []
239
+
240
+ # Apply manual type mapping
241
+ if type_mapping:
242
+ for col, dtype in type_mapping.items():
243
+ if col in self.df.columns:
244
+ try:
245
+ self.df[col] = self.df[col].astype(dtype)
246
+ changes.append(f"{col} → {dtype}")
247
+ except (ValueError, TypeError) as e:
248
+ self.cleaning_log.append(f"Could not convert {col} to {dtype}: {e}")
249
+
250
+ if infer_types:
251
+ for col in self.df.columns:
252
+ # Try to convert object columns to numeric
253
+ if self.df[col].dtype == 'object':
254
+ # Try numeric conversion
255
+ try:
256
+ numeric_series = pd.to_numeric(self.df[col], errors='coerce')
257
+ if numeric_series.notna().sum() / len(self.df) > 0.9:
258
+ self.df[col] = numeric_series
259
+ changes.append(f"{col} → numeric")
260
+ continue
261
+ except:
262
+ pass
263
+
264
+ # Try datetime conversion
265
+ try:
266
+ datetime_series = pd.to_datetime(self.df[col], errors='coerce')
267
+ if datetime_series.notna().sum() / len(self.df) > 0.9:
268
+ self.df[col] = datetime_series
269
+ changes.append(f"{col} → datetime")
270
+ continue
271
+ except:
272
+ pass
273
+
274
+ # Convert to category if low cardinality
275
+ if self.df[col].nunique() / len(self.df) < 0.05:
276
+ self.df[col] = self.df[col].astype('category')
277
+ changes.append(f"{col} → category")
278
+
279
+ if changes:
280
+ msg = f"Fixed data types: {len(changes)} columns"
281
+ self.cleaning_log.append(msg)
282
+ self.transformations.append({
283
+ "operation": "fix_data_types",
284
+ "changes": changes
285
+ })
286
+
287
+ return self
288
+
289
+ def handle_outliers(
290
+ self,
291
+ method: str = 'iqr',
292
+ columns: Optional[List[str]] = None,
293
+ threshold: float = 1.5,
294
+ action: str = 'clip'
295
+ ) -> 'DataCleaner':
296
+ """
297
+ Detect and handle outliers in numeric columns.
298
+
299
+ Parameters:
300
+ -----------
301
+ method : str
302
+ 'iqr' or 'zscore'
303
+ columns : list, optional
304
+ Specific columns to check (default: all numeric)
305
+ threshold : float
306
+ IQR multiplier (1.5) or Z-score threshold (3.0)
307
+ action : str
308
+ 'clip', 'remove', or 'nan'
309
+ """
310
+ if columns is None:
311
+ columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
312
+
313
+ outliers_handled = {}
314
+
315
+ for col in columns:
316
+ if col not in self.df.columns:
317
+ continue
318
+
319
+ series = self.df[col].dropna()
320
+
321
+ if method == 'iqr':
322
+ Q1 = series.quantile(0.25)
323
+ Q3 = series.quantile(0.75)
324
+ IQR = Q3 - Q1
325
+ lower_bound = Q1 - threshold * IQR
326
+ upper_bound = Q3 + threshold * IQR
327
+ elif method == 'zscore':
328
+ mean = series.mean()
329
+ std = series.std()
330
+ lower_bound = mean - threshold * std
331
+ upper_bound = mean + threshold * std
332
+ else:
333
+ raise ValueError(f"Unknown method: {method}")
334
+
335
+ # Count outliers
336
+ outlier_mask = (self.df[col] < lower_bound) | (self.df[col] > upper_bound)
337
+ outlier_count = outlier_mask.sum()
338
+
339
+ if outlier_count > 0:
340
+ # Capture examples BEFORE modifying
341
+ outlier_indices = self.df[outlier_mask].index.tolist()
342
+ for idx in outlier_indices[:2]:
343
+ # Get the actual value
344
+ old_val = self.df.loc[idx, col]
345
+ new_val_str = "NaN"
346
+ action_desc = "Removed"
347
+
348
+ if action == 'clip':
349
+ if old_val < lower_bound:
350
+ new_val = lower_bound
351
+ else:
352
+ new_val = upper_bound
353
+ new_val_str = f"{new_val:.4f}"
354
+ action_desc = "Clipped w/ IQR"
355
+ elif action == 'nan':
356
+ new_val_str = "NaN"
357
+ action_desc = "Set to NaN"
358
+
359
+ self.row_changes.append({
360
+ "index": int(idx),
361
+ "column": col,
362
+ "old_value": f"{old_val:.4f}",
363
+ "new_value": new_val_str,
364
+ "operation": action_desc,
365
+ "reason": "Outlier"
366
+ })
367
+
368
+ if action == 'clip':
369
+ self.df[col] = self.df[col].clip(lower_bound, upper_bound)
370
+ elif action == 'remove':
371
+ self.df = self.df[~outlier_mask]
372
+ elif action == 'nan':
373
+ self.df.loc[outlier_mask, col] = np.nan
374
+
375
+ outliers_handled[col] = int(outlier_count)
376
+
377
+ if outliers_handled:
378
+ total = sum(outliers_handled.values())
379
+ msg = f"Handled {total:,} outliers in {len(outliers_handled)} columns using {method}/{action}"
380
+ self.cleaning_log.append(msg)
381
+ self.transformations.append({
382
+ "operation": "handle_outliers",
383
+ "method": method,
384
+ "action": action,
385
+ "outliers_per_column": outliers_handled
386
+ })
387
+
388
+ return self
389
+
390
+ def clean_categorical_values(
391
+ self,
392
+ columns: Optional[List[str]] = None,
393
+ lowercase: bool = True,
394
+ strip_whitespace: bool = True,
395
+ replace_mapping: Optional[Dict[str, Dict[str, str]]] = None
396
+ ) -> 'DataCleaner':
397
+ """
398
+ Clean and standardize categorical values.
399
+
400
+ Parameters:
401
+ -----------
402
+ columns : list, optional
403
+ Specific columns to clean (default: all object/category)
404
+ lowercase : bool
405
+ Convert to lowercase
406
+ strip_whitespace : bool
407
+ Remove leading/trailing whitespace
408
+ replace_mapping : dict, optional
409
+ Column-specific value replacements
410
+ """
411
+ if columns is None:
412
+ columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
413
+
414
+ changes = []
415
+
416
+ for col in columns:
417
+ if col not in self.df.columns:
418
+ continue
419
+
420
+ # Original series for comparison
421
+ original_series = self.df[col].copy()
422
+ original_unique = self.df[col].nunique()
423
+
424
+ if self.df[col].dtype == 'category':
425
+ self.df[col] = self.df[col].astype(str)
426
+
427
+ if strip_whitespace:
428
+ self.df[col] = self.df[col].str.strip()
429
+
430
+ if lowercase:
431
+ self.df[col] = self.df[col].str.lower()
432
+
433
+ # Apply custom replacements
434
+ if replace_mapping and col in replace_mapping:
435
+ self.df[col] = self.df[col].replace(replace_mapping[col])
436
+
437
+ # Detect and log row-level changes
438
+ # We only care about non-null values that actually changed
439
+ changed_mask = (original_series != self.df[col]) & original_series.notna()
440
+ if changed_mask.any():
441
+ changed_indices = self.df[changed_mask].index[:1000]
442
+ for idx in changed_indices:
443
+ self.row_changes.append({
444
+ "index": int(idx),
445
+ "column": col,
446
+ "old_value": str(original_series.loc[idx]),
447
+ "new_value": str(self.df.loc[idx, col]),
448
+ "operation": "Text Normalized",
449
+ "reason": "Categorical Standardized"
450
+ })
451
+
452
+ new_unique = self.df[col].nunique()
453
+ if new_unique < original_unique:
454
+ changes.append(f"{col}: {original_unique} → {new_unique} unique values")
455
+
456
+ if changes:
457
+ msg = f"Cleaned categorical values in {len(columns)} columns"
458
+ self.cleaning_log.append(msg)
459
+ self.transformations.append({
460
+ "operation": "clean_categorical",
461
+ "changes": changes
462
+ })
463
+
464
+ return self
465
+
466
+ def normalize_features(
467
+ self,
468
+ method: str = 'standard',
469
+ columns: Optional[List[str]] = None
470
+ ) -> 'DataCleaner':
471
+ """
472
+ Scale numerical features.
473
+
474
+ Parameters:
475
+ -----------
476
+ method : str
477
+ 'standard', 'minmax', or 'robust'
478
+ columns : list, optional
479
+ Columns to scale (default: all numeric)
480
+ """
481
+ if columns is None:
482
+ columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
483
+
484
+ # Filter columns present in df
485
+ columns = [c for c in columns if c in self.df.columns]
486
+
487
+ if not columns:
488
+ return self
489
+
490
+ scaler = None
491
+ if method == 'standard':
492
+ scaler = StandardScaler()
493
+ elif method == 'minmax':
494
+ scaler = MinMaxScaler()
495
+ elif method == 'robust':
496
+ scaler = RobustScaler()
497
+ else:
498
+ raise ValueError(f"Unknown scaling method: {method}")
499
+
500
+ try:
501
+ self.df[columns] = scaler.fit_transform(self.df[columns])
502
+
503
+ msg = f"Scaled {len(columns)} columns using {method} scaler"
504
+ self.cleaning_log.append(msg)
505
+ self.transformations.append({
506
+ "operation": "normalize_features",
507
+ "method": method,
508
+ "columns": columns
509
+ })
510
+ except Exception as e:
511
+ msg = f"Failed to scale columns: {str(e)}"
512
+ self.cleaning_log.append(msg)
513
+
514
+ return self
515
+
516
+ def encode_categorical(
517
+ self,
518
+ method: str = 'onehot',
519
+ columns: Optional[List[str]] = None,
520
+ max_categories: int = 20
521
+ ) -> 'DataCleaner':
522
+ """
523
+ Encode categorical features.
524
+
525
+ Parameters:
526
+ -----------
527
+ method : str
528
+ 'onehot' or 'label'
529
+ columns : list, optional
530
+ Columns to encode
531
+ max_categories : int
532
+ Max unique values for one-hot encoding
533
+ """
534
+ if columns is None:
535
+ columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
536
+
537
+ columns = [c for c in columns if c in self.df.columns]
538
+
539
+ if not columns:
540
+ return self
541
+
542
+ changes = []
543
+
544
+ if method == 'label':
545
+ le = LabelEncoder()
546
+ for col in columns:
547
+ try:
548
+ # Handle nulls first (fill with 'Unknown')
549
+ if self.df[col].isnull().any():
550
+ self.df[col] = self.df[col].fillna('Unknown')
551
+
552
+ self.df[col] = le.fit_transform(self.df[col].astype(str))
553
+ changes.append(f"{col} (LabelEncoded)")
554
+ except Exception as e:
555
+ self.cleaning_log.append(f"Label encoding failed for {col}: {e}")
556
+
557
+ elif method == 'onehot':
558
+ for col in columns:
559
+ if self.df[col].nunique() > max_categories:
560
+ self.cleaning_log.append(f"Skipping OneHot for {col}: >{max_categories} categories")
561
+ continue
562
+
563
+ try:
564
+ dummies = pd.get_dummies(self.df[col], prefix=col, dummy_na=True)
565
+ self.df = pd.concat([self.df, dummies], axis=1)
566
+ self.df.drop(columns=[col], inplace=True)
567
+ changes.append(f"{col} -> {dummies.shape[1]} columns")
568
+ except Exception as e:
569
+ self.cleaning_log.append(f"OneHot encoding failed for {col}: {e}")
570
+
571
+ if changes:
572
+ msg = f"Encoded {len(changes)} features using {method}"
573
+ self.cleaning_log.append(msg)
574
+ self.transformations.append({
575
+ "operation": "encode_categorical",
576
+ "method": method,
577
+ "columns": columns,
578
+ "changes": changes
579
+ })
580
+
581
+ return self
582
+
583
+ def clean_numeric_text(
584
+ self,
585
+ columns: Optional[List[str]] = None,
586
+ remove_symbols: bool = True,
587
+ handle_shorthand: bool = True
588
+ ) -> 'DataCleaner':
589
+ """
590
+ Clean text columns containing numbers (e.g. '$1,200', '1.5k').
591
+
592
+ Parameters:
593
+ -----------
594
+ columns : list, optional
595
+ Columns to clean
596
+ remove_symbols : bool
597
+ Remove currency symbols and commas
598
+ handle_shorthand : bool
599
+ Convert 'k', 'M', 'B' suffixes (e.g. 1.5k -> 1500)
600
+ """
601
+ if columns is None:
602
+ # Try to guess columns that look like numeric text
603
+ columns = []
604
+ for col in self.df.select_dtypes(include=['object', 'string']).columns:
605
+ # Sample check
606
+ # Keep an object dtype: pandas 3's ``astype(str)`` creates a
607
+ # strict Arrow-backed string array, whose regex engine rejects
608
+ # some valid Python regex escapes used below.
609
+ sample = self.df[col].dropna().astype('object').sample(min(20, len(self.df)), random_state=42)
610
+ if sample.str.contains(r'[\$\€\Ā£\,kKmMbB]').any() and sample.str.contains(r'\d').all():
611
+ columns.append(col)
612
+
613
+ changes = []
614
+
615
+ for col in columns:
616
+ if col not in self.df.columns:
617
+ continue
618
+
619
+ original_nans = self.df[col].isna().sum()
620
+
621
+ # Work on a copy
622
+ # Do not use ``astype(str)`` here. In pandas 3 it creates the
623
+ # strict ``str`` dtype, which can reject later non-string writes
624
+ # and uses Arrow's more limited regex implementation.
625
+ series = self.df[col].astype('object').str.strip()
626
+
627
+ if remove_symbols:
628
+ # Remove typical currency symbols and commas
629
+ series = series.str.replace(r'[\$\€\Ā£\,\s]', '', regex=True)
630
+
631
+ if handle_shorthand:
632
+ def parse_shorthand(val):
633
+ if pd.isna(val) or val == 'nan': return np.nan
634
+ val = val.lower()
635
+ multiplier = 1
636
+ if val.endswith('k'):
637
+ multiplier = 1000
638
+ val = val[:-1]
639
+ elif val.endswith('m'):
640
+ multiplier = 1000000
641
+ val = val[:-1]
642
+ elif val.endswith('b'):
643
+ multiplier = 1000000000
644
+ val = val[:-1]
645
+
646
+ try:
647
+ return float(val) * multiplier
648
+ except:
649
+ return np.nan
650
+
651
+ if handle_shorthand:
652
+ # Use the defined internal function
653
+ self.df[col] = series.apply(parse_shorthand)
654
+ else:
655
+ self.df[col] = pd.to_numeric(series, errors='coerce')
656
+
657
+ new_nans = self.df[col].isna().sum()
658
+ valid_converted = len(self.df) - new_nans
659
+
660
+ # Log examples of successful conversions
661
+ if valid_converted > 0:
662
+ # Find indices where it wasn't null before but is now a valid number
663
+ # AND the string representation looks different (e.g., "$100" vs 100.0)
664
+ # or just log any valid conversion to show off
665
+ valid_mask = self.df[col].notna() & (self.df[col].astype(str) != series)
666
+ if valid_mask.any():
667
+ # Capture all changes (limit to 1000 safety)
668
+ sample_indices = self.df[valid_mask].index[:1000]
669
+ for idx in sample_indices:
670
+ val_old = self.df.loc[idx, col] # This is already the NEW value in self.df
671
+ # We need the OLD value from 'series' variable (which was copies)
672
+ # Wait, 'series' was `self.df[col].astype(str).str.strip()` ...
673
+ # but we also did regex replacement on it if remove_symbols=True
674
+ # So let's grab the raw original from a temp var if we can, or just use the 'series' which is "cleaned string"
675
+
676
+ # Actually, let's just grab the original raw value from a backup if needed,
677
+ # but 'series' is close enough to show "cleaned string" vs "final number".
678
+ # Better yet: usage `self.df.loc[idx, col]` is the NEW value.
679
+ # The OLD value is in `series[idx]` (if index aligns, which it should).
680
+ # BUT series was modified by remove_symbols.
681
+
682
+ # Let's just say:
683
+ old_val_str = series.loc[idx]
684
+ new_val = self.df.loc[idx, col]
685
+
686
+ self.row_changes.append({
687
+ "index": int(idx),
688
+ "column": col,
689
+ "old_value": str(old_val_str),
690
+ "new_value": str(new_val),
691
+ "operation": "Text Cleaned",
692
+ "reason": "Numeric Text"
693
+ })
694
+
695
+ changes.append(f"{col}: Converted to numeric ({valid_converted} valid)")
696
+
697
+ if changes:
698
+ self.cleaning_log.append(f"Cleaned numeric text in {len(changes)} columns")
699
+ self.transformations.append({
700
+ "operation": "clean_numeric_text",
701
+ "columns": columns,
702
+ "changes": changes
703
+ })
704
+
705
+ return self
706
+
707
+ def rename_columns(
708
+ self,
709
+ mapping: Dict[str, str]
710
+ ) -> 'DataCleaner':
711
+ """
712
+ Rename columns.
713
+
714
+ Parameters:
715
+ -----------
716
+ mapping : dict
717
+ Dictionary of {old_name: new_name}
718
+ """
719
+ # Filter mapping to existing columns
720
+ valid_mapping = {k: v for k, v in mapping.items() if k in self.df.columns}
721
+
722
+ if valid_mapping:
723
+ self.df.rename(columns=valid_mapping, inplace=True)
724
+ msg = f"Renamed {len(valid_mapping)} columns: {valid_mapping}"
725
+ self.cleaning_log.append(msg)
726
+ self.transformations.append({
727
+ "operation": "rename_columns",
728
+ "mapping": valid_mapping
729
+ })
730
+
731
+ return self
732
+
733
+ def extract_regex_feature(
734
+ self,
735
+ source_col: str,
736
+ pattern: str,
737
+ new_col_name: str
738
+ ) -> 'DataCleaner':
739
+ r"""
740
+ Extract text using regex capture group.
741
+
742
+ Parameters:
743
+ -----------
744
+ source_col : str
745
+ Source column name
746
+ pattern : str
747
+ Regex pattern with one capture group (e.g. r'ID: (\d+)')
748
+ new_col_name : str
749
+ Name for the new column
750
+ """
751
+ if source_col not in self.df.columns:
752
+ return self
753
+
754
+ try:
755
+ # Ensure pattern is raw string if possible, generally passed as string here
756
+ extracted = self.df[source_col].astype(str).str.extract(pattern, expand=False)
757
+
758
+ self.df[new_col_name] = extracted
759
+
760
+ matched_count = extracted.notna().sum()
761
+ msg = f"Extracted '{new_col_name}' from '{source_col}' ({matched_count} matches)"
762
+ self.cleaning_log.append(msg)
763
+ self.transformations.append({
764
+ "operation": "extract_regex",
765
+ "source": source_col,
766
+ "target": new_col_name,
767
+ "pattern": pattern,
768
+ "matches": int(matched_count)
769
+ })
770
+
771
+ except Exception as e:
772
+ self.cleaning_log.append(f"Regex extraction failed: {e}")
773
+
774
+ return self
775
+
776
+ def drop_columns(
777
+ self,
778
+ columns: Optional[List[str]] = None,
779
+ drop_constant: bool = True,
780
+ drop_id_like: bool = False
781
+ ) -> 'DataCleaner':
782
+ """
783
+ Drop specified or problematic columns.
784
+
785
+ Parameters:
786
+ -----------
787
+ columns : list, optional
788
+ Specific columns to drop
789
+ drop_constant : bool
790
+ Drop columns with only one unique value
791
+ drop_id_like : bool
792
+ Drop columns that appear to be IDs
793
+ """
794
+ cols_to_drop = set(columns or [])
795
+
796
+ if drop_constant:
797
+ for col in self.df.columns:
798
+ if self.df[col].nunique() <= 1:
799
+ cols_to_drop.add(col)
800
+
801
+ if drop_id_like:
802
+ for col in self.df.columns:
803
+ if self.df[col].nunique() == len(self.df):
804
+ cols_to_drop.add(col)
805
+
806
+ cols_to_drop = [c for c in cols_to_drop if c in self.df.columns]
807
+
808
+ if cols_to_drop:
809
+ self.df = self.df.drop(columns=cols_to_drop)
810
+ msg = f"Dropped {len(cols_to_drop)} columns: {cols_to_drop}"
811
+ self.cleaning_log.append(msg)
812
+ self.transformations.append({
813
+ "operation": "drop_columns",
814
+ "columns": cols_to_drop
815
+ })
816
+
817
+ return self
818
+
819
+ def get_cleaned_data(self) -> pd.DataFrame:
820
+ """Return the cleaned DataFrame."""
821
+ return self.df
822
+
823
+ def get_cleaning_summary(self) -> Dict[str, Any]:
824
+ """Return summary of all cleaning operations."""
825
+ return {
826
+ "original_shape": self.original_shape,
827
+ "final_shape": self.df.shape,
828
+ "rows_changed": self.original_shape[0] - self.df.shape[0],
829
+ "columns_changed": self.original_shape[1] - self.df.shape[1],
830
+ "operations": self.cleaning_log,
831
+ "transformations": self.transformations,
832
+ "row_changes": self.row_changes
833
+ }
834
+
835
+ def validate_quality(self) -> Dict[str, Any]:
836
+ """
837
+ Perform data quality checks and calculate a 0-100 Quality Score.
838
+
839
+ Returns:
840
+ --------
841
+ dict
842
+ Report containing quality metrics, score, and warnings.
843
+ """
844
+ n_rows = len(self.df)
845
+ n_cols = len(self.df.columns)
846
+
847
+ if n_rows == 0 or n_cols == 0:
848
+ return {"score": 0, "grade": "F", "warnings": ["Empty dataset"]}
849
+
850
+ # 1. Completeness (0-40 pts)
851
+ missing_total = self.df.isnull().sum().sum()
852
+ missing_pct = missing_total / (n_rows * n_cols)
853
+ completeness_score = max(0, 40 * (1 - missing_pct * 2)) # Penalize missing heavily
854
+
855
+ # 2. Uniqueness (0-30 pts)
856
+ # Check duplicates
857
+ n_dupes = self.df.duplicated().sum()
858
+ dupe_pct = n_dupes / n_rows
859
+ uniqueness_score = max(0, 30 * (1 - dupe_pct * 2))
860
+
861
+ # 3. Consistency/Validity (0-30 pts)
862
+ # Check for constant columns (0 variance)
863
+ n_constant = sum([1 for c in self.df.columns if self.df[c].nunique() <= 1])
864
+ const_pct = n_constant / n_cols
865
+ consistency_score = max(0, 30 * (1 - const_pct * 3))
866
+
867
+ final_score = int(completeness_score + uniqueness_score + consistency_score)
868
+
869
+ grade = 'A' if final_score >= 90 else 'B' if final_score >= 80 else 'C' if final_score >= 60 else 'D' if final_score >= 40 else 'F'
870
+
871
+ report: Dict[str, Any] = {
872
+ "score": final_score,
873
+ "grade": grade,
874
+ "rows": n_rows,
875
+ "columns": n_cols,
876
+ "missing_values": int(missing_total),
877
+ "missing_percentage": round(float(missing_pct * 100), 1),
878
+ "duplicate_rows": int(n_dupes),
879
+ "constant_columns": [],
880
+ "warnings": []
881
+ }
882
+
883
+ # Check for constant columns
884
+ for col in self.df.columns:
885
+ if self.df[col].nunique() <= 1:
886
+ report["constant_columns"].append(col)
887
+ report["warnings"].append(f"Column '{col}' is constant (1 unique value)")
888
+
889
+ # Check for extreme missing values
890
+ high_missing = self.df.columns[self.df.isnull().mean() > 0.5].tolist()
891
+ if high_missing:
892
+ report["warnings"].append(f"{len(high_missing)} columns have >50% missing values")
893
+
894
+ return report
895
+
896
+ def generate_suggestions(self) -> List[Dict[str, str]]:
897
+ """
898
+ Generate AI-like cleaning suggestions based on data issues.
899
+ """
900
+ suggestions = []
901
+
902
+ # Missing Values
903
+ missing = self.df.isnull().sum()
904
+ missing_cols = missing[missing > 0]
905
+
906
+ for col, count in missing_cols.items():
907
+ pct = count / len(self.df)
908
+ if pct > 0.4:
909
+ suggestions.append({
910
+ "column": col,
911
+ "issue": f"{pct:.0%} missing values",
912
+ "action": "Drop Column",
913
+ "reason": "Too much missing data to impute reliably."
914
+ })
915
+ else:
916
+ method = "Median Imputation" if pd.api.types.is_numeric_dtype(self.df[col]) else "Mode Imputation"
917
+ suggestions.append({
918
+ "column": col,
919
+ "issue": f"{pct:.0%} missing values",
920
+ "action": method,
921
+ "reason": "Standard strategy for filling gaps."
922
+ })
923
+
924
+ # Duplicates
925
+ n_dupes = self.df.duplicated().sum()
926
+ if n_dupes > 0:
927
+ suggestions.append({
928
+ "column": "Dataset",
929
+ "issue": f"{n_dupes} duplicate rows",
930
+ "action": "Remove Duplicates",
931
+ "reason": "Duplicate data skews model training."
932
+ })
933
+
934
+ # Outliers (Numeric)
935
+ numeric_cols = self.df.select_dtypes(include=[np.number]).columns
936
+ for col in numeric_cols:
937
+ if self.df[col].nunique() < 10: continue # Skip categorical-like
938
+
939
+ Q1 = self.df[col].quantile(0.25)
940
+ Q3 = self.df[col].quantile(0.75)
941
+ IQR = Q3 - Q1
942
+ lower = Q1 - 1.5 * IQR
943
+ upper = Q3 + 1.5 * IQR
944
+ outliers = ((self.df[col] < lower) | (self.df[col] > upper)).sum()
945
+
946
+ if outliers > 0:
947
+ pct = outliers / len(self.df)
948
+ if pct < 0.05:
949
+ action = "Clip (Winsorize)"
950
+ else:
951
+ action = "Log Transform" if (self.df[col] > 0).all() else "Standard Scaling"
952
+
953
+ suggestions.append({
954
+ "column": col,
955
+ "issue": f"{outliers} outliers detected",
956
+ "action": action,
957
+ "reason": "Outliers can distort linear models."
958
+ })
959
+
960
+ # ID Columns
961
+ for col in self.df.columns:
962
+ if col.lower() in ['id', 'uuid', 'guid', 'index'] or \
963
+ (self.df[col].nunique() == len(self.df) and pd.api.types.is_string_dtype(self.df[col])):
964
+ suggestions.append({
965
+ "column": col,
966
+ "issue": "High cardinality / ID-like",
967
+ "action": "Drop Column",
968
+ "reason": "Identifiers do not predict the target."
969
+ })
970
+
971
+ return suggestions
972
+
973
+ def print_summary(self) -> None:
974
+ """Print cleaning summary."""
975
+ summary = self.get_cleaning_summary()
976
+
977
+ print("=" * 60)
978
+ logger.info("DATA CLEANING SUMMARY")
979
+ print("=" * 60)
980
+ logger.info(f"\nšŸ“Š Shape: {summary['original_shape']} → {summary['final_shape']}")
981
+ logger.info(f" Rows changed: {summary['rows_changed']:+,}")
982
+ logger.info(f" Columns changed: {summary['columns_changed']:+,}")
983
+
984
+ logger.info("\nšŸ”§ Operations performed:")
985
+ for op in summary['operations']:
986
+ logger.info(f" • {op}")
987
+
988
+ print("\n" + "=" * 60)