cleanflow-kit 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.1.0.dist-info/METADATA +119 -0
- cleanflow_kit-1.1.0.dist-info/RECORD +17 -0
- cleanflow_kit-1.1.0.dist-info/WHEEL +5 -0
- cleanflow_kit-1.1.0.dist-info/licenses/LICENSE +21 -0
- cleanflow_kit-1.1.0.dist-info/top_level.txt +1 -0
- dataclean/__init__.py +44 -0
- dataclean/_compat.py +24 -0
- dataclean/data_cleaner.py +988 -0
- dataclean/data_loader.py +194 -0
- dataclean/drift_detector.py +111 -0
- dataclean/eda.py +400 -0
- dataclean/feature_engineer.py +608 -0
- dataclean/model_trainer.py +874 -0
- dataclean/pipeline.py +548 -0
- dataclean/py.typed +0 -0
- dataclean/report_generator.py +365 -0
- dataclean/synthetic_generator.py +97 -0
|
@@ -0,0 +1,988 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Cleaner Module
|
|
3
|
+
===================
|
|
4
|
+
Handles data cleaning operations including missing values,
|
|
5
|
+
duplicates, outliers, and data type corrections.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
import pandas as pd
|
|
11
|
+
import numpy as np
|
|
12
|
+
from typing import Optional, Dict, List, Any, Union
|
|
13
|
+
from sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler, LabelEncoder, OneHotEncoder
|
|
14
|
+
from sklearn.feature_selection import VarianceThreshold
|
|
15
|
+
import re
|
|
16
|
+
from ._compat import normalize_string_columns
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class DataCleaner:
|
|
20
|
+
"""Clean and preprocess datasets."""
|
|
21
|
+
|
|
22
|
+
def __init__(self, df: pd.DataFrame):
|
|
23
|
+
"""
|
|
24
|
+
Initialize cleaner with a DataFrame.
|
|
25
|
+
|
|
26
|
+
Parameters:
|
|
27
|
+
-----------
|
|
28
|
+
df : pd.DataFrame
|
|
29
|
+
Input DataFrame to clean
|
|
30
|
+
"""
|
|
31
|
+
self.df = normalize_string_columns(df.copy())
|
|
32
|
+
# Convert nullable integer/float extension types to standard NumPy
|
|
33
|
+
# dtypes for compatibility with NumPy/scikit-learn.
|
|
34
|
+
for col in self.df.columns:
|
|
35
|
+
if pd.api.types.is_integer_dtype(self.df[col]) and hasattr(self.df[col].dtype, 'numpy_dtype'):
|
|
36
|
+
self.df[col] = self.df[col].astype(self.df[col].dtype.numpy_dtype)
|
|
37
|
+
elif pd.api.types.is_float_dtype(self.df[col]) and hasattr(self.df[col].dtype, 'numpy_dtype'):
|
|
38
|
+
self.df[col] = self.df[col].astype(self.df[col].dtype.numpy_dtype)
|
|
39
|
+
self.original_shape = df.shape
|
|
40
|
+
self.cleaning_log: List[str] = []
|
|
41
|
+
self.transformations: List[Dict[str, Any]] = []
|
|
42
|
+
self.row_changes: List[Dict[str, Any]] = []
|
|
43
|
+
|
|
44
|
+
def remove_duplicates(
|
|
45
|
+
self,
|
|
46
|
+
subset: Optional[List[str]] = None,
|
|
47
|
+
keep: str = 'first'
|
|
48
|
+
) -> 'DataCleaner':
|
|
49
|
+
"""
|
|
50
|
+
Remove duplicate rows.
|
|
51
|
+
|
|
52
|
+
Parameters:
|
|
53
|
+
-----------
|
|
54
|
+
subset : list, optional
|
|
55
|
+
Columns to consider for duplicates
|
|
56
|
+
keep : str
|
|
57
|
+
Which duplicate to keep ('first', 'last', False)
|
|
58
|
+
"""
|
|
59
|
+
before = len(self.df)
|
|
60
|
+
|
|
61
|
+
# Identify duplicates for logging before removing
|
|
62
|
+
dupes = self.df[self.df.duplicated(subset=subset, keep=keep)]
|
|
63
|
+
if not dupes.empty:
|
|
64
|
+
for idx in dupes.index[:2]:
|
|
65
|
+
# Create a string representation of the row (first 3 cols)
|
|
66
|
+
row_str = ", ".join([str(x) for x in self.df.loc[idx].values[:3]]) + "..."
|
|
67
|
+
self.row_changes.append({
|
|
68
|
+
"index": int(idx),
|
|
69
|
+
"column": "Row",
|
|
70
|
+
"old_value": row_str,
|
|
71
|
+
"new_value": "Deleted",
|
|
72
|
+
"operation": "Row Removed",
|
|
73
|
+
"reason": "Duplicate"
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
self.df = self.df.drop_duplicates(subset=subset, keep=keep)
|
|
77
|
+
removed = before - len(self.df)
|
|
78
|
+
|
|
79
|
+
if removed > 0:
|
|
80
|
+
msg = f"Removed {removed:,} duplicate rows"
|
|
81
|
+
self.cleaning_log.append(msg)
|
|
82
|
+
self.transformations.append({
|
|
83
|
+
"operation": "remove_duplicates",
|
|
84
|
+
"rows_removed": removed
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
return self
|
|
88
|
+
|
|
89
|
+
def handle_missing_values(
|
|
90
|
+
self,
|
|
91
|
+
strategy: str = 'auto',
|
|
92
|
+
numeric_strategy: str = 'median',
|
|
93
|
+
categorical_strategy: str = 'mode',
|
|
94
|
+
drop_threshold: float = 0.4,
|
|
95
|
+
fill_value: Optional[Any] = None
|
|
96
|
+
) -> 'DataCleaner':
|
|
97
|
+
"""
|
|
98
|
+
Handle missing values in the dataset.
|
|
99
|
+
|
|
100
|
+
Parameters:
|
|
101
|
+
-----------
|
|
102
|
+
strategy : str
|
|
103
|
+
'auto', 'drop_rows', 'drop_cols', 'fill'
|
|
104
|
+
numeric_strategy : str
|
|
105
|
+
Strategy for numeric columns: 'mean', 'median', 'zero'
|
|
106
|
+
categorical_strategy : str
|
|
107
|
+
Strategy for categorical: 'mode', 'unknown'
|
|
108
|
+
drop_threshold : float
|
|
109
|
+
Drop columns with missing > threshold (0-1)
|
|
110
|
+
fill_value : any, optional
|
|
111
|
+
Custom fill value when strategy='fill'
|
|
112
|
+
"""
|
|
113
|
+
missing_before = self.df.isnull().sum().sum()
|
|
114
|
+
|
|
115
|
+
if missing_before == 0:
|
|
116
|
+
self.cleaning_log.append("No missing values to handle")
|
|
117
|
+
return self
|
|
118
|
+
|
|
119
|
+
# Drop columns with too many missing values
|
|
120
|
+
missing_pct = self.df.isnull().sum() / len(self.df)
|
|
121
|
+
cols_to_drop = missing_pct[missing_pct > drop_threshold].index.tolist()
|
|
122
|
+
|
|
123
|
+
if cols_to_drop:
|
|
124
|
+
self.df = self.df.drop(columns=cols_to_drop)
|
|
125
|
+
msg = f"Dropped {len(cols_to_drop)} columns with >{drop_threshold*100}% missing: {cols_to_drop}"
|
|
126
|
+
self.cleaning_log.append(msg)
|
|
127
|
+
self.transformations.append({
|
|
128
|
+
"operation": "drop_high_missing_columns",
|
|
129
|
+
"columns": cols_to_drop,
|
|
130
|
+
"threshold": drop_threshold
|
|
131
|
+
})
|
|
132
|
+
|
|
133
|
+
if strategy == 'drop_rows':
|
|
134
|
+
before = len(self.df)
|
|
135
|
+
self.df = self.df.dropna()
|
|
136
|
+
msg = f"Dropped {before - len(self.df):,} rows with missing values"
|
|
137
|
+
self.cleaning_log.append(msg)
|
|
138
|
+
return self
|
|
139
|
+
|
|
140
|
+
# Fill numeric columns
|
|
141
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns
|
|
142
|
+
for col in numeric_cols:
|
|
143
|
+
if self.df[col].isnull().any():
|
|
144
|
+
# Log usage
|
|
145
|
+
missing_indices = self.df[self.df[col].isnull()].index.tolist()
|
|
146
|
+
|
|
147
|
+
# Determine strategy name and value
|
|
148
|
+
strat_name = "Imputed (Custom)"
|
|
149
|
+
val_used = fill_value
|
|
150
|
+
|
|
151
|
+
if fill_value is not None:
|
|
152
|
+
self.df[col] = self.df[col].fillna(fill_value)
|
|
153
|
+
strat_name = f"Filled {fill_value}"
|
|
154
|
+
elif numeric_strategy == 'mean':
|
|
155
|
+
val_used = self.df[col].mean()
|
|
156
|
+
self.df[col] = self.df[col].fillna(val_used)
|
|
157
|
+
strat_name = "Imputed Mean"
|
|
158
|
+
elif numeric_strategy == 'median':
|
|
159
|
+
val_used = self.df[col].median()
|
|
160
|
+
self.df[col] = self.df[col].fillna(val_used)
|
|
161
|
+
strat_name = "Imputed Median"
|
|
162
|
+
elif numeric_strategy == 'zero':
|
|
163
|
+
val_used = 0
|
|
164
|
+
self.df[col] = self.df[col].fillna(0)
|
|
165
|
+
strat_name = "Filled Zero"
|
|
166
|
+
|
|
167
|
+
# Capture 2 examples
|
|
168
|
+
for idx in missing_indices[:2]:
|
|
169
|
+
self.row_changes.append({
|
|
170
|
+
"index": int(idx),
|
|
171
|
+
"column": col,
|
|
172
|
+
"old_value": "NaN",
|
|
173
|
+
"new_value": round(float(val_used), 4) if isinstance(val_used, (float, int)) else str(val_used),
|
|
174
|
+
"operation": strat_name,
|
|
175
|
+
"reason": "Missing Value"
|
|
176
|
+
})
|
|
177
|
+
|
|
178
|
+
# Fill categorical columns
|
|
179
|
+
categorical_cols = self.df.select_dtypes(include=['object', 'category']).columns
|
|
180
|
+
for col in categorical_cols:
|
|
181
|
+
if self.df[col].isnull().any():
|
|
182
|
+
missing_indices = self.df[self.df[col].isnull()].index.tolist()
|
|
183
|
+
strat_name = "Imputed (Custom)"
|
|
184
|
+
val_used = fill_value
|
|
185
|
+
|
|
186
|
+
if fill_value is not None:
|
|
187
|
+
self.df[col] = self.df[col].fillna(fill_value)
|
|
188
|
+
strat_name = f"Filled {fill_value}"
|
|
189
|
+
elif categorical_strategy == 'mode':
|
|
190
|
+
mode_val = self.df[col].mode()
|
|
191
|
+
if len(mode_val) > 0:
|
|
192
|
+
val_used = mode_val[0]
|
|
193
|
+
self.df[col] = self.df[col].fillna(val_used)
|
|
194
|
+
strat_name = "Imputed Mode"
|
|
195
|
+
elif categorical_strategy == 'unknown':
|
|
196
|
+
val_used = "Unknown"
|
|
197
|
+
self.df[col] = self.df[col].fillna('Unknown')
|
|
198
|
+
strat_name = "Filled 'Unknown'"
|
|
199
|
+
|
|
200
|
+
# Capture 2 examples
|
|
201
|
+
for idx in missing_indices[:2]:
|
|
202
|
+
self.row_changes.append({
|
|
203
|
+
"index": int(idx),
|
|
204
|
+
"column": col,
|
|
205
|
+
"old_value": "NaN",
|
|
206
|
+
"new_value": str(val_used),
|
|
207
|
+
"operation": strat_name,
|
|
208
|
+
"reason": "Missing Value"
|
|
209
|
+
})
|
|
210
|
+
|
|
211
|
+
missing_after = self.df.isnull().sum().sum()
|
|
212
|
+
msg = f"Handled missing values: {missing_before:,} ā {missing_after:,}"
|
|
213
|
+
self.cleaning_log.append(msg)
|
|
214
|
+
self.transformations.append({
|
|
215
|
+
"operation": "handle_missing",
|
|
216
|
+
"numeric_strategy": numeric_strategy,
|
|
217
|
+
"categorical_strategy": categorical_strategy,
|
|
218
|
+
"values_filled": missing_before - missing_after
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
return self
|
|
222
|
+
|
|
223
|
+
def fix_data_types(
|
|
224
|
+
self,
|
|
225
|
+
type_mapping: Optional[Dict[str, str]] = None,
|
|
226
|
+
infer_types: bool = True
|
|
227
|
+
) -> 'DataCleaner':
|
|
228
|
+
"""
|
|
229
|
+
Fix and optimize data types.
|
|
230
|
+
|
|
231
|
+
Parameters:
|
|
232
|
+
-----------
|
|
233
|
+
type_mapping : dict, optional
|
|
234
|
+
Manual mapping of column names to types
|
|
235
|
+
infer_types : bool
|
|
236
|
+
Whether to automatically infer types
|
|
237
|
+
"""
|
|
238
|
+
changes = []
|
|
239
|
+
|
|
240
|
+
# Apply manual type mapping
|
|
241
|
+
if type_mapping:
|
|
242
|
+
for col, dtype in type_mapping.items():
|
|
243
|
+
if col in self.df.columns:
|
|
244
|
+
try:
|
|
245
|
+
self.df[col] = self.df[col].astype(dtype)
|
|
246
|
+
changes.append(f"{col} ā {dtype}")
|
|
247
|
+
except (ValueError, TypeError) as e:
|
|
248
|
+
self.cleaning_log.append(f"Could not convert {col} to {dtype}: {e}")
|
|
249
|
+
|
|
250
|
+
if infer_types:
|
|
251
|
+
for col in self.df.columns:
|
|
252
|
+
# Try to convert object columns to numeric
|
|
253
|
+
if self.df[col].dtype == 'object':
|
|
254
|
+
# Try numeric conversion
|
|
255
|
+
try:
|
|
256
|
+
numeric_series = pd.to_numeric(self.df[col], errors='coerce')
|
|
257
|
+
if numeric_series.notna().sum() / len(self.df) > 0.9:
|
|
258
|
+
self.df[col] = numeric_series
|
|
259
|
+
changes.append(f"{col} ā numeric")
|
|
260
|
+
continue
|
|
261
|
+
except:
|
|
262
|
+
pass
|
|
263
|
+
|
|
264
|
+
# Try datetime conversion
|
|
265
|
+
try:
|
|
266
|
+
datetime_series = pd.to_datetime(self.df[col], errors='coerce')
|
|
267
|
+
if datetime_series.notna().sum() / len(self.df) > 0.9:
|
|
268
|
+
self.df[col] = datetime_series
|
|
269
|
+
changes.append(f"{col} ā datetime")
|
|
270
|
+
continue
|
|
271
|
+
except:
|
|
272
|
+
pass
|
|
273
|
+
|
|
274
|
+
# Convert to category if low cardinality
|
|
275
|
+
if self.df[col].nunique() / len(self.df) < 0.05:
|
|
276
|
+
self.df[col] = self.df[col].astype('category')
|
|
277
|
+
changes.append(f"{col} ā category")
|
|
278
|
+
|
|
279
|
+
if changes:
|
|
280
|
+
msg = f"Fixed data types: {len(changes)} columns"
|
|
281
|
+
self.cleaning_log.append(msg)
|
|
282
|
+
self.transformations.append({
|
|
283
|
+
"operation": "fix_data_types",
|
|
284
|
+
"changes": changes
|
|
285
|
+
})
|
|
286
|
+
|
|
287
|
+
return self
|
|
288
|
+
|
|
289
|
+
def handle_outliers(
|
|
290
|
+
self,
|
|
291
|
+
method: str = 'iqr',
|
|
292
|
+
columns: Optional[List[str]] = None,
|
|
293
|
+
threshold: float = 1.5,
|
|
294
|
+
action: str = 'clip'
|
|
295
|
+
) -> 'DataCleaner':
|
|
296
|
+
"""
|
|
297
|
+
Detect and handle outliers in numeric columns.
|
|
298
|
+
|
|
299
|
+
Parameters:
|
|
300
|
+
-----------
|
|
301
|
+
method : str
|
|
302
|
+
'iqr' or 'zscore'
|
|
303
|
+
columns : list, optional
|
|
304
|
+
Specific columns to check (default: all numeric)
|
|
305
|
+
threshold : float
|
|
306
|
+
IQR multiplier (1.5) or Z-score threshold (3.0)
|
|
307
|
+
action : str
|
|
308
|
+
'clip', 'remove', or 'nan'
|
|
309
|
+
"""
|
|
310
|
+
if columns is None:
|
|
311
|
+
columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
312
|
+
|
|
313
|
+
outliers_handled = {}
|
|
314
|
+
|
|
315
|
+
for col in columns:
|
|
316
|
+
if col not in self.df.columns:
|
|
317
|
+
continue
|
|
318
|
+
|
|
319
|
+
series = self.df[col].dropna()
|
|
320
|
+
|
|
321
|
+
if method == 'iqr':
|
|
322
|
+
Q1 = series.quantile(0.25)
|
|
323
|
+
Q3 = series.quantile(0.75)
|
|
324
|
+
IQR = Q3 - Q1
|
|
325
|
+
lower_bound = Q1 - threshold * IQR
|
|
326
|
+
upper_bound = Q3 + threshold * IQR
|
|
327
|
+
elif method == 'zscore':
|
|
328
|
+
mean = series.mean()
|
|
329
|
+
std = series.std()
|
|
330
|
+
lower_bound = mean - threshold * std
|
|
331
|
+
upper_bound = mean + threshold * std
|
|
332
|
+
else:
|
|
333
|
+
raise ValueError(f"Unknown method: {method}")
|
|
334
|
+
|
|
335
|
+
# Count outliers
|
|
336
|
+
outlier_mask = (self.df[col] < lower_bound) | (self.df[col] > upper_bound)
|
|
337
|
+
outlier_count = outlier_mask.sum()
|
|
338
|
+
|
|
339
|
+
if outlier_count > 0:
|
|
340
|
+
# Capture examples BEFORE modifying
|
|
341
|
+
outlier_indices = self.df[outlier_mask].index.tolist()
|
|
342
|
+
for idx in outlier_indices[:2]:
|
|
343
|
+
# Get the actual value
|
|
344
|
+
old_val = self.df.loc[idx, col]
|
|
345
|
+
new_val_str = "NaN"
|
|
346
|
+
action_desc = "Removed"
|
|
347
|
+
|
|
348
|
+
if action == 'clip':
|
|
349
|
+
if old_val < lower_bound:
|
|
350
|
+
new_val = lower_bound
|
|
351
|
+
else:
|
|
352
|
+
new_val = upper_bound
|
|
353
|
+
new_val_str = f"{new_val:.4f}"
|
|
354
|
+
action_desc = "Clipped w/ IQR"
|
|
355
|
+
elif action == 'nan':
|
|
356
|
+
new_val_str = "NaN"
|
|
357
|
+
action_desc = "Set to NaN"
|
|
358
|
+
|
|
359
|
+
self.row_changes.append({
|
|
360
|
+
"index": int(idx),
|
|
361
|
+
"column": col,
|
|
362
|
+
"old_value": f"{old_val:.4f}",
|
|
363
|
+
"new_value": new_val_str,
|
|
364
|
+
"operation": action_desc,
|
|
365
|
+
"reason": "Outlier"
|
|
366
|
+
})
|
|
367
|
+
|
|
368
|
+
if action == 'clip':
|
|
369
|
+
self.df[col] = self.df[col].clip(lower_bound, upper_bound)
|
|
370
|
+
elif action == 'remove':
|
|
371
|
+
self.df = self.df[~outlier_mask]
|
|
372
|
+
elif action == 'nan':
|
|
373
|
+
self.df.loc[outlier_mask, col] = np.nan
|
|
374
|
+
|
|
375
|
+
outliers_handled[col] = int(outlier_count)
|
|
376
|
+
|
|
377
|
+
if outliers_handled:
|
|
378
|
+
total = sum(outliers_handled.values())
|
|
379
|
+
msg = f"Handled {total:,} outliers in {len(outliers_handled)} columns using {method}/{action}"
|
|
380
|
+
self.cleaning_log.append(msg)
|
|
381
|
+
self.transformations.append({
|
|
382
|
+
"operation": "handle_outliers",
|
|
383
|
+
"method": method,
|
|
384
|
+
"action": action,
|
|
385
|
+
"outliers_per_column": outliers_handled
|
|
386
|
+
})
|
|
387
|
+
|
|
388
|
+
return self
|
|
389
|
+
|
|
390
|
+
def clean_categorical_values(
|
|
391
|
+
self,
|
|
392
|
+
columns: Optional[List[str]] = None,
|
|
393
|
+
lowercase: bool = True,
|
|
394
|
+
strip_whitespace: bool = True,
|
|
395
|
+
replace_mapping: Optional[Dict[str, Dict[str, str]]] = None
|
|
396
|
+
) -> 'DataCleaner':
|
|
397
|
+
"""
|
|
398
|
+
Clean and standardize categorical values.
|
|
399
|
+
|
|
400
|
+
Parameters:
|
|
401
|
+
-----------
|
|
402
|
+
columns : list, optional
|
|
403
|
+
Specific columns to clean (default: all object/category)
|
|
404
|
+
lowercase : bool
|
|
405
|
+
Convert to lowercase
|
|
406
|
+
strip_whitespace : bool
|
|
407
|
+
Remove leading/trailing whitespace
|
|
408
|
+
replace_mapping : dict, optional
|
|
409
|
+
Column-specific value replacements
|
|
410
|
+
"""
|
|
411
|
+
if columns is None:
|
|
412
|
+
columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
413
|
+
|
|
414
|
+
changes = []
|
|
415
|
+
|
|
416
|
+
for col in columns:
|
|
417
|
+
if col not in self.df.columns:
|
|
418
|
+
continue
|
|
419
|
+
|
|
420
|
+
# Original series for comparison
|
|
421
|
+
original_series = self.df[col].copy()
|
|
422
|
+
original_unique = self.df[col].nunique()
|
|
423
|
+
|
|
424
|
+
if self.df[col].dtype == 'category':
|
|
425
|
+
self.df[col] = self.df[col].astype(str)
|
|
426
|
+
|
|
427
|
+
if strip_whitespace:
|
|
428
|
+
self.df[col] = self.df[col].str.strip()
|
|
429
|
+
|
|
430
|
+
if lowercase:
|
|
431
|
+
self.df[col] = self.df[col].str.lower()
|
|
432
|
+
|
|
433
|
+
# Apply custom replacements
|
|
434
|
+
if replace_mapping and col in replace_mapping:
|
|
435
|
+
self.df[col] = self.df[col].replace(replace_mapping[col])
|
|
436
|
+
|
|
437
|
+
# Detect and log row-level changes
|
|
438
|
+
# We only care about non-null values that actually changed
|
|
439
|
+
changed_mask = (original_series != self.df[col]) & original_series.notna()
|
|
440
|
+
if changed_mask.any():
|
|
441
|
+
changed_indices = self.df[changed_mask].index[:1000]
|
|
442
|
+
for idx in changed_indices:
|
|
443
|
+
self.row_changes.append({
|
|
444
|
+
"index": int(idx),
|
|
445
|
+
"column": col,
|
|
446
|
+
"old_value": str(original_series.loc[idx]),
|
|
447
|
+
"new_value": str(self.df.loc[idx, col]),
|
|
448
|
+
"operation": "Text Normalized",
|
|
449
|
+
"reason": "Categorical Standardized"
|
|
450
|
+
})
|
|
451
|
+
|
|
452
|
+
new_unique = self.df[col].nunique()
|
|
453
|
+
if new_unique < original_unique:
|
|
454
|
+
changes.append(f"{col}: {original_unique} ā {new_unique} unique values")
|
|
455
|
+
|
|
456
|
+
if changes:
|
|
457
|
+
msg = f"Cleaned categorical values in {len(columns)} columns"
|
|
458
|
+
self.cleaning_log.append(msg)
|
|
459
|
+
self.transformations.append({
|
|
460
|
+
"operation": "clean_categorical",
|
|
461
|
+
"changes": changes
|
|
462
|
+
})
|
|
463
|
+
|
|
464
|
+
return self
|
|
465
|
+
|
|
466
|
+
def normalize_features(
|
|
467
|
+
self,
|
|
468
|
+
method: str = 'standard',
|
|
469
|
+
columns: Optional[List[str]] = None
|
|
470
|
+
) -> 'DataCleaner':
|
|
471
|
+
"""
|
|
472
|
+
Scale numerical features.
|
|
473
|
+
|
|
474
|
+
Parameters:
|
|
475
|
+
-----------
|
|
476
|
+
method : str
|
|
477
|
+
'standard', 'minmax', or 'robust'
|
|
478
|
+
columns : list, optional
|
|
479
|
+
Columns to scale (default: all numeric)
|
|
480
|
+
"""
|
|
481
|
+
if columns is None:
|
|
482
|
+
columns = self.df.select_dtypes(include=[np.number]).columns.tolist()
|
|
483
|
+
|
|
484
|
+
# Filter columns present in df
|
|
485
|
+
columns = [c for c in columns if c in self.df.columns]
|
|
486
|
+
|
|
487
|
+
if not columns:
|
|
488
|
+
return self
|
|
489
|
+
|
|
490
|
+
scaler = None
|
|
491
|
+
if method == 'standard':
|
|
492
|
+
scaler = StandardScaler()
|
|
493
|
+
elif method == 'minmax':
|
|
494
|
+
scaler = MinMaxScaler()
|
|
495
|
+
elif method == 'robust':
|
|
496
|
+
scaler = RobustScaler()
|
|
497
|
+
else:
|
|
498
|
+
raise ValueError(f"Unknown scaling method: {method}")
|
|
499
|
+
|
|
500
|
+
try:
|
|
501
|
+
self.df[columns] = scaler.fit_transform(self.df[columns])
|
|
502
|
+
|
|
503
|
+
msg = f"Scaled {len(columns)} columns using {method} scaler"
|
|
504
|
+
self.cleaning_log.append(msg)
|
|
505
|
+
self.transformations.append({
|
|
506
|
+
"operation": "normalize_features",
|
|
507
|
+
"method": method,
|
|
508
|
+
"columns": columns
|
|
509
|
+
})
|
|
510
|
+
except Exception as e:
|
|
511
|
+
msg = f"Failed to scale columns: {str(e)}"
|
|
512
|
+
self.cleaning_log.append(msg)
|
|
513
|
+
|
|
514
|
+
return self
|
|
515
|
+
|
|
516
|
+
def encode_categorical(
|
|
517
|
+
self,
|
|
518
|
+
method: str = 'onehot',
|
|
519
|
+
columns: Optional[List[str]] = None,
|
|
520
|
+
max_categories: int = 20
|
|
521
|
+
) -> 'DataCleaner':
|
|
522
|
+
"""
|
|
523
|
+
Encode categorical features.
|
|
524
|
+
|
|
525
|
+
Parameters:
|
|
526
|
+
-----------
|
|
527
|
+
method : str
|
|
528
|
+
'onehot' or 'label'
|
|
529
|
+
columns : list, optional
|
|
530
|
+
Columns to encode
|
|
531
|
+
max_categories : int
|
|
532
|
+
Max unique values for one-hot encoding
|
|
533
|
+
"""
|
|
534
|
+
if columns is None:
|
|
535
|
+
columns = self.df.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
536
|
+
|
|
537
|
+
columns = [c for c in columns if c in self.df.columns]
|
|
538
|
+
|
|
539
|
+
if not columns:
|
|
540
|
+
return self
|
|
541
|
+
|
|
542
|
+
changes = []
|
|
543
|
+
|
|
544
|
+
if method == 'label':
|
|
545
|
+
le = LabelEncoder()
|
|
546
|
+
for col in columns:
|
|
547
|
+
try:
|
|
548
|
+
# Handle nulls first (fill with 'Unknown')
|
|
549
|
+
if self.df[col].isnull().any():
|
|
550
|
+
self.df[col] = self.df[col].fillna('Unknown')
|
|
551
|
+
|
|
552
|
+
self.df[col] = le.fit_transform(self.df[col].astype(str))
|
|
553
|
+
changes.append(f"{col} (LabelEncoded)")
|
|
554
|
+
except Exception as e:
|
|
555
|
+
self.cleaning_log.append(f"Label encoding failed for {col}: {e}")
|
|
556
|
+
|
|
557
|
+
elif method == 'onehot':
|
|
558
|
+
for col in columns:
|
|
559
|
+
if self.df[col].nunique() > max_categories:
|
|
560
|
+
self.cleaning_log.append(f"Skipping OneHot for {col}: >{max_categories} categories")
|
|
561
|
+
continue
|
|
562
|
+
|
|
563
|
+
try:
|
|
564
|
+
dummies = pd.get_dummies(self.df[col], prefix=col, dummy_na=True)
|
|
565
|
+
self.df = pd.concat([self.df, dummies], axis=1)
|
|
566
|
+
self.df.drop(columns=[col], inplace=True)
|
|
567
|
+
changes.append(f"{col} -> {dummies.shape[1]} columns")
|
|
568
|
+
except Exception as e:
|
|
569
|
+
self.cleaning_log.append(f"OneHot encoding failed for {col}: {e}")
|
|
570
|
+
|
|
571
|
+
if changes:
|
|
572
|
+
msg = f"Encoded {len(changes)} features using {method}"
|
|
573
|
+
self.cleaning_log.append(msg)
|
|
574
|
+
self.transformations.append({
|
|
575
|
+
"operation": "encode_categorical",
|
|
576
|
+
"method": method,
|
|
577
|
+
"columns": columns,
|
|
578
|
+
"changes": changes
|
|
579
|
+
})
|
|
580
|
+
|
|
581
|
+
return self
|
|
582
|
+
|
|
583
|
+
def clean_numeric_text(
|
|
584
|
+
self,
|
|
585
|
+
columns: Optional[List[str]] = None,
|
|
586
|
+
remove_symbols: bool = True,
|
|
587
|
+
handle_shorthand: bool = True
|
|
588
|
+
) -> 'DataCleaner':
|
|
589
|
+
"""
|
|
590
|
+
Clean text columns containing numbers (e.g. '$1,200', '1.5k').
|
|
591
|
+
|
|
592
|
+
Parameters:
|
|
593
|
+
-----------
|
|
594
|
+
columns : list, optional
|
|
595
|
+
Columns to clean
|
|
596
|
+
remove_symbols : bool
|
|
597
|
+
Remove currency symbols and commas
|
|
598
|
+
handle_shorthand : bool
|
|
599
|
+
Convert 'k', 'M', 'B' suffixes (e.g. 1.5k -> 1500)
|
|
600
|
+
"""
|
|
601
|
+
if columns is None:
|
|
602
|
+
# Try to guess columns that look like numeric text
|
|
603
|
+
columns = []
|
|
604
|
+
for col in self.df.select_dtypes(include=['object', 'string']).columns:
|
|
605
|
+
# Sample check
|
|
606
|
+
# Keep an object dtype: pandas 3's ``astype(str)`` creates a
|
|
607
|
+
# strict Arrow-backed string array, whose regex engine rejects
|
|
608
|
+
# some valid Python regex escapes used below.
|
|
609
|
+
sample = self.df[col].dropna().astype('object').sample(min(20, len(self.df)), random_state=42)
|
|
610
|
+
if sample.str.contains(r'[\$\ā¬\Ā£\,kKmMbB]').any() and sample.str.contains(r'\d').all():
|
|
611
|
+
columns.append(col)
|
|
612
|
+
|
|
613
|
+
changes = []
|
|
614
|
+
|
|
615
|
+
for col in columns:
|
|
616
|
+
if col not in self.df.columns:
|
|
617
|
+
continue
|
|
618
|
+
|
|
619
|
+
original_nans = self.df[col].isna().sum()
|
|
620
|
+
|
|
621
|
+
# Work on a copy
|
|
622
|
+
# Do not use ``astype(str)`` here. In pandas 3 it creates the
|
|
623
|
+
# strict ``str`` dtype, which can reject later non-string writes
|
|
624
|
+
# and uses Arrow's more limited regex implementation.
|
|
625
|
+
series = self.df[col].astype('object').str.strip()
|
|
626
|
+
|
|
627
|
+
if remove_symbols:
|
|
628
|
+
# Remove typical currency symbols and commas
|
|
629
|
+
series = series.str.replace(r'[\$\ā¬\Ā£\,\s]', '', regex=True)
|
|
630
|
+
|
|
631
|
+
if handle_shorthand:
|
|
632
|
+
def parse_shorthand(val):
|
|
633
|
+
if pd.isna(val) or val == 'nan': return np.nan
|
|
634
|
+
val = val.lower()
|
|
635
|
+
multiplier = 1
|
|
636
|
+
if val.endswith('k'):
|
|
637
|
+
multiplier = 1000
|
|
638
|
+
val = val[:-1]
|
|
639
|
+
elif val.endswith('m'):
|
|
640
|
+
multiplier = 1000000
|
|
641
|
+
val = val[:-1]
|
|
642
|
+
elif val.endswith('b'):
|
|
643
|
+
multiplier = 1000000000
|
|
644
|
+
val = val[:-1]
|
|
645
|
+
|
|
646
|
+
try:
|
|
647
|
+
return float(val) * multiplier
|
|
648
|
+
except:
|
|
649
|
+
return np.nan
|
|
650
|
+
|
|
651
|
+
if handle_shorthand:
|
|
652
|
+
# Use the defined internal function
|
|
653
|
+
self.df[col] = series.apply(parse_shorthand)
|
|
654
|
+
else:
|
|
655
|
+
self.df[col] = pd.to_numeric(series, errors='coerce')
|
|
656
|
+
|
|
657
|
+
new_nans = self.df[col].isna().sum()
|
|
658
|
+
valid_converted = len(self.df) - new_nans
|
|
659
|
+
|
|
660
|
+
# Log examples of successful conversions
|
|
661
|
+
if valid_converted > 0:
|
|
662
|
+
# Find indices where it wasn't null before but is now a valid number
|
|
663
|
+
# AND the string representation looks different (e.g., "$100" vs 100.0)
|
|
664
|
+
# or just log any valid conversion to show off
|
|
665
|
+
valid_mask = self.df[col].notna() & (self.df[col].astype(str) != series)
|
|
666
|
+
if valid_mask.any():
|
|
667
|
+
# Capture all changes (limit to 1000 safety)
|
|
668
|
+
sample_indices = self.df[valid_mask].index[:1000]
|
|
669
|
+
for idx in sample_indices:
|
|
670
|
+
val_old = self.df.loc[idx, col] # This is already the NEW value in self.df
|
|
671
|
+
# We need the OLD value from 'series' variable (which was copies)
|
|
672
|
+
# Wait, 'series' was `self.df[col].astype(str).str.strip()` ...
|
|
673
|
+
# but we also did regex replacement on it if remove_symbols=True
|
|
674
|
+
# So let's grab the raw original from a temp var if we can, or just use the 'series' which is "cleaned string"
|
|
675
|
+
|
|
676
|
+
# Actually, let's just grab the original raw value from a backup if needed,
|
|
677
|
+
# but 'series' is close enough to show "cleaned string" vs "final number".
|
|
678
|
+
# Better yet: usage `self.df.loc[idx, col]` is the NEW value.
|
|
679
|
+
# The OLD value is in `series[idx]` (if index aligns, which it should).
|
|
680
|
+
# BUT series was modified by remove_symbols.
|
|
681
|
+
|
|
682
|
+
# Let's just say:
|
|
683
|
+
old_val_str = series.loc[idx]
|
|
684
|
+
new_val = self.df.loc[idx, col]
|
|
685
|
+
|
|
686
|
+
self.row_changes.append({
|
|
687
|
+
"index": int(idx),
|
|
688
|
+
"column": col,
|
|
689
|
+
"old_value": str(old_val_str),
|
|
690
|
+
"new_value": str(new_val),
|
|
691
|
+
"operation": "Text Cleaned",
|
|
692
|
+
"reason": "Numeric Text"
|
|
693
|
+
})
|
|
694
|
+
|
|
695
|
+
changes.append(f"{col}: Converted to numeric ({valid_converted} valid)")
|
|
696
|
+
|
|
697
|
+
if changes:
|
|
698
|
+
self.cleaning_log.append(f"Cleaned numeric text in {len(changes)} columns")
|
|
699
|
+
self.transformations.append({
|
|
700
|
+
"operation": "clean_numeric_text",
|
|
701
|
+
"columns": columns,
|
|
702
|
+
"changes": changes
|
|
703
|
+
})
|
|
704
|
+
|
|
705
|
+
return self
|
|
706
|
+
|
|
707
|
+
def rename_columns(
|
|
708
|
+
self,
|
|
709
|
+
mapping: Dict[str, str]
|
|
710
|
+
) -> 'DataCleaner':
|
|
711
|
+
"""
|
|
712
|
+
Rename columns.
|
|
713
|
+
|
|
714
|
+
Parameters:
|
|
715
|
+
-----------
|
|
716
|
+
mapping : dict
|
|
717
|
+
Dictionary of {old_name: new_name}
|
|
718
|
+
"""
|
|
719
|
+
# Filter mapping to existing columns
|
|
720
|
+
valid_mapping = {k: v for k, v in mapping.items() if k in self.df.columns}
|
|
721
|
+
|
|
722
|
+
if valid_mapping:
|
|
723
|
+
self.df.rename(columns=valid_mapping, inplace=True)
|
|
724
|
+
msg = f"Renamed {len(valid_mapping)} columns: {valid_mapping}"
|
|
725
|
+
self.cleaning_log.append(msg)
|
|
726
|
+
self.transformations.append({
|
|
727
|
+
"operation": "rename_columns",
|
|
728
|
+
"mapping": valid_mapping
|
|
729
|
+
})
|
|
730
|
+
|
|
731
|
+
return self
|
|
732
|
+
|
|
733
|
+
def extract_regex_feature(
|
|
734
|
+
self,
|
|
735
|
+
source_col: str,
|
|
736
|
+
pattern: str,
|
|
737
|
+
new_col_name: str
|
|
738
|
+
) -> 'DataCleaner':
|
|
739
|
+
r"""
|
|
740
|
+
Extract text using regex capture group.
|
|
741
|
+
|
|
742
|
+
Parameters:
|
|
743
|
+
-----------
|
|
744
|
+
source_col : str
|
|
745
|
+
Source column name
|
|
746
|
+
pattern : str
|
|
747
|
+
Regex pattern with one capture group (e.g. r'ID: (\d+)')
|
|
748
|
+
new_col_name : str
|
|
749
|
+
Name for the new column
|
|
750
|
+
"""
|
|
751
|
+
if source_col not in self.df.columns:
|
|
752
|
+
return self
|
|
753
|
+
|
|
754
|
+
try:
|
|
755
|
+
# Ensure pattern is raw string if possible, generally passed as string here
|
|
756
|
+
extracted = self.df[source_col].astype(str).str.extract(pattern, expand=False)
|
|
757
|
+
|
|
758
|
+
self.df[new_col_name] = extracted
|
|
759
|
+
|
|
760
|
+
matched_count = extracted.notna().sum()
|
|
761
|
+
msg = f"Extracted '{new_col_name}' from '{source_col}' ({matched_count} matches)"
|
|
762
|
+
self.cleaning_log.append(msg)
|
|
763
|
+
self.transformations.append({
|
|
764
|
+
"operation": "extract_regex",
|
|
765
|
+
"source": source_col,
|
|
766
|
+
"target": new_col_name,
|
|
767
|
+
"pattern": pattern,
|
|
768
|
+
"matches": int(matched_count)
|
|
769
|
+
})
|
|
770
|
+
|
|
771
|
+
except Exception as e:
|
|
772
|
+
self.cleaning_log.append(f"Regex extraction failed: {e}")
|
|
773
|
+
|
|
774
|
+
return self
|
|
775
|
+
|
|
776
|
+
def drop_columns(
|
|
777
|
+
self,
|
|
778
|
+
columns: Optional[List[str]] = None,
|
|
779
|
+
drop_constant: bool = True,
|
|
780
|
+
drop_id_like: bool = False
|
|
781
|
+
) -> 'DataCleaner':
|
|
782
|
+
"""
|
|
783
|
+
Drop specified or problematic columns.
|
|
784
|
+
|
|
785
|
+
Parameters:
|
|
786
|
+
-----------
|
|
787
|
+
columns : list, optional
|
|
788
|
+
Specific columns to drop
|
|
789
|
+
drop_constant : bool
|
|
790
|
+
Drop columns with only one unique value
|
|
791
|
+
drop_id_like : bool
|
|
792
|
+
Drop columns that appear to be IDs
|
|
793
|
+
"""
|
|
794
|
+
cols_to_drop = set(columns or [])
|
|
795
|
+
|
|
796
|
+
if drop_constant:
|
|
797
|
+
for col in self.df.columns:
|
|
798
|
+
if self.df[col].nunique() <= 1:
|
|
799
|
+
cols_to_drop.add(col)
|
|
800
|
+
|
|
801
|
+
if drop_id_like:
|
|
802
|
+
for col in self.df.columns:
|
|
803
|
+
if self.df[col].nunique() == len(self.df):
|
|
804
|
+
cols_to_drop.add(col)
|
|
805
|
+
|
|
806
|
+
cols_to_drop = [c for c in cols_to_drop if c in self.df.columns]
|
|
807
|
+
|
|
808
|
+
if cols_to_drop:
|
|
809
|
+
self.df = self.df.drop(columns=cols_to_drop)
|
|
810
|
+
msg = f"Dropped {len(cols_to_drop)} columns: {cols_to_drop}"
|
|
811
|
+
self.cleaning_log.append(msg)
|
|
812
|
+
self.transformations.append({
|
|
813
|
+
"operation": "drop_columns",
|
|
814
|
+
"columns": cols_to_drop
|
|
815
|
+
})
|
|
816
|
+
|
|
817
|
+
return self
|
|
818
|
+
|
|
819
|
+
def get_cleaned_data(self) -> pd.DataFrame:
|
|
820
|
+
"""Return the cleaned DataFrame."""
|
|
821
|
+
return self.df
|
|
822
|
+
|
|
823
|
+
def get_cleaning_summary(self) -> Dict[str, Any]:
|
|
824
|
+
"""Return summary of all cleaning operations."""
|
|
825
|
+
return {
|
|
826
|
+
"original_shape": self.original_shape,
|
|
827
|
+
"final_shape": self.df.shape,
|
|
828
|
+
"rows_changed": self.original_shape[0] - self.df.shape[0],
|
|
829
|
+
"columns_changed": self.original_shape[1] - self.df.shape[1],
|
|
830
|
+
"operations": self.cleaning_log,
|
|
831
|
+
"transformations": self.transformations,
|
|
832
|
+
"row_changes": self.row_changes
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
def validate_quality(self) -> Dict[str, Any]:
|
|
836
|
+
"""
|
|
837
|
+
Perform data quality checks and calculate a 0-100 Quality Score.
|
|
838
|
+
|
|
839
|
+
Returns:
|
|
840
|
+
--------
|
|
841
|
+
dict
|
|
842
|
+
Report containing quality metrics, score, and warnings.
|
|
843
|
+
"""
|
|
844
|
+
n_rows = len(self.df)
|
|
845
|
+
n_cols = len(self.df.columns)
|
|
846
|
+
|
|
847
|
+
if n_rows == 0 or n_cols == 0:
|
|
848
|
+
return {"score": 0, "grade": "F", "warnings": ["Empty dataset"]}
|
|
849
|
+
|
|
850
|
+
# 1. Completeness (0-40 pts)
|
|
851
|
+
missing_total = self.df.isnull().sum().sum()
|
|
852
|
+
missing_pct = missing_total / (n_rows * n_cols)
|
|
853
|
+
completeness_score = max(0, 40 * (1 - missing_pct * 2)) # Penalize missing heavily
|
|
854
|
+
|
|
855
|
+
# 2. Uniqueness (0-30 pts)
|
|
856
|
+
# Check duplicates
|
|
857
|
+
n_dupes = self.df.duplicated().sum()
|
|
858
|
+
dupe_pct = n_dupes / n_rows
|
|
859
|
+
uniqueness_score = max(0, 30 * (1 - dupe_pct * 2))
|
|
860
|
+
|
|
861
|
+
# 3. Consistency/Validity (0-30 pts)
|
|
862
|
+
# Check for constant columns (0 variance)
|
|
863
|
+
n_constant = sum([1 for c in self.df.columns if self.df[c].nunique() <= 1])
|
|
864
|
+
const_pct = n_constant / n_cols
|
|
865
|
+
consistency_score = max(0, 30 * (1 - const_pct * 3))
|
|
866
|
+
|
|
867
|
+
final_score = int(completeness_score + uniqueness_score + consistency_score)
|
|
868
|
+
|
|
869
|
+
grade = 'A' if final_score >= 90 else 'B' if final_score >= 80 else 'C' if final_score >= 60 else 'D' if final_score >= 40 else 'F'
|
|
870
|
+
|
|
871
|
+
report: Dict[str, Any] = {
|
|
872
|
+
"score": final_score,
|
|
873
|
+
"grade": grade,
|
|
874
|
+
"rows": n_rows,
|
|
875
|
+
"columns": n_cols,
|
|
876
|
+
"missing_values": int(missing_total),
|
|
877
|
+
"missing_percentage": round(float(missing_pct * 100), 1),
|
|
878
|
+
"duplicate_rows": int(n_dupes),
|
|
879
|
+
"constant_columns": [],
|
|
880
|
+
"warnings": []
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
# Check for constant columns
|
|
884
|
+
for col in self.df.columns:
|
|
885
|
+
if self.df[col].nunique() <= 1:
|
|
886
|
+
report["constant_columns"].append(col)
|
|
887
|
+
report["warnings"].append(f"Column '{col}' is constant (1 unique value)")
|
|
888
|
+
|
|
889
|
+
# Check for extreme missing values
|
|
890
|
+
high_missing = self.df.columns[self.df.isnull().mean() > 0.5].tolist()
|
|
891
|
+
if high_missing:
|
|
892
|
+
report["warnings"].append(f"{len(high_missing)} columns have >50% missing values")
|
|
893
|
+
|
|
894
|
+
return report
|
|
895
|
+
|
|
896
|
+
def generate_suggestions(self) -> List[Dict[str, str]]:
|
|
897
|
+
"""
|
|
898
|
+
Generate AI-like cleaning suggestions based on data issues.
|
|
899
|
+
"""
|
|
900
|
+
suggestions = []
|
|
901
|
+
|
|
902
|
+
# Missing Values
|
|
903
|
+
missing = self.df.isnull().sum()
|
|
904
|
+
missing_cols = missing[missing > 0]
|
|
905
|
+
|
|
906
|
+
for col, count in missing_cols.items():
|
|
907
|
+
pct = count / len(self.df)
|
|
908
|
+
if pct > 0.4:
|
|
909
|
+
suggestions.append({
|
|
910
|
+
"column": col,
|
|
911
|
+
"issue": f"{pct:.0%} missing values",
|
|
912
|
+
"action": "Drop Column",
|
|
913
|
+
"reason": "Too much missing data to impute reliably."
|
|
914
|
+
})
|
|
915
|
+
else:
|
|
916
|
+
method = "Median Imputation" if pd.api.types.is_numeric_dtype(self.df[col]) else "Mode Imputation"
|
|
917
|
+
suggestions.append({
|
|
918
|
+
"column": col,
|
|
919
|
+
"issue": f"{pct:.0%} missing values",
|
|
920
|
+
"action": method,
|
|
921
|
+
"reason": "Standard strategy for filling gaps."
|
|
922
|
+
})
|
|
923
|
+
|
|
924
|
+
# Duplicates
|
|
925
|
+
n_dupes = self.df.duplicated().sum()
|
|
926
|
+
if n_dupes > 0:
|
|
927
|
+
suggestions.append({
|
|
928
|
+
"column": "Dataset",
|
|
929
|
+
"issue": f"{n_dupes} duplicate rows",
|
|
930
|
+
"action": "Remove Duplicates",
|
|
931
|
+
"reason": "Duplicate data skews model training."
|
|
932
|
+
})
|
|
933
|
+
|
|
934
|
+
# Outliers (Numeric)
|
|
935
|
+
numeric_cols = self.df.select_dtypes(include=[np.number]).columns
|
|
936
|
+
for col in numeric_cols:
|
|
937
|
+
if self.df[col].nunique() < 10: continue # Skip categorical-like
|
|
938
|
+
|
|
939
|
+
Q1 = self.df[col].quantile(0.25)
|
|
940
|
+
Q3 = self.df[col].quantile(0.75)
|
|
941
|
+
IQR = Q3 - Q1
|
|
942
|
+
lower = Q1 - 1.5 * IQR
|
|
943
|
+
upper = Q3 + 1.5 * IQR
|
|
944
|
+
outliers = ((self.df[col] < lower) | (self.df[col] > upper)).sum()
|
|
945
|
+
|
|
946
|
+
if outliers > 0:
|
|
947
|
+
pct = outliers / len(self.df)
|
|
948
|
+
if pct < 0.05:
|
|
949
|
+
action = "Clip (Winsorize)"
|
|
950
|
+
else:
|
|
951
|
+
action = "Log Transform" if (self.df[col] > 0).all() else "Standard Scaling"
|
|
952
|
+
|
|
953
|
+
suggestions.append({
|
|
954
|
+
"column": col,
|
|
955
|
+
"issue": f"{outliers} outliers detected",
|
|
956
|
+
"action": action,
|
|
957
|
+
"reason": "Outliers can distort linear models."
|
|
958
|
+
})
|
|
959
|
+
|
|
960
|
+
# ID Columns
|
|
961
|
+
for col in self.df.columns:
|
|
962
|
+
if col.lower() in ['id', 'uuid', 'guid', 'index'] or \
|
|
963
|
+
(self.df[col].nunique() == len(self.df) and pd.api.types.is_string_dtype(self.df[col])):
|
|
964
|
+
suggestions.append({
|
|
965
|
+
"column": col,
|
|
966
|
+
"issue": "High cardinality / ID-like",
|
|
967
|
+
"action": "Drop Column",
|
|
968
|
+
"reason": "Identifiers do not predict the target."
|
|
969
|
+
})
|
|
970
|
+
|
|
971
|
+
return suggestions
|
|
972
|
+
|
|
973
|
+
def print_summary(self) -> None:
|
|
974
|
+
"""Print cleaning summary."""
|
|
975
|
+
summary = self.get_cleaning_summary()
|
|
976
|
+
|
|
977
|
+
print("=" * 60)
|
|
978
|
+
logger.info("DATA CLEANING SUMMARY")
|
|
979
|
+
print("=" * 60)
|
|
980
|
+
logger.info(f"\nš Shape: {summary['original_shape']} ā {summary['final_shape']}")
|
|
981
|
+
logger.info(f" Rows changed: {summary['rows_changed']:+,}")
|
|
982
|
+
logger.info(f" Columns changed: {summary['columns_changed']:+,}")
|
|
983
|
+
|
|
984
|
+
logger.info("\nš§ Operations performed:")
|
|
985
|
+
for op in summary['operations']:
|
|
986
|
+
logger.info(f" ⢠{op}")
|
|
987
|
+
|
|
988
|
+
print("\n" + "=" * 60)
|