cleanflow-kit 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.1.0.dist-info/METADATA +119 -0
- cleanflow_kit-1.1.0.dist-info/RECORD +17 -0
- cleanflow_kit-1.1.0.dist-info/WHEEL +5 -0
- cleanflow_kit-1.1.0.dist-info/licenses/LICENSE +21 -0
- cleanflow_kit-1.1.0.dist-info/top_level.txt +1 -0
- dataclean/__init__.py +44 -0
- dataclean/_compat.py +24 -0
- dataclean/data_cleaner.py +988 -0
- dataclean/data_loader.py +194 -0
- dataclean/drift_detector.py +111 -0
- dataclean/eda.py +400 -0
- dataclean/feature_engineer.py +608 -0
- dataclean/model_trainer.py +874 -0
- dataclean/pipeline.py +548 -0
- dataclean/py.typed +0 -0
- dataclean/report_generator.py +365 -0
- dataclean/synthetic_generator.py +97 -0
dataclean/data_loader.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Loader Module
|
|
3
|
+
==================
|
|
4
|
+
Handles loading and initial validation of datasets.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import numpy as np
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Union, Optional, Dict, Any
|
|
13
|
+
from ._compat import normalize_string_columns
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class DataLoader:
|
|
17
|
+
"""Load and validate datasets from various file formats."""
|
|
18
|
+
|
|
19
|
+
def __init__(self):
|
|
20
|
+
self.df: Optional[pd.DataFrame] = None
|
|
21
|
+
self.file_path: Optional[str] = None
|
|
22
|
+
self.validation_report: Dict[str, Any] = {}
|
|
23
|
+
|
|
24
|
+
def load(
|
|
25
|
+
self,
|
|
26
|
+
source: Union[str, pd.DataFrame],
|
|
27
|
+
**kwargs
|
|
28
|
+
) -> pd.DataFrame:
|
|
29
|
+
"""
|
|
30
|
+
Load dataset from file path or DataFrame.
|
|
31
|
+
|
|
32
|
+
Parameters:
|
|
33
|
+
-----------
|
|
34
|
+
source : str or pd.DataFrame
|
|
35
|
+
File path (CSV/Excel) or existing DataFrame
|
|
36
|
+
**kwargs : dict
|
|
37
|
+
Additional arguments passed to pandas read functions
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
--------
|
|
41
|
+
pd.DataFrame : Loaded dataset
|
|
42
|
+
"""
|
|
43
|
+
if isinstance(source, pd.DataFrame):
|
|
44
|
+
self.df = source.copy()
|
|
45
|
+
self.file_path = "DataFrame input"
|
|
46
|
+
elif isinstance(source, str):
|
|
47
|
+
self.file_path = source
|
|
48
|
+
self.df = self._load_from_file(source, **kwargs)
|
|
49
|
+
else:
|
|
50
|
+
raise ValueError(f"Unsupported source type: {type(source)}")
|
|
51
|
+
|
|
52
|
+
return normalize_string_columns(self.df)
|
|
53
|
+
|
|
54
|
+
def _load_from_file(self, file_path: str, **kwargs) -> pd.DataFrame:
|
|
55
|
+
"""Load data from file based on extension."""
|
|
56
|
+
path = Path(file_path)
|
|
57
|
+
|
|
58
|
+
if not path.exists():
|
|
59
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
60
|
+
|
|
61
|
+
extension = path.suffix.lower()
|
|
62
|
+
|
|
63
|
+
if extension == '.csv':
|
|
64
|
+
return pd.read_csv(file_path, **kwargs)
|
|
65
|
+
elif extension in ['.xlsx', '.xls']:
|
|
66
|
+
return pd.read_excel(file_path, **kwargs)
|
|
67
|
+
elif extension == '.json':
|
|
68
|
+
return pd.read_json(file_path, **kwargs)
|
|
69
|
+
elif extension == '.parquet':
|
|
70
|
+
return pd.read_parquet(file_path, **kwargs)
|
|
71
|
+
else:
|
|
72
|
+
raise ValueError(f"Unsupported file format: {extension}")
|
|
73
|
+
|
|
74
|
+
def validate(self) -> Dict[str, Any]:
|
|
75
|
+
"""
|
|
76
|
+
Perform initial validation on loaded dataset.
|
|
77
|
+
|
|
78
|
+
Returns:
|
|
79
|
+
--------
|
|
80
|
+
dict : Validation report with dataset information
|
|
81
|
+
"""
|
|
82
|
+
if self.df is None:
|
|
83
|
+
raise ValueError("No dataset loaded. Call load() first.")
|
|
84
|
+
|
|
85
|
+
df = self.df
|
|
86
|
+
|
|
87
|
+
# Basic info
|
|
88
|
+
self.validation_report = {
|
|
89
|
+
"shape": df.shape,
|
|
90
|
+
"columns": list(df.columns),
|
|
91
|
+
"dtypes": df.dtypes.to_dict(),
|
|
92
|
+
"memory_usage_mb": df.memory_usage(deep=True).sum() / (1024 * 1024),
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
# Missing values
|
|
96
|
+
missing = df.isnull().sum()
|
|
97
|
+
missing_pct = (missing / len(df) * 100).round(2)
|
|
98
|
+
self.validation_report["missing_values"] = {
|
|
99
|
+
"counts": missing[missing > 0].to_dict(),
|
|
100
|
+
"percentages": missing_pct[missing_pct > 0].to_dict(),
|
|
101
|
+
"total_missing_cells": int(missing.sum()),
|
|
102
|
+
"columns_with_missing": int((missing > 0).sum())
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
# Duplicates
|
|
106
|
+
duplicate_count = df.duplicated().sum()
|
|
107
|
+
self.validation_report["duplicates"] = {
|
|
108
|
+
"count": int(duplicate_count),
|
|
109
|
+
"percentage": round(duplicate_count / len(df) * 100, 2)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
# Data type analysis
|
|
113
|
+
numeric_cols = df.select_dtypes(include=[np.number]).columns.tolist()
|
|
114
|
+
categorical_cols = df.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
115
|
+
datetime_cols = df.select_dtypes(include=['datetime64']).columns.tolist()
|
|
116
|
+
boolean_cols = df.select_dtypes(include=['bool']).columns.tolist()
|
|
117
|
+
|
|
118
|
+
self.validation_report["column_types"] = {
|
|
119
|
+
"numeric": numeric_cols,
|
|
120
|
+
"categorical": categorical_cols,
|
|
121
|
+
"datetime": datetime_cols,
|
|
122
|
+
"boolean": boolean_cols
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
# Constant/near-constant columns
|
|
126
|
+
constant_cols = []
|
|
127
|
+
near_constant_cols = []
|
|
128
|
+
|
|
129
|
+
for col in df.columns:
|
|
130
|
+
nunique = df[col].nunique()
|
|
131
|
+
if nunique == 1:
|
|
132
|
+
constant_cols.append(col)
|
|
133
|
+
elif nunique <= 2 and len(df) > 100:
|
|
134
|
+
near_constant_cols.append(col)
|
|
135
|
+
|
|
136
|
+
self.validation_report["low_variance_columns"] = {
|
|
137
|
+
"constant": constant_cols,
|
|
138
|
+
"near_constant": near_constant_cols
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
# Potential ID columns (high cardinality)
|
|
142
|
+
potential_id_cols = []
|
|
143
|
+
for col in df.columns:
|
|
144
|
+
if df[col].nunique() == len(df):
|
|
145
|
+
potential_id_cols.append(col)
|
|
146
|
+
|
|
147
|
+
self.validation_report["potential_id_columns"] = potential_id_cols
|
|
148
|
+
|
|
149
|
+
return self.validation_report
|
|
150
|
+
|
|
151
|
+
def print_summary(self) -> None:
|
|
152
|
+
"""Print a formatted summary of the validation report."""
|
|
153
|
+
if not self.validation_report:
|
|
154
|
+
self.validate()
|
|
155
|
+
|
|
156
|
+
report = self.validation_report
|
|
157
|
+
|
|
158
|
+
print("=" * 60)
|
|
159
|
+
logger.info("DATASET VALIDATION SUMMARY")
|
|
160
|
+
print("=" * 60)
|
|
161
|
+
|
|
162
|
+
logger.info(f"\n📊 Shape: {report['shape'][0]:,} rows × {report['shape'][1]} columns")
|
|
163
|
+
logger.info(f"💾 Memory Usage: {report['memory_usage_mb']:.2f} MB")
|
|
164
|
+
|
|
165
|
+
logger.info("\n📋 Column Types:")
|
|
166
|
+
for ctype, cols in report['column_types'].items():
|
|
167
|
+
if cols:
|
|
168
|
+
logger.info(f" • {ctype.capitalize()}: {len(cols)} columns")
|
|
169
|
+
|
|
170
|
+
missing = report['missing_values']
|
|
171
|
+
if missing['columns_with_missing'] > 0:
|
|
172
|
+
logger.info("\n⚠️ Missing Values:")
|
|
173
|
+
logger.info(f" • Columns affected: {missing['columns_with_missing']}")
|
|
174
|
+
logger.info(f" • Total missing cells: {missing['total_missing_cells']:,}")
|
|
175
|
+
for col, pct in list(missing['percentages'].items())[:5]:
|
|
176
|
+
logger.info(f" • {col}: {pct}%")
|
|
177
|
+
if len(missing['percentages']) > 5:
|
|
178
|
+
logger.info(f" ... and {len(missing['percentages']) - 5} more columns")
|
|
179
|
+
else:
|
|
180
|
+
logger.info("\n✅ No missing values detected")
|
|
181
|
+
|
|
182
|
+
dups = report['duplicates']
|
|
183
|
+
if dups['count'] > 0:
|
|
184
|
+
logger.info(f"\n⚠️ Duplicates: {dups['count']:,} rows ({dups['percentage']}%)")
|
|
185
|
+
else:
|
|
186
|
+
logger.info("\n✅ No duplicate rows detected")
|
|
187
|
+
|
|
188
|
+
if report['low_variance_columns']['constant']:
|
|
189
|
+
logger.info(f"\n⚠️ Constant columns: {report['low_variance_columns']['constant']}")
|
|
190
|
+
|
|
191
|
+
if report['potential_id_columns']:
|
|
192
|
+
logger.info(f"\n🔑 Potential ID columns: {report['potential_id_columns']}")
|
|
193
|
+
|
|
194
|
+
print("\n" + "=" * 60)
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Drift Detector
|
|
3
|
+
===================
|
|
4
|
+
Detects schema changes and statistical drift between two datasets.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import numpy as np
|
|
11
|
+
from scipy import stats
|
|
12
|
+
from typing import Dict, Any, List, Optional
|
|
13
|
+
|
|
14
|
+
class DriftDetector:
|
|
15
|
+
"""
|
|
16
|
+
Detects data drift between a baseline dataset and a current dataset.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
def __init__(self, baseline_df: pd.DataFrame, current_df: pd.DataFrame):
|
|
20
|
+
self.baseline = baseline_df
|
|
21
|
+
self.current = current_df
|
|
22
|
+
self.report: Dict[str, Any] = {
|
|
23
|
+
'schema_drift': {},
|
|
24
|
+
'statistical_drift': {},
|
|
25
|
+
'score': 100
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
def run(self) -> Dict[str, Any]:
|
|
29
|
+
"""Run all drift checks."""
|
|
30
|
+
self._check_schema_drift()
|
|
31
|
+
self._check_statistical_drift()
|
|
32
|
+
self._calculate_score()
|
|
33
|
+
return self.report
|
|
34
|
+
|
|
35
|
+
def _check_schema_drift(self):
|
|
36
|
+
"""Check for missing or new columns."""
|
|
37
|
+
base_cols = set(self.baseline.columns)
|
|
38
|
+
curr_cols = set(self.current.columns)
|
|
39
|
+
|
|
40
|
+
missing = list(base_cols - curr_cols)
|
|
41
|
+
new = list(curr_cols - base_cols)
|
|
42
|
+
|
|
43
|
+
self.report['schema_drift'] = {
|
|
44
|
+
'missing_columns': missing,
|
|
45
|
+
'new_columns': new,
|
|
46
|
+
'has_drift': len(missing) > 0
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
def _check_statistical_drift(self):
|
|
50
|
+
"""
|
|
51
|
+
Check for statistical drift in shared numeric columns using KS-Test.
|
|
52
|
+
KS-Test (Kolmogorov-Smirnov) checks if two samples come from same distribution.
|
|
53
|
+
"""
|
|
54
|
+
base_cols = set(self.baseline.select_dtypes(include=[np.number]).columns)
|
|
55
|
+
curr_cols = set(self.current.select_dtypes(include=[np.number]).columns)
|
|
56
|
+
shared_cols = list(base_cols.intersection(curr_cols))
|
|
57
|
+
|
|
58
|
+
drifted_features = []
|
|
59
|
+
|
|
60
|
+
for col in shared_cols:
|
|
61
|
+
# Drop NaNs for valid test
|
|
62
|
+
b_data = self.baseline[col].dropna()
|
|
63
|
+
c_data = self.current[col].dropna()
|
|
64
|
+
|
|
65
|
+
if len(b_data) == 0 or len(c_data) == 0:
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
# KS Test
|
|
69
|
+
# Null hypothesis: samples are from same distribution.
|
|
70
|
+
# If p_value < 0.05, we reject null hypothesis -> DRIFT DETECTED.
|
|
71
|
+
stat, p_value = stats.ks_2samp(b_data, c_data)
|
|
72
|
+
|
|
73
|
+
is_drifted = p_value < 0.05
|
|
74
|
+
|
|
75
|
+
# Additional metrics
|
|
76
|
+
b_mean = b_data.mean()
|
|
77
|
+
c_mean = c_data.mean()
|
|
78
|
+
delta_mean = abs(b_mean - c_mean)
|
|
79
|
+
perc_change = (delta_mean / b_mean) * 100 if b_mean != 0 else 0
|
|
80
|
+
|
|
81
|
+
self.report['statistical_drift'][col] = {
|
|
82
|
+
'p_value': float(p_value),
|
|
83
|
+
'is_drifted': bool(is_drifted),
|
|
84
|
+
'baseline_mean': float(b_mean),
|
|
85
|
+
'current_mean': float(c_mean),
|
|
86
|
+
'pct_change': float(perc_change)
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
if is_drifted:
|
|
90
|
+
drifted_features.append(col)
|
|
91
|
+
|
|
92
|
+
self.report['drifted_features'] = drifted_features
|
|
93
|
+
self.report['drift_detected'] = len(drifted_features) > 0
|
|
94
|
+
|
|
95
|
+
def _calculate_score(self):
|
|
96
|
+
"""Calculate a simple health score (0-100)."""
|
|
97
|
+
# Penalty for schema drift
|
|
98
|
+
score = 100
|
|
99
|
+
if self.report['schema_drift']['has_drift']:
|
|
100
|
+
score -= 20
|
|
101
|
+
|
|
102
|
+
# Penalty for statistical drift
|
|
103
|
+
total_feats = len(self.report['statistical_drift'])
|
|
104
|
+
drifted = len(self.report['drifted_features'])
|
|
105
|
+
|
|
106
|
+
if total_feats > 0:
|
|
107
|
+
drift_ratio = drifted / total_feats
|
|
108
|
+
score -= (drift_ratio * 50)
|
|
109
|
+
|
|
110
|
+
self.report['score'] = max(0, int(score))
|
|
111
|
+
|