cleanflow-kit 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,194 @@
1
+ """
2
+ Data Loader Module
3
+ ==================
4
+ Handles loading and initial validation of datasets.
5
+ """
6
+
7
+ import logging
8
+ logger = logging.getLogger(__name__)
9
+ import pandas as pd
10
+ import numpy as np
11
+ from pathlib import Path
12
+ from typing import Union, Optional, Dict, Any
13
+ from ._compat import normalize_string_columns
14
+
15
+
16
+ class DataLoader:
17
+ """Load and validate datasets from various file formats."""
18
+
19
+ def __init__(self):
20
+ self.df: Optional[pd.DataFrame] = None
21
+ self.file_path: Optional[str] = None
22
+ self.validation_report: Dict[str, Any] = {}
23
+
24
+ def load(
25
+ self,
26
+ source: Union[str, pd.DataFrame],
27
+ **kwargs
28
+ ) -> pd.DataFrame:
29
+ """
30
+ Load dataset from file path or DataFrame.
31
+
32
+ Parameters:
33
+ -----------
34
+ source : str or pd.DataFrame
35
+ File path (CSV/Excel) or existing DataFrame
36
+ **kwargs : dict
37
+ Additional arguments passed to pandas read functions
38
+
39
+ Returns:
40
+ --------
41
+ pd.DataFrame : Loaded dataset
42
+ """
43
+ if isinstance(source, pd.DataFrame):
44
+ self.df = source.copy()
45
+ self.file_path = "DataFrame input"
46
+ elif isinstance(source, str):
47
+ self.file_path = source
48
+ self.df = self._load_from_file(source, **kwargs)
49
+ else:
50
+ raise ValueError(f"Unsupported source type: {type(source)}")
51
+
52
+ return normalize_string_columns(self.df)
53
+
54
+ def _load_from_file(self, file_path: str, **kwargs) -> pd.DataFrame:
55
+ """Load data from file based on extension."""
56
+ path = Path(file_path)
57
+
58
+ if not path.exists():
59
+ raise FileNotFoundError(f"File not found: {file_path}")
60
+
61
+ extension = path.suffix.lower()
62
+
63
+ if extension == '.csv':
64
+ return pd.read_csv(file_path, **kwargs)
65
+ elif extension in ['.xlsx', '.xls']:
66
+ return pd.read_excel(file_path, **kwargs)
67
+ elif extension == '.json':
68
+ return pd.read_json(file_path, **kwargs)
69
+ elif extension == '.parquet':
70
+ return pd.read_parquet(file_path, **kwargs)
71
+ else:
72
+ raise ValueError(f"Unsupported file format: {extension}")
73
+
74
+ def validate(self) -> Dict[str, Any]:
75
+ """
76
+ Perform initial validation on loaded dataset.
77
+
78
+ Returns:
79
+ --------
80
+ dict : Validation report with dataset information
81
+ """
82
+ if self.df is None:
83
+ raise ValueError("No dataset loaded. Call load() first.")
84
+
85
+ df = self.df
86
+
87
+ # Basic info
88
+ self.validation_report = {
89
+ "shape": df.shape,
90
+ "columns": list(df.columns),
91
+ "dtypes": df.dtypes.to_dict(),
92
+ "memory_usage_mb": df.memory_usage(deep=True).sum() / (1024 * 1024),
93
+ }
94
+
95
+ # Missing values
96
+ missing = df.isnull().sum()
97
+ missing_pct = (missing / len(df) * 100).round(2)
98
+ self.validation_report["missing_values"] = {
99
+ "counts": missing[missing > 0].to_dict(),
100
+ "percentages": missing_pct[missing_pct > 0].to_dict(),
101
+ "total_missing_cells": int(missing.sum()),
102
+ "columns_with_missing": int((missing > 0).sum())
103
+ }
104
+
105
+ # Duplicates
106
+ duplicate_count = df.duplicated().sum()
107
+ self.validation_report["duplicates"] = {
108
+ "count": int(duplicate_count),
109
+ "percentage": round(duplicate_count / len(df) * 100, 2)
110
+ }
111
+
112
+ # Data type analysis
113
+ numeric_cols = df.select_dtypes(include=[np.number]).columns.tolist()
114
+ categorical_cols = df.select_dtypes(include=['object', 'category']).columns.tolist()
115
+ datetime_cols = df.select_dtypes(include=['datetime64']).columns.tolist()
116
+ boolean_cols = df.select_dtypes(include=['bool']).columns.tolist()
117
+
118
+ self.validation_report["column_types"] = {
119
+ "numeric": numeric_cols,
120
+ "categorical": categorical_cols,
121
+ "datetime": datetime_cols,
122
+ "boolean": boolean_cols
123
+ }
124
+
125
+ # Constant/near-constant columns
126
+ constant_cols = []
127
+ near_constant_cols = []
128
+
129
+ for col in df.columns:
130
+ nunique = df[col].nunique()
131
+ if nunique == 1:
132
+ constant_cols.append(col)
133
+ elif nunique <= 2 and len(df) > 100:
134
+ near_constant_cols.append(col)
135
+
136
+ self.validation_report["low_variance_columns"] = {
137
+ "constant": constant_cols,
138
+ "near_constant": near_constant_cols
139
+ }
140
+
141
+ # Potential ID columns (high cardinality)
142
+ potential_id_cols = []
143
+ for col in df.columns:
144
+ if df[col].nunique() == len(df):
145
+ potential_id_cols.append(col)
146
+
147
+ self.validation_report["potential_id_columns"] = potential_id_cols
148
+
149
+ return self.validation_report
150
+
151
+ def print_summary(self) -> None:
152
+ """Print a formatted summary of the validation report."""
153
+ if not self.validation_report:
154
+ self.validate()
155
+
156
+ report = self.validation_report
157
+
158
+ print("=" * 60)
159
+ logger.info("DATASET VALIDATION SUMMARY")
160
+ print("=" * 60)
161
+
162
+ logger.info(f"\n📊 Shape: {report['shape'][0]:,} rows × {report['shape'][1]} columns")
163
+ logger.info(f"💾 Memory Usage: {report['memory_usage_mb']:.2f} MB")
164
+
165
+ logger.info("\n📋 Column Types:")
166
+ for ctype, cols in report['column_types'].items():
167
+ if cols:
168
+ logger.info(f" • {ctype.capitalize()}: {len(cols)} columns")
169
+
170
+ missing = report['missing_values']
171
+ if missing['columns_with_missing'] > 0:
172
+ logger.info("\n⚠️ Missing Values:")
173
+ logger.info(f" • Columns affected: {missing['columns_with_missing']}")
174
+ logger.info(f" • Total missing cells: {missing['total_missing_cells']:,}")
175
+ for col, pct in list(missing['percentages'].items())[:5]:
176
+ logger.info(f" • {col}: {pct}%")
177
+ if len(missing['percentages']) > 5:
178
+ logger.info(f" ... and {len(missing['percentages']) - 5} more columns")
179
+ else:
180
+ logger.info("\n✅ No missing values detected")
181
+
182
+ dups = report['duplicates']
183
+ if dups['count'] > 0:
184
+ logger.info(f"\n⚠️ Duplicates: {dups['count']:,} rows ({dups['percentage']}%)")
185
+ else:
186
+ logger.info("\n✅ No duplicate rows detected")
187
+
188
+ if report['low_variance_columns']['constant']:
189
+ logger.info(f"\n⚠️ Constant columns: {report['low_variance_columns']['constant']}")
190
+
191
+ if report['potential_id_columns']:
192
+ logger.info(f"\n🔑 Potential ID columns: {report['potential_id_columns']}")
193
+
194
+ print("\n" + "=" * 60)
@@ -0,0 +1,111 @@
1
+ """
2
+ Data Drift Detector
3
+ ===================
4
+ Detects schema changes and statistical drift between two datasets.
5
+ """
6
+
7
+ import logging
8
+ logger = logging.getLogger(__name__)
9
+ import pandas as pd
10
+ import numpy as np
11
+ from scipy import stats
12
+ from typing import Dict, Any, List, Optional
13
+
14
+ class DriftDetector:
15
+ """
16
+ Detects data drift between a baseline dataset and a current dataset.
17
+ """
18
+
19
+ def __init__(self, baseline_df: pd.DataFrame, current_df: pd.DataFrame):
20
+ self.baseline = baseline_df
21
+ self.current = current_df
22
+ self.report: Dict[str, Any] = {
23
+ 'schema_drift': {},
24
+ 'statistical_drift': {},
25
+ 'score': 100
26
+ }
27
+
28
+ def run(self) -> Dict[str, Any]:
29
+ """Run all drift checks."""
30
+ self._check_schema_drift()
31
+ self._check_statistical_drift()
32
+ self._calculate_score()
33
+ return self.report
34
+
35
+ def _check_schema_drift(self):
36
+ """Check for missing or new columns."""
37
+ base_cols = set(self.baseline.columns)
38
+ curr_cols = set(self.current.columns)
39
+
40
+ missing = list(base_cols - curr_cols)
41
+ new = list(curr_cols - base_cols)
42
+
43
+ self.report['schema_drift'] = {
44
+ 'missing_columns': missing,
45
+ 'new_columns': new,
46
+ 'has_drift': len(missing) > 0
47
+ }
48
+
49
+ def _check_statistical_drift(self):
50
+ """
51
+ Check for statistical drift in shared numeric columns using KS-Test.
52
+ KS-Test (Kolmogorov-Smirnov) checks if two samples come from same distribution.
53
+ """
54
+ base_cols = set(self.baseline.select_dtypes(include=[np.number]).columns)
55
+ curr_cols = set(self.current.select_dtypes(include=[np.number]).columns)
56
+ shared_cols = list(base_cols.intersection(curr_cols))
57
+
58
+ drifted_features = []
59
+
60
+ for col in shared_cols:
61
+ # Drop NaNs for valid test
62
+ b_data = self.baseline[col].dropna()
63
+ c_data = self.current[col].dropna()
64
+
65
+ if len(b_data) == 0 or len(c_data) == 0:
66
+ continue
67
+
68
+ # KS Test
69
+ # Null hypothesis: samples are from same distribution.
70
+ # If p_value < 0.05, we reject null hypothesis -> DRIFT DETECTED.
71
+ stat, p_value = stats.ks_2samp(b_data, c_data)
72
+
73
+ is_drifted = p_value < 0.05
74
+
75
+ # Additional metrics
76
+ b_mean = b_data.mean()
77
+ c_mean = c_data.mean()
78
+ delta_mean = abs(b_mean - c_mean)
79
+ perc_change = (delta_mean / b_mean) * 100 if b_mean != 0 else 0
80
+
81
+ self.report['statistical_drift'][col] = {
82
+ 'p_value': float(p_value),
83
+ 'is_drifted': bool(is_drifted),
84
+ 'baseline_mean': float(b_mean),
85
+ 'current_mean': float(c_mean),
86
+ 'pct_change': float(perc_change)
87
+ }
88
+
89
+ if is_drifted:
90
+ drifted_features.append(col)
91
+
92
+ self.report['drifted_features'] = drifted_features
93
+ self.report['drift_detected'] = len(drifted_features) > 0
94
+
95
+ def _calculate_score(self):
96
+ """Calculate a simple health score (0-100)."""
97
+ # Penalty for schema drift
98
+ score = 100
99
+ if self.report['schema_drift']['has_drift']:
100
+ score -= 20
101
+
102
+ # Penalty for statistical drift
103
+ total_feats = len(self.report['statistical_drift'])
104
+ drifted = len(self.report['drifted_features'])
105
+
106
+ if total_feats > 0:
107
+ drift_ratio = drifted / total_feats
108
+ score -= (drift_ratio * 50)
109
+
110
+ self.report['score'] = max(0, int(score))
111
+