cleanflow-kit 1.2.1__tar.gz → 1.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.2.3/PKG-INFO +622 -0
- cleanflow_kit-1.2.3/README.md +586 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/__init__.py +1 -1
- cleanflow_kit-1.2.3/cleanflow_kit.egg-info/PKG-INFO +622 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/pyproject.toml +1 -1
- cleanflow_kit-1.2.1/PKG-INFO +0 -119
- cleanflow_kit-1.2.1/README.md +0 -83
- cleanflow_kit-1.2.1/cleanflow_kit.egg-info/PKG-INFO +0 -119
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/LICENSE +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/_compat.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/data_cleaner.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/data_loader.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/drift_detector.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/eda.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/feature_engineer.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/model_trainer.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/pipeline.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/py.typed +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/report_generator.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow/synthetic_generator.py +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow_kit.egg-info/SOURCES.txt +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow_kit.egg-info/dependency_links.txt +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow_kit.egg-info/requires.txt +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/cleanflow_kit.egg-info/top_level.txt +0 -0
- {cleanflow_kit-1.2.1 → cleanflow_kit-1.2.3}/setup.cfg +0 -0
|
@@ -0,0 +1,622 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cleanflow-kit
|
|
3
|
+
Version: 1.2.3
|
|
4
|
+
Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
|
|
5
|
+
Author: CleanFlow Contributors
|
|
6
|
+
License: MIT License
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.8
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: pandas>=2.0.0
|
|
14
|
+
Requires-Dist: numpy>=1.20.0
|
|
15
|
+
Requires-Dist: scikit-learn>=1.0.0
|
|
16
|
+
Requires-Dist: matplotlib>=3.5.0
|
|
17
|
+
Requires-Dist: joblib>=1.1.0
|
|
18
|
+
Provides-Extra: smote
|
|
19
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "smote"
|
|
20
|
+
Provides-Extra: drift
|
|
21
|
+
Requires-Dist: scipy>=1.7.0; extra == "drift"
|
|
22
|
+
Provides-Extra: all
|
|
23
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "all"
|
|
24
|
+
Requires-Dist: scipy>=1.7.0; extra == "all"
|
|
25
|
+
Provides-Extra: web
|
|
26
|
+
Requires-Dist: Flask; extra == "web"
|
|
27
|
+
Requires-Dist: Flask-SQLAlchemy; extra == "web"
|
|
28
|
+
Requires-Dist: PyJWT; extra == "web"
|
|
29
|
+
Requires-Dist: gunicorn; extra == "web"
|
|
30
|
+
Requires-Dist: openpyxl; extra == "web"
|
|
31
|
+
Provides-Extra: api
|
|
32
|
+
Requires-Dist: FastAPI; extra == "api"
|
|
33
|
+
Requires-Dist: uvicorn; extra == "api"
|
|
34
|
+
Requires-Dist: pydantic; extra == "api"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# CleanFlow (`cleanflow-kit`)
|
|
38
|
+
|
|
39
|
+
A comprehensive, intelligent Python package for automated **data cleaning**, **exploratory data analysis (EDA)**, **feature engineering**, **AutoML model training**, **data drift detection**, **synthetic data generation**, and **HTML/Markdown report generation**.
|
|
40
|
+
|
|
41
|
+
CleanFlow eliminates the repetitive 80% of data science work — replacing hundreds of lines of boilerplate Pandas/Scikit-learn code with a clean, chainable API.
|
|
42
|
+
|
|
43
|
+
---
|
|
44
|
+
|
|
45
|
+
## 📦 Installation
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install cleanflow-kit
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Install with optional extras:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
# SMOTE for class imbalance handling
|
|
55
|
+
pip install "cleanflow-kit[smote]"
|
|
56
|
+
|
|
57
|
+
# Drift detection (requires scipy)
|
|
58
|
+
pip install "cleanflow-kit[drift]"
|
|
59
|
+
|
|
60
|
+
# Everything
|
|
61
|
+
pip install "cleanflow-kit[all]"
|
|
62
|
+
|
|
63
|
+
# Web application (Flask UI)
|
|
64
|
+
pip install "cleanflow-kit[web,api]"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
**Requirements:** Python 3.8+, pandas ≥ 2.0, numpy ≥ 1.20, scikit-learn ≥ 1.0, matplotlib ≥ 3.5
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## 🚀 Quick Start (3 Lines)
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from cleanflow import DataPipeline
|
|
75
|
+
|
|
76
|
+
pipeline = DataPipeline()
|
|
77
|
+
pipeline.load("your_data.csv")
|
|
78
|
+
|
|
79
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
80
|
+
target_col="price",
|
|
81
|
+
problem_type="regression" # or "classification" — auto-detected if omitted
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
pipeline.save_data("./output")
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
That's it. Your data is cleaned, analyzed, feature-engineered, and saved.
|
|
88
|
+
|
|
89
|
+
---
|
|
90
|
+
|
|
91
|
+
## 📖 Complete Usage Guide
|
|
92
|
+
|
|
93
|
+
### 1. One-Click Full Pipeline
|
|
94
|
+
|
|
95
|
+
The simplest way to use CleanFlow. Runs validation → cleaning → EDA → feature engineering automatically.
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from cleanflow import DataPipeline
|
|
99
|
+
|
|
100
|
+
pipeline = DataPipeline()
|
|
101
|
+
pipeline.load("housing_prices.csv")
|
|
102
|
+
|
|
103
|
+
# Run everything
|
|
104
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
105
|
+
target_col="price",
|
|
106
|
+
problem_type="regression",
|
|
107
|
+
show_eda_plots=True,
|
|
108
|
+
cleaning_config={ # Optional overrides
|
|
109
|
+
"missing_numeric_strategy": "mean",
|
|
110
|
+
"outlier_method": "zscore"
|
|
111
|
+
},
|
|
112
|
+
feature_config={ # Optional overrides
|
|
113
|
+
"scale_method": "minmax",
|
|
114
|
+
"create_polynomial_features": True,
|
|
115
|
+
"auto_evolve_features": True
|
|
116
|
+
}
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
# Access reports
|
|
120
|
+
print(pipeline.pipeline_report['validation'])
|
|
121
|
+
print(pipeline.pipeline_report['cleaning'])
|
|
122
|
+
print(pipeline.pipeline_report['eda'])
|
|
123
|
+
print(pipeline.pipeline_report['feature_engineering'])
|
|
124
|
+
|
|
125
|
+
# Save outputs
|
|
126
|
+
pipeline.save_data("./output", format="csv") # or "parquet"
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
### 2. Step-by-Step Pipeline (Full Control)
|
|
132
|
+
|
|
133
|
+
Call each stage individually for granular control over every parameter.
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
from cleanflow import DataPipeline
|
|
137
|
+
|
|
138
|
+
pipeline = DataPipeline()
|
|
139
|
+
pipeline.load("customer_churn.csv")
|
|
140
|
+
|
|
141
|
+
# ── Step 1: Validate ──
|
|
142
|
+
report = pipeline.validate()
|
|
143
|
+
# Returns: shape, dtypes, missing values, duplicates, constant columns, potential ID columns
|
|
144
|
+
|
|
145
|
+
# ── Step 2: Clean ──
|
|
146
|
+
pipeline.clean(
|
|
147
|
+
remove_duplicates=True,
|
|
148
|
+
handle_missing=True,
|
|
149
|
+
missing_numeric_strategy='median', # 'mean', 'median', 'zero'
|
|
150
|
+
missing_categorical_strategy='mode', # 'mode', 'unknown'
|
|
151
|
+
missing_drop_threshold=0.4, # Drop columns with >40% missing
|
|
152
|
+
fix_types=True, # Auto-infer numeric, datetime, category types
|
|
153
|
+
handle_outliers=True,
|
|
154
|
+
outlier_method='iqr', # 'iqr' or 'zscore'
|
|
155
|
+
outlier_action='clip', # 'clip', 'remove', or 'nan'
|
|
156
|
+
clean_categorical=True, # Strip whitespace, lowercase
|
|
157
|
+
drop_constant=True # Remove constant/single-value columns
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
# ── Step 3: Exploratory Data Analysis ──
|
|
161
|
+
eda_results = pipeline.analyze(
|
|
162
|
+
target_col="churn",
|
|
163
|
+
show_plots=True,
|
|
164
|
+
save_plots=True,
|
|
165
|
+
output_dir="./eda_plots"
|
|
166
|
+
)
|
|
167
|
+
# Generates: distributions.png, boxplots.png, correlation.png, categorical.png, target_relationships.png
|
|
168
|
+
|
|
169
|
+
# ── Step 4: Feature Engineering ──
|
|
170
|
+
pipeline.engineer_features(
|
|
171
|
+
target_col="churn",
|
|
172
|
+
problem_type="classification",
|
|
173
|
+
encode_categorical=True, # Auto: OneHot for low-cardinality, Label for high
|
|
174
|
+
scale_features=True,
|
|
175
|
+
scale_method='standard', # 'standard' or 'minmax'
|
|
176
|
+
create_datetime_features=True, # Extracts year, month, day, dayofweek, is_weekend
|
|
177
|
+
create_polynomial_features=False, # Generates degree-2 polynomial terms
|
|
178
|
+
polynomial_degree=2,
|
|
179
|
+
drop_low_variance=True, # Removes near-zero variance features
|
|
180
|
+
drop_high_correlation=True, # Removes one of each highly correlated pair
|
|
181
|
+
correlation_threshold=0.95,
|
|
182
|
+
handle_imbalance=True, # Applies SMOTE for imbalanced classes
|
|
183
|
+
imbalance_method='smote', # 'smote', 'oversample', 'undersample'
|
|
184
|
+
auto_evolve_features=False # Auto-generates & selects best interaction features
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
# ── Step 5: Train Models (AutoML) ──
|
|
188
|
+
model_results = pipeline.train_model(
|
|
189
|
+
target_col="churn",
|
|
190
|
+
problem_type="classification",
|
|
191
|
+
export_path="./best_model.pkl" # Optional: export trained model
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
print("Best Model:", model_results['best_model']['name'])
|
|
195
|
+
print("Metrics:", model_results['best_model']['metrics'])
|
|
196
|
+
print("Reliability:", model_results['reliability'])
|
|
197
|
+
print("Feature Importance:", model_results['feature_importance'])
|
|
198
|
+
|
|
199
|
+
# ── Step 6: Generate Reports ──
|
|
200
|
+
pipeline.generate_html_report("./report.html")
|
|
201
|
+
pipeline.generate_markdown_report("./report.md")
|
|
202
|
+
|
|
203
|
+
# ── Access Data at Any Stage ──
|
|
204
|
+
raw_df = pipeline.get_raw_data()
|
|
205
|
+
cleaned_df = pipeline.get_cleaned_data()
|
|
206
|
+
final_df = pipeline.get_final_data()
|
|
207
|
+
full_report = pipeline.get_report()
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
---
|
|
211
|
+
|
|
212
|
+
### 3. Using Individual Modules Directly
|
|
213
|
+
|
|
214
|
+
For maximum flexibility, use the standalone classes without the pipeline orchestrator.
|
|
215
|
+
|
|
216
|
+
```python
|
|
217
|
+
from cleanflow import DataLoader, CleanFlower, EDAAnalyzer, FeatureEngineer, ModelTrainer
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
#### DataLoader
|
|
221
|
+
|
|
222
|
+
```python
|
|
223
|
+
loader = DataLoader()
|
|
224
|
+
|
|
225
|
+
# Load from file (CSV, Excel, JSON, Parquet)
|
|
226
|
+
df = loader.load("data.csv")
|
|
227
|
+
df = loader.load("data.xlsx")
|
|
228
|
+
df = loader.load("data.json")
|
|
229
|
+
df = loader.load("data.parquet")
|
|
230
|
+
|
|
231
|
+
# Load from existing DataFrame
|
|
232
|
+
df = loader.load(my_dataframe)
|
|
233
|
+
|
|
234
|
+
# Validate and print profile
|
|
235
|
+
report = loader.validate()
|
|
236
|
+
loader.print_summary()
|
|
237
|
+
# Reports: shape, memory usage, column types, missing values, duplicates,
|
|
238
|
+
# constant columns, near-constant columns, potential ID columns
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
#### CleanFlower (Data Cleaner)
|
|
242
|
+
|
|
243
|
+
```python
|
|
244
|
+
cleaner = CleanFlower(df)
|
|
245
|
+
|
|
246
|
+
# Chain operations fluently
|
|
247
|
+
cleaner.remove_duplicates()
|
|
248
|
+
|
|
249
|
+
cleaner.handle_missing_values(
|
|
250
|
+
numeric_strategy='median', # 'mean', 'median', 'zero'
|
|
251
|
+
categorical_strategy='mode', # 'mode', 'unknown'
|
|
252
|
+
drop_threshold=0.4 # Drop cols with >40% missing
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
cleaner.fix_data_types() # Auto-infer numeric, datetime, category
|
|
256
|
+
|
|
257
|
+
cleaner.handle_outliers(
|
|
258
|
+
method='iqr', # 'iqr' or 'zscore'
|
|
259
|
+
threshold=1.5, # IQR multiplier or Z-score threshold
|
|
260
|
+
action='clip' # 'clip', 'remove', 'nan'
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
cleaner.clean_categorical_values(
|
|
264
|
+
lowercase=True,
|
|
265
|
+
strip_whitespace=True,
|
|
266
|
+
replace_mapping={"city": {"nyc": "new york", "sf": "san francisco"}}
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
cleaner.clean_numeric_text() # Converts "$1,200" → 1200, "1.5k" → 1500
|
|
270
|
+
|
|
271
|
+
cleaner.normalize_features(method='standard') # 'standard', 'minmax', 'robust'
|
|
272
|
+
|
|
273
|
+
cleaner.encode_categorical(method='onehot') # 'onehot' or 'label'
|
|
274
|
+
|
|
275
|
+
cleaner.rename_columns({"old_name": "new_name"})
|
|
276
|
+
|
|
277
|
+
cleaner.extract_regex_feature(
|
|
278
|
+
source_col="description",
|
|
279
|
+
pattern=r'ID: (\d+)',
|
|
280
|
+
new_col_name="extracted_id"
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
cleaner.drop_columns(
|
|
284
|
+
columns=["unwanted_col"],
|
|
285
|
+
drop_constant=True, # Remove single-value columns
|
|
286
|
+
drop_id_like=True # Remove columns where every value is unique
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
# Data quality scoring (0–100)
|
|
290
|
+
quality = cleaner.validate_quality()
|
|
291
|
+
# Returns: score, grade (A/B/C/D/F), missing_values, duplicate_rows, constant_columns, warnings
|
|
292
|
+
|
|
293
|
+
# AI-like cleaning suggestions
|
|
294
|
+
suggestions = cleaner.generate_suggestions()
|
|
295
|
+
# Returns list of: {column, issue, action, reason} for every detected problem
|
|
296
|
+
|
|
297
|
+
# Get results
|
|
298
|
+
cleaned_df = cleaner.get_cleaned_data()
|
|
299
|
+
summary = cleaner.get_cleaning_summary()
|
|
300
|
+
# Summary includes: original_shape, final_shape, operations log, row-level change audit trail
|
|
301
|
+
cleaner.print_summary()
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
#### EDAAnalyzer
|
|
305
|
+
|
|
306
|
+
```python
|
|
307
|
+
eda = EDAAnalyzer(df, target_col="price")
|
|
308
|
+
|
|
309
|
+
# Individual analyses
|
|
310
|
+
stats = eda.summary_statistics() # Descriptive stats + skewness + kurtosis
|
|
311
|
+
cat_summary = eda.categorical_summary() # Value counts & percentages per categorical col
|
|
312
|
+
corr_matrix = eda.correlation_analysis(
|
|
313
|
+
method='pearson', # 'pearson', 'kendall', 'spearman'
|
|
314
|
+
threshold=0.7 # Flag pairs above this correlation
|
|
315
|
+
)
|
|
316
|
+
distributions = eda.distribution_analysis() # Mean, median, std, skewness, kurtosis, IQR, range
|
|
317
|
+
target_info = eda.target_analysis() # Stats for continuous, class distribution for categorical
|
|
318
|
+
|
|
319
|
+
# Generate visualizations
|
|
320
|
+
eda.plot_distributions(save_path="dist.png")
|
|
321
|
+
eda.plot_boxplots(save_path="box.png")
|
|
322
|
+
eda.plot_correlation_heatmap(save_path="corr.png")
|
|
323
|
+
eda.plot_categorical(save_path="cat.png")
|
|
324
|
+
eda.plot_target_relationships(save_path="target.png")
|
|
325
|
+
|
|
326
|
+
# Or run everything at once
|
|
327
|
+
results = eda.run_full_analysis(show_plots=True, save_plots=True, output_dir="./plots")
|
|
328
|
+
|
|
329
|
+
# Get auto-generated insights
|
|
330
|
+
insights = eda.get_insights()
|
|
331
|
+
# Examples: "Highly skewed features detected: [col1, col2]"
|
|
332
|
+
# "High cardinality in 'city': 342 unique values"
|
|
333
|
+
# "Class imbalance detected in target: ratio = 0.12"
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
#### FeatureEngineer
|
|
337
|
+
|
|
338
|
+
```python
|
|
339
|
+
engineer = FeatureEngineer(df, target_col="churn", problem_type="classification")
|
|
340
|
+
|
|
341
|
+
# Encoding
|
|
342
|
+
engineer.encode_categorical(
|
|
343
|
+
method='auto', # 'auto', 'onehot', 'label'
|
|
344
|
+
max_categories=10, # OneHot threshold; Label above this
|
|
345
|
+
drop_first=True
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
# Scaling
|
|
349
|
+
engineer.scale_features(method='standard') # 'standard' or 'minmax'
|
|
350
|
+
|
|
351
|
+
# Datetime extraction
|
|
352
|
+
engineer.create_datetime_features(
|
|
353
|
+
features=['year', 'month', 'day', 'dayofweek', 'quarter', 'hour', 'is_weekend']
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
# Polynomial & interaction features
|
|
357
|
+
engineer.create_polynomial_features(degree=2, interaction_only=False)
|
|
358
|
+
|
|
359
|
+
# Binned features
|
|
360
|
+
engineer.create_binned_features(n_bins=5, strategy='quantile') # 'quantile' or 'uniform'
|
|
361
|
+
|
|
362
|
+
# Feature selection
|
|
363
|
+
engineer.compute_feature_importance(n_features=20) # Uses mutual information
|
|
364
|
+
engineer.drop_low_importance_features(threshold=0.01)
|
|
365
|
+
engineer.drop_low_variance_features(threshold=0.01)
|
|
366
|
+
engineer.drop_highly_correlated(threshold=0.95)
|
|
367
|
+
|
|
368
|
+
# Class imbalance handling
|
|
369
|
+
engineer.handle_class_imbalance(
|
|
370
|
+
method='smote', # 'smote', 'oversample', 'undersample'
|
|
371
|
+
sampling_strategy='auto'
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
# Auto-evolve: automatically generates interaction features, evaluates them
|
|
375
|
+
# with mutual information, and keeps only the best ones
|
|
376
|
+
engineer.auto_evolve(max_new_features=10)
|
|
377
|
+
|
|
378
|
+
# Get results
|
|
379
|
+
final_df = engineer.get_transformed_data()
|
|
380
|
+
summary = engineer.get_summary()
|
|
381
|
+
engineer.print_summary()
|
|
382
|
+
```
|
|
383
|
+
|
|
384
|
+
#### ModelTrainer
|
|
385
|
+
|
|
386
|
+
```python
|
|
387
|
+
trainer = ModelTrainer(df, target_col="price", problem_type="regression")
|
|
388
|
+
|
|
389
|
+
# Full pipeline
|
|
390
|
+
results = trainer.run()
|
|
391
|
+
# Internally runs: detect_target → detect_problem_type → prepare_features →
|
|
392
|
+
# train_all_models → get_best_model → get_metrics_dashboard
|
|
393
|
+
|
|
394
|
+
# Or step by step:
|
|
395
|
+
trainer.detect_target() # Auto-detects target using heuristics
|
|
396
|
+
trainer.detect_problem_type() # Classification vs regression by cardinality
|
|
397
|
+
trainer.prepare_features() # Encode, scale, impute, drop non-numeric
|
|
398
|
+
trainer.train_all_models() # CV + RandomizedSearchCV for all models
|
|
399
|
+
best = trainer.get_best_model() # Select winner by F1 (clf) or R² (reg)
|
|
400
|
+
importance = trainer.get_feature_importance(top_n=10)
|
|
401
|
+
reliability = trainer.get_reliability_score() # 0–100 score + grade (A/B/C/D)
|
|
402
|
+
dashboard = trainer.get_metrics_dashboard() # Complete results dict
|
|
403
|
+
|
|
404
|
+
# Export model as .pkl
|
|
405
|
+
trainer.export_model("./model.pkl")
|
|
406
|
+
# Saved: model, scaler, label_encoder, feature_names, metrics, reliability
|
|
407
|
+
|
|
408
|
+
# Models trained:
|
|
409
|
+
# Classification: Logistic Regression, Random Forest (with class_weight='balanced' if imbalanced)
|
|
410
|
+
# Regression: Ridge Regression, Random Forest
|
|
411
|
+
# All with hyperparameter tuning via RandomizedSearchCV and 3-fold Stratified/KFold CV
|
|
412
|
+
```
|
|
413
|
+
|
|
414
|
+
---
|
|
415
|
+
|
|
416
|
+
### 4. Data Drift Detection
|
|
417
|
+
|
|
418
|
+
Compare a baseline dataset against new production data to detect schema and statistical drift.
|
|
419
|
+
|
|
420
|
+
```python
|
|
421
|
+
from cleanflow.drift_detector import DriftDetector
|
|
422
|
+
|
|
423
|
+
detector = DriftDetector(baseline_df=train_df, current_df=production_df)
|
|
424
|
+
report = detector.run()
|
|
425
|
+
|
|
426
|
+
print(report['schema_drift']) # Missing or new columns
|
|
427
|
+
print(report['statistical_drift']) # Per-column KS-test p-values
|
|
428
|
+
print(report['drifted_features']) # List of columns with drift (p < 0.05)
|
|
429
|
+
print(report['drift_detected']) # True/False
|
|
430
|
+
print(report['score']) # Health score 0–100
|
|
431
|
+
```
|
|
432
|
+
|
|
433
|
+
Uses the **Kolmogorov-Smirnov test** on shared numeric columns. Penalizes schema changes and statistical deviations to produce a single health score.
|
|
434
|
+
|
|
435
|
+
---
|
|
436
|
+
|
|
437
|
+
### 5. Synthetic Data Generation
|
|
438
|
+
|
|
439
|
+
Generate privacy-safe statistical clones of your dataset using Kernel Density Estimation (KDE).
|
|
440
|
+
|
|
441
|
+
```python
|
|
442
|
+
from cleanflow.synthetic_generator import SyntheticGenerator
|
|
443
|
+
|
|
444
|
+
gen = SyntheticGenerator(df)
|
|
445
|
+
gen.fit() # Learn distributions
|
|
446
|
+
synthetic_df = gen.generate(n_rows=1000) # Generate 1000 synthetic rows
|
|
447
|
+
|
|
448
|
+
# Numeric columns: sampled from KDE (Gaussian kernel)
|
|
449
|
+
# Categorical columns: sampled from observed probability distribution
|
|
450
|
+
# Constant columns: replicated as-is
|
|
451
|
+
```
|
|
452
|
+
|
|
453
|
+
---
|
|
454
|
+
|
|
455
|
+
### 6. Working with DataFrames Directly
|
|
456
|
+
|
|
457
|
+
No file needed — pass a DataFrame directly.
|
|
458
|
+
|
|
459
|
+
```python
|
|
460
|
+
import pandas as pd
|
|
461
|
+
from cleanflow import DataPipeline
|
|
462
|
+
|
|
463
|
+
df = pd.DataFrame({
|
|
464
|
+
'feature1': [1.0, 2.5, None, 4.0, 100.0],
|
|
465
|
+
'category': ['A', 'B', 'A', None, 'B'],
|
|
466
|
+
'target': [0, 1, 0, 1, 0]
|
|
467
|
+
})
|
|
468
|
+
|
|
469
|
+
pipeline = DataPipeline()
|
|
470
|
+
pipeline.load(df)
|
|
471
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
472
|
+
target_col='target',
|
|
473
|
+
problem_type='classification'
|
|
474
|
+
)
|
|
475
|
+
```
|
|
476
|
+
|
|
477
|
+
---
|
|
478
|
+
|
|
479
|
+
### 7. Flask Web Application
|
|
480
|
+
|
|
481
|
+
An optional browser-based UI with drag-and-drop uploads, user authentication, and an iOS 26-inspired glassmorphism design.
|
|
482
|
+
|
|
483
|
+
```bash
|
|
484
|
+
pip install "cleanflow-kit[web,api]"
|
|
485
|
+
python app.py
|
|
486
|
+
```
|
|
487
|
+
|
|
488
|
+
---
|
|
489
|
+
|
|
490
|
+
## 📋 Complete Feature Reference
|
|
491
|
+
|
|
492
|
+
### Data Loading & Validation
|
|
493
|
+
| Feature | Details |
|
|
494
|
+
|---------|---------|
|
|
495
|
+
| Multi-format support | CSV, Excel (.xlsx/.xls), JSON, Parquet, DataFrame |
|
|
496
|
+
| Smart profiling | Auto-detects dtypes, missing values, duplicates |
|
|
497
|
+
| Noise detection | Flags constant columns, near-constant columns, potential ID columns |
|
|
498
|
+
| Memory reporting | Reports dataset memory usage in MB |
|
|
499
|
+
|
|
500
|
+
### Data Cleaning (CleanFlower)
|
|
501
|
+
| Feature | Details |
|
|
502
|
+
|---------|---------|
|
|
503
|
+
| Duplicate removal | Exact row deduplication |
|
|
504
|
+
| Missing value imputation | mean / median / zero (numeric), mode / "Unknown" (categorical) |
|
|
505
|
+
| Threshold-based column dropping | Drops columns exceeding a configurable missing % (default 40%) |
|
|
506
|
+
| Auto type inference | Converts object → numeric, object → datetime, low-cardinality → category |
|
|
507
|
+
| Outlier handling | IQR or Z-score detection with clip / remove / NaN actions |
|
|
508
|
+
| Categorical normalization | Strip whitespace, lowercase, custom value mapping |
|
|
509
|
+
| Numeric text parsing | Converts "$1,200" → 1200, "1.5k" → 1500, "2M" → 2000000 |
|
|
510
|
+
| Regex feature extraction | Extract structured data from text using capture groups |
|
|
511
|
+
| Column renaming | Batch rename via dictionary mapping |
|
|
512
|
+
| Column dropping | Drop by name, drop constants, drop ID-like columns |
|
|
513
|
+
| Feature scaling | Standard (Z-score), MinMax, Robust scalers |
|
|
514
|
+
| Categorical encoding | One-Hot or Label encoding |
|
|
515
|
+
| Quality scoring | 0–100 data quality score (A/B/C/D/F grade) based on completeness, uniqueness, consistency |
|
|
516
|
+
| Cleaning suggestions | Auto-generated per-column recommendations with issue, action, and reason |
|
|
517
|
+
| Row-level audit trail | Logs every individual cell change (old value → new value, operation, reason) |
|
|
518
|
+
|
|
519
|
+
### Exploratory Data Analysis (EDA)
|
|
520
|
+
| Feature | Details |
|
|
521
|
+
|---------|---------|
|
|
522
|
+
| Summary statistics | Count, mean, std, min/max, quartiles, skewness, kurtosis |
|
|
523
|
+
| Categorical profiling | Value counts and percentages per category |
|
|
524
|
+
| Correlation analysis | Pearson/Kendall/Spearman with high-correlation pair detection |
|
|
525
|
+
| Distribution analysis | Mean, median, std, skewness, kurtosis, IQR, range per feature |
|
|
526
|
+
| Target analysis | Continuous stats or class distribution + imbalance detection |
|
|
527
|
+
| Auto-generated insights | Flags skewed features, high-cardinality columns, class imbalance |
|
|
528
|
+
| Visualizations | Histograms, Boxplots, Correlation Heatmaps, Category Bar Charts, Feature-Target Scatters |
|
|
529
|
+
|
|
530
|
+
### Feature Engineering
|
|
531
|
+
| Feature | Details |
|
|
532
|
+
|---------|---------|
|
|
533
|
+
| Smart encoding | Auto-selects OneHot (≤ threshold) or Label (above threshold) per column |
|
|
534
|
+
| Feature scaling | StandardScaler or MinMaxScaler on numeric features |
|
|
535
|
+
| Datetime extraction | year, month, day, dayofweek, quarter, hour, minute, is_weekend |
|
|
536
|
+
| Polynomial features | Degree-N polynomial and/or interaction-only terms |
|
|
537
|
+
| Binned features | Quantile or uniform binning of numeric columns |
|
|
538
|
+
| Mutual information importance | Ranks features by mutual information with the target |
|
|
539
|
+
| Low-importance dropping | Removes features below an importance threshold |
|
|
540
|
+
| Low-variance dropping | Removes near-zero variance features via VarianceThreshold |
|
|
541
|
+
| High-correlation dropping | Removes one of each pair exceeding a correlation threshold |
|
|
542
|
+
| Class imbalance handling | SMOTE, random oversampling, or random undersampling |
|
|
543
|
+
| Auto-evolve | Automatically generates interaction features, evaluates with MI, keeps the best |
|
|
544
|
+
|
|
545
|
+
### AutoML Model Training
|
|
546
|
+
| Feature | Details |
|
|
547
|
+
|---------|---------|
|
|
548
|
+
| Auto target detection | Heuristic scan for columns named "target", "label", "price", "churn", etc. |
|
|
549
|
+
| Auto problem type detection | Classification (categorical/few unique) vs Regression (continuous) |
|
|
550
|
+
| Classification models | Logistic Regression, Random Forest (with balanced class weights if needed) |
|
|
551
|
+
| Regression models | Ridge Regression, Random Forest |
|
|
552
|
+
| Hyperparameter tuning | RandomizedSearchCV with configurable iteration budget |
|
|
553
|
+
| Cross-validation | 3-fold StratifiedKFold (classification) or KFold (regression) |
|
|
554
|
+
| Multi-metric evaluation | Accuracy, Precision, Recall, F1, ROC-AUC (clf) / R², MAE, RMSE (reg) |
|
|
555
|
+
| Imbalance detection | Flags majority class > 70% and auto-applies SMOTE if available |
|
|
556
|
+
| Reliability score | Custom 0–100 score based on dataset size, CV variance, metric quality (grade A/B/C/D) |
|
|
557
|
+
| Feature importance | Extracts from tree `feature_importances_` or linear `coef_` |
|
|
558
|
+
| Baseline comparison | Trains a naive Decision Tree on raw data to quantify cleaning improvement |
|
|
559
|
+
| Model export | Saves model + scaler + encoder + feature names + metrics as `.pkl` |
|
|
560
|
+
|
|
561
|
+
### Drift Detection
|
|
562
|
+
| Feature | Details |
|
|
563
|
+
|---------|---------|
|
|
564
|
+
| Schema drift | Detects missing or new columns between baseline and current data |
|
|
565
|
+
| Statistical drift | Kolmogorov-Smirnov test on shared numeric columns (p < 0.05 = drift) |
|
|
566
|
+
| Health score | 0–100 composite score penalizing schema and statistical drift |
|
|
567
|
+
|
|
568
|
+
### Synthetic Data Generation
|
|
569
|
+
| Feature | Details |
|
|
570
|
+
|---------|---------|
|
|
571
|
+
| Numeric columns | Kernel Density Estimation (Gaussian) with automatic bandwidth selection |
|
|
572
|
+
| Categorical columns | Probabilistic sampling from observed frequency distributions |
|
|
573
|
+
| Integer preservation | Rounds generated values for originally integer-typed columns |
|
|
574
|
+
|
|
575
|
+
### Report Generation
|
|
576
|
+
| Feature | Details |
|
|
577
|
+
|---------|---------|
|
|
578
|
+
| HTML report | Publication-quality report with executive summary, methodology, data profiling, model architecture, performance evaluation, and future recommendations |
|
|
579
|
+
| Markdown report | Lightweight report with data transformation summary, model performance table, and key observations |
|
|
580
|
+
|
|
581
|
+
---
|
|
582
|
+
|
|
583
|
+
## 🗂️ Package Structure
|
|
584
|
+
|
|
585
|
+
```
|
|
586
|
+
cleanflow/
|
|
587
|
+
├── __init__.py # Package exports (DataPipeline, DataLoader, CleanFlower, etc.)
|
|
588
|
+
├── pipeline.py # DataPipeline orchestrator
|
|
589
|
+
├── data_loader.py # Multi-format data loading & validation
|
|
590
|
+
├── data_cleaner.py # CleanFlower — 30+ cleaning operations
|
|
591
|
+
├── eda.py # EDAAnalyzer — statistics & visualizations
|
|
592
|
+
├── feature_engineer.py # FeatureEngineer — encoding, scaling, selection
|
|
593
|
+
├── model_trainer.py # ModelTrainer — AutoML with CV, tuning, reliability
|
|
594
|
+
├── drift_detector.py # DriftDetector — schema & statistical drift
|
|
595
|
+
├── synthetic_generator.py # SyntheticGenerator — KDE-based data generation
|
|
596
|
+
├── report_generator.py # ReportGenerator — HTML/Markdown reports
|
|
597
|
+
└── _compat.py # String normalization compatibility layer
|
|
598
|
+
```
|
|
599
|
+
|
|
600
|
+
---
|
|
601
|
+
|
|
602
|
+
## 🔌 API Quick Reference
|
|
603
|
+
|
|
604
|
+
```python
|
|
605
|
+
from cleanflow import DataPipeline
|
|
606
|
+
|
|
607
|
+
pipeline = DataPipeline()
|
|
608
|
+
pipeline.load(source) # str path or pd.DataFrame
|
|
609
|
+
pipeline.validate() # → dict
|
|
610
|
+
pipeline.clean(...) # → DataPipeline (chainable)
|
|
611
|
+
pipeline.analyze(...) # → dict
|
|
612
|
+
pipeline.engineer_features(...) # → DataPipeline (chainable)
|
|
613
|
+
pipeline.run_full_pipeline(...) # → (cleaned_df, final_df)
|
|
614
|
+
pipeline.train_model(...) # → dict (metrics dashboard)
|
|
615
|
+
pipeline.generate_html_report(path) # → str (file path)
|
|
616
|
+
pipeline.generate_markdown_report(path) # → str (file path)
|
|
617
|
+
pipeline.save_data(dir, format='csv') # saves cleaned_data + final_data
|
|
618
|
+
pipeline.get_raw_data() # → pd.DataFrame
|
|
619
|
+
pipeline.get_cleaned_data() # → pd.DataFrame
|
|
620
|
+
pipeline.get_final_data() # → pd.DataFrame
|
|
621
|
+
pipeline.get_report() # → dict (full pipeline report)
|
|
622
|
+
```
|