cleanflow-kit 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dataclean/pipeline.py ADDED
@@ -0,0 +1,548 @@
1
+ """
2
+ Data Pipeline Module
3
+ ====================
4
+ Main orchestrator that combines all modules into a unified pipeline.
5
+ """
6
+
7
+ import logging
8
+ logger = logging.getLogger(__name__)
9
+ import pandas as pd
10
+ import numpy as np
11
+ import os
12
+ import io
13
+ import json
14
+ from datetime import datetime
15
+ from typing import Optional, Dict, Any, List, Tuple
16
+ from pathlib import Path
17
+
18
+ from .data_loader import DataLoader
19
+ from .data_cleaner import DataCleaner
20
+ # EDAAnalyzer and FeatureEngineer are imported lazily inside methods
21
+ # to avoid loading matplotlib/seaborn/scipy at module import time
22
+ from .model_trainer import ModelTrainer
23
+ from .report_generator import ReportGenerator
24
+
25
+
26
+ class DataPipeline:
27
+ """
28
+ Complete data pipeline for cleaning, EDA, and feature engineering.
29
+
30
+ This class orchestrates all data processing steps from raw data
31
+ to model-ready features.
32
+
33
+ Example:
34
+ --------
35
+ >>> pipeline = DataPipeline()
36
+ >>> pipeline.load("data.csv")
37
+ >>> pipeline.run_full_pipeline(target_col="price", problem_type="regression")
38
+ >>> cleaned_df = pipeline.get_cleaned_data()
39
+ >>> final_df = pipeline.get_final_data()
40
+ """
41
+
42
+ def __init__(self):
43
+ """Initialize the data pipeline."""
44
+ self.loader: Optional[DataLoader] = None
45
+ self.cleaner: Optional[DataCleaner] = None
46
+ self.eda: Optional['EDAAnalyzer'] = None
47
+ self.engineer: Optional['FeatureEngineer'] = None
48
+ self.trainer: Optional[ModelTrainer] = None
49
+
50
+ self.raw_df: Optional[pd.DataFrame] = None
51
+ self.cleaned_df: Optional[pd.DataFrame] = None
52
+ self.final_df: Optional[pd.DataFrame] = None
53
+
54
+ self.target_col: Optional[str] = None
55
+ self.problem_type: Optional[str] = None
56
+
57
+ self.pipeline_report: Dict[str, Any] = {}
58
+ self.model_results: Optional[Dict[str, Any]] = None
59
+
60
+ def load(
61
+ self,
62
+ source,
63
+ **kwargs
64
+ ) -> 'DataPipeline':
65
+ """
66
+ Load dataset from file or DataFrame.
67
+
68
+ Parameters:
69
+ -----------
70
+ source : str or pd.DataFrame
71
+ File path or DataFrame
72
+ **kwargs : dict
73
+ Additional arguments for pandas read functions
74
+ """
75
+ self.loader = DataLoader()
76
+ self.raw_df = self.loader.load(source, **kwargs)
77
+
78
+ logger.info(f"✅ Loaded dataset: {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
79
+
80
+ return self
81
+
82
+ def validate(self) -> Dict[str, Any]:
83
+ """
84
+ Validate the loaded dataset.
85
+
86
+ Returns:
87
+ --------
88
+ dict : Validation report
89
+ """
90
+ if self.loader is None:
91
+ raise ValueError("No data loaded. Call load() first.")
92
+
93
+ report = self.loader.validate()
94
+ self.loader.print_summary()
95
+ self.pipeline_report['validation'] = report
96
+
97
+ return report
98
+
99
+ def clean(
100
+ self,
101
+ remove_duplicates: bool = True,
102
+ handle_missing: bool = True,
103
+ missing_numeric_strategy: str = 'median',
104
+ missing_categorical_strategy: str = 'mode',
105
+ missing_drop_threshold: float = 0.4,
106
+ fix_types: bool = True,
107
+ handle_outliers: bool = True,
108
+ outlier_method: str = 'iqr',
109
+ outlier_action: str = 'clip',
110
+ clean_categorical: bool = True,
111
+ drop_constant: bool = True
112
+ ) -> 'DataPipeline':
113
+ """
114
+ Clean the dataset.
115
+
116
+ Parameters:
117
+ -----------
118
+ (See DataCleaner for parameter descriptions)
119
+ """
120
+ if self.raw_df is None:
121
+ raise ValueError("No data loaded. Call load() first.")
122
+
123
+ self.cleaner = DataCleaner(self.raw_df)
124
+
125
+ if remove_duplicates:
126
+ self.cleaner.remove_duplicates()
127
+
128
+ if handle_missing:
129
+ self.cleaner.handle_missing_values(
130
+ numeric_strategy=missing_numeric_strategy,
131
+ categorical_strategy=missing_categorical_strategy,
132
+ drop_threshold=missing_drop_threshold
133
+ )
134
+
135
+ if fix_types:
136
+ self.cleaner.fix_data_types()
137
+
138
+ if handle_outliers:
139
+ self.cleaner.handle_outliers(
140
+ method=outlier_method,
141
+ action=outlier_action
142
+ )
143
+
144
+ if clean_categorical:
145
+ self.cleaner.clean_categorical_values()
146
+
147
+ if drop_constant:
148
+ self.cleaner.drop_columns(drop_constant=True)
149
+
150
+ self.cleaned_df = self.cleaner.get_cleaned_data()
151
+ self.cleaner.print_summary()
152
+ self.pipeline_report['cleaning'] = self.cleaner.get_cleaning_summary()
153
+
154
+ return self
155
+
156
+ def analyze(
157
+ self,
158
+ target_col: Optional[str] = None,
159
+ show_plots: bool = True,
160
+ save_plots: bool = False,
161
+ output_dir: Optional[str] = None
162
+ ) -> Dict[str, Any]:
163
+ """
164
+ Perform exploratory data analysis.
165
+
166
+ Parameters:
167
+ -----------
168
+ target_col : str, optional
169
+ Target column for analysis
170
+ show_plots : bool
171
+ Whether to display plots
172
+ save_plots : bool
173
+ Whether to save plots to files
174
+ output_dir : str, optional
175
+ Directory to save plots
176
+ """
177
+ df = self.cleaned_df if self.cleaned_df is not None else self.raw_df
178
+
179
+ if df is None:
180
+ raise ValueError("No data available. Call load() first.")
181
+
182
+ if target_col:
183
+ self.target_col = target_col
184
+
185
+ from .eda import EDAAnalyzer
186
+ self.eda = EDAAnalyzer(df, target_col=self.target_col)
187
+ results = self.eda.run_full_analysis(
188
+ show_plots=show_plots,
189
+ save_plots=save_plots,
190
+ output_dir=output_dir
191
+ )
192
+
193
+ self.pipeline_report['eda'] = results
194
+
195
+ return results
196
+
197
+ def engineer_features(
198
+ self,
199
+ target_col: Optional[str] = None,
200
+ problem_type: Optional[str] = None,
201
+ encode_categorical: bool = True,
202
+ scale_features: bool = True,
203
+ scale_method: str = 'standard',
204
+ create_datetime_features: bool = True,
205
+ create_polynomial_features: bool = False,
206
+ polynomial_degree: int = 2,
207
+ drop_low_variance: bool = True,
208
+ drop_high_correlation: bool = True,
209
+ correlation_threshold: float = 0.95,
210
+ handle_imbalance: bool = False,
211
+ imbalance_method: str = 'smote',
212
+ auto_evolve_features: bool = False
213
+ ) -> 'DataPipeline':
214
+ """
215
+ Perform feature engineering.
216
+
217
+ Parameters:
218
+ -----------
219
+ (See FeatureEngineer for parameter descriptions)
220
+ """
221
+ df = self.cleaned_df if self.cleaned_df is not None else self.raw_df
222
+
223
+ if df is None:
224
+ raise ValueError("No data available. Call load() first.")
225
+
226
+ if target_col:
227
+ self.target_col = target_col
228
+ if problem_type:
229
+ self.problem_type = problem_type
230
+
231
+ from .feature_engineer import FeatureEngineer
232
+ self.engineer = FeatureEngineer(
233
+ df,
234
+ target_col=self.target_col,
235
+ problem_type=self.problem_type
236
+ )
237
+
238
+ if create_datetime_features:
239
+ self.engineer.create_datetime_features()
240
+
241
+ if encode_categorical:
242
+ self.engineer.encode_categorical()
243
+
244
+ if create_polynomial_features:
245
+ self.engineer.create_polynomial_features(degree=polynomial_degree)
246
+
247
+ if auto_evolve_features:
248
+ self.engineer.auto_evolve()
249
+
250
+ if drop_low_variance:
251
+ self.engineer.drop_low_variance_features()
252
+
253
+ if drop_high_correlation:
254
+ self.engineer.drop_highly_correlated(threshold=correlation_threshold)
255
+
256
+ if scale_features:
257
+ self.engineer.scale_features(method=scale_method)
258
+
259
+ if handle_imbalance and self.problem_type == 'classification':
260
+ self.engineer.handle_class_imbalance(method=imbalance_method)
261
+
262
+ # Compute feature importance if target specified
263
+ if self.target_col:
264
+ self.engineer.compute_feature_importance()
265
+
266
+ self.final_df = self.engineer.get_transformed_data()
267
+ self.engineer.print_summary()
268
+ self.pipeline_report['feature_engineering'] = self.engineer.get_summary()
269
+
270
+ return self
271
+
272
+ def run_full_pipeline(
273
+ self,
274
+ target_col: Optional[str] = None,
275
+ problem_type: Optional[str] = None,
276
+ show_eda_plots: bool = True,
277
+ cleaning_config: Optional[Dict[str, Any]] = None,
278
+ feature_config: Optional[Dict[str, Any]] = None
279
+ ) -> Tuple[pd.DataFrame, pd.DataFrame]:
280
+ """
281
+ Run the complete pipeline from validation to feature engineering.
282
+
283
+ Parameters:
284
+ -----------
285
+ target_col : str, optional
286
+ Target column name
287
+ problem_type : str, optional
288
+ 'classification', 'regression', or 'clustering'
289
+ show_eda_plots : bool
290
+ Whether to show EDA visualizations
291
+ cleaning_config : dict, optional
292
+ Override cleaning parameters
293
+ feature_config : dict, optional
294
+ Override feature engineering parameters
295
+
296
+ Returns:
297
+ --------
298
+ tuple : (cleaned_df, final_df)
299
+ """
300
+ self.target_col = target_col
301
+ self.problem_type = problem_type
302
+
303
+ print("\n" + "=" * 70)
304
+ logger.info("🚀 STARTING DATA PIPELINE")
305
+ print("=" * 70)
306
+
307
+ # Step 1: Validation
308
+ logger.info("\n📋 STEP 1: DATA VALIDATION")
309
+ print("-" * 40)
310
+ self.validate()
311
+
312
+ # Step 2: Cleaning
313
+ logger.info("\n🧹 STEP 2: DATA CLEANING")
314
+ print("-" * 40)
315
+ clean_params = cleaning_config or {}
316
+ self.clean(**clean_params)
317
+
318
+ # Step 3: EDA
319
+ logger.info("\n📊 STEP 3: EXPLORATORY DATA ANALYSIS")
320
+ print("-" * 40)
321
+ self.analyze(target_col=target_col, show_plots=show_eda_plots)
322
+
323
+ # Step 4: Feature Engineering
324
+ logger.info("\n⚙️ STEP 4: FEATURE ENGINEERING")
325
+ print("-" * 40)
326
+ feature_params = feature_config or {}
327
+ feature_params['target_col'] = target_col
328
+ feature_params['problem_type'] = problem_type
329
+ self.engineer_features(**feature_params)
330
+
331
+ # Summary
332
+ self._print_pipeline_summary()
333
+
334
+ return self.cleaned_df, self.final_df
335
+
336
+ def _print_pipeline_summary(self) -> None:
337
+ """Print final pipeline summary."""
338
+ print("\n" + "=" * 70)
339
+ logger.info("✅ PIPELINE COMPLETE")
340
+ print("=" * 70)
341
+
342
+ if self.raw_df is not None:
343
+ logger.info("\n📊 Data Transformation:")
344
+ logger.info(f" Raw: {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
345
+
346
+ if self.cleaned_df is not None:
347
+ logger.info(f" Cleaned: {self.cleaned_df.shape[0]:,} rows × {self.cleaned_df.shape[1]} columns")
348
+
349
+ if self.final_df is not None:
350
+ logger.info(f" Final: {self.final_df.shape[0]:,} rows × {self.final_df.shape[1]} columns")
351
+
352
+ logger.info(f"\n🎯 Target: {self.target_col or 'Not specified'}")
353
+ logger.info(f"📈 Problem Type: {self.problem_type or 'Not specified'}")
354
+
355
+ # Data quality check
356
+ if self.final_df is not None:
357
+ missing = self.final_df.isnull().sum().sum()
358
+ logger.info("\n✅ Final Data Quality:")
359
+ logger.info(f" • Missing values: {missing}")
360
+ logger.info(f" • Ready for modeling: {'Yes' if missing == 0 else 'No (handle remaining missing)'}")
361
+
362
+ print("\n" + "=" * 70)
363
+
364
+ def get_raw_data(self) -> Optional[pd.DataFrame]:
365
+ """Return the raw DataFrame."""
366
+ return self.raw_df
367
+
368
+ def get_cleaned_data(self) -> Optional[pd.DataFrame]:
369
+ """Return the cleaned DataFrame."""
370
+ return self.cleaned_df
371
+
372
+ def get_final_data(self) -> Optional[pd.DataFrame]:
373
+ """Return the feature-engineered DataFrame (model-ready)."""
374
+ return self.final_df
375
+
376
+ def get_report(self) -> Dict[str, Any]:
377
+ """Return the complete pipeline report."""
378
+ return self.pipeline_report
379
+
380
+ def save_data(
381
+ self,
382
+ output_dir: str,
383
+ save_cleaned: bool = True,
384
+ save_final: bool = True,
385
+ format: str = 'csv'
386
+ ) -> None:
387
+ """
388
+ Save processed datasets to files.
389
+
390
+ Parameters:
391
+ -----------
392
+ output_dir : str
393
+ Output directory path
394
+ save_cleaned : bool
395
+ Save cleaned data
396
+ save_final : bool
397
+ Save final engineered data
398
+ format : str
399
+ Output format: 'csv' or 'parquet'
400
+ """
401
+ output_path = Path(output_dir)
402
+ output_path.mkdir(parents=True, exist_ok=True)
403
+
404
+ if save_cleaned and self.cleaned_df is not None:
405
+ if format == 'csv':
406
+ self.cleaned_df.to_csv(output_path / 'cleaned_data.csv', index=False)
407
+ else:
408
+ self.cleaned_df.to_parquet(output_path / 'cleaned_data.parquet', index=False)
409
+ logger.info(f"✅ Saved cleaned data to {output_path / f'cleaned_data.{format}'}")
410
+
411
+ if save_final and self.final_df is not None:
412
+ if format == 'csv':
413
+ self.final_df.to_csv(output_path / 'final_data.csv', index=False)
414
+ else:
415
+ self.final_df.to_parquet(output_path / 'final_data.parquet', index=False)
416
+ logger.info(f"✅ Saved final data to {output_path / f'final_data.{format}'}")
417
+
418
+ def train_model(
419
+ self,
420
+ target_col: Optional[str] = None,
421
+ problem_type: Optional[str] = None,
422
+ export_path: Optional[str] = None
423
+ ) -> Dict[str, Any]:
424
+ """
425
+ Train ML models on the processed data.
426
+
427
+ Parameters:
428
+ -----------
429
+ target_col : str, optional
430
+ Target column name (auto-detected if None)
431
+ problem_type : str, optional
432
+ 'classification' or 'regression' (auto-detected if None)
433
+ export_path : str, optional
434
+ Path to export the best model (.pkl)
435
+
436
+ Returns:
437
+ --------
438
+ dict : Training results dashboard
439
+ """
440
+ df = self.final_df if self.final_df is not None else self.cleaned_df
441
+ if df is None:
442
+ raise ValueError("No data available. Run clean() or load() first.")
443
+
444
+ t_col = target_col or self.target_col
445
+ p_type = problem_type or self.problem_type
446
+
447
+ self.trainer = ModelTrainer(df, target_col=t_col, problem_type=p_type, raw_df=self.raw_df)
448
+ self.model_results = self.trainer.run()
449
+
450
+ if export_path:
451
+ self.trainer.export_model(export_path)
452
+
453
+ self.pipeline_report['model_training'] = self.model_results
454
+ return self.model_results
455
+
456
+ def generate_html_report(self, output_path: str) -> str:
457
+ """
458
+ Generate a detailed HTML report using ReportGenerator.
459
+ """
460
+ generator = ReportGenerator(self, self.model_results)
461
+ html_content = generator.generate_html()
462
+
463
+ with open(output_path, 'w', encoding='utf-8') as f:
464
+ f.write(html_content)
465
+
466
+ logger.info(f"✅ Generated detailed HTML report: {output_path}")
467
+ return output_path
468
+
469
+ def generate_markdown_report(self, output_path: str) -> str:
470
+ """
471
+ Generate a comprehensive Markdown report of the pipeline run.
472
+ """
473
+ report = []
474
+ report.append(f"# Mini Data Clean Tool - Model Report")
475
+ report.append(f"**Generated:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n")
476
+
477
+ # 1. Dataset Overview
478
+ report.append("## 1. Data Processing Journey")
479
+ if self.raw_df is not None:
480
+ report.append(f"- **Raw Dataset:** {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
481
+ if self.cleaned_df is not None:
482
+ report.append(f"- **Cleaned Dataset:** {self.cleaned_df.shape[0]:,} rows × {self.cleaned_df.shape[1]} columns")
483
+
484
+ # Pipeline Steps
485
+ if 'preprocessing' in self.pipeline_report:
486
+ report.append("\n### Cleaning Steps Executed:")
487
+ steps = self.pipeline_report['preprocessing'].get('steps_executed', [])
488
+ if steps:
489
+ for step in steps:
490
+ report.append(f"- {step}")
491
+ else:
492
+ report.append("- No major cleaning issues found.")
493
+
494
+ # 2. Model Performance
495
+ report.append("\n## 2. Model Performance Analysis")
496
+
497
+ if self.model_results and 'comparison' in self.model_results:
498
+ comp = self.model_results['comparison']
499
+ metric_name = comp.get('metric', 'Metric')
500
+
501
+ report.append("| Model | Score | Reliability |")
502
+ report.append("| :--- | :--- | :--- |")
503
+
504
+ # Raw Row
505
+ raw_score = comp.get('raw_score', 'N/A')
506
+ raw_rel = comp.get('raw_reliability', {})
507
+ raw_grade = raw_rel.get('grade', 'N/A')
508
+ raw_pts = raw_rel.get('score', 0)
509
+ report.append(f"| **Raw Baseline** | {raw_score}% ({metric_name}) | **{raw_grade}** ({raw_pts}/100) |")
510
+
511
+ # Cleaned Row
512
+ clean_score = comp.get('cleaned_score', 'N/A')
513
+ # Get reliability from best model metadata or calculate
514
+ # For now, let's grab it from export metadata if available, or approximate
515
+ # Actually, model_trainer export includes it.
516
+ # We can grab it from best_model dict if available
517
+ clean_grade = 'N/A'
518
+ clean_pts = 0
519
+
520
+ # Try to dig reliability out of best model
521
+ # This is a bit tricky as it's not directly in comparison dict usually
522
+ # But we can look at reliability score if we re-calculated it or stored it
523
+ # The backend calc logic is in ModelTrainer.
524
+ # For the report sake, let's look at the 'model_training' specific section
525
+ best_model_info = self.model_results.get('best_model', {})
526
+ # We don't have reliability directly here unless we added it to the dict in ModelTrainer.run
527
+ # Let's check ModelTrainer.run output structure.
528
+ # It returns 'best_model': {... 'metrics': ...}
529
+ # The 'reliability' key is in the *export*, not immediately in run return?
530
+ # Wait, line 832 in ModelTrainer adds 'reliability' to export.
531
+ # We should probably expose it in the main result dict too for easy access.
532
+
533
+ report.append(f"| **Cleaned Model** | **{clean_score}%** ({metric_name}) | *(See Dashboard)* |")
534
+
535
+ report.append(f"\n**Improvement:** {comp.get('improvement_pct', 0)}% improvement over baseline.")
536
+
537
+ # 3. Key Observations
538
+ report.append("\n## 3. Key Observations")
539
+ report.append("> This model report was generated automatically by the Mini Data Clean Tool.")
540
+ report.append(f"- **Target Variable:** `{self.target_col}`")
541
+ report.append(f"- **Problem Type:** `{self.problem_type}`")
542
+
543
+ # Save
544
+ with open(output_path, 'w', encoding='utf-8') as f:
545
+ f.write('\n'.join(report))
546
+
547
+ logger.info(f"✅ Generated model report: {output_path}")
548
+ return output_path
dataclean/py.typed ADDED
File without changes