cleanflow-kit 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.1.0.dist-info/METADATA +119 -0
- cleanflow_kit-1.1.0.dist-info/RECORD +17 -0
- cleanflow_kit-1.1.0.dist-info/WHEEL +5 -0
- cleanflow_kit-1.1.0.dist-info/licenses/LICENSE +21 -0
- cleanflow_kit-1.1.0.dist-info/top_level.txt +1 -0
- dataclean/__init__.py +44 -0
- dataclean/_compat.py +24 -0
- dataclean/data_cleaner.py +988 -0
- dataclean/data_loader.py +194 -0
- dataclean/drift_detector.py +111 -0
- dataclean/eda.py +400 -0
- dataclean/feature_engineer.py +608 -0
- dataclean/model_trainer.py +874 -0
- dataclean/pipeline.py +548 -0
- dataclean/py.typed +0 -0
- dataclean/report_generator.py +365 -0
- dataclean/synthetic_generator.py +97 -0
dataclean/pipeline.py
ADDED
|
@@ -0,0 +1,548 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Pipeline Module
|
|
3
|
+
====================
|
|
4
|
+
Main orchestrator that combines all modules into a unified pipeline.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import numpy as np
|
|
11
|
+
import os
|
|
12
|
+
import io
|
|
13
|
+
import json
|
|
14
|
+
from datetime import datetime
|
|
15
|
+
from typing import Optional, Dict, Any, List, Tuple
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from .data_loader import DataLoader
|
|
19
|
+
from .data_cleaner import DataCleaner
|
|
20
|
+
# EDAAnalyzer and FeatureEngineer are imported lazily inside methods
|
|
21
|
+
# to avoid loading matplotlib/seaborn/scipy at module import time
|
|
22
|
+
from .model_trainer import ModelTrainer
|
|
23
|
+
from .report_generator import ReportGenerator
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class DataPipeline:
|
|
27
|
+
"""
|
|
28
|
+
Complete data pipeline for cleaning, EDA, and feature engineering.
|
|
29
|
+
|
|
30
|
+
This class orchestrates all data processing steps from raw data
|
|
31
|
+
to model-ready features.
|
|
32
|
+
|
|
33
|
+
Example:
|
|
34
|
+
--------
|
|
35
|
+
>>> pipeline = DataPipeline()
|
|
36
|
+
>>> pipeline.load("data.csv")
|
|
37
|
+
>>> pipeline.run_full_pipeline(target_col="price", problem_type="regression")
|
|
38
|
+
>>> cleaned_df = pipeline.get_cleaned_data()
|
|
39
|
+
>>> final_df = pipeline.get_final_data()
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
"""Initialize the data pipeline."""
|
|
44
|
+
self.loader: Optional[DataLoader] = None
|
|
45
|
+
self.cleaner: Optional[DataCleaner] = None
|
|
46
|
+
self.eda: Optional['EDAAnalyzer'] = None
|
|
47
|
+
self.engineer: Optional['FeatureEngineer'] = None
|
|
48
|
+
self.trainer: Optional[ModelTrainer] = None
|
|
49
|
+
|
|
50
|
+
self.raw_df: Optional[pd.DataFrame] = None
|
|
51
|
+
self.cleaned_df: Optional[pd.DataFrame] = None
|
|
52
|
+
self.final_df: Optional[pd.DataFrame] = None
|
|
53
|
+
|
|
54
|
+
self.target_col: Optional[str] = None
|
|
55
|
+
self.problem_type: Optional[str] = None
|
|
56
|
+
|
|
57
|
+
self.pipeline_report: Dict[str, Any] = {}
|
|
58
|
+
self.model_results: Optional[Dict[str, Any]] = None
|
|
59
|
+
|
|
60
|
+
def load(
|
|
61
|
+
self,
|
|
62
|
+
source,
|
|
63
|
+
**kwargs
|
|
64
|
+
) -> 'DataPipeline':
|
|
65
|
+
"""
|
|
66
|
+
Load dataset from file or DataFrame.
|
|
67
|
+
|
|
68
|
+
Parameters:
|
|
69
|
+
-----------
|
|
70
|
+
source : str or pd.DataFrame
|
|
71
|
+
File path or DataFrame
|
|
72
|
+
**kwargs : dict
|
|
73
|
+
Additional arguments for pandas read functions
|
|
74
|
+
"""
|
|
75
|
+
self.loader = DataLoader()
|
|
76
|
+
self.raw_df = self.loader.load(source, **kwargs)
|
|
77
|
+
|
|
78
|
+
logger.info(f"✅ Loaded dataset: {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
|
|
79
|
+
|
|
80
|
+
return self
|
|
81
|
+
|
|
82
|
+
def validate(self) -> Dict[str, Any]:
|
|
83
|
+
"""
|
|
84
|
+
Validate the loaded dataset.
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
--------
|
|
88
|
+
dict : Validation report
|
|
89
|
+
"""
|
|
90
|
+
if self.loader is None:
|
|
91
|
+
raise ValueError("No data loaded. Call load() first.")
|
|
92
|
+
|
|
93
|
+
report = self.loader.validate()
|
|
94
|
+
self.loader.print_summary()
|
|
95
|
+
self.pipeline_report['validation'] = report
|
|
96
|
+
|
|
97
|
+
return report
|
|
98
|
+
|
|
99
|
+
def clean(
|
|
100
|
+
self,
|
|
101
|
+
remove_duplicates: bool = True,
|
|
102
|
+
handle_missing: bool = True,
|
|
103
|
+
missing_numeric_strategy: str = 'median',
|
|
104
|
+
missing_categorical_strategy: str = 'mode',
|
|
105
|
+
missing_drop_threshold: float = 0.4,
|
|
106
|
+
fix_types: bool = True,
|
|
107
|
+
handle_outliers: bool = True,
|
|
108
|
+
outlier_method: str = 'iqr',
|
|
109
|
+
outlier_action: str = 'clip',
|
|
110
|
+
clean_categorical: bool = True,
|
|
111
|
+
drop_constant: bool = True
|
|
112
|
+
) -> 'DataPipeline':
|
|
113
|
+
"""
|
|
114
|
+
Clean the dataset.
|
|
115
|
+
|
|
116
|
+
Parameters:
|
|
117
|
+
-----------
|
|
118
|
+
(See DataCleaner for parameter descriptions)
|
|
119
|
+
"""
|
|
120
|
+
if self.raw_df is None:
|
|
121
|
+
raise ValueError("No data loaded. Call load() first.")
|
|
122
|
+
|
|
123
|
+
self.cleaner = DataCleaner(self.raw_df)
|
|
124
|
+
|
|
125
|
+
if remove_duplicates:
|
|
126
|
+
self.cleaner.remove_duplicates()
|
|
127
|
+
|
|
128
|
+
if handle_missing:
|
|
129
|
+
self.cleaner.handle_missing_values(
|
|
130
|
+
numeric_strategy=missing_numeric_strategy,
|
|
131
|
+
categorical_strategy=missing_categorical_strategy,
|
|
132
|
+
drop_threshold=missing_drop_threshold
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
if fix_types:
|
|
136
|
+
self.cleaner.fix_data_types()
|
|
137
|
+
|
|
138
|
+
if handle_outliers:
|
|
139
|
+
self.cleaner.handle_outliers(
|
|
140
|
+
method=outlier_method,
|
|
141
|
+
action=outlier_action
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
if clean_categorical:
|
|
145
|
+
self.cleaner.clean_categorical_values()
|
|
146
|
+
|
|
147
|
+
if drop_constant:
|
|
148
|
+
self.cleaner.drop_columns(drop_constant=True)
|
|
149
|
+
|
|
150
|
+
self.cleaned_df = self.cleaner.get_cleaned_data()
|
|
151
|
+
self.cleaner.print_summary()
|
|
152
|
+
self.pipeline_report['cleaning'] = self.cleaner.get_cleaning_summary()
|
|
153
|
+
|
|
154
|
+
return self
|
|
155
|
+
|
|
156
|
+
def analyze(
|
|
157
|
+
self,
|
|
158
|
+
target_col: Optional[str] = None,
|
|
159
|
+
show_plots: bool = True,
|
|
160
|
+
save_plots: bool = False,
|
|
161
|
+
output_dir: Optional[str] = None
|
|
162
|
+
) -> Dict[str, Any]:
|
|
163
|
+
"""
|
|
164
|
+
Perform exploratory data analysis.
|
|
165
|
+
|
|
166
|
+
Parameters:
|
|
167
|
+
-----------
|
|
168
|
+
target_col : str, optional
|
|
169
|
+
Target column for analysis
|
|
170
|
+
show_plots : bool
|
|
171
|
+
Whether to display plots
|
|
172
|
+
save_plots : bool
|
|
173
|
+
Whether to save plots to files
|
|
174
|
+
output_dir : str, optional
|
|
175
|
+
Directory to save plots
|
|
176
|
+
"""
|
|
177
|
+
df = self.cleaned_df if self.cleaned_df is not None else self.raw_df
|
|
178
|
+
|
|
179
|
+
if df is None:
|
|
180
|
+
raise ValueError("No data available. Call load() first.")
|
|
181
|
+
|
|
182
|
+
if target_col:
|
|
183
|
+
self.target_col = target_col
|
|
184
|
+
|
|
185
|
+
from .eda import EDAAnalyzer
|
|
186
|
+
self.eda = EDAAnalyzer(df, target_col=self.target_col)
|
|
187
|
+
results = self.eda.run_full_analysis(
|
|
188
|
+
show_plots=show_plots,
|
|
189
|
+
save_plots=save_plots,
|
|
190
|
+
output_dir=output_dir
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
self.pipeline_report['eda'] = results
|
|
194
|
+
|
|
195
|
+
return results
|
|
196
|
+
|
|
197
|
+
def engineer_features(
|
|
198
|
+
self,
|
|
199
|
+
target_col: Optional[str] = None,
|
|
200
|
+
problem_type: Optional[str] = None,
|
|
201
|
+
encode_categorical: bool = True,
|
|
202
|
+
scale_features: bool = True,
|
|
203
|
+
scale_method: str = 'standard',
|
|
204
|
+
create_datetime_features: bool = True,
|
|
205
|
+
create_polynomial_features: bool = False,
|
|
206
|
+
polynomial_degree: int = 2,
|
|
207
|
+
drop_low_variance: bool = True,
|
|
208
|
+
drop_high_correlation: bool = True,
|
|
209
|
+
correlation_threshold: float = 0.95,
|
|
210
|
+
handle_imbalance: bool = False,
|
|
211
|
+
imbalance_method: str = 'smote',
|
|
212
|
+
auto_evolve_features: bool = False
|
|
213
|
+
) -> 'DataPipeline':
|
|
214
|
+
"""
|
|
215
|
+
Perform feature engineering.
|
|
216
|
+
|
|
217
|
+
Parameters:
|
|
218
|
+
-----------
|
|
219
|
+
(See FeatureEngineer for parameter descriptions)
|
|
220
|
+
"""
|
|
221
|
+
df = self.cleaned_df if self.cleaned_df is not None else self.raw_df
|
|
222
|
+
|
|
223
|
+
if df is None:
|
|
224
|
+
raise ValueError("No data available. Call load() first.")
|
|
225
|
+
|
|
226
|
+
if target_col:
|
|
227
|
+
self.target_col = target_col
|
|
228
|
+
if problem_type:
|
|
229
|
+
self.problem_type = problem_type
|
|
230
|
+
|
|
231
|
+
from .feature_engineer import FeatureEngineer
|
|
232
|
+
self.engineer = FeatureEngineer(
|
|
233
|
+
df,
|
|
234
|
+
target_col=self.target_col,
|
|
235
|
+
problem_type=self.problem_type
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
if create_datetime_features:
|
|
239
|
+
self.engineer.create_datetime_features()
|
|
240
|
+
|
|
241
|
+
if encode_categorical:
|
|
242
|
+
self.engineer.encode_categorical()
|
|
243
|
+
|
|
244
|
+
if create_polynomial_features:
|
|
245
|
+
self.engineer.create_polynomial_features(degree=polynomial_degree)
|
|
246
|
+
|
|
247
|
+
if auto_evolve_features:
|
|
248
|
+
self.engineer.auto_evolve()
|
|
249
|
+
|
|
250
|
+
if drop_low_variance:
|
|
251
|
+
self.engineer.drop_low_variance_features()
|
|
252
|
+
|
|
253
|
+
if drop_high_correlation:
|
|
254
|
+
self.engineer.drop_highly_correlated(threshold=correlation_threshold)
|
|
255
|
+
|
|
256
|
+
if scale_features:
|
|
257
|
+
self.engineer.scale_features(method=scale_method)
|
|
258
|
+
|
|
259
|
+
if handle_imbalance and self.problem_type == 'classification':
|
|
260
|
+
self.engineer.handle_class_imbalance(method=imbalance_method)
|
|
261
|
+
|
|
262
|
+
# Compute feature importance if target specified
|
|
263
|
+
if self.target_col:
|
|
264
|
+
self.engineer.compute_feature_importance()
|
|
265
|
+
|
|
266
|
+
self.final_df = self.engineer.get_transformed_data()
|
|
267
|
+
self.engineer.print_summary()
|
|
268
|
+
self.pipeline_report['feature_engineering'] = self.engineer.get_summary()
|
|
269
|
+
|
|
270
|
+
return self
|
|
271
|
+
|
|
272
|
+
def run_full_pipeline(
|
|
273
|
+
self,
|
|
274
|
+
target_col: Optional[str] = None,
|
|
275
|
+
problem_type: Optional[str] = None,
|
|
276
|
+
show_eda_plots: bool = True,
|
|
277
|
+
cleaning_config: Optional[Dict[str, Any]] = None,
|
|
278
|
+
feature_config: Optional[Dict[str, Any]] = None
|
|
279
|
+
) -> Tuple[pd.DataFrame, pd.DataFrame]:
|
|
280
|
+
"""
|
|
281
|
+
Run the complete pipeline from validation to feature engineering.
|
|
282
|
+
|
|
283
|
+
Parameters:
|
|
284
|
+
-----------
|
|
285
|
+
target_col : str, optional
|
|
286
|
+
Target column name
|
|
287
|
+
problem_type : str, optional
|
|
288
|
+
'classification', 'regression', or 'clustering'
|
|
289
|
+
show_eda_plots : bool
|
|
290
|
+
Whether to show EDA visualizations
|
|
291
|
+
cleaning_config : dict, optional
|
|
292
|
+
Override cleaning parameters
|
|
293
|
+
feature_config : dict, optional
|
|
294
|
+
Override feature engineering parameters
|
|
295
|
+
|
|
296
|
+
Returns:
|
|
297
|
+
--------
|
|
298
|
+
tuple : (cleaned_df, final_df)
|
|
299
|
+
"""
|
|
300
|
+
self.target_col = target_col
|
|
301
|
+
self.problem_type = problem_type
|
|
302
|
+
|
|
303
|
+
print("\n" + "=" * 70)
|
|
304
|
+
logger.info("🚀 STARTING DATA PIPELINE")
|
|
305
|
+
print("=" * 70)
|
|
306
|
+
|
|
307
|
+
# Step 1: Validation
|
|
308
|
+
logger.info("\n📋 STEP 1: DATA VALIDATION")
|
|
309
|
+
print("-" * 40)
|
|
310
|
+
self.validate()
|
|
311
|
+
|
|
312
|
+
# Step 2: Cleaning
|
|
313
|
+
logger.info("\n🧹 STEP 2: DATA CLEANING")
|
|
314
|
+
print("-" * 40)
|
|
315
|
+
clean_params = cleaning_config or {}
|
|
316
|
+
self.clean(**clean_params)
|
|
317
|
+
|
|
318
|
+
# Step 3: EDA
|
|
319
|
+
logger.info("\n📊 STEP 3: EXPLORATORY DATA ANALYSIS")
|
|
320
|
+
print("-" * 40)
|
|
321
|
+
self.analyze(target_col=target_col, show_plots=show_eda_plots)
|
|
322
|
+
|
|
323
|
+
# Step 4: Feature Engineering
|
|
324
|
+
logger.info("\n⚙️ STEP 4: FEATURE ENGINEERING")
|
|
325
|
+
print("-" * 40)
|
|
326
|
+
feature_params = feature_config or {}
|
|
327
|
+
feature_params['target_col'] = target_col
|
|
328
|
+
feature_params['problem_type'] = problem_type
|
|
329
|
+
self.engineer_features(**feature_params)
|
|
330
|
+
|
|
331
|
+
# Summary
|
|
332
|
+
self._print_pipeline_summary()
|
|
333
|
+
|
|
334
|
+
return self.cleaned_df, self.final_df
|
|
335
|
+
|
|
336
|
+
def _print_pipeline_summary(self) -> None:
|
|
337
|
+
"""Print final pipeline summary."""
|
|
338
|
+
print("\n" + "=" * 70)
|
|
339
|
+
logger.info("✅ PIPELINE COMPLETE")
|
|
340
|
+
print("=" * 70)
|
|
341
|
+
|
|
342
|
+
if self.raw_df is not None:
|
|
343
|
+
logger.info("\n📊 Data Transformation:")
|
|
344
|
+
logger.info(f" Raw: {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
|
|
345
|
+
|
|
346
|
+
if self.cleaned_df is not None:
|
|
347
|
+
logger.info(f" Cleaned: {self.cleaned_df.shape[0]:,} rows × {self.cleaned_df.shape[1]} columns")
|
|
348
|
+
|
|
349
|
+
if self.final_df is not None:
|
|
350
|
+
logger.info(f" Final: {self.final_df.shape[0]:,} rows × {self.final_df.shape[1]} columns")
|
|
351
|
+
|
|
352
|
+
logger.info(f"\n🎯 Target: {self.target_col or 'Not specified'}")
|
|
353
|
+
logger.info(f"📈 Problem Type: {self.problem_type or 'Not specified'}")
|
|
354
|
+
|
|
355
|
+
# Data quality check
|
|
356
|
+
if self.final_df is not None:
|
|
357
|
+
missing = self.final_df.isnull().sum().sum()
|
|
358
|
+
logger.info("\n✅ Final Data Quality:")
|
|
359
|
+
logger.info(f" • Missing values: {missing}")
|
|
360
|
+
logger.info(f" • Ready for modeling: {'Yes' if missing == 0 else 'No (handle remaining missing)'}")
|
|
361
|
+
|
|
362
|
+
print("\n" + "=" * 70)
|
|
363
|
+
|
|
364
|
+
def get_raw_data(self) -> Optional[pd.DataFrame]:
|
|
365
|
+
"""Return the raw DataFrame."""
|
|
366
|
+
return self.raw_df
|
|
367
|
+
|
|
368
|
+
def get_cleaned_data(self) -> Optional[pd.DataFrame]:
|
|
369
|
+
"""Return the cleaned DataFrame."""
|
|
370
|
+
return self.cleaned_df
|
|
371
|
+
|
|
372
|
+
def get_final_data(self) -> Optional[pd.DataFrame]:
|
|
373
|
+
"""Return the feature-engineered DataFrame (model-ready)."""
|
|
374
|
+
return self.final_df
|
|
375
|
+
|
|
376
|
+
def get_report(self) -> Dict[str, Any]:
|
|
377
|
+
"""Return the complete pipeline report."""
|
|
378
|
+
return self.pipeline_report
|
|
379
|
+
|
|
380
|
+
def save_data(
|
|
381
|
+
self,
|
|
382
|
+
output_dir: str,
|
|
383
|
+
save_cleaned: bool = True,
|
|
384
|
+
save_final: bool = True,
|
|
385
|
+
format: str = 'csv'
|
|
386
|
+
) -> None:
|
|
387
|
+
"""
|
|
388
|
+
Save processed datasets to files.
|
|
389
|
+
|
|
390
|
+
Parameters:
|
|
391
|
+
-----------
|
|
392
|
+
output_dir : str
|
|
393
|
+
Output directory path
|
|
394
|
+
save_cleaned : bool
|
|
395
|
+
Save cleaned data
|
|
396
|
+
save_final : bool
|
|
397
|
+
Save final engineered data
|
|
398
|
+
format : str
|
|
399
|
+
Output format: 'csv' or 'parquet'
|
|
400
|
+
"""
|
|
401
|
+
output_path = Path(output_dir)
|
|
402
|
+
output_path.mkdir(parents=True, exist_ok=True)
|
|
403
|
+
|
|
404
|
+
if save_cleaned and self.cleaned_df is not None:
|
|
405
|
+
if format == 'csv':
|
|
406
|
+
self.cleaned_df.to_csv(output_path / 'cleaned_data.csv', index=False)
|
|
407
|
+
else:
|
|
408
|
+
self.cleaned_df.to_parquet(output_path / 'cleaned_data.parquet', index=False)
|
|
409
|
+
logger.info(f"✅ Saved cleaned data to {output_path / f'cleaned_data.{format}'}")
|
|
410
|
+
|
|
411
|
+
if save_final and self.final_df is not None:
|
|
412
|
+
if format == 'csv':
|
|
413
|
+
self.final_df.to_csv(output_path / 'final_data.csv', index=False)
|
|
414
|
+
else:
|
|
415
|
+
self.final_df.to_parquet(output_path / 'final_data.parquet', index=False)
|
|
416
|
+
logger.info(f"✅ Saved final data to {output_path / f'final_data.{format}'}")
|
|
417
|
+
|
|
418
|
+
def train_model(
|
|
419
|
+
self,
|
|
420
|
+
target_col: Optional[str] = None,
|
|
421
|
+
problem_type: Optional[str] = None,
|
|
422
|
+
export_path: Optional[str] = None
|
|
423
|
+
) -> Dict[str, Any]:
|
|
424
|
+
"""
|
|
425
|
+
Train ML models on the processed data.
|
|
426
|
+
|
|
427
|
+
Parameters:
|
|
428
|
+
-----------
|
|
429
|
+
target_col : str, optional
|
|
430
|
+
Target column name (auto-detected if None)
|
|
431
|
+
problem_type : str, optional
|
|
432
|
+
'classification' or 'regression' (auto-detected if None)
|
|
433
|
+
export_path : str, optional
|
|
434
|
+
Path to export the best model (.pkl)
|
|
435
|
+
|
|
436
|
+
Returns:
|
|
437
|
+
--------
|
|
438
|
+
dict : Training results dashboard
|
|
439
|
+
"""
|
|
440
|
+
df = self.final_df if self.final_df is not None else self.cleaned_df
|
|
441
|
+
if df is None:
|
|
442
|
+
raise ValueError("No data available. Run clean() or load() first.")
|
|
443
|
+
|
|
444
|
+
t_col = target_col or self.target_col
|
|
445
|
+
p_type = problem_type or self.problem_type
|
|
446
|
+
|
|
447
|
+
self.trainer = ModelTrainer(df, target_col=t_col, problem_type=p_type, raw_df=self.raw_df)
|
|
448
|
+
self.model_results = self.trainer.run()
|
|
449
|
+
|
|
450
|
+
if export_path:
|
|
451
|
+
self.trainer.export_model(export_path)
|
|
452
|
+
|
|
453
|
+
self.pipeline_report['model_training'] = self.model_results
|
|
454
|
+
return self.model_results
|
|
455
|
+
|
|
456
|
+
def generate_html_report(self, output_path: str) -> str:
|
|
457
|
+
"""
|
|
458
|
+
Generate a detailed HTML report using ReportGenerator.
|
|
459
|
+
"""
|
|
460
|
+
generator = ReportGenerator(self, self.model_results)
|
|
461
|
+
html_content = generator.generate_html()
|
|
462
|
+
|
|
463
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
464
|
+
f.write(html_content)
|
|
465
|
+
|
|
466
|
+
logger.info(f"✅ Generated detailed HTML report: {output_path}")
|
|
467
|
+
return output_path
|
|
468
|
+
|
|
469
|
+
def generate_markdown_report(self, output_path: str) -> str:
|
|
470
|
+
"""
|
|
471
|
+
Generate a comprehensive Markdown report of the pipeline run.
|
|
472
|
+
"""
|
|
473
|
+
report = []
|
|
474
|
+
report.append(f"# Mini Data Clean Tool - Model Report")
|
|
475
|
+
report.append(f"**Generated:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n")
|
|
476
|
+
|
|
477
|
+
# 1. Dataset Overview
|
|
478
|
+
report.append("## 1. Data Processing Journey")
|
|
479
|
+
if self.raw_df is not None:
|
|
480
|
+
report.append(f"- **Raw Dataset:** {self.raw_df.shape[0]:,} rows × {self.raw_df.shape[1]} columns")
|
|
481
|
+
if self.cleaned_df is not None:
|
|
482
|
+
report.append(f"- **Cleaned Dataset:** {self.cleaned_df.shape[0]:,} rows × {self.cleaned_df.shape[1]} columns")
|
|
483
|
+
|
|
484
|
+
# Pipeline Steps
|
|
485
|
+
if 'preprocessing' in self.pipeline_report:
|
|
486
|
+
report.append("\n### Cleaning Steps Executed:")
|
|
487
|
+
steps = self.pipeline_report['preprocessing'].get('steps_executed', [])
|
|
488
|
+
if steps:
|
|
489
|
+
for step in steps:
|
|
490
|
+
report.append(f"- {step}")
|
|
491
|
+
else:
|
|
492
|
+
report.append("- No major cleaning issues found.")
|
|
493
|
+
|
|
494
|
+
# 2. Model Performance
|
|
495
|
+
report.append("\n## 2. Model Performance Analysis")
|
|
496
|
+
|
|
497
|
+
if self.model_results and 'comparison' in self.model_results:
|
|
498
|
+
comp = self.model_results['comparison']
|
|
499
|
+
metric_name = comp.get('metric', 'Metric')
|
|
500
|
+
|
|
501
|
+
report.append("| Model | Score | Reliability |")
|
|
502
|
+
report.append("| :--- | :--- | :--- |")
|
|
503
|
+
|
|
504
|
+
# Raw Row
|
|
505
|
+
raw_score = comp.get('raw_score', 'N/A')
|
|
506
|
+
raw_rel = comp.get('raw_reliability', {})
|
|
507
|
+
raw_grade = raw_rel.get('grade', 'N/A')
|
|
508
|
+
raw_pts = raw_rel.get('score', 0)
|
|
509
|
+
report.append(f"| **Raw Baseline** | {raw_score}% ({metric_name}) | **{raw_grade}** ({raw_pts}/100) |")
|
|
510
|
+
|
|
511
|
+
# Cleaned Row
|
|
512
|
+
clean_score = comp.get('cleaned_score', 'N/A')
|
|
513
|
+
# Get reliability from best model metadata or calculate
|
|
514
|
+
# For now, let's grab it from export metadata if available, or approximate
|
|
515
|
+
# Actually, model_trainer export includes it.
|
|
516
|
+
# We can grab it from best_model dict if available
|
|
517
|
+
clean_grade = 'N/A'
|
|
518
|
+
clean_pts = 0
|
|
519
|
+
|
|
520
|
+
# Try to dig reliability out of best model
|
|
521
|
+
# This is a bit tricky as it's not directly in comparison dict usually
|
|
522
|
+
# But we can look at reliability score if we re-calculated it or stored it
|
|
523
|
+
# The backend calc logic is in ModelTrainer.
|
|
524
|
+
# For the report sake, let's look at the 'model_training' specific section
|
|
525
|
+
best_model_info = self.model_results.get('best_model', {})
|
|
526
|
+
# We don't have reliability directly here unless we added it to the dict in ModelTrainer.run
|
|
527
|
+
# Let's check ModelTrainer.run output structure.
|
|
528
|
+
# It returns 'best_model': {... 'metrics': ...}
|
|
529
|
+
# The 'reliability' key is in the *export*, not immediately in run return?
|
|
530
|
+
# Wait, line 832 in ModelTrainer adds 'reliability' to export.
|
|
531
|
+
# We should probably expose it in the main result dict too for easy access.
|
|
532
|
+
|
|
533
|
+
report.append(f"| **Cleaned Model** | **{clean_score}%** ({metric_name}) | *(See Dashboard)* |")
|
|
534
|
+
|
|
535
|
+
report.append(f"\n**Improvement:** {comp.get('improvement_pct', 0)}% improvement over baseline.")
|
|
536
|
+
|
|
537
|
+
# 3. Key Observations
|
|
538
|
+
report.append("\n## 3. Key Observations")
|
|
539
|
+
report.append("> This model report was generated automatically by the Mini Data Clean Tool.")
|
|
540
|
+
report.append(f"- **Target Variable:** `{self.target_col}`")
|
|
541
|
+
report.append(f"- **Problem Type:** `{self.problem_type}`")
|
|
542
|
+
|
|
543
|
+
# Save
|
|
544
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
545
|
+
f.write('\n'.join(report))
|
|
546
|
+
|
|
547
|
+
logger.info(f"✅ Generated model report: {output_path}")
|
|
548
|
+
return output_path
|
dataclean/py.typed
ADDED
|
File without changes
|