autoprepml 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoprepml/__init__.py +33 -0
- autoprepml/cleaning.py +176 -0
- autoprepml/cli.py +166 -0
- autoprepml/config.py +95 -0
- autoprepml/core.py +194 -0
- autoprepml/detection.py +129 -0
- autoprepml/graph.py +374 -0
- autoprepml/llm_suggest.py +89 -0
- autoprepml/reports.py +381 -0
- autoprepml/text.py +304 -0
- autoprepml/timeseries.py +336 -0
- autoprepml/utils.py +8 -0
- autoprepml/visualization.py +196 -0
- autoprepml-1.0.0.dist-info/METADATA +1035 -0
- autoprepml-1.0.0.dist-info/RECORD +19 -0
- autoprepml-1.0.0.dist-info/WHEEL +5 -0
- autoprepml-1.0.0.dist-info/entry_points.txt +2 -0
- autoprepml-1.0.0.dist-info/licenses/LICENSE +21 -0
- autoprepml-1.0.0.dist-info/top_level.txt +1 -0
autoprepml/__init__.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""AutoPrepML - Automated Data Preprocessing Pipeline
|
|
2
|
+
|
|
3
|
+
A Python library for automatic detection, cleaning, and reporting of common
|
|
4
|
+
data quality issues in machine learning pipelines.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
__version__ = "0.1.0"
|
|
8
|
+
__author__ = "AutoPrepML Contributors"
|
|
9
|
+
__license__ = "MIT"
|
|
10
|
+
|
|
11
|
+
from .core import AutoPrepML
|
|
12
|
+
from .text import TextPrepML
|
|
13
|
+
from .timeseries import TimeSeriesPrepML
|
|
14
|
+
from .graph import GraphPrepML
|
|
15
|
+
from . import detection
|
|
16
|
+
from . import cleaning
|
|
17
|
+
from . import visualization
|
|
18
|
+
from . import reports
|
|
19
|
+
from . import config
|
|
20
|
+
from . import llm_suggest
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"AutoPrepML",
|
|
24
|
+
"TextPrepML",
|
|
25
|
+
"TimeSeriesPrepML",
|
|
26
|
+
"GraphPrepML",
|
|
27
|
+
"detection",
|
|
28
|
+
"cleaning",
|
|
29
|
+
"visualization",
|
|
30
|
+
"reports",
|
|
31
|
+
"config",
|
|
32
|
+
"llm_suggest",
|
|
33
|
+
]
|
autoprepml/cleaning.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Cleaning and transformation functions for AutoPrepML"""
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import numpy as np
|
|
4
|
+
from sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def impute_missing(df: pd.DataFrame, strategy: str = 'auto',
|
|
8
|
+
numeric_strategy: str = 'median',
|
|
9
|
+
categorical_strategy: str = 'mode') -> pd.DataFrame:
|
|
10
|
+
"""Impute missing values in DataFrame.
|
|
11
|
+
|
|
12
|
+
Args:
|
|
13
|
+
df: Input DataFrame
|
|
14
|
+
strategy: 'auto' (smart detection), 'median', 'mean', 'mode', 'drop'
|
|
15
|
+
numeric_strategy: Strategy for numeric columns when strategy='auto'
|
|
16
|
+
categorical_strategy: Strategy for categorical columns when strategy='auto'
|
|
17
|
+
|
|
18
|
+
Returns:
|
|
19
|
+
DataFrame with imputed values
|
|
20
|
+
"""
|
|
21
|
+
df_clean = df.copy()
|
|
22
|
+
|
|
23
|
+
for col in df_clean.columns:
|
|
24
|
+
if df_clean[col].isnull().sum() == 0:
|
|
25
|
+
continue
|
|
26
|
+
|
|
27
|
+
if strategy == 'drop':
|
|
28
|
+
df_clean = df_clean.dropna(subset=[col])
|
|
29
|
+
continue
|
|
30
|
+
|
|
31
|
+
# Auto-detect strategy based on dtype
|
|
32
|
+
if strategy == 'auto':
|
|
33
|
+
if pd.api.types.is_numeric_dtype(df_clean[col]):
|
|
34
|
+
current_strategy = numeric_strategy
|
|
35
|
+
else:
|
|
36
|
+
current_strategy = categorical_strategy
|
|
37
|
+
else:
|
|
38
|
+
current_strategy = strategy
|
|
39
|
+
|
|
40
|
+
# Apply imputation
|
|
41
|
+
if current_strategy == 'median' and pd.api.types.is_numeric_dtype(df_clean[col]):
|
|
42
|
+
fill_value = df_clean[col].median()
|
|
43
|
+
elif current_strategy == 'mean' and pd.api.types.is_numeric_dtype(df_clean[col]):
|
|
44
|
+
fill_value = df_clean[col].mean()
|
|
45
|
+
elif current_strategy == 'mode':
|
|
46
|
+
mode_values = df_clean[col].mode()
|
|
47
|
+
fill_value = mode_values.iloc[0] if len(mode_values) > 0 else ''
|
|
48
|
+
else:
|
|
49
|
+
fill_value = 0 if pd.api.types.is_numeric_dtype(df_clean[col]) else ''
|
|
50
|
+
|
|
51
|
+
df_clean[col] = df_clean[col].fillna(fill_value)
|
|
52
|
+
|
|
53
|
+
return df_clean
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def scale_features(df: pd.DataFrame, method: str = 'standard',
|
|
57
|
+
exclude_cols: list = None) -> pd.DataFrame:
|
|
58
|
+
"""Scale numeric features.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
df: Input DataFrame
|
|
62
|
+
method: 'standard' (z-score normalization) or 'minmax' (0-1 scaling)
|
|
63
|
+
exclude_cols: List of column names to exclude from scaling
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
DataFrame with scaled numeric features
|
|
67
|
+
"""
|
|
68
|
+
df_scaled = df.copy()
|
|
69
|
+
exclude_cols = exclude_cols or []
|
|
70
|
+
|
|
71
|
+
numeric_cols = df_scaled.select_dtypes(include=[np.number]).columns.tolist()
|
|
72
|
+
cols_to_scale = [col for col in numeric_cols if col not in exclude_cols]
|
|
73
|
+
|
|
74
|
+
if not cols_to_scale:
|
|
75
|
+
return df_scaled
|
|
76
|
+
|
|
77
|
+
if method == 'standard':
|
|
78
|
+
scaler = StandardScaler()
|
|
79
|
+
elif method == 'minmax':
|
|
80
|
+
scaler = MinMaxScaler()
|
|
81
|
+
else:
|
|
82
|
+
raise ValueError(f"Unknown scaling method: {method}")
|
|
83
|
+
|
|
84
|
+
df_scaled[cols_to_scale] = scaler.fit_transform(df_scaled[cols_to_scale])
|
|
85
|
+
|
|
86
|
+
return df_scaled
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def encode_categorical(df: pd.DataFrame, method: str = 'label',
|
|
90
|
+
exclude_cols: list = None) -> pd.DataFrame:
|
|
91
|
+
"""Encode categorical features.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
df: Input DataFrame
|
|
95
|
+
method: 'label' (label encoding) or 'onehot' (one-hot encoding)
|
|
96
|
+
exclude_cols: List of column names to exclude from encoding
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
DataFrame with encoded categorical features
|
|
100
|
+
"""
|
|
101
|
+
df_encoded = df.copy()
|
|
102
|
+
exclude_cols = exclude_cols or []
|
|
103
|
+
|
|
104
|
+
categorical_cols = df_encoded.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
105
|
+
cols_to_encode = [col for col in categorical_cols if col not in exclude_cols]
|
|
106
|
+
|
|
107
|
+
if not cols_to_encode:
|
|
108
|
+
return df_encoded
|
|
109
|
+
|
|
110
|
+
if method == 'label':
|
|
111
|
+
for col in cols_to_encode:
|
|
112
|
+
le = LabelEncoder()
|
|
113
|
+
df_encoded[col] = le.fit_transform(df_encoded[col].astype(str))
|
|
114
|
+
|
|
115
|
+
elif method == 'onehot':
|
|
116
|
+
df_encoded = pd.get_dummies(df_encoded, columns=cols_to_encode, drop_first=True)
|
|
117
|
+
|
|
118
|
+
else:
|
|
119
|
+
raise ValueError(f"Unknown encoding method: {method}")
|
|
120
|
+
|
|
121
|
+
return df_encoded
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def balance_classes(df: pd.DataFrame, target_col: str, method: str = 'oversample') -> pd.DataFrame:
|
|
125
|
+
"""Balance class distribution in target column.
|
|
126
|
+
|
|
127
|
+
Args:
|
|
128
|
+
df: Input DataFrame
|
|
129
|
+
target_col: Name of target column
|
|
130
|
+
method: 'oversample' (duplicate minority), 'undersample' (reduce majority)
|
|
131
|
+
|
|
132
|
+
Returns:
|
|
133
|
+
DataFrame with balanced classes
|
|
134
|
+
"""
|
|
135
|
+
if target_col not in df.columns:
|
|
136
|
+
raise ValueError(f"Target column '{target_col}' not found in DataFrame")
|
|
137
|
+
|
|
138
|
+
df_balanced = df.copy()
|
|
139
|
+
value_counts = df_balanced[target_col].value_counts()
|
|
140
|
+
|
|
141
|
+
if method == 'oversample':
|
|
142
|
+
max_count = value_counts.max()
|
|
143
|
+
dfs = []
|
|
144
|
+
for class_value in value_counts.index:
|
|
145
|
+
class_df = df_balanced[df_balanced[target_col] == class_value]
|
|
146
|
+
count_diff = max_count - len(class_df)
|
|
147
|
+
if count_diff > 0:
|
|
148
|
+
oversampled = class_df.sample(n=count_diff, replace=True, random_state=42)
|
|
149
|
+
dfs.extend([class_df, oversampled])
|
|
150
|
+
else:
|
|
151
|
+
dfs.append(class_df)
|
|
152
|
+
df_balanced = pd.concat(dfs, ignore_index=True)
|
|
153
|
+
elif method == 'undersample':
|
|
154
|
+
min_count = value_counts.min()
|
|
155
|
+
dfs = [
|
|
156
|
+
df_balanced[df_balanced[target_col] == class_value].sample(n=min_count, random_state=42)
|
|
157
|
+
for class_value in value_counts.index
|
|
158
|
+
]
|
|
159
|
+
df_balanced = pd.concat(dfs, ignore_index=True)
|
|
160
|
+
else:
|
|
161
|
+
raise ValueError(f"Unknown balancing method: {method}")
|
|
162
|
+
|
|
163
|
+
return df_balanced.sample(frac=1, random_state=42).reset_index(drop=True)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def remove_outliers(df: pd.DataFrame, outlier_indices: list) -> pd.DataFrame:
|
|
167
|
+
"""Remove rows identified as outliers.
|
|
168
|
+
|
|
169
|
+
Args:
|
|
170
|
+
df: Input DataFrame
|
|
171
|
+
outlier_indices: List of row indices to remove
|
|
172
|
+
|
|
173
|
+
Returns:
|
|
174
|
+
DataFrame with outliers removed
|
|
175
|
+
"""
|
|
176
|
+
return df.drop(index=outlier_indices).reset_index(drop=True)
|
autoprepml/cli.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Command-line interface for AutoPrepML"""
|
|
2
|
+
import argparse
|
|
3
|
+
import sys
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from .core import AutoPrepML
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main():
|
|
10
|
+
"""Main CLI entrypoint."""
|
|
11
|
+
parser = argparse.ArgumentParser(
|
|
12
|
+
description='AutoPrepML - Automated Data Preprocessing Pipeline',
|
|
13
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
14
|
+
epilog="""
|
|
15
|
+
Examples:
|
|
16
|
+
# Basic cleaning
|
|
17
|
+
autoprepml --input data.csv --output cleaned.csv
|
|
18
|
+
|
|
19
|
+
# Classification task with target column
|
|
20
|
+
autoprepml --input train.csv --output clean_train.csv --task classification --target label
|
|
21
|
+
|
|
22
|
+
# Generate HTML report
|
|
23
|
+
autoprepml --input data.csv --output cleaned.csv --report report.html
|
|
24
|
+
|
|
25
|
+
# Use custom config
|
|
26
|
+
autoprepml --input data.csv --output cleaned.csv --config config.yaml
|
|
27
|
+
"""
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
parser.add_argument(
|
|
31
|
+
'--input', '-i',
|
|
32
|
+
required=True,
|
|
33
|
+
help='Input CSV file path'
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
parser.add_argument(
|
|
37
|
+
'--output', '-o',
|
|
38
|
+
required=True,
|
|
39
|
+
help='Output cleaned CSV file path'
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
'--report', '-r',
|
|
44
|
+
help='Output report file path (.html or .json)'
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
parser.add_argument(
|
|
48
|
+
'--task',
|
|
49
|
+
choices=['classification', 'regression'],
|
|
50
|
+
help='ML task type (affects preprocessing strategy)'
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
parser.add_argument(
|
|
54
|
+
'--target',
|
|
55
|
+
help='Name of target column (for classification/regression tasks)'
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
parser.add_argument(
|
|
59
|
+
'--config', '-c',
|
|
60
|
+
help='Path to YAML/JSON configuration file'
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
'--no-plots',
|
|
65
|
+
action='store_true',
|
|
66
|
+
help='Disable plot generation in reports'
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
parser.add_argument(
|
|
70
|
+
'--detect-only',
|
|
71
|
+
action='store_true',
|
|
72
|
+
help='Only run detection, do not clean data'
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
'--verbose', '-v',
|
|
77
|
+
action='store_true',
|
|
78
|
+
help='Enable verbose logging'
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
args = parser.parse_args()
|
|
82
|
+
|
|
83
|
+
# Validate inputs
|
|
84
|
+
input_path = Path(args.input)
|
|
85
|
+
if not input_path.exists():
|
|
86
|
+
print(f"❌ Error: Input file not found: {args.input}", file=sys.stderr)
|
|
87
|
+
sys.exit(1)
|
|
88
|
+
|
|
89
|
+
if input_path.suffix.lower() != '.csv':
|
|
90
|
+
print("❌ Error: Input file must be a CSV file", file=sys.stderr)
|
|
91
|
+
sys.exit(1)
|
|
92
|
+
|
|
93
|
+
# Validate report format
|
|
94
|
+
if args.report:
|
|
95
|
+
report_path = Path(args.report)
|
|
96
|
+
if report_path.suffix.lower() not in ['.html', '.json']:
|
|
97
|
+
print("❌ Error: Report file must be .html or .json", file=sys.stderr)
|
|
98
|
+
sys.exit(1)
|
|
99
|
+
|
|
100
|
+
# Load data
|
|
101
|
+
try:
|
|
102
|
+
print(f"📂 Loading data from {args.input}...")
|
|
103
|
+
df = pd.read_csv(args.input)
|
|
104
|
+
print(f"✅ Loaded {len(df)} rows, {len(df.columns)} columns")
|
|
105
|
+
except Exception as e:
|
|
106
|
+
print(f"❌ Error loading CSV: {e}", file=sys.stderr)
|
|
107
|
+
sys.exit(1)
|
|
108
|
+
|
|
109
|
+
# Initialize AutoPrepML
|
|
110
|
+
try:
|
|
111
|
+
prep = AutoPrepML(df, config_path=args.config)
|
|
112
|
+
|
|
113
|
+
if args.no_plots:
|
|
114
|
+
prep.config['reporting']['include_plots'] = False
|
|
115
|
+
|
|
116
|
+
# Detection phase
|
|
117
|
+
print("\n🔍 Running detection...")
|
|
118
|
+
detection_results = prep.detect(target_col=args.target)
|
|
119
|
+
|
|
120
|
+
# Print detection summary
|
|
121
|
+
missing_count = len(detection_results.get('missing_values', {}))
|
|
122
|
+
outlier_count = detection_results.get('outliers', {}).get('outlier_count', 0)
|
|
123
|
+
|
|
124
|
+
print(f" • Missing values: {missing_count} columns affected")
|
|
125
|
+
print(f" • Outliers detected: {outlier_count} rows")
|
|
126
|
+
|
|
127
|
+
if args.target and 'class_imbalance' in detection_results:
|
|
128
|
+
imbalance = detection_results['class_imbalance']
|
|
129
|
+
status = "⚠ Imbalanced" if imbalance['is_imbalanced'] else "✓ Balanced"
|
|
130
|
+
print(f" • Class distribution: {status}")
|
|
131
|
+
|
|
132
|
+
if args.detect_only:
|
|
133
|
+
print("\n✅ Detection complete (--detect-only mode)")
|
|
134
|
+
if args.report:
|
|
135
|
+
prep.save_report(args.report)
|
|
136
|
+
print(f"📄 Report saved to {args.report}")
|
|
137
|
+
sys.exit(0)
|
|
138
|
+
|
|
139
|
+
# Cleaning phase
|
|
140
|
+
print("\n🧹 Cleaning data...")
|
|
141
|
+
clean_df, report = prep.clean(task=args.task, target_col=args.target)
|
|
142
|
+
|
|
143
|
+
# Save cleaned data
|
|
144
|
+
output_path = Path(args.output)
|
|
145
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
146
|
+
clean_df.to_csv(args.output, index=False)
|
|
147
|
+
print(f"✅ Cleaned data saved to {args.output}")
|
|
148
|
+
print(f" Shape: {clean_df.shape[0]} rows × {clean_df.shape[1]} columns")
|
|
149
|
+
|
|
150
|
+
# Save report
|
|
151
|
+
if args.report:
|
|
152
|
+
prep.save_report(args.report)
|
|
153
|
+
print(f"📄 Report saved to {args.report}")
|
|
154
|
+
|
|
155
|
+
print("\n🎉 AutoPrepML completed successfully!")
|
|
156
|
+
|
|
157
|
+
except Exception as e:
|
|
158
|
+
print(f"❌ Error during preprocessing: {e}", file=sys.stderr)
|
|
159
|
+
if args.verbose:
|
|
160
|
+
import traceback
|
|
161
|
+
traceback.print_exc()
|
|
162
|
+
sys.exit(1)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
if __name__ == '__main__':
|
|
166
|
+
main()
|
autoprepml/config.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Configuration management for AutoPrepML"""
|
|
2
|
+
import os
|
|
3
|
+
import json
|
|
4
|
+
import yaml
|
|
5
|
+
from typing import Dict, Any, Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
DEFAULT_CONFIG = {
|
|
9
|
+
'cleaning': {
|
|
10
|
+
'missing_strategy': 'auto',
|
|
11
|
+
'numeric_strategy': 'median',
|
|
12
|
+
'categorical_strategy': 'mode',
|
|
13
|
+
'outlier_method': 'iforest',
|
|
14
|
+
'outlier_contamination': 0.05,
|
|
15
|
+
'remove_outliers': False,
|
|
16
|
+
'scale_method': 'standard',
|
|
17
|
+
'encode_method': 'label',
|
|
18
|
+
'balance_method': 'oversample'
|
|
19
|
+
},
|
|
20
|
+
'detection': {
|
|
21
|
+
'outlier_method': 'iforest',
|
|
22
|
+
'contamination': 0.05,
|
|
23
|
+
'zscore_threshold': 3.0,
|
|
24
|
+
'imbalance_threshold': 0.3
|
|
25
|
+
},
|
|
26
|
+
'reporting': {
|
|
27
|
+
'format': 'html',
|
|
28
|
+
'include_plots': True,
|
|
29
|
+
'plot_style': 'seaborn',
|
|
30
|
+
'output_dir': './reports'
|
|
31
|
+
},
|
|
32
|
+
'logging': {
|
|
33
|
+
'enabled': True,
|
|
34
|
+
'level': 'INFO',
|
|
35
|
+
'file': 'autoprepml.log'
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def load_config(config_path: Optional[str] = None) -> Dict[str, Any]:
|
|
41
|
+
"""Load configuration from YAML or JSON file.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
config_path: Path to config file. If None, returns default config.
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
Configuration dictionary
|
|
48
|
+
"""
|
|
49
|
+
if config_path is None:
|
|
50
|
+
return DEFAULT_CONFIG.copy()
|
|
51
|
+
|
|
52
|
+
if not os.path.exists(config_path):
|
|
53
|
+
raise FileNotFoundError(f"Config file not found: {config_path}")
|
|
54
|
+
|
|
55
|
+
with open(config_path, 'r', encoding='utf-8') as f:
|
|
56
|
+
if config_path.endswith('.yaml') or config_path.endswith('.yml'):
|
|
57
|
+
user_config = yaml.safe_load(f)
|
|
58
|
+
elif config_path.endswith('.json'):
|
|
59
|
+
user_config = json.load(f)
|
|
60
|
+
else:
|
|
61
|
+
raise ValueError("Config file must be YAML or JSON")
|
|
62
|
+
|
|
63
|
+
# Merge with defaults (user config overrides defaults)
|
|
64
|
+
config = DEFAULT_CONFIG.copy()
|
|
65
|
+
if user_config:
|
|
66
|
+
for section, values in user_config.items():
|
|
67
|
+
if section in config and isinstance(config[section], dict):
|
|
68
|
+
config[section].update(values)
|
|
69
|
+
else:
|
|
70
|
+
config[section] = values
|
|
71
|
+
|
|
72
|
+
return config
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def save_config(config: Dict[str, Any], output_path: str) -> None:
|
|
76
|
+
"""Save configuration to YAML file.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
config: Configuration dictionary
|
|
80
|
+
output_path: Path to save config file
|
|
81
|
+
"""
|
|
82
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
83
|
+
if output_path.endswith('.json'):
|
|
84
|
+
json.dump(config, f, indent=2)
|
|
85
|
+
else:
|
|
86
|
+
yaml.dump(config, f, default_flow_style=False, sort_keys=False)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def get_default_config() -> Dict[str, Any]:
|
|
90
|
+
"""Return a copy of the default configuration.
|
|
91
|
+
|
|
92
|
+
Returns:
|
|
93
|
+
Default configuration dictionary
|
|
94
|
+
"""
|
|
95
|
+
return DEFAULT_CONFIG.copy()
|
autoprepml/core.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Core high-level interface for AutoPrepML"""
|
|
2
|
+
from typing import Optional, Dict, Any, Tuple
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import logging
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
|
|
7
|
+
from . import detection
|
|
8
|
+
from . import cleaning
|
|
9
|
+
from . import visualization
|
|
10
|
+
from .config import load_config, DEFAULT_CONFIG
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# Setup logging
|
|
14
|
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
|
15
|
+
logger = logging.getLogger('autoprepml')
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AutoPrepML:
|
|
19
|
+
"""Main AutoPrepML class for automated data preprocessing.
|
|
20
|
+
|
|
21
|
+
Example:
|
|
22
|
+
>>> df = pd.read_csv('data.csv')
|
|
23
|
+
>>> prep = AutoPrepML(df)
|
|
24
|
+
>>> clean_df, report = prep.clean(task='classification', target_col='label')
|
|
25
|
+
>>> prep.save_report('report.html')
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self, df: pd.DataFrame, config: Optional[Dict[str, Any]] = None,
|
|
29
|
+
config_path: Optional[str] = None):
|
|
30
|
+
"""Initialize AutoPrepML with a DataFrame.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
df: Input DataFrame to preprocess
|
|
34
|
+
config: Optional configuration dictionary
|
|
35
|
+
config_path: Optional path to YAML/JSON config file
|
|
36
|
+
"""
|
|
37
|
+
self.original_df = df.copy()
|
|
38
|
+
self.df = df.copy()
|
|
39
|
+
self.cleaned_df = None
|
|
40
|
+
self.log = []
|
|
41
|
+
self.detection_results = {}
|
|
42
|
+
self.plots = {}
|
|
43
|
+
|
|
44
|
+
# Load configuration
|
|
45
|
+
if config_path:
|
|
46
|
+
self.config = load_config(config_path)
|
|
47
|
+
elif config:
|
|
48
|
+
self.config = config
|
|
49
|
+
else:
|
|
50
|
+
self.config = DEFAULT_CONFIG.copy()
|
|
51
|
+
|
|
52
|
+
self._log_action('initialized', {'shape': df.shape, 'columns': list(df.columns)})
|
|
53
|
+
|
|
54
|
+
def _log_action(self, action: str, details: Any) -> None:
|
|
55
|
+
"""Internal logging helper."""
|
|
56
|
+
entry = {
|
|
57
|
+
'timestamp': datetime.now().isoformat(),
|
|
58
|
+
'action': action,
|
|
59
|
+
'details': details
|
|
60
|
+
}
|
|
61
|
+
self.log.append(entry)
|
|
62
|
+
logger.info(f"{action}: {details}")
|
|
63
|
+
|
|
64
|
+
def detect(self, target_col: Optional[str] = None) -> Dict[str, Any]:
|
|
65
|
+
"""Run all detection functions.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
target_col: Optional target column for imbalance detection
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
Dictionary containing all detection results
|
|
72
|
+
"""
|
|
73
|
+
self.detection_results = detection.detect_all(self.df, target_col)
|
|
74
|
+
self._log_action('detection_complete', self.detection_results)
|
|
75
|
+
return self.detection_results
|
|
76
|
+
|
|
77
|
+
def clean(self, task: Optional[str] = None, target_col: Optional[str] = None,
|
|
78
|
+
auto: bool = True) -> Tuple[pd.DataFrame, Dict[str, Any]]:
|
|
79
|
+
"""Clean the dataset automatically.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
task: 'classification', 'regression', or None
|
|
83
|
+
target_col: Name of target column (for imbalance handling)
|
|
84
|
+
auto: If True, apply all cleaning steps automatically
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
Tuple of (cleaned_df, report_dict)
|
|
88
|
+
"""
|
|
89
|
+
df_clean = self.df.copy()
|
|
90
|
+
|
|
91
|
+
# Run detection first
|
|
92
|
+
if not self.detection_results:
|
|
93
|
+
self.detect(target_col)
|
|
94
|
+
|
|
95
|
+
# Step 1: Handle missing values
|
|
96
|
+
if self.detection_results.get('missing_values'):
|
|
97
|
+
strategy = self.config['cleaning']['missing_strategy']
|
|
98
|
+
df_clean = cleaning.impute_missing(df_clean, strategy=strategy)
|
|
99
|
+
self._log_action('imputed_missing', {'strategy': strategy})
|
|
100
|
+
|
|
101
|
+
# Step 2: Handle outliers (optional)
|
|
102
|
+
outliers = self.detection_results.get('outliers', {})
|
|
103
|
+
if outliers.get('outlier_count', 0) > 0 and self.config['cleaning']['remove_outliers']:
|
|
104
|
+
df_clean = cleaning.remove_outliers(df_clean, outliers['outlier_indices'])
|
|
105
|
+
self._log_action('removed_outliers', {'count': outliers['outlier_count']})
|
|
106
|
+
|
|
107
|
+
# Step 3: Encode categorical variables
|
|
108
|
+
df_clean = cleaning.encode_categorical(df_clean,
|
|
109
|
+
method=self.config['cleaning']['encode_method'],
|
|
110
|
+
exclude_cols=[target_col] if target_col else None)
|
|
111
|
+
self._log_action('encoded_categorical', {'method': self.config['cleaning']['encode_method']})
|
|
112
|
+
|
|
113
|
+
# Step 4: Scale features
|
|
114
|
+
df_clean = cleaning.scale_features(df_clean,
|
|
115
|
+
method=self.config['cleaning']['scale_method'],
|
|
116
|
+
exclude_cols=[target_col] if target_col else None)
|
|
117
|
+
self._log_action('scaled_features', {'method': self.config['cleaning']['scale_method']})
|
|
118
|
+
|
|
119
|
+
# Step 5: Balance classes (if classification task and imbalanced)
|
|
120
|
+
if task == 'classification' and target_col:
|
|
121
|
+
imbalance = self.detection_results.get('class_imbalance', {})
|
|
122
|
+
if imbalance.get('is_imbalanced'):
|
|
123
|
+
df_clean = cleaning.balance_classes(df_clean, target_col,
|
|
124
|
+
method=self.config['cleaning']['balance_method'])
|
|
125
|
+
self._log_action('balanced_classes', {'method': self.config['cleaning']['balance_method']})
|
|
126
|
+
|
|
127
|
+
self.cleaned_df = df_clean
|
|
128
|
+
|
|
129
|
+
# Generate report
|
|
130
|
+
report = self.report()
|
|
131
|
+
|
|
132
|
+
return df_clean, report
|
|
133
|
+
|
|
134
|
+
def summary(self) -> Dict[str, Any]:
|
|
135
|
+
"""Get a quick summary of the dataset.
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
Dictionary with basic dataset statistics
|
|
139
|
+
"""
|
|
140
|
+
return {
|
|
141
|
+
'shape': self.df.shape,
|
|
142
|
+
'columns': list(self.df.columns),
|
|
143
|
+
'dtypes': self.df.dtypes.astype(str).to_dict(),
|
|
144
|
+
'missing_values': detection.detect_missing(self.df),
|
|
145
|
+
'numeric_columns': self.df.select_dtypes(include=['number']).columns.tolist(),
|
|
146
|
+
'categorical_columns': self.df.select_dtypes(include=['object', 'category']).columns.tolist()
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
def report(self, include_plots: bool = True) -> Dict[str, Any]:
|
|
150
|
+
"""Generate comprehensive preprocessing report.
|
|
151
|
+
|
|
152
|
+
Args:
|
|
153
|
+
include_plots: Whether to generate visualization plots
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
Complete report dictionary
|
|
157
|
+
"""
|
|
158
|
+
report = {
|
|
159
|
+
'timestamp': datetime.now().isoformat(),
|
|
160
|
+
'original_shape': self.original_df.shape,
|
|
161
|
+
'cleaned_shape': self.cleaned_df.shape if self.cleaned_df is not None else None,
|
|
162
|
+
'detection_results': self.detection_results,
|
|
163
|
+
'logs': self.log,
|
|
164
|
+
'config': self.config
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
# Generate plots if requested
|
|
168
|
+
if include_plots and self.config['reporting']['include_plots']:
|
|
169
|
+
outlier_indices = self.detection_results.get('outliers', {}).get('outlier_indices', [])
|
|
170
|
+
self.plots = visualization.generate_all_plots(self.original_df, outlier_indices)
|
|
171
|
+
report['plots'] = self.plots
|
|
172
|
+
|
|
173
|
+
return report
|
|
174
|
+
|
|
175
|
+
def save_report(self, output_path: str) -> None:
|
|
176
|
+
"""Save report to file.
|
|
177
|
+
|
|
178
|
+
Args:
|
|
179
|
+
output_path: Path to save report (supports .json, .html)
|
|
180
|
+
"""
|
|
181
|
+
from .reports import generate_json_report, generate_html_report
|
|
182
|
+
|
|
183
|
+
report = self.report(include_plots=True)
|
|
184
|
+
|
|
185
|
+
if output_path.endswith('.json'):
|
|
186
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
187
|
+
f.write(generate_json_report(report))
|
|
188
|
+
elif output_path.endswith('.html'):
|
|
189
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
190
|
+
f.write(generate_html_report(report))
|
|
191
|
+
else:
|
|
192
|
+
raise ValueError("Output path must end with .json or .html")
|
|
193
|
+
|
|
194
|
+
self._log_action('saved_report', {'path': output_path})
|