cleanflow-kit 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cleanflow_kit-1.1.0/LICENSE +21 -0
- cleanflow_kit-1.1.0/PKG-INFO +119 -0
- cleanflow_kit-1.1.0/README.md +83 -0
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/PKG-INFO +119 -0
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/SOURCES.txt +20 -0
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/dependency_links.txt +1 -0
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/requires.txt +27 -0
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/top_level.txt +1 -0
- cleanflow_kit-1.1.0/dataclean/__init__.py +44 -0
- cleanflow_kit-1.1.0/dataclean/_compat.py +24 -0
- cleanflow_kit-1.1.0/dataclean/data_cleaner.py +988 -0
- cleanflow_kit-1.1.0/dataclean/data_loader.py +194 -0
- cleanflow_kit-1.1.0/dataclean/drift_detector.py +111 -0
- cleanflow_kit-1.1.0/dataclean/eda.py +400 -0
- cleanflow_kit-1.1.0/dataclean/feature_engineer.py +608 -0
- cleanflow_kit-1.1.0/dataclean/model_trainer.py +874 -0
- cleanflow_kit-1.1.0/dataclean/pipeline.py +548 -0
- cleanflow_kit-1.1.0/dataclean/py.typed +0 -0
- cleanflow_kit-1.1.0/dataclean/report_generator.py +365 -0
- cleanflow_kit-1.1.0/dataclean/synthetic_generator.py +97 -0
- cleanflow_kit-1.1.0/pyproject.toml +43 -0
- cleanflow_kit-1.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DataClean Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cleanflow-kit
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
|
|
5
|
+
Author: DataClean Contributors
|
|
6
|
+
License: MIT License
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.8
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: pandas>=2.0.0
|
|
14
|
+
Requires-Dist: numpy>=1.20.0
|
|
15
|
+
Requires-Dist: scikit-learn>=1.0.0
|
|
16
|
+
Requires-Dist: matplotlib>=3.5.0
|
|
17
|
+
Requires-Dist: joblib>=1.1.0
|
|
18
|
+
Provides-Extra: smote
|
|
19
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "smote"
|
|
20
|
+
Provides-Extra: drift
|
|
21
|
+
Requires-Dist: scipy>=1.7.0; extra == "drift"
|
|
22
|
+
Provides-Extra: all
|
|
23
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "all"
|
|
24
|
+
Requires-Dist: scipy>=1.7.0; extra == "all"
|
|
25
|
+
Provides-Extra: web
|
|
26
|
+
Requires-Dist: Flask; extra == "web"
|
|
27
|
+
Requires-Dist: Flask-SQLAlchemy; extra == "web"
|
|
28
|
+
Requires-Dist: PyJWT; extra == "web"
|
|
29
|
+
Requires-Dist: gunicorn; extra == "web"
|
|
30
|
+
Requires-Dist: openpyxl; extra == "web"
|
|
31
|
+
Provides-Extra: api
|
|
32
|
+
Requires-Dist: FastAPI; extra == "api"
|
|
33
|
+
Requires-Dist: uvicorn; extra == "api"
|
|
34
|
+
Requires-Dist: pydantic; extra == "api"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# DataClean
|
|
38
|
+
|
|
39
|
+
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
40
|
+
|
|
41
|
+
## 🚀 Quick Start
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from dataclean import DataPipeline
|
|
45
|
+
|
|
46
|
+
# Initialize and load data
|
|
47
|
+
pipeline = DataPipeline()
|
|
48
|
+
pipeline.load("data.csv")
|
|
49
|
+
|
|
50
|
+
# Run full pipeline
|
|
51
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
52
|
+
target_col="target_column",
|
|
53
|
+
problem_type="classification" # or "regression"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Save results
|
|
57
|
+
pipeline.save_data("./output")
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## 📦 Installation
|
|
61
|
+
|
|
62
|
+
Install via pip:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install dataclean
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
69
|
+
```bash
|
|
70
|
+
pip install dataclean[all]
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## 📋 Features
|
|
74
|
+
|
|
75
|
+
### 1. Data Loading & Validation
|
|
76
|
+
- Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
|
|
77
|
+
- Detect duplicates, missing values, data types.
|
|
78
|
+
- Identify constant/near-constant and potential ID columns.
|
|
79
|
+
|
|
80
|
+
### 2. Data Cleaning
|
|
81
|
+
- Remove duplicate rows.
|
|
82
|
+
- Handle missing values (mean/median/mode/drop).
|
|
83
|
+
- Fix incorrect data types.
|
|
84
|
+
- Handle outliers (IQR/Z-score).
|
|
85
|
+
- Clean categorical values (whitespace, case).
|
|
86
|
+
|
|
87
|
+
### 3. Exploratory Data Analysis (EDA)
|
|
88
|
+
- Summary statistics, Correlation analysis, Distribution analysis.
|
|
89
|
+
- Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
|
|
90
|
+
|
|
91
|
+
### 4. Feature Engineering
|
|
92
|
+
- Categorical encoding (One-Hot, Label).
|
|
93
|
+
- Feature scaling (Standard, MinMax).
|
|
94
|
+
- Datetime feature extraction.
|
|
95
|
+
- Polynomial and interaction features.
|
|
96
|
+
- Handling class imbalance (SMOTE).
|
|
97
|
+
|
|
98
|
+
### 5. ML Model Training
|
|
99
|
+
- Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
|
|
100
|
+
- Reliability scoring, cross-validation, and metrics comparison.
|
|
101
|
+
|
|
102
|
+
## 🎯 Supported Problem Types
|
|
103
|
+
- **Classification**
|
|
104
|
+
- **Regression**
|
|
105
|
+
*(Clustering is a planned future feature)*
|
|
106
|
+
|
|
107
|
+
## 🌐 Optional Web Application
|
|
108
|
+
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
109
|
+
Install web dependencies:
|
|
110
|
+
```bash
|
|
111
|
+
pip install dataclean[web,api]
|
|
112
|
+
```
|
|
113
|
+
Run the web application:
|
|
114
|
+
```bash
|
|
115
|
+
python app.py
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
## 📝 License
|
|
119
|
+
MIT License
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# DataClean
|
|
2
|
+
|
|
3
|
+
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
4
|
+
|
|
5
|
+
## 🚀 Quick Start
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from dataclean import DataPipeline
|
|
9
|
+
|
|
10
|
+
# Initialize and load data
|
|
11
|
+
pipeline = DataPipeline()
|
|
12
|
+
pipeline.load("data.csv")
|
|
13
|
+
|
|
14
|
+
# Run full pipeline
|
|
15
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
16
|
+
target_col="target_column",
|
|
17
|
+
problem_type="classification" # or "regression"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
# Save results
|
|
21
|
+
pipeline.save_data("./output")
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## 📦 Installation
|
|
25
|
+
|
|
26
|
+
Install via pip:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install dataclean
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
33
|
+
```bash
|
|
34
|
+
pip install dataclean[all]
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## 📋 Features
|
|
38
|
+
|
|
39
|
+
### 1. Data Loading & Validation
|
|
40
|
+
- Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
|
|
41
|
+
- Detect duplicates, missing values, data types.
|
|
42
|
+
- Identify constant/near-constant and potential ID columns.
|
|
43
|
+
|
|
44
|
+
### 2. Data Cleaning
|
|
45
|
+
- Remove duplicate rows.
|
|
46
|
+
- Handle missing values (mean/median/mode/drop).
|
|
47
|
+
- Fix incorrect data types.
|
|
48
|
+
- Handle outliers (IQR/Z-score).
|
|
49
|
+
- Clean categorical values (whitespace, case).
|
|
50
|
+
|
|
51
|
+
### 3. Exploratory Data Analysis (EDA)
|
|
52
|
+
- Summary statistics, Correlation analysis, Distribution analysis.
|
|
53
|
+
- Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
|
|
54
|
+
|
|
55
|
+
### 4. Feature Engineering
|
|
56
|
+
- Categorical encoding (One-Hot, Label).
|
|
57
|
+
- Feature scaling (Standard, MinMax).
|
|
58
|
+
- Datetime feature extraction.
|
|
59
|
+
- Polynomial and interaction features.
|
|
60
|
+
- Handling class imbalance (SMOTE).
|
|
61
|
+
|
|
62
|
+
### 5. ML Model Training
|
|
63
|
+
- Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
|
|
64
|
+
- Reliability scoring, cross-validation, and metrics comparison.
|
|
65
|
+
|
|
66
|
+
## 🎯 Supported Problem Types
|
|
67
|
+
- **Classification**
|
|
68
|
+
- **Regression**
|
|
69
|
+
*(Clustering is a planned future feature)*
|
|
70
|
+
|
|
71
|
+
## 🌐 Optional Web Application
|
|
72
|
+
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
73
|
+
Install web dependencies:
|
|
74
|
+
```bash
|
|
75
|
+
pip install dataclean[web,api]
|
|
76
|
+
```
|
|
77
|
+
Run the web application:
|
|
78
|
+
```bash
|
|
79
|
+
python app.py
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## 📝 License
|
|
83
|
+
MIT License
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cleanflow-kit
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
|
|
5
|
+
Author: DataClean Contributors
|
|
6
|
+
License: MIT License
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.8
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: pandas>=2.0.0
|
|
14
|
+
Requires-Dist: numpy>=1.20.0
|
|
15
|
+
Requires-Dist: scikit-learn>=1.0.0
|
|
16
|
+
Requires-Dist: matplotlib>=3.5.0
|
|
17
|
+
Requires-Dist: joblib>=1.1.0
|
|
18
|
+
Provides-Extra: smote
|
|
19
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "smote"
|
|
20
|
+
Provides-Extra: drift
|
|
21
|
+
Requires-Dist: scipy>=1.7.0; extra == "drift"
|
|
22
|
+
Provides-Extra: all
|
|
23
|
+
Requires-Dist: imbalanced-learn>=0.9.0; extra == "all"
|
|
24
|
+
Requires-Dist: scipy>=1.7.0; extra == "all"
|
|
25
|
+
Provides-Extra: web
|
|
26
|
+
Requires-Dist: Flask; extra == "web"
|
|
27
|
+
Requires-Dist: Flask-SQLAlchemy; extra == "web"
|
|
28
|
+
Requires-Dist: PyJWT; extra == "web"
|
|
29
|
+
Requires-Dist: gunicorn; extra == "web"
|
|
30
|
+
Requires-Dist: openpyxl; extra == "web"
|
|
31
|
+
Provides-Extra: api
|
|
32
|
+
Requires-Dist: FastAPI; extra == "api"
|
|
33
|
+
Requires-Dist: uvicorn; extra == "api"
|
|
34
|
+
Requires-Dist: pydantic; extra == "api"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# DataClean
|
|
38
|
+
|
|
39
|
+
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
40
|
+
|
|
41
|
+
## 🚀 Quick Start
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from dataclean import DataPipeline
|
|
45
|
+
|
|
46
|
+
# Initialize and load data
|
|
47
|
+
pipeline = DataPipeline()
|
|
48
|
+
pipeline.load("data.csv")
|
|
49
|
+
|
|
50
|
+
# Run full pipeline
|
|
51
|
+
cleaned_df, final_df = pipeline.run_full_pipeline(
|
|
52
|
+
target_col="target_column",
|
|
53
|
+
problem_type="classification" # or "regression"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Save results
|
|
57
|
+
pipeline.save_data("./output")
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## 📦 Installation
|
|
61
|
+
|
|
62
|
+
Install via pip:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install dataclean
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
69
|
+
```bash
|
|
70
|
+
pip install dataclean[all]
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## 📋 Features
|
|
74
|
+
|
|
75
|
+
### 1. Data Loading & Validation
|
|
76
|
+
- Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
|
|
77
|
+
- Detect duplicates, missing values, data types.
|
|
78
|
+
- Identify constant/near-constant and potential ID columns.
|
|
79
|
+
|
|
80
|
+
### 2. Data Cleaning
|
|
81
|
+
- Remove duplicate rows.
|
|
82
|
+
- Handle missing values (mean/median/mode/drop).
|
|
83
|
+
- Fix incorrect data types.
|
|
84
|
+
- Handle outliers (IQR/Z-score).
|
|
85
|
+
- Clean categorical values (whitespace, case).
|
|
86
|
+
|
|
87
|
+
### 3. Exploratory Data Analysis (EDA)
|
|
88
|
+
- Summary statistics, Correlation analysis, Distribution analysis.
|
|
89
|
+
- Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
|
|
90
|
+
|
|
91
|
+
### 4. Feature Engineering
|
|
92
|
+
- Categorical encoding (One-Hot, Label).
|
|
93
|
+
- Feature scaling (Standard, MinMax).
|
|
94
|
+
- Datetime feature extraction.
|
|
95
|
+
- Polynomial and interaction features.
|
|
96
|
+
- Handling class imbalance (SMOTE).
|
|
97
|
+
|
|
98
|
+
### 5. ML Model Training
|
|
99
|
+
- Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
|
|
100
|
+
- Reliability scoring, cross-validation, and metrics comparison.
|
|
101
|
+
|
|
102
|
+
## 🎯 Supported Problem Types
|
|
103
|
+
- **Classification**
|
|
104
|
+
- **Regression**
|
|
105
|
+
*(Clustering is a planned future feature)*
|
|
106
|
+
|
|
107
|
+
## 🌐 Optional Web Application
|
|
108
|
+
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
109
|
+
Install web dependencies:
|
|
110
|
+
```bash
|
|
111
|
+
pip install dataclean[web,api]
|
|
112
|
+
```
|
|
113
|
+
Run the web application:
|
|
114
|
+
```bash
|
|
115
|
+
python app.py
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
## 📝 License
|
|
119
|
+
MIT License
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
cleanflow_kit.egg-info/PKG-INFO
|
|
5
|
+
cleanflow_kit.egg-info/SOURCES.txt
|
|
6
|
+
cleanflow_kit.egg-info/dependency_links.txt
|
|
7
|
+
cleanflow_kit.egg-info/requires.txt
|
|
8
|
+
cleanflow_kit.egg-info/top_level.txt
|
|
9
|
+
dataclean/__init__.py
|
|
10
|
+
dataclean/_compat.py
|
|
11
|
+
dataclean/data_cleaner.py
|
|
12
|
+
dataclean/data_loader.py
|
|
13
|
+
dataclean/drift_detector.py
|
|
14
|
+
dataclean/eda.py
|
|
15
|
+
dataclean/feature_engineer.py
|
|
16
|
+
dataclean/model_trainer.py
|
|
17
|
+
dataclean/pipeline.py
|
|
18
|
+
dataclean/py.typed
|
|
19
|
+
dataclean/report_generator.py
|
|
20
|
+
dataclean/synthetic_generator.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
pandas>=2.0.0
|
|
2
|
+
numpy>=1.20.0
|
|
3
|
+
scikit-learn>=1.0.0
|
|
4
|
+
matplotlib>=3.5.0
|
|
5
|
+
joblib>=1.1.0
|
|
6
|
+
|
|
7
|
+
[all]
|
|
8
|
+
imbalanced-learn>=0.9.0
|
|
9
|
+
scipy>=1.7.0
|
|
10
|
+
|
|
11
|
+
[api]
|
|
12
|
+
FastAPI
|
|
13
|
+
uvicorn
|
|
14
|
+
pydantic
|
|
15
|
+
|
|
16
|
+
[drift]
|
|
17
|
+
scipy>=1.7.0
|
|
18
|
+
|
|
19
|
+
[smote]
|
|
20
|
+
imbalanced-learn>=0.9.0
|
|
21
|
+
|
|
22
|
+
[web]
|
|
23
|
+
Flask
|
|
24
|
+
Flask-SQLAlchemy
|
|
25
|
+
PyJWT
|
|
26
|
+
gunicorn
|
|
27
|
+
openpyxl
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
dataclean
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Pipeline Tool
|
|
3
|
+
==================
|
|
4
|
+
A comprehensive, modular Python tool for automated data cleaning,
|
|
5
|
+
exploratory data analysis (EDA), and feature engineering.
|
|
6
|
+
|
|
7
|
+
Modules:
|
|
8
|
+
- data_loader: Load and validate datasets
|
|
9
|
+
- data_cleaner: Clean and preprocess data
|
|
10
|
+
- eda: Exploratory data analysis and visualizations
|
|
11
|
+
- feature_engineer: Feature engineering and transformation
|
|
12
|
+
- pipeline: Main orchestrator combining all modules
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
from .data_loader import DataLoader
|
|
18
|
+
from .data_cleaner import DataCleaner
|
|
19
|
+
from .model_trainer import ModelTrainer
|
|
20
|
+
|
|
21
|
+
# Lazy imports to avoid loading matplotlib/seaborn/scipy at import time
|
|
22
|
+
# Use: from .eda import EDAAnalyzer
|
|
23
|
+
# Use: from .feature_engineer import FeatureEngineer
|
|
24
|
+
# Use: from .pipeline import DataPipeline
|
|
25
|
+
|
|
26
|
+
def _get_pipeline():
|
|
27
|
+
from .pipeline import DataPipeline as _DP
|
|
28
|
+
return _DP
|
|
29
|
+
|
|
30
|
+
# Make DataPipeline available but lazy
|
|
31
|
+
import importlib
|
|
32
|
+
def __getattr__(name):
|
|
33
|
+
if name == "DataPipeline":
|
|
34
|
+
return _get_pipeline()
|
|
35
|
+
if name == "EDAAnalyzer":
|
|
36
|
+
from .eda import EDAAnalyzer
|
|
37
|
+
return EDAAnalyzer
|
|
38
|
+
if name == "FeatureEngineer":
|
|
39
|
+
from .feature_engineer import FeatureEngineer
|
|
40
|
+
return FeatureEngineer
|
|
41
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
42
|
+
|
|
43
|
+
__version__ = "1.1.0"
|
|
44
|
+
__all__ = ["DataPipeline", "DataLoader", "DataCleaner", "EDAAnalyzer", "FeatureEngineer", "ModelTrainer"]
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Pandas dtype compatibility helpers.
|
|
2
|
+
|
|
3
|
+
Pandas 3 infers text columns as the dedicated ``str`` dtype. Unlike the
|
|
4
|
+
legacy ``object`` dtype, it rejects non-string assignments. The data-cleaning
|
|
5
|
+
and ML code intentionally converts some text columns to numeric values, so we
|
|
6
|
+
normalise inferred string columns to ``object`` at each public entry point.
|
|
7
|
+
|
|
8
|
+
This module intentionally does not import pandas. Importing it from ``app.py``
|
|
9
|
+
therefore preserves the app's lazy-loading behaviour for data libraries.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def normalize_string_columns(df):
|
|
14
|
+
"""Convert pandas string extension columns to object columns in-place.
|
|
15
|
+
|
|
16
|
+
The dtype names cover pandas 2.x's optional ``StringDtype`` and pandas
|
|
17
|
+
3.x's default ``str`` dtype. ``object`` is appropriate because a column
|
|
18
|
+
can later be transformed from text into numeric values.
|
|
19
|
+
"""
|
|
20
|
+
for column in df.columns:
|
|
21
|
+
series = df[column]
|
|
22
|
+
if getattr(series.dtype, "name", None) in {"str", "string", "String"}:
|
|
23
|
+
df[column] = series.astype(object)
|
|
24
|
+
return df
|