cleanflow-kit 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DataClean Contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,119 @@
1
+ Metadata-Version: 2.4
2
+ Name: cleanflow-kit
3
+ Version: 1.1.0
4
+ Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
5
+ Author: DataClean Contributors
6
+ License: MIT License
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.8
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Requires-Dist: pandas>=2.0.0
14
+ Requires-Dist: numpy>=1.20.0
15
+ Requires-Dist: scikit-learn>=1.0.0
16
+ Requires-Dist: matplotlib>=3.5.0
17
+ Requires-Dist: joblib>=1.1.0
18
+ Provides-Extra: smote
19
+ Requires-Dist: imbalanced-learn>=0.9.0; extra == "smote"
20
+ Provides-Extra: drift
21
+ Requires-Dist: scipy>=1.7.0; extra == "drift"
22
+ Provides-Extra: all
23
+ Requires-Dist: imbalanced-learn>=0.9.0; extra == "all"
24
+ Requires-Dist: scipy>=1.7.0; extra == "all"
25
+ Provides-Extra: web
26
+ Requires-Dist: Flask; extra == "web"
27
+ Requires-Dist: Flask-SQLAlchemy; extra == "web"
28
+ Requires-Dist: PyJWT; extra == "web"
29
+ Requires-Dist: gunicorn; extra == "web"
30
+ Requires-Dist: openpyxl; extra == "web"
31
+ Provides-Extra: api
32
+ Requires-Dist: FastAPI; extra == "api"
33
+ Requires-Dist: uvicorn; extra == "api"
34
+ Requires-Dist: pydantic; extra == "api"
35
+ Dynamic: license-file
36
+
37
+ # DataClean
38
+
39
+ A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
40
+
41
+ ## 🚀 Quick Start
42
+
43
+ ```python
44
+ from dataclean import DataPipeline
45
+
46
+ # Initialize and load data
47
+ pipeline = DataPipeline()
48
+ pipeline.load("data.csv")
49
+
50
+ # Run full pipeline
51
+ cleaned_df, final_df = pipeline.run_full_pipeline(
52
+ target_col="target_column",
53
+ problem_type="classification" # or "regression"
54
+ )
55
+
56
+ # Save results
57
+ pipeline.save_data("./output")
58
+ ```
59
+
60
+ ## 📦 Installation
61
+
62
+ Install via pip:
63
+
64
+ ```bash
65
+ pip install dataclean
66
+ ```
67
+
68
+ For optional features (SMOTE for class imbalance handling and Drift detection):
69
+ ```bash
70
+ pip install dataclean[all]
71
+ ```
72
+
73
+ ## 📋 Features
74
+
75
+ ### 1. Data Loading & Validation
76
+ - Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
77
+ - Detect duplicates, missing values, data types.
78
+ - Identify constant/near-constant and potential ID columns.
79
+
80
+ ### 2. Data Cleaning
81
+ - Remove duplicate rows.
82
+ - Handle missing values (mean/median/mode/drop).
83
+ - Fix incorrect data types.
84
+ - Handle outliers (IQR/Z-score).
85
+ - Clean categorical values (whitespace, case).
86
+
87
+ ### 3. Exploratory Data Analysis (EDA)
88
+ - Summary statistics, Correlation analysis, Distribution analysis.
89
+ - Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
90
+
91
+ ### 4. Feature Engineering
92
+ - Categorical encoding (One-Hot, Label).
93
+ - Feature scaling (Standard, MinMax).
94
+ - Datetime feature extraction.
95
+ - Polynomial and interaction features.
96
+ - Handling class imbalance (SMOTE).
97
+
98
+ ### 5. ML Model Training
99
+ - Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
100
+ - Reliability scoring, cross-validation, and metrics comparison.
101
+
102
+ ## 🎯 Supported Problem Types
103
+ - **Classification**
104
+ - **Regression**
105
+ *(Clustering is a planned future feature)*
106
+
107
+ ## 🌐 Optional Web Application
108
+ The repository includes an optional Flask web application to interact with the pipeline via a browser.
109
+ Install web dependencies:
110
+ ```bash
111
+ pip install dataclean[web,api]
112
+ ```
113
+ Run the web application:
114
+ ```bash
115
+ python app.py
116
+ ```
117
+
118
+ ## 📝 License
119
+ MIT License
@@ -0,0 +1,83 @@
1
+ # DataClean
2
+
3
+ A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
4
+
5
+ ## 🚀 Quick Start
6
+
7
+ ```python
8
+ from dataclean import DataPipeline
9
+
10
+ # Initialize and load data
11
+ pipeline = DataPipeline()
12
+ pipeline.load("data.csv")
13
+
14
+ # Run full pipeline
15
+ cleaned_df, final_df = pipeline.run_full_pipeline(
16
+ target_col="target_column",
17
+ problem_type="classification" # or "regression"
18
+ )
19
+
20
+ # Save results
21
+ pipeline.save_data("./output")
22
+ ```
23
+
24
+ ## 📦 Installation
25
+
26
+ Install via pip:
27
+
28
+ ```bash
29
+ pip install dataclean
30
+ ```
31
+
32
+ For optional features (SMOTE for class imbalance handling and Drift detection):
33
+ ```bash
34
+ pip install dataclean[all]
35
+ ```
36
+
37
+ ## 📋 Features
38
+
39
+ ### 1. Data Loading & Validation
40
+ - Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
41
+ - Detect duplicates, missing values, data types.
42
+ - Identify constant/near-constant and potential ID columns.
43
+
44
+ ### 2. Data Cleaning
45
+ - Remove duplicate rows.
46
+ - Handle missing values (mean/median/mode/drop).
47
+ - Fix incorrect data types.
48
+ - Handle outliers (IQR/Z-score).
49
+ - Clean categorical values (whitespace, case).
50
+
51
+ ### 3. Exploratory Data Analysis (EDA)
52
+ - Summary statistics, Correlation analysis, Distribution analysis.
53
+ - Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
54
+
55
+ ### 4. Feature Engineering
56
+ - Categorical encoding (One-Hot, Label).
57
+ - Feature scaling (Standard, MinMax).
58
+ - Datetime feature extraction.
59
+ - Polynomial and interaction features.
60
+ - Handling class imbalance (SMOTE).
61
+
62
+ ### 5. ML Model Training
63
+ - Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
64
+ - Reliability scoring, cross-validation, and metrics comparison.
65
+
66
+ ## 🎯 Supported Problem Types
67
+ - **Classification**
68
+ - **Regression**
69
+ *(Clustering is a planned future feature)*
70
+
71
+ ## 🌐 Optional Web Application
72
+ The repository includes an optional Flask web application to interact with the pipeline via a browser.
73
+ Install web dependencies:
74
+ ```bash
75
+ pip install dataclean[web,api]
76
+ ```
77
+ Run the web application:
78
+ ```bash
79
+ python app.py
80
+ ```
81
+
82
+ ## 📝 License
83
+ MIT License
@@ -0,0 +1,119 @@
1
+ Metadata-Version: 2.4
2
+ Name: cleanflow-kit
3
+ Version: 1.1.0
4
+ Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
5
+ Author: DataClean Contributors
6
+ License: MIT License
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.8
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Requires-Dist: pandas>=2.0.0
14
+ Requires-Dist: numpy>=1.20.0
15
+ Requires-Dist: scikit-learn>=1.0.0
16
+ Requires-Dist: matplotlib>=3.5.0
17
+ Requires-Dist: joblib>=1.1.0
18
+ Provides-Extra: smote
19
+ Requires-Dist: imbalanced-learn>=0.9.0; extra == "smote"
20
+ Provides-Extra: drift
21
+ Requires-Dist: scipy>=1.7.0; extra == "drift"
22
+ Provides-Extra: all
23
+ Requires-Dist: imbalanced-learn>=0.9.0; extra == "all"
24
+ Requires-Dist: scipy>=1.7.0; extra == "all"
25
+ Provides-Extra: web
26
+ Requires-Dist: Flask; extra == "web"
27
+ Requires-Dist: Flask-SQLAlchemy; extra == "web"
28
+ Requires-Dist: PyJWT; extra == "web"
29
+ Requires-Dist: gunicorn; extra == "web"
30
+ Requires-Dist: openpyxl; extra == "web"
31
+ Provides-Extra: api
32
+ Requires-Dist: FastAPI; extra == "api"
33
+ Requires-Dist: uvicorn; extra == "api"
34
+ Requires-Dist: pydantic; extra == "api"
35
+ Dynamic: license-file
36
+
37
+ # DataClean
38
+
39
+ A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
40
+
41
+ ## 🚀 Quick Start
42
+
43
+ ```python
44
+ from dataclean import DataPipeline
45
+
46
+ # Initialize and load data
47
+ pipeline = DataPipeline()
48
+ pipeline.load("data.csv")
49
+
50
+ # Run full pipeline
51
+ cleaned_df, final_df = pipeline.run_full_pipeline(
52
+ target_col="target_column",
53
+ problem_type="classification" # or "regression"
54
+ )
55
+
56
+ # Save results
57
+ pipeline.save_data("./output")
58
+ ```
59
+
60
+ ## 📦 Installation
61
+
62
+ Install via pip:
63
+
64
+ ```bash
65
+ pip install dataclean
66
+ ```
67
+
68
+ For optional features (SMOTE for class imbalance handling and Drift detection):
69
+ ```bash
70
+ pip install dataclean[all]
71
+ ```
72
+
73
+ ## 📋 Features
74
+
75
+ ### 1. Data Loading & Validation
76
+ - Load CSV, Excel, JSON, Parquet files or pandas DataFrames directly.
77
+ - Detect duplicates, missing values, data types.
78
+ - Identify constant/near-constant and potential ID columns.
79
+
80
+ ### 2. Data Cleaning
81
+ - Remove duplicate rows.
82
+ - Handle missing values (mean/median/mode/drop).
83
+ - Fix incorrect data types.
84
+ - Handle outliers (IQR/Z-score).
85
+ - Clean categorical values (whitespace, case).
86
+
87
+ ### 3. Exploratory Data Analysis (EDA)
88
+ - Summary statistics, Correlation analysis, Distribution analysis.
89
+ - Visualizations: Histograms, Boxplots, Heatmaps, Feature-target scatter plots.
90
+
91
+ ### 4. Feature Engineering
92
+ - Categorical encoding (One-Hot, Label).
93
+ - Feature scaling (Standard, MinMax).
94
+ - Datetime feature extraction.
95
+ - Polynomial and interaction features.
96
+ - Handling class imbalance (SMOTE).
97
+
98
+ ### 5. ML Model Training
99
+ - Auto-train classification (RandomForest, LogisticRegression) or regression models (RandomForest, Ridge).
100
+ - Reliability scoring, cross-validation, and metrics comparison.
101
+
102
+ ## 🎯 Supported Problem Types
103
+ - **Classification**
104
+ - **Regression**
105
+ *(Clustering is a planned future feature)*
106
+
107
+ ## 🌐 Optional Web Application
108
+ The repository includes an optional Flask web application to interact with the pipeline via a browser.
109
+ Install web dependencies:
110
+ ```bash
111
+ pip install dataclean[web,api]
112
+ ```
113
+ Run the web application:
114
+ ```bash
115
+ python app.py
116
+ ```
117
+
118
+ ## 📝 License
119
+ MIT License
@@ -0,0 +1,20 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ cleanflow_kit.egg-info/PKG-INFO
5
+ cleanflow_kit.egg-info/SOURCES.txt
6
+ cleanflow_kit.egg-info/dependency_links.txt
7
+ cleanflow_kit.egg-info/requires.txt
8
+ cleanflow_kit.egg-info/top_level.txt
9
+ dataclean/__init__.py
10
+ dataclean/_compat.py
11
+ dataclean/data_cleaner.py
12
+ dataclean/data_loader.py
13
+ dataclean/drift_detector.py
14
+ dataclean/eda.py
15
+ dataclean/feature_engineer.py
16
+ dataclean/model_trainer.py
17
+ dataclean/pipeline.py
18
+ dataclean/py.typed
19
+ dataclean/report_generator.py
20
+ dataclean/synthetic_generator.py
@@ -0,0 +1,27 @@
1
+ pandas>=2.0.0
2
+ numpy>=1.20.0
3
+ scikit-learn>=1.0.0
4
+ matplotlib>=3.5.0
5
+ joblib>=1.1.0
6
+
7
+ [all]
8
+ imbalanced-learn>=0.9.0
9
+ scipy>=1.7.0
10
+
11
+ [api]
12
+ FastAPI
13
+ uvicorn
14
+ pydantic
15
+
16
+ [drift]
17
+ scipy>=1.7.0
18
+
19
+ [smote]
20
+ imbalanced-learn>=0.9.0
21
+
22
+ [web]
23
+ Flask
24
+ Flask-SQLAlchemy
25
+ PyJWT
26
+ gunicorn
27
+ openpyxl
@@ -0,0 +1 @@
1
+ dataclean
@@ -0,0 +1,44 @@
1
+ """
2
+ Data Pipeline Tool
3
+ ==================
4
+ A comprehensive, modular Python tool for automated data cleaning,
5
+ exploratory data analysis (EDA), and feature engineering.
6
+
7
+ Modules:
8
+ - data_loader: Load and validate datasets
9
+ - data_cleaner: Clean and preprocess data
10
+ - eda: Exploratory data analysis and visualizations
11
+ - feature_engineer: Feature engineering and transformation
12
+ - pipeline: Main orchestrator combining all modules
13
+ """
14
+
15
+ import logging
16
+ logger = logging.getLogger(__name__)
17
+ from .data_loader import DataLoader
18
+ from .data_cleaner import DataCleaner
19
+ from .model_trainer import ModelTrainer
20
+
21
+ # Lazy imports to avoid loading matplotlib/seaborn/scipy at import time
22
+ # Use: from .eda import EDAAnalyzer
23
+ # Use: from .feature_engineer import FeatureEngineer
24
+ # Use: from .pipeline import DataPipeline
25
+
26
+ def _get_pipeline():
27
+ from .pipeline import DataPipeline as _DP
28
+ return _DP
29
+
30
+ # Make DataPipeline available but lazy
31
+ import importlib
32
+ def __getattr__(name):
33
+ if name == "DataPipeline":
34
+ return _get_pipeline()
35
+ if name == "EDAAnalyzer":
36
+ from .eda import EDAAnalyzer
37
+ return EDAAnalyzer
38
+ if name == "FeatureEngineer":
39
+ from .feature_engineer import FeatureEngineer
40
+ return FeatureEngineer
41
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
42
+
43
+ __version__ = "1.1.0"
44
+ __all__ = ["DataPipeline", "DataLoader", "DataCleaner", "EDAAnalyzer", "FeatureEngineer", "ModelTrainer"]
@@ -0,0 +1,24 @@
1
+ """Pandas dtype compatibility helpers.
2
+
3
+ Pandas 3 infers text columns as the dedicated ``str`` dtype. Unlike the
4
+ legacy ``object`` dtype, it rejects non-string assignments. The data-cleaning
5
+ and ML code intentionally converts some text columns to numeric values, so we
6
+ normalise inferred string columns to ``object`` at each public entry point.
7
+
8
+ This module intentionally does not import pandas. Importing it from ``app.py``
9
+ therefore preserves the app's lazy-loading behaviour for data libraries.
10
+ """
11
+
12
+
13
+ def normalize_string_columns(df):
14
+ """Convert pandas string extension columns to object columns in-place.
15
+
16
+ The dtype names cover pandas 2.x's optional ``StringDtype`` and pandas
17
+ 3.x's default ``str`` dtype. ``object`` is appropriate because a column
18
+ can later be transformed from text into numeric values.
19
+ """
20
+ for column in df.columns:
21
+ series = df[column]
22
+ if getattr(series.dtype, "name", None) in {"str", "string", "String"}:
23
+ df[column] = series.astype(object)
24
+ return df