cleanflow-kit 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/PKG-INFO +7 -7
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/README.md +5 -5
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/__init__.py +2 -2
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/data_cleaner.py +12 -12
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/model_trainer.py +1 -1
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/pipeline.py +4 -4
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/cleanflow_kit.egg-info/PKG-INFO +7 -7
- cleanflow_kit-1.2.0/cleanflow_kit.egg-info/SOURCES.txt +20 -0
- cleanflow_kit-1.2.0/cleanflow_kit.egg-info/top_level.txt +1 -0
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/pyproject.toml +3 -3
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/SOURCES.txt +0 -20
- cleanflow_kit-1.1.0/cleanflow_kit.egg-info/top_level.txt +0 -1
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/LICENSE +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/_compat.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/data_loader.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/drift_detector.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/eda.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/feature_engineer.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/py.typed +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/report_generator.py +0 -0
- {cleanflow_kit-1.1.0/dataclean → cleanflow_kit-1.2.0/cleanflow}/synthetic_generator.py +0 -0
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/cleanflow_kit.egg-info/dependency_links.txt +0 -0
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/cleanflow_kit.egg-info/requires.txt +0 -0
- {cleanflow_kit-1.1.0 → cleanflow_kit-1.2.0}/setup.cfg +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cleanflow-kit
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
|
|
5
|
-
Author:
|
|
5
|
+
Author: CleanFlow Contributors
|
|
6
6
|
License: MIT License
|
|
7
7
|
Classifier: Programming Language :: Python :: 3
|
|
8
8
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -34,14 +34,14 @@ Requires-Dist: uvicorn; extra == "api"
|
|
|
34
34
|
Requires-Dist: pydantic; extra == "api"
|
|
35
35
|
Dynamic: license-file
|
|
36
36
|
|
|
37
|
-
#
|
|
37
|
+
# CleanFlow
|
|
38
38
|
|
|
39
39
|
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
40
40
|
|
|
41
41
|
## 🚀 Quick Start
|
|
42
42
|
|
|
43
43
|
```python
|
|
44
|
-
from
|
|
44
|
+
from cleanflow import DataPipeline
|
|
45
45
|
|
|
46
46
|
# Initialize and load data
|
|
47
47
|
pipeline = DataPipeline()
|
|
@@ -62,12 +62,12 @@ pipeline.save_data("./output")
|
|
|
62
62
|
Install via pip:
|
|
63
63
|
|
|
64
64
|
```bash
|
|
65
|
-
pip install
|
|
65
|
+
pip install cleanflow
|
|
66
66
|
```
|
|
67
67
|
|
|
68
68
|
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
69
69
|
```bash
|
|
70
|
-
pip install
|
|
70
|
+
pip install cleanflow[all]
|
|
71
71
|
```
|
|
72
72
|
|
|
73
73
|
## 📋 Features
|
|
@@ -108,7 +108,7 @@ pip install dataclean[all]
|
|
|
108
108
|
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
109
109
|
Install web dependencies:
|
|
110
110
|
```bash
|
|
111
|
-
pip install
|
|
111
|
+
pip install cleanflow[web,api]
|
|
112
112
|
```
|
|
113
113
|
Run the web application:
|
|
114
114
|
```bash
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
#
|
|
1
|
+
# CleanFlow
|
|
2
2
|
|
|
3
3
|
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
4
4
|
|
|
5
5
|
## 🚀 Quick Start
|
|
6
6
|
|
|
7
7
|
```python
|
|
8
|
-
from
|
|
8
|
+
from cleanflow import DataPipeline
|
|
9
9
|
|
|
10
10
|
# Initialize and load data
|
|
11
11
|
pipeline = DataPipeline()
|
|
@@ -26,12 +26,12 @@ pipeline.save_data("./output")
|
|
|
26
26
|
Install via pip:
|
|
27
27
|
|
|
28
28
|
```bash
|
|
29
|
-
pip install
|
|
29
|
+
pip install cleanflow
|
|
30
30
|
```
|
|
31
31
|
|
|
32
32
|
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
33
33
|
```bash
|
|
34
|
-
pip install
|
|
34
|
+
pip install cleanflow[all]
|
|
35
35
|
```
|
|
36
36
|
|
|
37
37
|
## 📋 Features
|
|
@@ -72,7 +72,7 @@ pip install dataclean[all]
|
|
|
72
72
|
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
73
73
|
Install web dependencies:
|
|
74
74
|
```bash
|
|
75
|
-
pip install
|
|
75
|
+
pip install cleanflow[web,api]
|
|
76
76
|
```
|
|
77
77
|
Run the web application:
|
|
78
78
|
```bash
|
|
@@ -15,7 +15,7 @@ Modules:
|
|
|
15
15
|
import logging
|
|
16
16
|
logger = logging.getLogger(__name__)
|
|
17
17
|
from .data_loader import DataLoader
|
|
18
|
-
from .data_cleaner import
|
|
18
|
+
from .data_cleaner import CleanFlower
|
|
19
19
|
from .model_trainer import ModelTrainer
|
|
20
20
|
|
|
21
21
|
# Lazy imports to avoid loading matplotlib/seaborn/scipy at import time
|
|
@@ -41,4 +41,4 @@ def __getattr__(name):
|
|
|
41
41
|
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
42
42
|
|
|
43
43
|
__version__ = "1.1.0"
|
|
44
|
-
__all__ = ["DataPipeline", "DataLoader", "
|
|
44
|
+
__all__ = ["DataPipeline", "DataLoader", "CleanFlower", "EDAAnalyzer", "FeatureEngineer", "ModelTrainer"]
|
|
@@ -16,7 +16,7 @@ import re
|
|
|
16
16
|
from ._compat import normalize_string_columns
|
|
17
17
|
|
|
18
18
|
|
|
19
|
-
class
|
|
19
|
+
class CleanFlower:
|
|
20
20
|
"""Clean and preprocess datasets."""
|
|
21
21
|
|
|
22
22
|
def __init__(self, df: pd.DataFrame):
|
|
@@ -45,7 +45,7 @@ class DataCleaner:
|
|
|
45
45
|
self,
|
|
46
46
|
subset: Optional[List[str]] = None,
|
|
47
47
|
keep: str = 'first'
|
|
48
|
-
) -> '
|
|
48
|
+
) -> 'CleanFlower':
|
|
49
49
|
"""
|
|
50
50
|
Remove duplicate rows.
|
|
51
51
|
|
|
@@ -93,7 +93,7 @@ class DataCleaner:
|
|
|
93
93
|
categorical_strategy: str = 'mode',
|
|
94
94
|
drop_threshold: float = 0.4,
|
|
95
95
|
fill_value: Optional[Any] = None
|
|
96
|
-
) -> '
|
|
96
|
+
) -> 'CleanFlower':
|
|
97
97
|
"""
|
|
98
98
|
Handle missing values in the dataset.
|
|
99
99
|
|
|
@@ -224,7 +224,7 @@ class DataCleaner:
|
|
|
224
224
|
self,
|
|
225
225
|
type_mapping: Optional[Dict[str, str]] = None,
|
|
226
226
|
infer_types: bool = True
|
|
227
|
-
) -> '
|
|
227
|
+
) -> 'CleanFlower':
|
|
228
228
|
"""
|
|
229
229
|
Fix and optimize data types.
|
|
230
230
|
|
|
@@ -292,7 +292,7 @@ class DataCleaner:
|
|
|
292
292
|
columns: Optional[List[str]] = None,
|
|
293
293
|
threshold: float = 1.5,
|
|
294
294
|
action: str = 'clip'
|
|
295
|
-
) -> '
|
|
295
|
+
) -> 'CleanFlower':
|
|
296
296
|
"""
|
|
297
297
|
Detect and handle outliers in numeric columns.
|
|
298
298
|
|
|
@@ -393,7 +393,7 @@ class DataCleaner:
|
|
|
393
393
|
lowercase: bool = True,
|
|
394
394
|
strip_whitespace: bool = True,
|
|
395
395
|
replace_mapping: Optional[Dict[str, Dict[str, str]]] = None
|
|
396
|
-
) -> '
|
|
396
|
+
) -> 'CleanFlower':
|
|
397
397
|
"""
|
|
398
398
|
Clean and standardize categorical values.
|
|
399
399
|
|
|
@@ -467,7 +467,7 @@ class DataCleaner:
|
|
|
467
467
|
self,
|
|
468
468
|
method: str = 'standard',
|
|
469
469
|
columns: Optional[List[str]] = None
|
|
470
|
-
) -> '
|
|
470
|
+
) -> 'CleanFlower':
|
|
471
471
|
"""
|
|
472
472
|
Scale numerical features.
|
|
473
473
|
|
|
@@ -518,7 +518,7 @@ class DataCleaner:
|
|
|
518
518
|
method: str = 'onehot',
|
|
519
519
|
columns: Optional[List[str]] = None,
|
|
520
520
|
max_categories: int = 20
|
|
521
|
-
) -> '
|
|
521
|
+
) -> 'CleanFlower':
|
|
522
522
|
"""
|
|
523
523
|
Encode categorical features.
|
|
524
524
|
|
|
@@ -585,7 +585,7 @@ class DataCleaner:
|
|
|
585
585
|
columns: Optional[List[str]] = None,
|
|
586
586
|
remove_symbols: bool = True,
|
|
587
587
|
handle_shorthand: bool = True
|
|
588
|
-
) -> '
|
|
588
|
+
) -> 'CleanFlower':
|
|
589
589
|
"""
|
|
590
590
|
Clean text columns containing numbers (e.g. '$1,200', '1.5k').
|
|
591
591
|
|
|
@@ -707,7 +707,7 @@ class DataCleaner:
|
|
|
707
707
|
def rename_columns(
|
|
708
708
|
self,
|
|
709
709
|
mapping: Dict[str, str]
|
|
710
|
-
) -> '
|
|
710
|
+
) -> 'CleanFlower':
|
|
711
711
|
"""
|
|
712
712
|
Rename columns.
|
|
713
713
|
|
|
@@ -735,7 +735,7 @@ class DataCleaner:
|
|
|
735
735
|
source_col: str,
|
|
736
736
|
pattern: str,
|
|
737
737
|
new_col_name: str
|
|
738
|
-
) -> '
|
|
738
|
+
) -> 'CleanFlower':
|
|
739
739
|
r"""
|
|
740
740
|
Extract text using regex capture group.
|
|
741
741
|
|
|
@@ -778,7 +778,7 @@ class DataCleaner:
|
|
|
778
778
|
columns: Optional[List[str]] = None,
|
|
779
779
|
drop_constant: bool = True,
|
|
780
780
|
drop_id_like: bool = False
|
|
781
|
-
) -> '
|
|
781
|
+
) -> 'CleanFlower':
|
|
782
782
|
"""
|
|
783
783
|
Drop specified or problematic columns.
|
|
784
784
|
|
|
@@ -18,7 +18,7 @@ import warnings
|
|
|
18
18
|
from datetime import datetime
|
|
19
19
|
from typing import Optional, Dict, Any, List, Tuple
|
|
20
20
|
from .feature_engineer import FeatureEngineer
|
|
21
|
-
from .data_cleaner import
|
|
21
|
+
from .data_cleaner import CleanFlower
|
|
22
22
|
|
|
23
23
|
from sklearn.model_selection import (
|
|
24
24
|
cross_validate, StratifiedKFold, KFold, RandomizedSearchCV
|
|
@@ -16,7 +16,7 @@ from typing import Optional, Dict, Any, List, Tuple
|
|
|
16
16
|
from pathlib import Path
|
|
17
17
|
|
|
18
18
|
from .data_loader import DataLoader
|
|
19
|
-
from .data_cleaner import
|
|
19
|
+
from .data_cleaner import CleanFlower
|
|
20
20
|
# EDAAnalyzer and FeatureEngineer are imported lazily inside methods
|
|
21
21
|
# to avoid loading matplotlib/seaborn/scipy at module import time
|
|
22
22
|
from .model_trainer import ModelTrainer
|
|
@@ -42,7 +42,7 @@ class DataPipeline:
|
|
|
42
42
|
def __init__(self):
|
|
43
43
|
"""Initialize the data pipeline."""
|
|
44
44
|
self.loader: Optional[DataLoader] = None
|
|
45
|
-
self.cleaner: Optional[
|
|
45
|
+
self.cleaner: Optional[CleanFlower] = None
|
|
46
46
|
self.eda: Optional['EDAAnalyzer'] = None
|
|
47
47
|
self.engineer: Optional['FeatureEngineer'] = None
|
|
48
48
|
self.trainer: Optional[ModelTrainer] = None
|
|
@@ -115,12 +115,12 @@ class DataPipeline:
|
|
|
115
115
|
|
|
116
116
|
Parameters:
|
|
117
117
|
-----------
|
|
118
|
-
(See
|
|
118
|
+
(See CleanFlower for parameter descriptions)
|
|
119
119
|
"""
|
|
120
120
|
if self.raw_df is None:
|
|
121
121
|
raise ValueError("No data loaded. Call load() first.")
|
|
122
122
|
|
|
123
|
-
self.cleaner =
|
|
123
|
+
self.cleaner = CleanFlower(self.raw_df)
|
|
124
124
|
|
|
125
125
|
if remove_duplicates:
|
|
126
126
|
self.cleaner.remove_duplicates()
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cleanflow-kit
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering.
|
|
5
|
-
Author:
|
|
5
|
+
Author: CleanFlow Contributors
|
|
6
6
|
License: MIT License
|
|
7
7
|
Classifier: Programming Language :: Python :: 3
|
|
8
8
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -34,14 +34,14 @@ Requires-Dist: uvicorn; extra == "api"
|
|
|
34
34
|
Requires-Dist: pydantic; extra == "api"
|
|
35
35
|
Dynamic: license-file
|
|
36
36
|
|
|
37
|
-
#
|
|
37
|
+
# CleanFlow
|
|
38
38
|
|
|
39
39
|
A comprehensive, modular Python tool for automated **data cleaning**, **exploratory data analysis (EDA)**, and **feature engineering**.
|
|
40
40
|
|
|
41
41
|
## 🚀 Quick Start
|
|
42
42
|
|
|
43
43
|
```python
|
|
44
|
-
from
|
|
44
|
+
from cleanflow import DataPipeline
|
|
45
45
|
|
|
46
46
|
# Initialize and load data
|
|
47
47
|
pipeline = DataPipeline()
|
|
@@ -62,12 +62,12 @@ pipeline.save_data("./output")
|
|
|
62
62
|
Install via pip:
|
|
63
63
|
|
|
64
64
|
```bash
|
|
65
|
-
pip install
|
|
65
|
+
pip install cleanflow
|
|
66
66
|
```
|
|
67
67
|
|
|
68
68
|
For optional features (SMOTE for class imbalance handling and Drift detection):
|
|
69
69
|
```bash
|
|
70
|
-
pip install
|
|
70
|
+
pip install cleanflow[all]
|
|
71
71
|
```
|
|
72
72
|
|
|
73
73
|
## 📋 Features
|
|
@@ -108,7 +108,7 @@ pip install dataclean[all]
|
|
|
108
108
|
The repository includes an optional Flask web application to interact with the pipeline via a browser.
|
|
109
109
|
Install web dependencies:
|
|
110
110
|
```bash
|
|
111
|
-
pip install
|
|
111
|
+
pip install cleanflow[web,api]
|
|
112
112
|
```
|
|
113
113
|
Run the web application:
|
|
114
114
|
```bash
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
cleanflow/__init__.py
|
|
5
|
+
cleanflow/_compat.py
|
|
6
|
+
cleanflow/data_cleaner.py
|
|
7
|
+
cleanflow/data_loader.py
|
|
8
|
+
cleanflow/drift_detector.py
|
|
9
|
+
cleanflow/eda.py
|
|
10
|
+
cleanflow/feature_engineer.py
|
|
11
|
+
cleanflow/model_trainer.py
|
|
12
|
+
cleanflow/pipeline.py
|
|
13
|
+
cleanflow/py.typed
|
|
14
|
+
cleanflow/report_generator.py
|
|
15
|
+
cleanflow/synthetic_generator.py
|
|
16
|
+
cleanflow_kit.egg-info/PKG-INFO
|
|
17
|
+
cleanflow_kit.egg-info/SOURCES.txt
|
|
18
|
+
cleanflow_kit.egg-info/dependency_links.txt
|
|
19
|
+
cleanflow_kit.egg-info/requires.txt
|
|
20
|
+
cleanflow_kit.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
cleanflow
|
|
@@ -4,11 +4,11 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "cleanflow-kit"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.2.0"
|
|
8
8
|
description = "A comprehensive, modular Python tool for automated data cleaning, EDA, and feature engineering."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
11
|
-
{ name = "
|
|
11
|
+
{ name = "CleanFlow Contributors" }
|
|
12
12
|
]
|
|
13
13
|
license = { text = "MIT License" }
|
|
14
14
|
requires-python = ">=3.8"
|
|
@@ -33,7 +33,7 @@ web = ["Flask", "Flask-SQLAlchemy", "PyJWT", "gunicorn", "openpyxl"]
|
|
|
33
33
|
api = ["FastAPI", "uvicorn", "pydantic"]
|
|
34
34
|
|
|
35
35
|
[tool.setuptools.packages.find]
|
|
36
|
-
include = ["
|
|
36
|
+
include = ["cleanflow*"]
|
|
37
37
|
|
|
38
38
|
[tool.pytest.ini_options]
|
|
39
39
|
minversion = "6.0"
|
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
LICENSE
|
|
2
|
-
README.md
|
|
3
|
-
pyproject.toml
|
|
4
|
-
cleanflow_kit.egg-info/PKG-INFO
|
|
5
|
-
cleanflow_kit.egg-info/SOURCES.txt
|
|
6
|
-
cleanflow_kit.egg-info/dependency_links.txt
|
|
7
|
-
cleanflow_kit.egg-info/requires.txt
|
|
8
|
-
cleanflow_kit.egg-info/top_level.txt
|
|
9
|
-
dataclean/__init__.py
|
|
10
|
-
dataclean/_compat.py
|
|
11
|
-
dataclean/data_cleaner.py
|
|
12
|
-
dataclean/data_loader.py
|
|
13
|
-
dataclean/drift_detector.py
|
|
14
|
-
dataclean/eda.py
|
|
15
|
-
dataclean/feature_engineer.py
|
|
16
|
-
dataclean/model_trainer.py
|
|
17
|
-
dataclean/pipeline.py
|
|
18
|
-
dataclean/py.typed
|
|
19
|
-
dataclean/report_generator.py
|
|
20
|
-
dataclean/synthetic_generator.py
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
dataclean
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|