easyclassifier 0.8.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- easyclassifier/__init__.py +23 -0
- easyclassifier/__main__.py +61 -0
- easyclassifier/dataset.py +142 -0
- easyclassifier/demo_data.py +32 -0
- easyclassifier/diagnostics.py +95 -0
- easyclassifier/distances.py +152 -0
- easyclassifier/evaluation.py +340 -0
- easyclassifier/figures.py +510 -0
- easyclassifier/help_texts.py +113 -0
- easyclassifier/importance.py +79 -0
- easyclassifier/latex_report.py +555 -0
- easyclassifier/logbook.py +31 -0
- easyclassifier/models.py +94 -0
- easyclassifier/preprocessing.py +219 -0
- easyclassifier/recommend.py +128 -0
- easyclassifier/reporting.py +70 -0
- easyclassifier/target.py +202 -0
- easyclassifier/ui.py +281 -0
- easyclassifier/wizard.py +1266 -0
- easyclassifier-0.8.1.dist-info/METADATA +267 -0
- easyclassifier-0.8.1.dist-info/RECORD +25 -0
- easyclassifier-0.8.1.dist-info/WHEEL +5 -0
- easyclassifier-0.8.1.dist-info/entry_points.txt +2 -0
- easyclassifier-0.8.1.dist-info/licenses/LICENSE +21 -0
- easyclassifier-0.8.1.dist-info/top_level.txt +1 -0
easyclassifier/models.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Classifier registry (Phase 10 & 23).
|
|
2
|
+
|
|
3
|
+
Each classifier is registered with a key, a friendly name, and a factory. New
|
|
4
|
+
algorithms can be added here without touching the wizard. Optional
|
|
5
|
+
dependencies (XGBoost, LightGBM) are detected gracefully.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Callable, Dict, List
|
|
12
|
+
|
|
13
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
14
|
+
from sklearn.linear_model import LogisticRegression
|
|
15
|
+
from sklearn.naive_bayes import GaussianNB
|
|
16
|
+
from sklearn.neural_network import MLPClassifier
|
|
17
|
+
from sklearn.svm import SVC
|
|
18
|
+
from sklearn.tree import DecisionTreeClassifier
|
|
19
|
+
|
|
20
|
+
from .distances import make_knn
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class ClassifierSpec:
|
|
25
|
+
key: str
|
|
26
|
+
name: str
|
|
27
|
+
factory: Callable[[], object]
|
|
28
|
+
help_key: str
|
|
29
|
+
available: bool = True
|
|
30
|
+
reason: str = ""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# Every classifier that uses randomness gets a fixed seed, so the same data
|
|
34
|
+
# and EasyClassifier version always give exactly the same results.
|
|
35
|
+
SEED = 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _svc():
|
|
39
|
+
# probability=True lets us compute ROC AUC / PR curves.
|
|
40
|
+
return SVC(probability=True, random_state=SEED)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _xgb_factory():
|
|
44
|
+
from xgboost import XGBClassifier
|
|
45
|
+
return XGBClassifier(eval_metric="logloss", verbosity=0,
|
|
46
|
+
random_state=SEED)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _lgbm_factory():
|
|
50
|
+
from lightgbm import LGBMClassifier
|
|
51
|
+
return LGBMClassifier(verbose=-1, random_state=SEED)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _optional(key, name, factory, help_key, module) -> ClassifierSpec:
|
|
55
|
+
try:
|
|
56
|
+
__import__(module)
|
|
57
|
+
return ClassifierSpec(key, name, factory, help_key, True)
|
|
58
|
+
except Exception: # noqa: BLE001
|
|
59
|
+
return ClassifierSpec(
|
|
60
|
+
key, name, factory, help_key, False,
|
|
61
|
+
reason=f"install '{module}' to enable",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def build_registry() -> Dict[str, ClassifierSpec]:
|
|
66
|
+
specs = [
|
|
67
|
+
ClassifierSpec("decision_tree", "Decision Tree",
|
|
68
|
+
lambda: DecisionTreeClassifier(random_state=SEED),
|
|
69
|
+
"decision_tree"),
|
|
70
|
+
ClassifierSpec("random_forest", "Random Forest",
|
|
71
|
+
lambda: RandomForestClassifier(random_state=SEED),
|
|
72
|
+
"random_forest"),
|
|
73
|
+
ClassifierSpec("svm", "Support Vector Machine (SVM)",
|
|
74
|
+
_svc, "svm"),
|
|
75
|
+
ClassifierSpec("logistic_regression", "Logistic Regression",
|
|
76
|
+
lambda: LogisticRegression(max_iter=1000),
|
|
77
|
+
"logistic_regression"),
|
|
78
|
+
ClassifierSpec("knn", "K-Nearest Neighbours (KNN)",
|
|
79
|
+
lambda: make_knn("hassanat", 5), "knn"),
|
|
80
|
+
ClassifierSpec("naive_bayes", "Naive Bayes",
|
|
81
|
+
GaussianNB, "naive_bayes"),
|
|
82
|
+
_optional("xgboost", "XGBoost", _xgb_factory, "xgboost", "xgboost"),
|
|
83
|
+
_optional("lightgbm", "LightGBM", _lgbm_factory, "lightgbm",
|
|
84
|
+
"lightgbm"),
|
|
85
|
+
ClassifierSpec("neural_network", "Neural Network (MLP)",
|
|
86
|
+
lambda: MLPClassifier(max_iter=500,
|
|
87
|
+
random_state=SEED),
|
|
88
|
+
"neural_network"),
|
|
89
|
+
]
|
|
90
|
+
return {s.key: s for s in specs}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def available_specs(registry: Dict[str, ClassifierSpec]) -> List[ClassifierSpec]:
|
|
94
|
+
return list(registry.values())
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Data cleaning, encoding, and scaling - leakage-safe.
|
|
2
|
+
|
|
3
|
+
Two kinds of step are handled differently:
|
|
4
|
+
|
|
5
|
+
* Row-level steps that learn nothing from the data (removing duplicate rows,
|
|
6
|
+
removing rows with missing values, casting text columns to text) are applied
|
|
7
|
+
once, before any train/test split.
|
|
8
|
+
|
|
9
|
+
* Steps that *learn* something from the data (the mean/median used to fill
|
|
10
|
+
gaps, the scaling range, the list of categories) are placed inside a
|
|
11
|
+
scikit-learn Pipeline. During hold-out or cross-validation the pipeline is
|
|
12
|
+
re-fitted on each training part only, so no information from the test part
|
|
13
|
+
leaks into training.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from typing import Dict, List, Optional, Tuple
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
import pandas as pd
|
|
23
|
+
from sklearn.compose import ColumnTransformer
|
|
24
|
+
from sklearn.impute import SimpleImputer
|
|
25
|
+
from sklearn.pipeline import Pipeline
|
|
26
|
+
from sklearn.preprocessing import (
|
|
27
|
+
LabelEncoder,
|
|
28
|
+
MinMaxScaler,
|
|
29
|
+
OneHotEncoder,
|
|
30
|
+
OrdinalEncoder,
|
|
31
|
+
StandardScaler,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# --------------------------------------------------------------------------- #
|
|
36
|
+
# Row-level cleaning (no learned statistics -> safe before splitting)
|
|
37
|
+
# --------------------------------------------------------------------------- #
|
|
38
|
+
|
|
39
|
+
def drop_missing_target(df: pd.DataFrame, target: str) -> Tuple[pd.DataFrame, int]:
|
|
40
|
+
"""Rows without a class label cannot be used for training or testing."""
|
|
41
|
+
before = len(df)
|
|
42
|
+
df = df.dropna(subset=[target]).reset_index(drop=True)
|
|
43
|
+
return df, before - len(df)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def drop_missing_rows(df: pd.DataFrame) -> Tuple[pd.DataFrame, int]:
|
|
47
|
+
before = len(df)
|
|
48
|
+
df = df.dropna().reset_index(drop=True)
|
|
49
|
+
return df, before - len(df)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def duplicates_expected_by_chance(df: pd.DataFrame,
|
|
53
|
+
factor: float = 10.0) -> bool:
|
|
54
|
+
"""True if identical rows are expected between *different* cases.
|
|
55
|
+
|
|
56
|
+
With few columns that each take few values (e.g. sex, region, a 0-24
|
|
57
|
+
score), different people often give identical answers; such rows are
|
|
58
|
+
real observations and must be kept. Only when the columns allow far more
|
|
59
|
+
combinations than there are rows (at least ``factor`` times as many) is
|
|
60
|
+
an identical row likely to be an accidental copy.
|
|
61
|
+
"""
|
|
62
|
+
combos = 1.0
|
|
63
|
+
for c in df.columns:
|
|
64
|
+
combos *= max(1, df[c].nunique(dropna=False))
|
|
65
|
+
if combos >= factor * len(df):
|
|
66
|
+
return False
|
|
67
|
+
return True
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def remove_duplicates(df: pd.DataFrame) -> Tuple[pd.DataFrame, int]:
|
|
71
|
+
before = len(df)
|
|
72
|
+
df = df.drop_duplicates().reset_index(drop=True)
|
|
73
|
+
return df, before - len(df)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def categorical_columns(X: pd.DataFrame) -> List[str]:
|
|
77
|
+
return [c for c in X.columns if not pd.api.types.is_numeric_dtype(X[c])]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def numeric_columns(X: pd.DataFrame) -> List[str]:
|
|
81
|
+
return [c for c in X.columns if pd.api.types.is_numeric_dtype(X[c])]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def cast_categoricals(X: pd.DataFrame) -> pd.DataFrame:
|
|
85
|
+
"""Make every non-numeric column plain text (missing values stay missing).
|
|
86
|
+
|
|
87
|
+
This is a type conversion only; nothing is learned from the data.
|
|
88
|
+
"""
|
|
89
|
+
X = X.copy()
|
|
90
|
+
for c in categorical_columns(X):
|
|
91
|
+
col = X[c].astype(object)
|
|
92
|
+
X[c] = col.where(col.isna(), col.astype(str))
|
|
93
|
+
return X
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
# --------------------------------------------------------------------------- #
|
|
97
|
+
# Target encoding (a fixed label mapping - not a learned statistic)
|
|
98
|
+
# --------------------------------------------------------------------------- #
|
|
99
|
+
|
|
100
|
+
def encode_target(y: pd.Series) -> Tuple[np.ndarray, List[str], LabelEncoder]:
|
|
101
|
+
"""Class labels -> 0..k-1. Alphabetical order, except for ordered
|
|
102
|
+
categories (e.g. Low / Medium / High groups), which keep their order."""
|
|
103
|
+
le = LabelEncoder()
|
|
104
|
+
if isinstance(y.dtype, pd.CategoricalDtype) and y.cat.ordered:
|
|
105
|
+
present = [c for c in y.cat.categories if (y == c).any()]
|
|
106
|
+
le.classes_ = np.array([str(c) for c in present], dtype=object)
|
|
107
|
+
mapping = {str(c): i for i, c in enumerate(present)}
|
|
108
|
+
y_enc = y.astype(str).map(mapping).to_numpy(dtype=int)
|
|
109
|
+
return y_enc, [str(c) for c in present], le
|
|
110
|
+
y_enc = le.fit_transform(y.astype(str))
|
|
111
|
+
return y_enc, [str(c) for c in le.classes_], le
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# --------------------------------------------------------------------------- #
|
|
115
|
+
# Learned preprocessing -> goes inside the Pipeline
|
|
116
|
+
# --------------------------------------------------------------------------- #
|
|
117
|
+
|
|
118
|
+
@dataclass
|
|
119
|
+
class PrepConfig:
|
|
120
|
+
"""User choices for the learned preprocessing steps."""
|
|
121
|
+
|
|
122
|
+
impute: str = "median" # "mean" | "median" | "mode"
|
|
123
|
+
encoding: str = "auto" # "auto" | "label" | "onehot"
|
|
124
|
+
scale_method: str = "standard" # "standard" | "minmax"
|
|
125
|
+
scale: Optional[bool] = None # True / False / None = decide per classifier
|
|
126
|
+
onehot_max_categories: int = 10
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _onehot():
|
|
130
|
+
try:
|
|
131
|
+
return OneHotEncoder(handle_unknown="ignore", sparse_output=False)
|
|
132
|
+
except TypeError: # scikit-learn < 1.2
|
|
133
|
+
return OneHotEncoder(handle_unknown="ignore", sparse=False)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _ordinal():
|
|
137
|
+
return OrdinalEncoder(handle_unknown="use_encoded_value", unknown_value=-1)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def build_preprocessor(X: pd.DataFrame, cfg: PrepConfig,
|
|
141
|
+
scale) -> ColumnTransformer:
|
|
142
|
+
"""Build an *unfitted* transformer for the columns of ``X``.
|
|
143
|
+
|
|
144
|
+
``scale``: False/None (no scaling), True (``cfg.scale_method``), or the
|
|
145
|
+
method itself, "standard" or "minmax".
|
|
146
|
+
|
|
147
|
+
Only column names and types are read from ``X`` here; all statistics are
|
|
148
|
+
learned later, when the pipeline is fitted on training data.
|
|
149
|
+
"""
|
|
150
|
+
num_cols = numeric_columns(X)
|
|
151
|
+
cat_cols = categorical_columns(X)
|
|
152
|
+
num_strategy = "most_frequent" if cfg.impute == "mode" else cfg.impute
|
|
153
|
+
method = scale if isinstance(scale, str) else (
|
|
154
|
+
cfg.scale_method if scale else None)
|
|
155
|
+
|
|
156
|
+
transformers = []
|
|
157
|
+
if num_cols:
|
|
158
|
+
steps = [("impute", SimpleImputer(strategy=num_strategy))]
|
|
159
|
+
if method:
|
|
160
|
+
scaler = (MinMaxScaler() if method == "minmax"
|
|
161
|
+
else StandardScaler())
|
|
162
|
+
steps.append(("scale", scaler))
|
|
163
|
+
transformers.append(("num", Pipeline(steps), num_cols))
|
|
164
|
+
|
|
165
|
+
if cat_cols:
|
|
166
|
+
if cfg.encoding == "label":
|
|
167
|
+
onehot_cols, ordinal_cols = [], cat_cols
|
|
168
|
+
elif cfg.encoding == "onehot":
|
|
169
|
+
onehot_cols, ordinal_cols = cat_cols, []
|
|
170
|
+
else: # auto: one-hot for few categories, ordinal for many
|
|
171
|
+
onehot_cols = [c for c in cat_cols
|
|
172
|
+
if X[c].nunique(dropna=True)
|
|
173
|
+
<= cfg.onehot_max_categories]
|
|
174
|
+
ordinal_cols = [c for c in cat_cols if c not in onehot_cols]
|
|
175
|
+
impute = ("impute", SimpleImputer(strategy="most_frequent"))
|
|
176
|
+
if onehot_cols:
|
|
177
|
+
transformers.append(("cat_onehot", Pipeline(
|
|
178
|
+
[impute, ("encode", _onehot())]), onehot_cols))
|
|
179
|
+
if ordinal_cols:
|
|
180
|
+
transformers.append(("cat_ordinal", Pipeline(
|
|
181
|
+
[("impute", SimpleImputer(strategy="most_frequent")),
|
|
182
|
+
("encode", _ordinal())]), ordinal_cols))
|
|
183
|
+
|
|
184
|
+
return ColumnTransformer(transformers, remainder="drop")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def build_pipeline(X: pd.DataFrame, cfg: PrepConfig, classifier,
|
|
188
|
+
scale: bool) -> Pipeline:
|
|
189
|
+
"""Preprocessing + classifier, fitted together on training data only."""
|
|
190
|
+
return Pipeline([
|
|
191
|
+
("prep", build_preprocessor(X, cfg, scale)),
|
|
192
|
+
("model", classifier),
|
|
193
|
+
])
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def describe_transformed(X: pd.DataFrame, cfg: PrepConfig,
|
|
197
|
+
scale: bool) -> np.ndarray:
|
|
198
|
+
"""Transform the full data for *description only* (feature counts,
|
|
199
|
+
whether negative values occur). Never used for scoring."""
|
|
200
|
+
return build_preprocessor(X, cfg, scale).fit_transform(X)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
# --------------------------------------------------------------------------- #
|
|
204
|
+
# Class balance
|
|
205
|
+
# --------------------------------------------------------------------------- #
|
|
206
|
+
|
|
207
|
+
def class_distribution(y) -> Dict[str, int]:
|
|
208
|
+
"""Count rows per class, ignoring missing labels and mixed types."""
|
|
209
|
+
counts = pd.Series(np.asarray(y, dtype=object)).dropna().astype(str) \
|
|
210
|
+
.value_counts(sort=False)
|
|
211
|
+
return {str(k): int(v) for k, v in counts.items()}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def imbalance_ratio(y) -> float:
|
|
215
|
+
"""Ratio of the largest class to the smallest. 1.0 means balanced."""
|
|
216
|
+
counts = list(class_distribution(y).values())
|
|
217
|
+
if not counts or min(counts) == 0:
|
|
218
|
+
return float("inf")
|
|
219
|
+
return max(counts) / min(counts)
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Smart recommendations (Phase 22).
|
|
2
|
+
|
|
3
|
+
Inspects the dataset and suggests sensible methods in plain language.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import List, Optional
|
|
9
|
+
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
from .dataset import Inspection
|
|
13
|
+
from .preprocessing import duplicates_expected_by_chance, imbalance_ratio
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _cat_cols(df, target) -> List[str]:
|
|
17
|
+
"""Non-numeric feature columns (works for object and string dtypes)."""
|
|
18
|
+
return [c for c in df.columns
|
|
19
|
+
if c != target and not pd.api.types.is_numeric_dtype(df[c])]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def missing_strategy(insp: Inspection) -> str:
|
|
23
|
+
"""Recommend a default missing-value strategy key."""
|
|
24
|
+
if not insp.has_missing:
|
|
25
|
+
return "none"
|
|
26
|
+
# If missingness is small, filling is safer than dropping.
|
|
27
|
+
frac = insp.missing_total / max(1, insp.n_rows * insp.n_cols)
|
|
28
|
+
return "median" if frac < 0.2 else "drop"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def encoding_method(df, target) -> str:
|
|
32
|
+
"""Recommend an encoding method key based on categorical cardinality."""
|
|
33
|
+
cat_cols = _cat_cols(df, target)
|
|
34
|
+
if not cat_cols:
|
|
35
|
+
return "auto"
|
|
36
|
+
high_card = any(df[c].nunique() > 10 for c in cat_cols)
|
|
37
|
+
return "auto" if high_card else "onehot"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# Distance- and margin-based classifiers depend on the units of the columns.
|
|
41
|
+
SCALED_CLASSIFIERS = {"svm", "logistic_regression", "neural_network", "knn"}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def scaling_for(key: str, knn_distance: Optional[str] = None,
|
|
45
|
+
choice: Optional[bool] = None,
|
|
46
|
+
method: str = "standard") -> Optional[str]:
|
|
47
|
+
"""How to scale numeric columns for one classifier.
|
|
48
|
+
|
|
49
|
+
Returns None (no scaling), "standard" (mean 0, SD 1) or "minmax" (0-1).
|
|
50
|
+
``choice`` is the user's answer: True / False, or None for automatic.
|
|
51
|
+
|
|
52
|
+
Automatic: tree-based models and Naive Bayes are not scaled; the others
|
|
53
|
+
are standardised; KNN with the Hassanat distance is scaled to 0-1, which
|
|
54
|
+
keeps all values non-negative (the formula's standard form) and gave
|
|
55
|
+
better results than unscaled data in EasyClassifier's benchmarks.
|
|
56
|
+
"""
|
|
57
|
+
if choice is False:
|
|
58
|
+
return None
|
|
59
|
+
if choice is True:
|
|
60
|
+
return method
|
|
61
|
+
if key == "knn" and knn_distance == "hassanat":
|
|
62
|
+
return "minmax"
|
|
63
|
+
return "standard" if key in SCALED_CLASSIFIERS else None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def validation_method(y) -> str:
|
|
67
|
+
"""Recommend a validation strategy key based on class balance."""
|
|
68
|
+
ratio = imbalance_ratio(y)
|
|
69
|
+
if ratio > 1.5:
|
|
70
|
+
return "stratified"
|
|
71
|
+
return "kfold5"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def build_notes(insp: Inspection, df, target, y) -> List[str]:
|
|
75
|
+
"""Return a list of plain-language recommendation strings to show once."""
|
|
76
|
+
notes: List[str] = []
|
|
77
|
+
|
|
78
|
+
if insp.has_missing:
|
|
79
|
+
rec = missing_strategy(insp)
|
|
80
|
+
pct = 100 * insp.missing_total / max(1, insp.n_rows * insp.n_cols)
|
|
81
|
+
if rec == "drop":
|
|
82
|
+
notes.append(
|
|
83
|
+
f"{insp.missing_total} missing values ({pct:.1f}% of cells) "
|
|
84
|
+
"detected. That's a lot, so removing incomplete rows is a "
|
|
85
|
+
"reasonable default."
|
|
86
|
+
)
|
|
87
|
+
else:
|
|
88
|
+
notes.append(
|
|
89
|
+
f"{insp.missing_total} missing values ({pct:.1f}% of cells) "
|
|
90
|
+
"detected. Filling them with the median keeps all your rows."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
if insp.has_duplicates:
|
|
94
|
+
if duplicates_expected_by_chance(df):
|
|
95
|
+
notes.append(
|
|
96
|
+
f"{insp.duplicate_rows} identical rows found. Your columns "
|
|
97
|
+
"allow only a few value combinations, so different cases "
|
|
98
|
+
"(e.g. people giving the same answers) are expected to "
|
|
99
|
+
"coincide. They will be kept.")
|
|
100
|
+
else:
|
|
101
|
+
notes.append(
|
|
102
|
+
f"{insp.duplicate_rows} duplicate rows detected. They are "
|
|
103
|
+
"probably accidental copies; removing them avoids "
|
|
104
|
+
"over-counting repeated records.")
|
|
105
|
+
|
|
106
|
+
cat_cols = _cat_cols(df, target)
|
|
107
|
+
if cat_cols:
|
|
108
|
+
high = [c for c in cat_cols if df[c].nunique() > 10]
|
|
109
|
+
if high:
|
|
110
|
+
notes.append(
|
|
111
|
+
"Some text columns have many categories. Automatic encoding "
|
|
112
|
+
"will label-encode those and one-hot encode the simpler ones."
|
|
113
|
+
)
|
|
114
|
+
else:
|
|
115
|
+
notes.append(
|
|
116
|
+
"Text columns detected with few categories each. One-hot "
|
|
117
|
+
"encoding is a safe choice."
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
ratio = imbalance_ratio(y)
|
|
121
|
+
if ratio > 1.5:
|
|
122
|
+
notes.append(
|
|
123
|
+
f"Classes are imbalanced (largest is {ratio:.1f}x the smallest). "
|
|
124
|
+
"Stratified cross-validation is recommended so each fold keeps the "
|
|
125
|
+
"same class mix."
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
return notes
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Output files: CSV/Excel summaries, predictions, model, importance table.
|
|
2
|
+
|
|
3
|
+
Figures are drawn by :mod:`easyclassifier.figures`.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from typing import List, Optional
|
|
10
|
+
|
|
11
|
+
import pandas as pd # noqa: E402
|
|
12
|
+
import joblib # noqa: E402
|
|
13
|
+
|
|
14
|
+
from .evaluation import Result, METRICS # noqa: E402
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def ensure_dir(path: str) -> str:
|
|
18
|
+
os.makedirs(path, exist_ok=True)
|
|
19
|
+
return path
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def save_feature_importance(table: pd.DataFrame, path: str) -> str:
|
|
23
|
+
table.round(4).to_csv(path, index=False)
|
|
24
|
+
return path
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# --------------------------------------------------------------------------- #
|
|
28
|
+
# Tabular summaries
|
|
29
|
+
# --------------------------------------------------------------------------- #
|
|
30
|
+
|
|
31
|
+
def results_dataframe(results: List[Result]) -> pd.DataFrame:
|
|
32
|
+
rows = []
|
|
33
|
+
for r in results:
|
|
34
|
+
row = {"Classifier": r.classifier_name}
|
|
35
|
+
for key, label in METRICS.items():
|
|
36
|
+
if key in r.metrics:
|
|
37
|
+
row[label] = round(r.metrics[key], 4)
|
|
38
|
+
rows.append(row)
|
|
39
|
+
return pd.DataFrame(rows)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def save_summary_csv(results: List[Result], path: str) -> str:
|
|
43
|
+
results_dataframe(results).to_csv(path, index=False)
|
|
44
|
+
return path
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def save_excel(results: List[Result], path: str) -> Optional[str]:
|
|
48
|
+
try:
|
|
49
|
+
results_dataframe(results).to_excel(path, index=False,
|
|
50
|
+
sheet_name="Results")
|
|
51
|
+
return path
|
|
52
|
+
except Exception: # noqa: BLE001 (openpyxl missing)
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def save_predictions(result: Result, path: str) -> str:
|
|
57
|
+
names = result.class_names
|
|
58
|
+
df = pd.DataFrame({
|
|
59
|
+
"actual": [names[int(v)] if int(v) < len(names) else v
|
|
60
|
+
for v in result.y_true],
|
|
61
|
+
"predicted": [names[int(v)] if int(v) < len(names) else v
|
|
62
|
+
for v in result.y_pred],
|
|
63
|
+
})
|
|
64
|
+
df.to_csv(path, index=False)
|
|
65
|
+
return path
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def save_model(model, path: str) -> str:
|
|
69
|
+
joblib.dump(model, path)
|
|
70
|
+
return path
|