easyclassifier 0.8.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,94 @@
1
+ """Classifier registry (Phase 10 & 23).
2
+
3
+ Each classifier is registered with a key, a friendly name, and a factory. New
4
+ algorithms can be added here without touching the wizard. Optional
5
+ dependencies (XGBoost, LightGBM) are detected gracefully.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Callable, Dict, List
12
+
13
+ from sklearn.ensemble import RandomForestClassifier
14
+ from sklearn.linear_model import LogisticRegression
15
+ from sklearn.naive_bayes import GaussianNB
16
+ from sklearn.neural_network import MLPClassifier
17
+ from sklearn.svm import SVC
18
+ from sklearn.tree import DecisionTreeClassifier
19
+
20
+ from .distances import make_knn
21
+
22
+
23
+ @dataclass
24
+ class ClassifierSpec:
25
+ key: str
26
+ name: str
27
+ factory: Callable[[], object]
28
+ help_key: str
29
+ available: bool = True
30
+ reason: str = ""
31
+
32
+
33
+ # Every classifier that uses randomness gets a fixed seed, so the same data
34
+ # and EasyClassifier version always give exactly the same results.
35
+ SEED = 0
36
+
37
+
38
+ def _svc():
39
+ # probability=True lets us compute ROC AUC / PR curves.
40
+ return SVC(probability=True, random_state=SEED)
41
+
42
+
43
+ def _xgb_factory():
44
+ from xgboost import XGBClassifier
45
+ return XGBClassifier(eval_metric="logloss", verbosity=0,
46
+ random_state=SEED)
47
+
48
+
49
+ def _lgbm_factory():
50
+ from lightgbm import LGBMClassifier
51
+ return LGBMClassifier(verbose=-1, random_state=SEED)
52
+
53
+
54
+ def _optional(key, name, factory, help_key, module) -> ClassifierSpec:
55
+ try:
56
+ __import__(module)
57
+ return ClassifierSpec(key, name, factory, help_key, True)
58
+ except Exception: # noqa: BLE001
59
+ return ClassifierSpec(
60
+ key, name, factory, help_key, False,
61
+ reason=f"install '{module}' to enable",
62
+ )
63
+
64
+
65
+ def build_registry() -> Dict[str, ClassifierSpec]:
66
+ specs = [
67
+ ClassifierSpec("decision_tree", "Decision Tree",
68
+ lambda: DecisionTreeClassifier(random_state=SEED),
69
+ "decision_tree"),
70
+ ClassifierSpec("random_forest", "Random Forest",
71
+ lambda: RandomForestClassifier(random_state=SEED),
72
+ "random_forest"),
73
+ ClassifierSpec("svm", "Support Vector Machine (SVM)",
74
+ _svc, "svm"),
75
+ ClassifierSpec("logistic_regression", "Logistic Regression",
76
+ lambda: LogisticRegression(max_iter=1000),
77
+ "logistic_regression"),
78
+ ClassifierSpec("knn", "K-Nearest Neighbours (KNN)",
79
+ lambda: make_knn("hassanat", 5), "knn"),
80
+ ClassifierSpec("naive_bayes", "Naive Bayes",
81
+ GaussianNB, "naive_bayes"),
82
+ _optional("xgboost", "XGBoost", _xgb_factory, "xgboost", "xgboost"),
83
+ _optional("lightgbm", "LightGBM", _lgbm_factory, "lightgbm",
84
+ "lightgbm"),
85
+ ClassifierSpec("neural_network", "Neural Network (MLP)",
86
+ lambda: MLPClassifier(max_iter=500,
87
+ random_state=SEED),
88
+ "neural_network"),
89
+ ]
90
+ return {s.key: s for s in specs}
91
+
92
+
93
+ def available_specs(registry: Dict[str, ClassifierSpec]) -> List[ClassifierSpec]:
94
+ return list(registry.values())
@@ -0,0 +1,219 @@
1
+ """Data cleaning, encoding, and scaling - leakage-safe.
2
+
3
+ Two kinds of step are handled differently:
4
+
5
+ * Row-level steps that learn nothing from the data (removing duplicate rows,
6
+ removing rows with missing values, casting text columns to text) are applied
7
+ once, before any train/test split.
8
+
9
+ * Steps that *learn* something from the data (the mean/median used to fill
10
+ gaps, the scaling range, the list of categories) are placed inside a
11
+ scikit-learn Pipeline. During hold-out or cross-validation the pipeline is
12
+ re-fitted on each training part only, so no information from the test part
13
+ leaks into training.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from dataclasses import dataclass
19
+ from typing import Dict, List, Optional, Tuple
20
+
21
+ import numpy as np
22
+ import pandas as pd
23
+ from sklearn.compose import ColumnTransformer
24
+ from sklearn.impute import SimpleImputer
25
+ from sklearn.pipeline import Pipeline
26
+ from sklearn.preprocessing import (
27
+ LabelEncoder,
28
+ MinMaxScaler,
29
+ OneHotEncoder,
30
+ OrdinalEncoder,
31
+ StandardScaler,
32
+ )
33
+
34
+
35
+ # --------------------------------------------------------------------------- #
36
+ # Row-level cleaning (no learned statistics -> safe before splitting)
37
+ # --------------------------------------------------------------------------- #
38
+
39
+ def drop_missing_target(df: pd.DataFrame, target: str) -> Tuple[pd.DataFrame, int]:
40
+ """Rows without a class label cannot be used for training or testing."""
41
+ before = len(df)
42
+ df = df.dropna(subset=[target]).reset_index(drop=True)
43
+ return df, before - len(df)
44
+
45
+
46
+ def drop_missing_rows(df: pd.DataFrame) -> Tuple[pd.DataFrame, int]:
47
+ before = len(df)
48
+ df = df.dropna().reset_index(drop=True)
49
+ return df, before - len(df)
50
+
51
+
52
+ def duplicates_expected_by_chance(df: pd.DataFrame,
53
+ factor: float = 10.0) -> bool:
54
+ """True if identical rows are expected between *different* cases.
55
+
56
+ With few columns that each take few values (e.g. sex, region, a 0-24
57
+ score), different people often give identical answers; such rows are
58
+ real observations and must be kept. Only when the columns allow far more
59
+ combinations than there are rows (at least ``factor`` times as many) is
60
+ an identical row likely to be an accidental copy.
61
+ """
62
+ combos = 1.0
63
+ for c in df.columns:
64
+ combos *= max(1, df[c].nunique(dropna=False))
65
+ if combos >= factor * len(df):
66
+ return False
67
+ return True
68
+
69
+
70
+ def remove_duplicates(df: pd.DataFrame) -> Tuple[pd.DataFrame, int]:
71
+ before = len(df)
72
+ df = df.drop_duplicates().reset_index(drop=True)
73
+ return df, before - len(df)
74
+
75
+
76
+ def categorical_columns(X: pd.DataFrame) -> List[str]:
77
+ return [c for c in X.columns if not pd.api.types.is_numeric_dtype(X[c])]
78
+
79
+
80
+ def numeric_columns(X: pd.DataFrame) -> List[str]:
81
+ return [c for c in X.columns if pd.api.types.is_numeric_dtype(X[c])]
82
+
83
+
84
+ def cast_categoricals(X: pd.DataFrame) -> pd.DataFrame:
85
+ """Make every non-numeric column plain text (missing values stay missing).
86
+
87
+ This is a type conversion only; nothing is learned from the data.
88
+ """
89
+ X = X.copy()
90
+ for c in categorical_columns(X):
91
+ col = X[c].astype(object)
92
+ X[c] = col.where(col.isna(), col.astype(str))
93
+ return X
94
+
95
+
96
+ # --------------------------------------------------------------------------- #
97
+ # Target encoding (a fixed label mapping - not a learned statistic)
98
+ # --------------------------------------------------------------------------- #
99
+
100
+ def encode_target(y: pd.Series) -> Tuple[np.ndarray, List[str], LabelEncoder]:
101
+ """Class labels -> 0..k-1. Alphabetical order, except for ordered
102
+ categories (e.g. Low / Medium / High groups), which keep their order."""
103
+ le = LabelEncoder()
104
+ if isinstance(y.dtype, pd.CategoricalDtype) and y.cat.ordered:
105
+ present = [c for c in y.cat.categories if (y == c).any()]
106
+ le.classes_ = np.array([str(c) for c in present], dtype=object)
107
+ mapping = {str(c): i for i, c in enumerate(present)}
108
+ y_enc = y.astype(str).map(mapping).to_numpy(dtype=int)
109
+ return y_enc, [str(c) for c in present], le
110
+ y_enc = le.fit_transform(y.astype(str))
111
+ return y_enc, [str(c) for c in le.classes_], le
112
+
113
+
114
+ # --------------------------------------------------------------------------- #
115
+ # Learned preprocessing -> goes inside the Pipeline
116
+ # --------------------------------------------------------------------------- #
117
+
118
+ @dataclass
119
+ class PrepConfig:
120
+ """User choices for the learned preprocessing steps."""
121
+
122
+ impute: str = "median" # "mean" | "median" | "mode"
123
+ encoding: str = "auto" # "auto" | "label" | "onehot"
124
+ scale_method: str = "standard" # "standard" | "minmax"
125
+ scale: Optional[bool] = None # True / False / None = decide per classifier
126
+ onehot_max_categories: int = 10
127
+
128
+
129
+ def _onehot():
130
+ try:
131
+ return OneHotEncoder(handle_unknown="ignore", sparse_output=False)
132
+ except TypeError: # scikit-learn < 1.2
133
+ return OneHotEncoder(handle_unknown="ignore", sparse=False)
134
+
135
+
136
+ def _ordinal():
137
+ return OrdinalEncoder(handle_unknown="use_encoded_value", unknown_value=-1)
138
+
139
+
140
+ def build_preprocessor(X: pd.DataFrame, cfg: PrepConfig,
141
+ scale) -> ColumnTransformer:
142
+ """Build an *unfitted* transformer for the columns of ``X``.
143
+
144
+ ``scale``: False/None (no scaling), True (``cfg.scale_method``), or the
145
+ method itself, "standard" or "minmax".
146
+
147
+ Only column names and types are read from ``X`` here; all statistics are
148
+ learned later, when the pipeline is fitted on training data.
149
+ """
150
+ num_cols = numeric_columns(X)
151
+ cat_cols = categorical_columns(X)
152
+ num_strategy = "most_frequent" if cfg.impute == "mode" else cfg.impute
153
+ method = scale if isinstance(scale, str) else (
154
+ cfg.scale_method if scale else None)
155
+
156
+ transformers = []
157
+ if num_cols:
158
+ steps = [("impute", SimpleImputer(strategy=num_strategy))]
159
+ if method:
160
+ scaler = (MinMaxScaler() if method == "minmax"
161
+ else StandardScaler())
162
+ steps.append(("scale", scaler))
163
+ transformers.append(("num", Pipeline(steps), num_cols))
164
+
165
+ if cat_cols:
166
+ if cfg.encoding == "label":
167
+ onehot_cols, ordinal_cols = [], cat_cols
168
+ elif cfg.encoding == "onehot":
169
+ onehot_cols, ordinal_cols = cat_cols, []
170
+ else: # auto: one-hot for few categories, ordinal for many
171
+ onehot_cols = [c for c in cat_cols
172
+ if X[c].nunique(dropna=True)
173
+ <= cfg.onehot_max_categories]
174
+ ordinal_cols = [c for c in cat_cols if c not in onehot_cols]
175
+ impute = ("impute", SimpleImputer(strategy="most_frequent"))
176
+ if onehot_cols:
177
+ transformers.append(("cat_onehot", Pipeline(
178
+ [impute, ("encode", _onehot())]), onehot_cols))
179
+ if ordinal_cols:
180
+ transformers.append(("cat_ordinal", Pipeline(
181
+ [("impute", SimpleImputer(strategy="most_frequent")),
182
+ ("encode", _ordinal())]), ordinal_cols))
183
+
184
+ return ColumnTransformer(transformers, remainder="drop")
185
+
186
+
187
+ def build_pipeline(X: pd.DataFrame, cfg: PrepConfig, classifier,
188
+ scale: bool) -> Pipeline:
189
+ """Preprocessing + classifier, fitted together on training data only."""
190
+ return Pipeline([
191
+ ("prep", build_preprocessor(X, cfg, scale)),
192
+ ("model", classifier),
193
+ ])
194
+
195
+
196
+ def describe_transformed(X: pd.DataFrame, cfg: PrepConfig,
197
+ scale: bool) -> np.ndarray:
198
+ """Transform the full data for *description only* (feature counts,
199
+ whether negative values occur). Never used for scoring."""
200
+ return build_preprocessor(X, cfg, scale).fit_transform(X)
201
+
202
+
203
+ # --------------------------------------------------------------------------- #
204
+ # Class balance
205
+ # --------------------------------------------------------------------------- #
206
+
207
+ def class_distribution(y) -> Dict[str, int]:
208
+ """Count rows per class, ignoring missing labels and mixed types."""
209
+ counts = pd.Series(np.asarray(y, dtype=object)).dropna().astype(str) \
210
+ .value_counts(sort=False)
211
+ return {str(k): int(v) for k, v in counts.items()}
212
+
213
+
214
+ def imbalance_ratio(y) -> float:
215
+ """Ratio of the largest class to the smallest. 1.0 means balanced."""
216
+ counts = list(class_distribution(y).values())
217
+ if not counts or min(counts) == 0:
218
+ return float("inf")
219
+ return max(counts) / min(counts)
@@ -0,0 +1,128 @@
1
+ """Smart recommendations (Phase 22).
2
+
3
+ Inspects the dataset and suggests sensible methods in plain language.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from typing import List, Optional
9
+
10
+ import pandas as pd
11
+
12
+ from .dataset import Inspection
13
+ from .preprocessing import duplicates_expected_by_chance, imbalance_ratio
14
+
15
+
16
+ def _cat_cols(df, target) -> List[str]:
17
+ """Non-numeric feature columns (works for object and string dtypes)."""
18
+ return [c for c in df.columns
19
+ if c != target and not pd.api.types.is_numeric_dtype(df[c])]
20
+
21
+
22
+ def missing_strategy(insp: Inspection) -> str:
23
+ """Recommend a default missing-value strategy key."""
24
+ if not insp.has_missing:
25
+ return "none"
26
+ # If missingness is small, filling is safer than dropping.
27
+ frac = insp.missing_total / max(1, insp.n_rows * insp.n_cols)
28
+ return "median" if frac < 0.2 else "drop"
29
+
30
+
31
+ def encoding_method(df, target) -> str:
32
+ """Recommend an encoding method key based on categorical cardinality."""
33
+ cat_cols = _cat_cols(df, target)
34
+ if not cat_cols:
35
+ return "auto"
36
+ high_card = any(df[c].nunique() > 10 for c in cat_cols)
37
+ return "auto" if high_card else "onehot"
38
+
39
+
40
+ # Distance- and margin-based classifiers depend on the units of the columns.
41
+ SCALED_CLASSIFIERS = {"svm", "logistic_regression", "neural_network", "knn"}
42
+
43
+
44
+ def scaling_for(key: str, knn_distance: Optional[str] = None,
45
+ choice: Optional[bool] = None,
46
+ method: str = "standard") -> Optional[str]:
47
+ """How to scale numeric columns for one classifier.
48
+
49
+ Returns None (no scaling), "standard" (mean 0, SD 1) or "minmax" (0-1).
50
+ ``choice`` is the user's answer: True / False, or None for automatic.
51
+
52
+ Automatic: tree-based models and Naive Bayes are not scaled; the others
53
+ are standardised; KNN with the Hassanat distance is scaled to 0-1, which
54
+ keeps all values non-negative (the formula's standard form) and gave
55
+ better results than unscaled data in EasyClassifier's benchmarks.
56
+ """
57
+ if choice is False:
58
+ return None
59
+ if choice is True:
60
+ return method
61
+ if key == "knn" and knn_distance == "hassanat":
62
+ return "minmax"
63
+ return "standard" if key in SCALED_CLASSIFIERS else None
64
+
65
+
66
+ def validation_method(y) -> str:
67
+ """Recommend a validation strategy key based on class balance."""
68
+ ratio = imbalance_ratio(y)
69
+ if ratio > 1.5:
70
+ return "stratified"
71
+ return "kfold5"
72
+
73
+
74
+ def build_notes(insp: Inspection, df, target, y) -> List[str]:
75
+ """Return a list of plain-language recommendation strings to show once."""
76
+ notes: List[str] = []
77
+
78
+ if insp.has_missing:
79
+ rec = missing_strategy(insp)
80
+ pct = 100 * insp.missing_total / max(1, insp.n_rows * insp.n_cols)
81
+ if rec == "drop":
82
+ notes.append(
83
+ f"{insp.missing_total} missing values ({pct:.1f}% of cells) "
84
+ "detected. That's a lot, so removing incomplete rows is a "
85
+ "reasonable default."
86
+ )
87
+ else:
88
+ notes.append(
89
+ f"{insp.missing_total} missing values ({pct:.1f}% of cells) "
90
+ "detected. Filling them with the median keeps all your rows."
91
+ )
92
+
93
+ if insp.has_duplicates:
94
+ if duplicates_expected_by_chance(df):
95
+ notes.append(
96
+ f"{insp.duplicate_rows} identical rows found. Your columns "
97
+ "allow only a few value combinations, so different cases "
98
+ "(e.g. people giving the same answers) are expected to "
99
+ "coincide. They will be kept.")
100
+ else:
101
+ notes.append(
102
+ f"{insp.duplicate_rows} duplicate rows detected. They are "
103
+ "probably accidental copies; removing them avoids "
104
+ "over-counting repeated records.")
105
+
106
+ cat_cols = _cat_cols(df, target)
107
+ if cat_cols:
108
+ high = [c for c in cat_cols if df[c].nunique() > 10]
109
+ if high:
110
+ notes.append(
111
+ "Some text columns have many categories. Automatic encoding "
112
+ "will label-encode those and one-hot encode the simpler ones."
113
+ )
114
+ else:
115
+ notes.append(
116
+ "Text columns detected with few categories each. One-hot "
117
+ "encoding is a safe choice."
118
+ )
119
+
120
+ ratio = imbalance_ratio(y)
121
+ if ratio > 1.5:
122
+ notes.append(
123
+ f"Classes are imbalanced (largest is {ratio:.1f}x the smallest). "
124
+ "Stratified cross-validation is recommended so each fold keeps the "
125
+ "same class mix."
126
+ )
127
+
128
+ return notes
@@ -0,0 +1,70 @@
1
+ """Output files: CSV/Excel summaries, predictions, model, importance table.
2
+
3
+ Figures are drawn by :mod:`easyclassifier.figures`.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ from typing import List, Optional
10
+
11
+ import pandas as pd # noqa: E402
12
+ import joblib # noqa: E402
13
+
14
+ from .evaluation import Result, METRICS # noqa: E402
15
+
16
+
17
+ def ensure_dir(path: str) -> str:
18
+ os.makedirs(path, exist_ok=True)
19
+ return path
20
+
21
+
22
+ def save_feature_importance(table: pd.DataFrame, path: str) -> str:
23
+ table.round(4).to_csv(path, index=False)
24
+ return path
25
+
26
+
27
+ # --------------------------------------------------------------------------- #
28
+ # Tabular summaries
29
+ # --------------------------------------------------------------------------- #
30
+
31
+ def results_dataframe(results: List[Result]) -> pd.DataFrame:
32
+ rows = []
33
+ for r in results:
34
+ row = {"Classifier": r.classifier_name}
35
+ for key, label in METRICS.items():
36
+ if key in r.metrics:
37
+ row[label] = round(r.metrics[key], 4)
38
+ rows.append(row)
39
+ return pd.DataFrame(rows)
40
+
41
+
42
+ def save_summary_csv(results: List[Result], path: str) -> str:
43
+ results_dataframe(results).to_csv(path, index=False)
44
+ return path
45
+
46
+
47
+ def save_excel(results: List[Result], path: str) -> Optional[str]:
48
+ try:
49
+ results_dataframe(results).to_excel(path, index=False,
50
+ sheet_name="Results")
51
+ return path
52
+ except Exception: # noqa: BLE001 (openpyxl missing)
53
+ return None
54
+
55
+
56
+ def save_predictions(result: Result, path: str) -> str:
57
+ names = result.class_names
58
+ df = pd.DataFrame({
59
+ "actual": [names[int(v)] if int(v) < len(names) else v
60
+ for v in result.y_true],
61
+ "predicted": [names[int(v)] if int(v) < len(names) else v
62
+ for v in result.y_pred],
63
+ })
64
+ df.to_csv(path, index=False)
65
+ return path
66
+
67
+
68
+ def save_model(model, path: str) -> str:
69
+ joblib.dump(model, path)
70
+ return path