easyclassifier 0.8.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- easyclassifier-0.8.1/LICENSE +21 -0
- easyclassifier-0.8.1/PKG-INFO +267 -0
- easyclassifier-0.8.1/README.md +231 -0
- easyclassifier-0.8.1/pyproject.toml +51 -0
- easyclassifier-0.8.1/setup.cfg +4 -0
- easyclassifier-0.8.1/src/easyclassifier/__init__.py +23 -0
- easyclassifier-0.8.1/src/easyclassifier/__main__.py +61 -0
- easyclassifier-0.8.1/src/easyclassifier/dataset.py +142 -0
- easyclassifier-0.8.1/src/easyclassifier/demo_data.py +32 -0
- easyclassifier-0.8.1/src/easyclassifier/diagnostics.py +95 -0
- easyclassifier-0.8.1/src/easyclassifier/distances.py +152 -0
- easyclassifier-0.8.1/src/easyclassifier/evaluation.py +340 -0
- easyclassifier-0.8.1/src/easyclassifier/figures.py +510 -0
- easyclassifier-0.8.1/src/easyclassifier/help_texts.py +113 -0
- easyclassifier-0.8.1/src/easyclassifier/importance.py +79 -0
- easyclassifier-0.8.1/src/easyclassifier/latex_report.py +555 -0
- easyclassifier-0.8.1/src/easyclassifier/logbook.py +31 -0
- easyclassifier-0.8.1/src/easyclassifier/models.py +94 -0
- easyclassifier-0.8.1/src/easyclassifier/preprocessing.py +219 -0
- easyclassifier-0.8.1/src/easyclassifier/recommend.py +128 -0
- easyclassifier-0.8.1/src/easyclassifier/reporting.py +70 -0
- easyclassifier-0.8.1/src/easyclassifier/target.py +202 -0
- easyclassifier-0.8.1/src/easyclassifier/ui.py +281 -0
- easyclassifier-0.8.1/src/easyclassifier/wizard.py +1266 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/PKG-INFO +267 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/SOURCES.txt +34 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/dependency_links.txt +1 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/entry_points.txt +2 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/requires.txt +13 -0
- easyclassifier-0.8.1/src/easyclassifier.egg-info/top_level.txt +1 -0
- easyclassifier-0.8.1/tests/test_figures.py +166 -0
- easyclassifier-0.8.1/tests/test_first_run.py +82 -0
- easyclassifier-0.8.1/tests/test_leakage.py +120 -0
- easyclassifier-0.8.1/tests/test_report.py +110 -0
- easyclassifier-0.8.1/tests/test_selection.py +165 -0
- easyclassifier-0.8.1/tests/test_target.py +90 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ahmad Hassanat
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: easyclassifier
|
|
3
|
+
Version: 0.8.1
|
|
4
|
+
Summary: Machine learning classification without programming - a guided, menu-driven wizard.
|
|
5
|
+
Author-email: Ahmad Hassanat <ahmad.hassanat@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/ahmadhassant/easyclassifier
|
|
8
|
+
Project-URL: Documentation, https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md
|
|
9
|
+
Project-URL: Issues, https://github.com/ahmadhassant/easyclassifier/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/ahmadhassant/easyclassifier/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: machine learning,classification,wizard,no-code,scikit-learn
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Classifier: Operating System :: OS Independent
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: pandas>=1.5
|
|
25
|
+
Requires-Dist: numpy>=1.23
|
|
26
|
+
Requires-Dist: scikit-learn>=1.2
|
|
27
|
+
Requires-Dist: matplotlib>=3.6
|
|
28
|
+
Requires-Dist: joblib>=1.1
|
|
29
|
+
Requires-Dist: openpyxl>=3.0.10
|
|
30
|
+
Provides-Extra: full
|
|
31
|
+
Requires-Dist: xgboost>=1.5; extra == "full"
|
|
32
|
+
Requires-Dist: lightgbm>=3.3; extra == "full"
|
|
33
|
+
Provides-Extra: test
|
|
34
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# EasyClassifier
|
|
38
|
+
|
|
39
|
+
[](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml)
|
|
40
|
+
[](https://pypi.org/project/easyclassifier/)
|
|
41
|
+
[](https://www.python.org/)
|
|
42
|
+
[](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE)
|
|
43
|
+
|
|
44
|
+
**Machine Learning without Programming.**
|
|
45
|
+
|
|
46
|
+
EasyClassifier is a guided, menu-driven assistant that lets anyone build and
|
|
47
|
+
evaluate machine learning classification models from a CSV or Excel file —
|
|
48
|
+
no Python code, no scripting, and no machine learning jargon required
|
|
49
|
+
(unless you want it).
|
|
50
|
+
|
|
51
|
+
**New to Python or the terminal? Read the step-by-step
|
|
52
|
+
[User Guide](https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md).** It covers installing Python, installing
|
|
53
|
+
EasyClassifier, running it, what each question means, and what the results
|
|
54
|
+
files contain.
|
|
55
|
+
|
|
56
|
+
## Quick start
|
|
57
|
+
|
|
58
|
+
**Requirements:** Python 3.10 or newer (3.12–3.14 recommended) on Windows,
|
|
59
|
+
macOS or Linux. About 400 MB of disk space for the scientific libraries.
|
|
60
|
+
|
|
61
|
+
| | Windows (Command Prompt) | macOS (Terminal) |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| Install | `py -m pip install easyclassifier` | `python3 -m pip install easyclassifier` |
|
|
64
|
+
| Check | `py -m easyclassifier --version` | `python3 -m easyclassifier --version` |
|
|
65
|
+
| Run | `py -m easyclassifier` | `python3 -m easyclassifier` |
|
|
66
|
+
|
|
67
|
+
On Linux (and with Homebrew Python on macOS), install into a separate
|
|
68
|
+
environment:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
python3 -m venv ~/easyclassifier-env
|
|
72
|
+
~/easyclassifier-env/bin/python -m pip install easyclassifier
|
|
73
|
+
~/easyclassifier-env/bin/python -m easyclassifier
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
The shorter command `easyclassifier` also works when Python's scripts folder
|
|
77
|
+
is on your PATH.
|
|
78
|
+
|
|
79
|
+
**First run:** type `demo` when asked for a data file, then press ENTER at
|
|
80
|
+
every question. It takes under a minute and produces a complete example
|
|
81
|
+
report.
|
|
82
|
+
|
|
83
|
+
**What you get:** a new folder `Results/<file>_<date>_<time>/` for every run,
|
|
84
|
+
with `report.pdf` (plain-language report with a ready-to-adapt Methods
|
|
85
|
+
paragraph), `report.tex`, score tables (`summary.csv`, `results.xlsx`),
|
|
86
|
+
predictions, column importance, figures, the trained model, `citations.txt`
|
|
87
|
+
and a full `log.txt`. The PDF is made when a LaTeX program is installed
|
|
88
|
+
(MiKTeX, MacTeX, TeX Live); otherwise `report.tex` can be opened in Overleaf.
|
|
89
|
+
An example is in [`examples/iris/report.pdf`](https://github.com/ahmadhassant/easyclassifier/blob/main/examples/iris/report.pdf).
|
|
90
|
+
|
|
91
|
+
**Optional extras:** `pip install "easyclassifier[full]"` adds XGBoost and
|
|
92
|
+
LightGBM.
|
|
93
|
+
|
|
94
|
+
## What the wizard does
|
|
95
|
+
|
|
96
|
+
One simple question at a time:
|
|
97
|
+
|
|
98
|
+
1. Load your data file (`.csv` or `.xlsx`; you can drag the file into the
|
|
99
|
+
window)
|
|
100
|
+
2. Inspect the data (rows, columns, missing values, duplicates, types)
|
|
101
|
+
3. Choose the column to predict, with checks that it makes sense
|
|
102
|
+
4. Prepare the data (missing values, duplicates, text columns, scaling)
|
|
103
|
+
5. Choose classifiers — or try them all
|
|
104
|
+
6. Review your choices, then train and evaluate with honest scores
|
|
105
|
+
7. Save everything in a new results folder
|
|
106
|
+
|
|
107
|
+
## Figures and colour themes
|
|
108
|
+
|
|
109
|
+
By default each run draws the five figures most used in classification
|
|
110
|
+
papers: class distribution, classifier comparison (with the score of every
|
|
111
|
+
test fold, so you can see whether the winner is clearly better), confusion
|
|
112
|
+
matrix (counts and percentages), ROC curves (one per class for three or more
|
|
113
|
+
classes) and feature importance. Precision-recall curves are added when
|
|
114
|
+
classes are imbalanced, and a missing-value map when the data have empty
|
|
115
|
+
cells. Advanced mode can add a correlation matrix, the most important
|
|
116
|
+
columns by class, and a learning curve ("would more data help?"). The report
|
|
117
|
+
explains how to read each figure, and states in words whether the winner is
|
|
118
|
+
clearly better and whether more data would help.
|
|
119
|
+
|
|
120
|
+
Figures are saved as PNG at 300 dpi, optionally also as PDF or SVG (vector).
|
|
121
|
+
Four colour themes: colour-blind safe (default, Okabe-Ito colours),
|
|
122
|
+
greyscale with patterns (for print), high contrast (slides) and soft. Each
|
|
123
|
+
class keeps the same colour in every figure.
|
|
124
|
+
|
|
125
|
+
## Choosing what to predict
|
|
126
|
+
|
|
127
|
+
Every column is listed with a short plain-language description, for example
|
|
128
|
+
`Purchased - 2 categories (No, Yes)` or `income - numbers from 39000 to
|
|
129
|
+
120000 (13 different)`. The most likely class column is marked as suggested
|
|
130
|
+
and chosen by pressing ENTER. Long lists are shown 20 at a time; typing part
|
|
131
|
+
of a name searches them.
|
|
132
|
+
|
|
133
|
+
The choice is checked before continuing:
|
|
134
|
+
|
|
135
|
+
* **One value only** - refused; there is nothing to predict.
|
|
136
|
+
* **Different in every row** (an ID or a name) - warning; choose another
|
|
137
|
+
column or continue anyway.
|
|
138
|
+
* **A measurement** (numbers with many different values) - EasyClassifier
|
|
139
|
+
predicts groups, not exact numbers (regression is not supported yet). The
|
|
140
|
+
user can split the numbers into 2, 3 or 4 equal-sized groups, split at a
|
|
141
|
+
value of their choice, or pick another column. The cut-points are shown and
|
|
142
|
+
recorded in the log and results.
|
|
143
|
+
* **Classes with a single row** - these can never be tested; the user can
|
|
144
|
+
remove those rows or pick another column. Classes with fewer than 5 rows
|
|
145
|
+
trigger a reliability note.
|
|
146
|
+
|
|
147
|
+
Predictor columns that cannot help (IDs, names, single-value columns) are
|
|
148
|
+
left out automatically, with a note; manual mode asks first.
|
|
149
|
+
|
|
150
|
+
## The report (LaTeX and PDF)
|
|
151
|
+
|
|
152
|
+
Every run writes `Results/report.tex` and, when a LaTeX program is installed
|
|
153
|
+
(TeX Live, MiKTeX, MacTeX or tectonic), compiles it to `Results/report.pdf`.
|
|
154
|
+
Without LaTeX, upload `report.tex` together with the `figures` folder to an
|
|
155
|
+
online editor such as Overleaf. The report contains:
|
|
156
|
+
|
|
157
|
+
* a short summary of the question and the honest final result;
|
|
158
|
+
* the data: rows, class sizes, columns used and left out;
|
|
159
|
+
* a **Methods paragraph ready to adapt** for a paper or thesis, describing
|
|
160
|
+
exactly what was done (cleaning, encoding, scaling, classifiers with default
|
|
161
|
+
parameters, validation, final-score method, KNN distance);
|
|
162
|
+
* the results, with a plain-language explanation of every measure, the
|
|
163
|
+
comparison of classifiers, and the figures;
|
|
164
|
+
* **which columns mattered**: permutation importance, measured on rows the
|
|
165
|
+
model was not trained on (also saved as `feature_importance.csv`);
|
|
166
|
+
* references to cite.
|
|
167
|
+
|
|
168
|
+
Column names in non-Latin scripts (e.g. Arabic) are shown as `?` in the PDF,
|
|
169
|
+
because pdfLaTeX cannot typeset them.
|
|
170
|
+
|
|
171
|
+
## Design choice: default parameters, no tuning
|
|
172
|
+
|
|
173
|
+
Classifiers use their default parameters, without hyperparameter tuning.
|
|
174
|
+
Default values were chosen by the methods' developers after evaluation across
|
|
175
|
+
many datasets, so they perform well on average; parameters tuned on a single
|
|
176
|
+
dataset tend to fit its particular characteristics and transfer poorly to new
|
|
177
|
+
data. Fixed defaults also make results exactly reproducible. For the same
|
|
178
|
+
reason, and to keep the tool simple, feature selection and extraction (e.g.
|
|
179
|
+
PCA) are not included; they are planned as future work.
|
|
180
|
+
|
|
181
|
+
## Honest evaluation (no data leakage)
|
|
182
|
+
|
|
183
|
+
Filling missing values, scaling and encoding are learned from the training
|
|
184
|
+
part only, separately for every hold-out split and every cross-validation
|
|
185
|
+
fold, using a scikit-learn Pipeline. Only steps that learn nothing from the
|
|
186
|
+
data (removing duplicate rows, removing incomplete rows) run before splitting.
|
|
187
|
+
The saved `trained_model.pkl` is the complete pipeline, so it can be applied
|
|
188
|
+
directly to new raw data with the same columns. The tests in `tests/` check
|
|
189
|
+
this (`python -m pytest tests`).
|
|
190
|
+
|
|
191
|
+
## An honest score for the "best" classifier
|
|
192
|
+
|
|
193
|
+
When several classifiers are compared, the winner's score is slightly too
|
|
194
|
+
optimistic: it partly won by luck on those particular splits. EasyClassifier
|
|
195
|
+
therefore reports a separate final score for the selected classifier:
|
|
196
|
+
|
|
197
|
+
* **Nested cross-validation** (datasets up to 2,000 rows): the whole
|
|
198
|
+
comparison is repeated inside each of 5 folds using only that fold's
|
|
199
|
+
training rows, and the winner is scored on the fold's test rows. In the
|
|
200
|
+
benchmarks it gave the most accurate final scores on small and medium data.
|
|
201
|
+
* **Final test set** (larger datasets): 20% of the rows are set aside before
|
|
202
|
+
anything else, the classifiers are compared on the remaining 80%, and the
|
|
203
|
+
winner is scored once on the untouched 20%. Accurate at this size and
|
|
204
|
+
about five times faster.
|
|
205
|
+
|
|
206
|
+
The choice is automatic; Advanced and Research modes can pick either one.
|
|
207
|
+
Classifiers are ranked by balanced accuracy. The comparison table is still
|
|
208
|
+
shown and saved, clearly marked as used only for choosing. The saved model is
|
|
209
|
+
the selected classifier refitted on all rows. The tests in
|
|
210
|
+
`tests/test_selection.py` check that no row is ever predicted by a model
|
|
211
|
+
trained on it, and that on random data the honest estimates stay near chance
|
|
212
|
+
while the winner's comparison score does not.
|
|
213
|
+
|
|
214
|
+
## KNN and the Hassanat distance
|
|
215
|
+
|
|
216
|
+
When you choose KNN, the wizard asks which distance to use: Hassanat (the
|
|
217
|
+
default), Euclidean, Manhattan, Chebyshev, Canberra, or Cosine. For the
|
|
218
|
+
Hassanat distance, EasyClassifier scales the numeric columns to 0–1 (learned
|
|
219
|
+
on training data only), which improved it on 9 of 10 benchmark datasets and
|
|
220
|
+
keeps all values non-negative; the other distances use standardised columns.
|
|
221
|
+
The report says whether the normal form (all values >= 0) or the signed form
|
|
222
|
+
(negative values present) of the formula was applied, and the citation is
|
|
223
|
+
shown on screen and saved to `citations.txt`:
|
|
224
|
+
|
|
225
|
+
> Hassanat, A. B. (2014). Dimensionality Invariant Similarity Measure.
|
|
226
|
+
> Journal of American Science, 10(8). arXiv:1409.0923.
|
|
227
|
+
|
|
228
|
+
## Benchmarks
|
|
229
|
+
|
|
230
|
+
On eleven public datasets from medicine, botany, chemistry, computer vision,
|
|
231
|
+
sociology, political science and psychology plus a random-label control
|
|
232
|
+
([details](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/README.md), [results](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/results/RESULTS.md)):
|
|
233
|
+
|
|
234
|
+
* The common practice of reporting the best cross-validation score after
|
|
235
|
+
fitting preprocessing on all rows was optimistic by 2.9 percentage points
|
|
236
|
+
of balanced accuracy on average (up to about 10 on small datasets).
|
|
237
|
+
EasyClassifier's final score was within 0.7 points on average and had the
|
|
238
|
+
smallest average error (2.3 vs 3.0 points). On random labels it reported
|
|
239
|
+
48.8% (truth: 50%), while the naive estimate said 52.9%.
|
|
240
|
+
* Among the KNN distances, with EasyClassifier's preprocessing, no distance
|
|
241
|
+
was significantly better than the others (Friedman p = 0.053); scaling to
|
|
242
|
+
0–1 improved the Hassanat distance on 9 of 10 datasets.
|
|
243
|
+
|
|
244
|
+
## Try it without your own data
|
|
245
|
+
|
|
246
|
+
At the first question, type `demo` to load a built-in sample dataset so you
|
|
247
|
+
can see the whole workflow immediately.
|
|
248
|
+
|
|
249
|
+
## Philosophy
|
|
250
|
+
|
|
251
|
+
> The user should never write Python code, never edit scripts, and never
|
|
252
|
+
> understand machine learning terminology unless they want to.
|
|
253
|
+
|
|
254
|
+
## Citing
|
|
255
|
+
|
|
256
|
+
If you use EasyClassifier in published work, please cite it (GitHub's
|
|
257
|
+
"Cite this repository" button uses [`CITATION.cff`](CITATION.cff)); each
|
|
258
|
+
run also writes the exact references to `citations.txt`.
|
|
259
|
+
|
|
260
|
+
## Contributing
|
|
261
|
+
|
|
262
|
+
Problem reports and suggestions are very welcome – see
|
|
263
|
+
[CONTRIBUTING.md](https://github.com/ahmadhassant/easyclassifier/blob/main/CONTRIBUTING.md).
|
|
264
|
+
|
|
265
|
+
## License
|
|
266
|
+
|
|
267
|
+
MIT – see [LICENSE](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE).
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
# EasyClassifier
|
|
2
|
+
|
|
3
|
+
[](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml)
|
|
4
|
+
[](https://pypi.org/project/easyclassifier/)
|
|
5
|
+
[](https://www.python.org/)
|
|
6
|
+
[](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE)
|
|
7
|
+
|
|
8
|
+
**Machine Learning without Programming.**
|
|
9
|
+
|
|
10
|
+
EasyClassifier is a guided, menu-driven assistant that lets anyone build and
|
|
11
|
+
evaluate machine learning classification models from a CSV or Excel file —
|
|
12
|
+
no Python code, no scripting, and no machine learning jargon required
|
|
13
|
+
(unless you want it).
|
|
14
|
+
|
|
15
|
+
**New to Python or the terminal? Read the step-by-step
|
|
16
|
+
[User Guide](https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md).** It covers installing Python, installing
|
|
17
|
+
EasyClassifier, running it, what each question means, and what the results
|
|
18
|
+
files contain.
|
|
19
|
+
|
|
20
|
+
## Quick start
|
|
21
|
+
|
|
22
|
+
**Requirements:** Python 3.10 or newer (3.12–3.14 recommended) on Windows,
|
|
23
|
+
macOS or Linux. About 400 MB of disk space for the scientific libraries.
|
|
24
|
+
|
|
25
|
+
| | Windows (Command Prompt) | macOS (Terminal) |
|
|
26
|
+
|---|---|---|
|
|
27
|
+
| Install | `py -m pip install easyclassifier` | `python3 -m pip install easyclassifier` |
|
|
28
|
+
| Check | `py -m easyclassifier --version` | `python3 -m easyclassifier --version` |
|
|
29
|
+
| Run | `py -m easyclassifier` | `python3 -m easyclassifier` |
|
|
30
|
+
|
|
31
|
+
On Linux (and with Homebrew Python on macOS), install into a separate
|
|
32
|
+
environment:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
python3 -m venv ~/easyclassifier-env
|
|
36
|
+
~/easyclassifier-env/bin/python -m pip install easyclassifier
|
|
37
|
+
~/easyclassifier-env/bin/python -m easyclassifier
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
The shorter command `easyclassifier` also works when Python's scripts folder
|
|
41
|
+
is on your PATH.
|
|
42
|
+
|
|
43
|
+
**First run:** type `demo` when asked for a data file, then press ENTER at
|
|
44
|
+
every question. It takes under a minute and produces a complete example
|
|
45
|
+
report.
|
|
46
|
+
|
|
47
|
+
**What you get:** a new folder `Results/<file>_<date>_<time>/` for every run,
|
|
48
|
+
with `report.pdf` (plain-language report with a ready-to-adapt Methods
|
|
49
|
+
paragraph), `report.tex`, score tables (`summary.csv`, `results.xlsx`),
|
|
50
|
+
predictions, column importance, figures, the trained model, `citations.txt`
|
|
51
|
+
and a full `log.txt`. The PDF is made when a LaTeX program is installed
|
|
52
|
+
(MiKTeX, MacTeX, TeX Live); otherwise `report.tex` can be opened in Overleaf.
|
|
53
|
+
An example is in [`examples/iris/report.pdf`](https://github.com/ahmadhassant/easyclassifier/blob/main/examples/iris/report.pdf).
|
|
54
|
+
|
|
55
|
+
**Optional extras:** `pip install "easyclassifier[full]"` adds XGBoost and
|
|
56
|
+
LightGBM.
|
|
57
|
+
|
|
58
|
+
## What the wizard does
|
|
59
|
+
|
|
60
|
+
One simple question at a time:
|
|
61
|
+
|
|
62
|
+
1. Load your data file (`.csv` or `.xlsx`; you can drag the file into the
|
|
63
|
+
window)
|
|
64
|
+
2. Inspect the data (rows, columns, missing values, duplicates, types)
|
|
65
|
+
3. Choose the column to predict, with checks that it makes sense
|
|
66
|
+
4. Prepare the data (missing values, duplicates, text columns, scaling)
|
|
67
|
+
5. Choose classifiers — or try them all
|
|
68
|
+
6. Review your choices, then train and evaluate with honest scores
|
|
69
|
+
7. Save everything in a new results folder
|
|
70
|
+
|
|
71
|
+
## Figures and colour themes
|
|
72
|
+
|
|
73
|
+
By default each run draws the five figures most used in classification
|
|
74
|
+
papers: class distribution, classifier comparison (with the score of every
|
|
75
|
+
test fold, so you can see whether the winner is clearly better), confusion
|
|
76
|
+
matrix (counts and percentages), ROC curves (one per class for three or more
|
|
77
|
+
classes) and feature importance. Precision-recall curves are added when
|
|
78
|
+
classes are imbalanced, and a missing-value map when the data have empty
|
|
79
|
+
cells. Advanced mode can add a correlation matrix, the most important
|
|
80
|
+
columns by class, and a learning curve ("would more data help?"). The report
|
|
81
|
+
explains how to read each figure, and states in words whether the winner is
|
|
82
|
+
clearly better and whether more data would help.
|
|
83
|
+
|
|
84
|
+
Figures are saved as PNG at 300 dpi, optionally also as PDF or SVG (vector).
|
|
85
|
+
Four colour themes: colour-blind safe (default, Okabe-Ito colours),
|
|
86
|
+
greyscale with patterns (for print), high contrast (slides) and soft. Each
|
|
87
|
+
class keeps the same colour in every figure.
|
|
88
|
+
|
|
89
|
+
## Choosing what to predict
|
|
90
|
+
|
|
91
|
+
Every column is listed with a short plain-language description, for example
|
|
92
|
+
`Purchased - 2 categories (No, Yes)` or `income - numbers from 39000 to
|
|
93
|
+
120000 (13 different)`. The most likely class column is marked as suggested
|
|
94
|
+
and chosen by pressing ENTER. Long lists are shown 20 at a time; typing part
|
|
95
|
+
of a name searches them.
|
|
96
|
+
|
|
97
|
+
The choice is checked before continuing:
|
|
98
|
+
|
|
99
|
+
* **One value only** - refused; there is nothing to predict.
|
|
100
|
+
* **Different in every row** (an ID or a name) - warning; choose another
|
|
101
|
+
column or continue anyway.
|
|
102
|
+
* **A measurement** (numbers with many different values) - EasyClassifier
|
|
103
|
+
predicts groups, not exact numbers (regression is not supported yet). The
|
|
104
|
+
user can split the numbers into 2, 3 or 4 equal-sized groups, split at a
|
|
105
|
+
value of their choice, or pick another column. The cut-points are shown and
|
|
106
|
+
recorded in the log and results.
|
|
107
|
+
* **Classes with a single row** - these can never be tested; the user can
|
|
108
|
+
remove those rows or pick another column. Classes with fewer than 5 rows
|
|
109
|
+
trigger a reliability note.
|
|
110
|
+
|
|
111
|
+
Predictor columns that cannot help (IDs, names, single-value columns) are
|
|
112
|
+
left out automatically, with a note; manual mode asks first.
|
|
113
|
+
|
|
114
|
+
## The report (LaTeX and PDF)
|
|
115
|
+
|
|
116
|
+
Every run writes `Results/report.tex` and, when a LaTeX program is installed
|
|
117
|
+
(TeX Live, MiKTeX, MacTeX or tectonic), compiles it to `Results/report.pdf`.
|
|
118
|
+
Without LaTeX, upload `report.tex` together with the `figures` folder to an
|
|
119
|
+
online editor such as Overleaf. The report contains:
|
|
120
|
+
|
|
121
|
+
* a short summary of the question and the honest final result;
|
|
122
|
+
* the data: rows, class sizes, columns used and left out;
|
|
123
|
+
* a **Methods paragraph ready to adapt** for a paper or thesis, describing
|
|
124
|
+
exactly what was done (cleaning, encoding, scaling, classifiers with default
|
|
125
|
+
parameters, validation, final-score method, KNN distance);
|
|
126
|
+
* the results, with a plain-language explanation of every measure, the
|
|
127
|
+
comparison of classifiers, and the figures;
|
|
128
|
+
* **which columns mattered**: permutation importance, measured on rows the
|
|
129
|
+
model was not trained on (also saved as `feature_importance.csv`);
|
|
130
|
+
* references to cite.
|
|
131
|
+
|
|
132
|
+
Column names in non-Latin scripts (e.g. Arabic) are shown as `?` in the PDF,
|
|
133
|
+
because pdfLaTeX cannot typeset them.
|
|
134
|
+
|
|
135
|
+
## Design choice: default parameters, no tuning
|
|
136
|
+
|
|
137
|
+
Classifiers use their default parameters, without hyperparameter tuning.
|
|
138
|
+
Default values were chosen by the methods' developers after evaluation across
|
|
139
|
+
many datasets, so they perform well on average; parameters tuned on a single
|
|
140
|
+
dataset tend to fit its particular characteristics and transfer poorly to new
|
|
141
|
+
data. Fixed defaults also make results exactly reproducible. For the same
|
|
142
|
+
reason, and to keep the tool simple, feature selection and extraction (e.g.
|
|
143
|
+
PCA) are not included; they are planned as future work.
|
|
144
|
+
|
|
145
|
+
## Honest evaluation (no data leakage)
|
|
146
|
+
|
|
147
|
+
Filling missing values, scaling and encoding are learned from the training
|
|
148
|
+
part only, separately for every hold-out split and every cross-validation
|
|
149
|
+
fold, using a scikit-learn Pipeline. Only steps that learn nothing from the
|
|
150
|
+
data (removing duplicate rows, removing incomplete rows) run before splitting.
|
|
151
|
+
The saved `trained_model.pkl` is the complete pipeline, so it can be applied
|
|
152
|
+
directly to new raw data with the same columns. The tests in `tests/` check
|
|
153
|
+
this (`python -m pytest tests`).
|
|
154
|
+
|
|
155
|
+
## An honest score for the "best" classifier
|
|
156
|
+
|
|
157
|
+
When several classifiers are compared, the winner's score is slightly too
|
|
158
|
+
optimistic: it partly won by luck on those particular splits. EasyClassifier
|
|
159
|
+
therefore reports a separate final score for the selected classifier:
|
|
160
|
+
|
|
161
|
+
* **Nested cross-validation** (datasets up to 2,000 rows): the whole
|
|
162
|
+
comparison is repeated inside each of 5 folds using only that fold's
|
|
163
|
+
training rows, and the winner is scored on the fold's test rows. In the
|
|
164
|
+
benchmarks it gave the most accurate final scores on small and medium data.
|
|
165
|
+
* **Final test set** (larger datasets): 20% of the rows are set aside before
|
|
166
|
+
anything else, the classifiers are compared on the remaining 80%, and the
|
|
167
|
+
winner is scored once on the untouched 20%. Accurate at this size and
|
|
168
|
+
about five times faster.
|
|
169
|
+
|
|
170
|
+
The choice is automatic; Advanced and Research modes can pick either one.
|
|
171
|
+
Classifiers are ranked by balanced accuracy. The comparison table is still
|
|
172
|
+
shown and saved, clearly marked as used only for choosing. The saved model is
|
|
173
|
+
the selected classifier refitted on all rows. The tests in
|
|
174
|
+
`tests/test_selection.py` check that no row is ever predicted by a model
|
|
175
|
+
trained on it, and that on random data the honest estimates stay near chance
|
|
176
|
+
while the winner's comparison score does not.
|
|
177
|
+
|
|
178
|
+
## KNN and the Hassanat distance
|
|
179
|
+
|
|
180
|
+
When you choose KNN, the wizard asks which distance to use: Hassanat (the
|
|
181
|
+
default), Euclidean, Manhattan, Chebyshev, Canberra, or Cosine. For the
|
|
182
|
+
Hassanat distance, EasyClassifier scales the numeric columns to 0–1 (learned
|
|
183
|
+
on training data only), which improved it on 9 of 10 benchmark datasets and
|
|
184
|
+
keeps all values non-negative; the other distances use standardised columns.
|
|
185
|
+
The report says whether the normal form (all values >= 0) or the signed form
|
|
186
|
+
(negative values present) of the formula was applied, and the citation is
|
|
187
|
+
shown on screen and saved to `citations.txt`:
|
|
188
|
+
|
|
189
|
+
> Hassanat, A. B. (2014). Dimensionality Invariant Similarity Measure.
|
|
190
|
+
> Journal of American Science, 10(8). arXiv:1409.0923.
|
|
191
|
+
|
|
192
|
+
## Benchmarks
|
|
193
|
+
|
|
194
|
+
On eleven public datasets from medicine, botany, chemistry, computer vision,
|
|
195
|
+
sociology, political science and psychology plus a random-label control
|
|
196
|
+
([details](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/README.md), [results](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/results/RESULTS.md)):
|
|
197
|
+
|
|
198
|
+
* The common practice of reporting the best cross-validation score after
|
|
199
|
+
fitting preprocessing on all rows was optimistic by 2.9 percentage points
|
|
200
|
+
of balanced accuracy on average (up to about 10 on small datasets).
|
|
201
|
+
EasyClassifier's final score was within 0.7 points on average and had the
|
|
202
|
+
smallest average error (2.3 vs 3.0 points). On random labels it reported
|
|
203
|
+
48.8% (truth: 50%), while the naive estimate said 52.9%.
|
|
204
|
+
* Among the KNN distances, with EasyClassifier's preprocessing, no distance
|
|
205
|
+
was significantly better than the others (Friedman p = 0.053); scaling to
|
|
206
|
+
0–1 improved the Hassanat distance on 9 of 10 datasets.
|
|
207
|
+
|
|
208
|
+
## Try it without your own data
|
|
209
|
+
|
|
210
|
+
At the first question, type `demo` to load a built-in sample dataset so you
|
|
211
|
+
can see the whole workflow immediately.
|
|
212
|
+
|
|
213
|
+
## Philosophy
|
|
214
|
+
|
|
215
|
+
> The user should never write Python code, never edit scripts, and never
|
|
216
|
+
> understand machine learning terminology unless they want to.
|
|
217
|
+
|
|
218
|
+
## Citing
|
|
219
|
+
|
|
220
|
+
If you use EasyClassifier in published work, please cite it (GitHub's
|
|
221
|
+
"Cite this repository" button uses [`CITATION.cff`](CITATION.cff)); each
|
|
222
|
+
run also writes the exact references to `citations.txt`.
|
|
223
|
+
|
|
224
|
+
## Contributing
|
|
225
|
+
|
|
226
|
+
Problem reports and suggestions are very welcome – see
|
|
227
|
+
[CONTRIBUTING.md](https://github.com/ahmadhassant/easyclassifier/blob/main/CONTRIBUTING.md).
|
|
228
|
+
|
|
229
|
+
## License
|
|
230
|
+
|
|
231
|
+
MIT – see [LICENSE](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE).
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "easyclassifier"
|
|
7
|
+
version = "0.8.1"
|
|
8
|
+
description = "Machine learning classification without programming - a guided, menu-driven wizard."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [{ name = "Ahmad Hassanat", email = "ahmad.hassanat@gmail.com" }]
|
|
13
|
+
keywords = ["machine learning", "classification", "wizard", "no-code", "scikit-learn"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.10",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Programming Language :: Python :: 3.14",
|
|
23
|
+
"Operating System :: OS Independent",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
"pandas>=1.5",
|
|
27
|
+
"numpy>=1.23",
|
|
28
|
+
"scikit-learn>=1.2",
|
|
29
|
+
"matplotlib>=3.6",
|
|
30
|
+
"joblib>=1.1",
|
|
31
|
+
"openpyxl>=3.0.10",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
full = ["xgboost>=1.5", "lightgbm>=3.3"]
|
|
36
|
+
test = ["pytest>=7"]
|
|
37
|
+
|
|
38
|
+
[project.scripts]
|
|
39
|
+
easyclassifier = "easyclassifier.__main__:main"
|
|
40
|
+
|
|
41
|
+
[project.urls]
|
|
42
|
+
Homepage = "https://github.com/ahmadhassant/easyclassifier"
|
|
43
|
+
Documentation = "https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md"
|
|
44
|
+
Issues = "https://github.com/ahmadhassant/easyclassifier/issues"
|
|
45
|
+
Changelog = "https://github.com/ahmadhassant/easyclassifier/blob/main/CHANGELOG.md"
|
|
46
|
+
|
|
47
|
+
[tool.setuptools.packages.find]
|
|
48
|
+
where = ["src"]
|
|
49
|
+
|
|
50
|
+
[tool.setuptools.package-data]
|
|
51
|
+
easyclassifier = ["data/*.csv"]
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""EasyClassifier - Machine Learning without Programming.
|
|
2
|
+
|
|
3
|
+
A guided, menu-driven wizard that lets non-programmers build and evaluate
|
|
4
|
+
classification models from a CSV file.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
__version__ = "0.8.1"
|
|
8
|
+
__author__ = "Ahmad Hassanat"
|
|
9
|
+
|
|
10
|
+
CITATION = (
|
|
11
|
+
"Hassanat, A. (2026). EasyClassifier: Machine Learning without "
|
|
12
|
+
"Programming (Version {version}) [Software].".format(version=__version__)
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from .wizard import Wizard # noqa: E402
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def run():
|
|
19
|
+
"""Launch the interactive wizard."""
|
|
20
|
+
Wizard().run()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
__all__ = ["Wizard", "run", "__version__", "CITATION"]
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Console entry point for EasyClassifier.
|
|
2
|
+
|
|
3
|
+
Start the wizard with either of:
|
|
4
|
+
|
|
5
|
+
easyclassifier
|
|
6
|
+
python -m easyclassifier
|
|
7
|
+
|
|
8
|
+
Options:
|
|
9
|
+
|
|
10
|
+
--version show the installed version and exit
|
|
11
|
+
--help show this help and exit
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
HELP = """EasyClassifier - machine learning classification without programming.
|
|
17
|
+
|
|
18
|
+
Usage:
|
|
19
|
+
easyclassifier start the step-by-step wizard
|
|
20
|
+
python -m easyclassifier the same, if the 'easyclassifier' command is
|
|
21
|
+
not found
|
|
22
|
+
|
|
23
|
+
Options:
|
|
24
|
+
--version show the installed version
|
|
25
|
+
--help show this help
|
|
26
|
+
|
|
27
|
+
The wizard asks for a data file (.csv or .xlsx). Type 'demo' at that
|
|
28
|
+
question to try it with a built-in example. Results are saved in a new
|
|
29
|
+
folder inside 'Results', in the folder you started from.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def main(argv=None):
|
|
34
|
+
args = sys.argv[1:] if argv is None else list(argv)
|
|
35
|
+
if any(a in ("-h", "--help", "/?") for a in args):
|
|
36
|
+
print(HELP)
|
|
37
|
+
return 0
|
|
38
|
+
if any(a in ("-V", "--version") for a in args):
|
|
39
|
+
from . import __version__
|
|
40
|
+
print(f"EasyClassifier {__version__} "
|
|
41
|
+
f"(Python {sys.version.split()[0]})")
|
|
42
|
+
return 0
|
|
43
|
+
if args:
|
|
44
|
+
print(f"Unknown option: {' '.join(args)}\n")
|
|
45
|
+
print(HELP)
|
|
46
|
+
return 2
|
|
47
|
+
|
|
48
|
+
import warnings
|
|
49
|
+
# Keep the screen clean for non-programmers; details go to the log.
|
|
50
|
+
warnings.filterwarnings("ignore")
|
|
51
|
+
from . import run
|
|
52
|
+
try:
|
|
53
|
+
run()
|
|
54
|
+
except KeyboardInterrupt:
|
|
55
|
+
print("\n\nExiting EasyClassifier. Goodbye!")
|
|
56
|
+
return 130
|
|
57
|
+
return 0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
if __name__ == "__main__":
|
|
61
|
+
sys.exit(main())
|