easyclassifier 0.8.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. easyclassifier-0.8.1/LICENSE +21 -0
  2. easyclassifier-0.8.1/PKG-INFO +267 -0
  3. easyclassifier-0.8.1/README.md +231 -0
  4. easyclassifier-0.8.1/pyproject.toml +51 -0
  5. easyclassifier-0.8.1/setup.cfg +4 -0
  6. easyclassifier-0.8.1/src/easyclassifier/__init__.py +23 -0
  7. easyclassifier-0.8.1/src/easyclassifier/__main__.py +61 -0
  8. easyclassifier-0.8.1/src/easyclassifier/dataset.py +142 -0
  9. easyclassifier-0.8.1/src/easyclassifier/demo_data.py +32 -0
  10. easyclassifier-0.8.1/src/easyclassifier/diagnostics.py +95 -0
  11. easyclassifier-0.8.1/src/easyclassifier/distances.py +152 -0
  12. easyclassifier-0.8.1/src/easyclassifier/evaluation.py +340 -0
  13. easyclassifier-0.8.1/src/easyclassifier/figures.py +510 -0
  14. easyclassifier-0.8.1/src/easyclassifier/help_texts.py +113 -0
  15. easyclassifier-0.8.1/src/easyclassifier/importance.py +79 -0
  16. easyclassifier-0.8.1/src/easyclassifier/latex_report.py +555 -0
  17. easyclassifier-0.8.1/src/easyclassifier/logbook.py +31 -0
  18. easyclassifier-0.8.1/src/easyclassifier/models.py +94 -0
  19. easyclassifier-0.8.1/src/easyclassifier/preprocessing.py +219 -0
  20. easyclassifier-0.8.1/src/easyclassifier/recommend.py +128 -0
  21. easyclassifier-0.8.1/src/easyclassifier/reporting.py +70 -0
  22. easyclassifier-0.8.1/src/easyclassifier/target.py +202 -0
  23. easyclassifier-0.8.1/src/easyclassifier/ui.py +281 -0
  24. easyclassifier-0.8.1/src/easyclassifier/wizard.py +1266 -0
  25. easyclassifier-0.8.1/src/easyclassifier.egg-info/PKG-INFO +267 -0
  26. easyclassifier-0.8.1/src/easyclassifier.egg-info/SOURCES.txt +34 -0
  27. easyclassifier-0.8.1/src/easyclassifier.egg-info/dependency_links.txt +1 -0
  28. easyclassifier-0.8.1/src/easyclassifier.egg-info/entry_points.txt +2 -0
  29. easyclassifier-0.8.1/src/easyclassifier.egg-info/requires.txt +13 -0
  30. easyclassifier-0.8.1/src/easyclassifier.egg-info/top_level.txt +1 -0
  31. easyclassifier-0.8.1/tests/test_figures.py +166 -0
  32. easyclassifier-0.8.1/tests/test_first_run.py +82 -0
  33. easyclassifier-0.8.1/tests/test_leakage.py +120 -0
  34. easyclassifier-0.8.1/tests/test_report.py +110 -0
  35. easyclassifier-0.8.1/tests/test_selection.py +165 -0
  36. easyclassifier-0.8.1/tests/test_target.py +90 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ahmad Hassanat
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,267 @@
1
+ Metadata-Version: 2.4
2
+ Name: easyclassifier
3
+ Version: 0.8.1
4
+ Summary: Machine learning classification without programming - a guided, menu-driven wizard.
5
+ Author-email: Ahmad Hassanat <ahmad.hassanat@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/ahmadhassant/easyclassifier
8
+ Project-URL: Documentation, https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md
9
+ Project-URL: Issues, https://github.com/ahmadhassant/easyclassifier/issues
10
+ Project-URL: Changelog, https://github.com/ahmadhassant/easyclassifier/blob/main/CHANGELOG.md
11
+ Keywords: machine learning,classification,wizard,no-code,scikit-learn
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Programming Language :: Python :: 3.14
20
+ Classifier: Operating System :: OS Independent
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: pandas>=1.5
25
+ Requires-Dist: numpy>=1.23
26
+ Requires-Dist: scikit-learn>=1.2
27
+ Requires-Dist: matplotlib>=3.6
28
+ Requires-Dist: joblib>=1.1
29
+ Requires-Dist: openpyxl>=3.0.10
30
+ Provides-Extra: full
31
+ Requires-Dist: xgboost>=1.5; extra == "full"
32
+ Requires-Dist: lightgbm>=3.3; extra == "full"
33
+ Provides-Extra: test
34
+ Requires-Dist: pytest>=7; extra == "test"
35
+ Dynamic: license-file
36
+
37
+ # EasyClassifier
38
+
39
+ [![tests](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml/badge.svg)](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml)
40
+ [![PyPI](https://img.shields.io/pypi/v/easyclassifier.svg)](https://pypi.org/project/easyclassifier/)
41
+ [![Python](https://img.shields.io/badge/python-3.10%E2%80%933.14-blue.svg)](https://www.python.org/)
42
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE)
43
+
44
+ **Machine Learning without Programming.**
45
+
46
+ EasyClassifier is a guided, menu-driven assistant that lets anyone build and
47
+ evaluate machine learning classification models from a CSV or Excel file —
48
+ no Python code, no scripting, and no machine learning jargon required
49
+ (unless you want it).
50
+
51
+ **New to Python or the terminal? Read the step-by-step
52
+ [User Guide](https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md).** It covers installing Python, installing
53
+ EasyClassifier, running it, what each question means, and what the results
54
+ files contain.
55
+
56
+ ## Quick start
57
+
58
+ **Requirements:** Python 3.10 or newer (3.12–3.14 recommended) on Windows,
59
+ macOS or Linux. About 400 MB of disk space for the scientific libraries.
60
+
61
+ | | Windows (Command Prompt) | macOS (Terminal) |
62
+ |---|---|---|
63
+ | Install | `py -m pip install easyclassifier` | `python3 -m pip install easyclassifier` |
64
+ | Check | `py -m easyclassifier --version` | `python3 -m easyclassifier --version` |
65
+ | Run | `py -m easyclassifier` | `python3 -m easyclassifier` |
66
+
67
+ On Linux (and with Homebrew Python on macOS), install into a separate
68
+ environment:
69
+
70
+ ```bash
71
+ python3 -m venv ~/easyclassifier-env
72
+ ~/easyclassifier-env/bin/python -m pip install easyclassifier
73
+ ~/easyclassifier-env/bin/python -m easyclassifier
74
+ ```
75
+
76
+ The shorter command `easyclassifier` also works when Python's scripts folder
77
+ is on your PATH.
78
+
79
+ **First run:** type `demo` when asked for a data file, then press ENTER at
80
+ every question. It takes under a minute and produces a complete example
81
+ report.
82
+
83
+ **What you get:** a new folder `Results/<file>_<date>_<time>/` for every run,
84
+ with `report.pdf` (plain-language report with a ready-to-adapt Methods
85
+ paragraph), `report.tex`, score tables (`summary.csv`, `results.xlsx`),
86
+ predictions, column importance, figures, the trained model, `citations.txt`
87
+ and a full `log.txt`. The PDF is made when a LaTeX program is installed
88
+ (MiKTeX, MacTeX, TeX Live); otherwise `report.tex` can be opened in Overleaf.
89
+ An example is in [`examples/iris/report.pdf`](https://github.com/ahmadhassant/easyclassifier/blob/main/examples/iris/report.pdf).
90
+
91
+ **Optional extras:** `pip install "easyclassifier[full]"` adds XGBoost and
92
+ LightGBM.
93
+
94
+ ## What the wizard does
95
+
96
+ One simple question at a time:
97
+
98
+ 1. Load your data file (`.csv` or `.xlsx`; you can drag the file into the
99
+ window)
100
+ 2. Inspect the data (rows, columns, missing values, duplicates, types)
101
+ 3. Choose the column to predict, with checks that it makes sense
102
+ 4. Prepare the data (missing values, duplicates, text columns, scaling)
103
+ 5. Choose classifiers — or try them all
104
+ 6. Review your choices, then train and evaluate with honest scores
105
+ 7. Save everything in a new results folder
106
+
107
+ ## Figures and colour themes
108
+
109
+ By default each run draws the five figures most used in classification
110
+ papers: class distribution, classifier comparison (with the score of every
111
+ test fold, so you can see whether the winner is clearly better), confusion
112
+ matrix (counts and percentages), ROC curves (one per class for three or more
113
+ classes) and feature importance. Precision-recall curves are added when
114
+ classes are imbalanced, and a missing-value map when the data have empty
115
+ cells. Advanced mode can add a correlation matrix, the most important
116
+ columns by class, and a learning curve ("would more data help?"). The report
117
+ explains how to read each figure, and states in words whether the winner is
118
+ clearly better and whether more data would help.
119
+
120
+ Figures are saved as PNG at 300 dpi, optionally also as PDF or SVG (vector).
121
+ Four colour themes: colour-blind safe (default, Okabe-Ito colours),
122
+ greyscale with patterns (for print), high contrast (slides) and soft. Each
123
+ class keeps the same colour in every figure.
124
+
125
+ ## Choosing what to predict
126
+
127
+ Every column is listed with a short plain-language description, for example
128
+ `Purchased - 2 categories (No, Yes)` or `income - numbers from 39000 to
129
+ 120000 (13 different)`. The most likely class column is marked as suggested
130
+ and chosen by pressing ENTER. Long lists are shown 20 at a time; typing part
131
+ of a name searches them.
132
+
133
+ The choice is checked before continuing:
134
+
135
+ * **One value only** - refused; there is nothing to predict.
136
+ * **Different in every row** (an ID or a name) - warning; choose another
137
+ column or continue anyway.
138
+ * **A measurement** (numbers with many different values) - EasyClassifier
139
+ predicts groups, not exact numbers (regression is not supported yet). The
140
+ user can split the numbers into 2, 3 or 4 equal-sized groups, split at a
141
+ value of their choice, or pick another column. The cut-points are shown and
142
+ recorded in the log and results.
143
+ * **Classes with a single row** - these can never be tested; the user can
144
+ remove those rows or pick another column. Classes with fewer than 5 rows
145
+ trigger a reliability note.
146
+
147
+ Predictor columns that cannot help (IDs, names, single-value columns) are
148
+ left out automatically, with a note; manual mode asks first.
149
+
150
+ ## The report (LaTeX and PDF)
151
+
152
+ Every run writes `Results/report.tex` and, when a LaTeX program is installed
153
+ (TeX Live, MiKTeX, MacTeX or tectonic), compiles it to `Results/report.pdf`.
154
+ Without LaTeX, upload `report.tex` together with the `figures` folder to an
155
+ online editor such as Overleaf. The report contains:
156
+
157
+ * a short summary of the question and the honest final result;
158
+ * the data: rows, class sizes, columns used and left out;
159
+ * a **Methods paragraph ready to adapt** for a paper or thesis, describing
160
+ exactly what was done (cleaning, encoding, scaling, classifiers with default
161
+ parameters, validation, final-score method, KNN distance);
162
+ * the results, with a plain-language explanation of every measure, the
163
+ comparison of classifiers, and the figures;
164
+ * **which columns mattered**: permutation importance, measured on rows the
165
+ model was not trained on (also saved as `feature_importance.csv`);
166
+ * references to cite.
167
+
168
+ Column names in non-Latin scripts (e.g. Arabic) are shown as `?` in the PDF,
169
+ because pdfLaTeX cannot typeset them.
170
+
171
+ ## Design choice: default parameters, no tuning
172
+
173
+ Classifiers use their default parameters, without hyperparameter tuning.
174
+ Default values were chosen by the methods' developers after evaluation across
175
+ many datasets, so they perform well on average; parameters tuned on a single
176
+ dataset tend to fit its particular characteristics and transfer poorly to new
177
+ data. Fixed defaults also make results exactly reproducible. For the same
178
+ reason, and to keep the tool simple, feature selection and extraction (e.g.
179
+ PCA) are not included; they are planned as future work.
180
+
181
+ ## Honest evaluation (no data leakage)
182
+
183
+ Filling missing values, scaling and encoding are learned from the training
184
+ part only, separately for every hold-out split and every cross-validation
185
+ fold, using a scikit-learn Pipeline. Only steps that learn nothing from the
186
+ data (removing duplicate rows, removing incomplete rows) run before splitting.
187
+ The saved `trained_model.pkl` is the complete pipeline, so it can be applied
188
+ directly to new raw data with the same columns. The tests in `tests/` check
189
+ this (`python -m pytest tests`).
190
+
191
+ ## An honest score for the "best" classifier
192
+
193
+ When several classifiers are compared, the winner's score is slightly too
194
+ optimistic: it partly won by luck on those particular splits. EasyClassifier
195
+ therefore reports a separate final score for the selected classifier:
196
+
197
+ * **Nested cross-validation** (datasets up to 2,000 rows): the whole
198
+ comparison is repeated inside each of 5 folds using only that fold's
199
+ training rows, and the winner is scored on the fold's test rows. In the
200
+ benchmarks it gave the most accurate final scores on small and medium data.
201
+ * **Final test set** (larger datasets): 20% of the rows are set aside before
202
+ anything else, the classifiers are compared on the remaining 80%, and the
203
+ winner is scored once on the untouched 20%. Accurate at this size and
204
+ about five times faster.
205
+
206
+ The choice is automatic; Advanced and Research modes can pick either one.
207
+ Classifiers are ranked by balanced accuracy. The comparison table is still
208
+ shown and saved, clearly marked as used only for choosing. The saved model is
209
+ the selected classifier refitted on all rows. The tests in
210
+ `tests/test_selection.py` check that no row is ever predicted by a model
211
+ trained on it, and that on random data the honest estimates stay near chance
212
+ while the winner's comparison score does not.
213
+
214
+ ## KNN and the Hassanat distance
215
+
216
+ When you choose KNN, the wizard asks which distance to use: Hassanat (the
217
+ default), Euclidean, Manhattan, Chebyshev, Canberra, or Cosine. For the
218
+ Hassanat distance, EasyClassifier scales the numeric columns to 0–1 (learned
219
+ on training data only), which improved it on 9 of 10 benchmark datasets and
220
+ keeps all values non-negative; the other distances use standardised columns.
221
+ The report says whether the normal form (all values >= 0) or the signed form
222
+ (negative values present) of the formula was applied, and the citation is
223
+ shown on screen and saved to `citations.txt`:
224
+
225
+ > Hassanat, A. B. (2014). Dimensionality Invariant Similarity Measure.
226
+ > Journal of American Science, 10(8). arXiv:1409.0923.
227
+
228
+ ## Benchmarks
229
+
230
+ On eleven public datasets from medicine, botany, chemistry, computer vision,
231
+ sociology, political science and psychology plus a random-label control
232
+ ([details](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/README.md), [results](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/results/RESULTS.md)):
233
+
234
+ * The common practice of reporting the best cross-validation score after
235
+ fitting preprocessing on all rows was optimistic by 2.9 percentage points
236
+ of balanced accuracy on average (up to about 10 on small datasets).
237
+ EasyClassifier's final score was within 0.7 points on average and had the
238
+ smallest average error (2.3 vs 3.0 points). On random labels it reported
239
+ 48.8% (truth: 50%), while the naive estimate said 52.9%.
240
+ * Among the KNN distances, with EasyClassifier's preprocessing, no distance
241
+ was significantly better than the others (Friedman p = 0.053); scaling to
242
+ 0–1 improved the Hassanat distance on 9 of 10 datasets.
243
+
244
+ ## Try it without your own data
245
+
246
+ At the first question, type `demo` to load a built-in sample dataset so you
247
+ can see the whole workflow immediately.
248
+
249
+ ## Philosophy
250
+
251
+ > The user should never write Python code, never edit scripts, and never
252
+ > understand machine learning terminology unless they want to.
253
+
254
+ ## Citing
255
+
256
+ If you use EasyClassifier in published work, please cite it (GitHub's
257
+ "Cite this repository" button uses [`CITATION.cff`](CITATION.cff)); each
258
+ run also writes the exact references to `citations.txt`.
259
+
260
+ ## Contributing
261
+
262
+ Problem reports and suggestions are very welcome – see
263
+ [CONTRIBUTING.md](https://github.com/ahmadhassant/easyclassifier/blob/main/CONTRIBUTING.md).
264
+
265
+ ## License
266
+
267
+ MIT – see [LICENSE](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE).
@@ -0,0 +1,231 @@
1
+ # EasyClassifier
2
+
3
+ [![tests](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml/badge.svg)](https://github.com/ahmadhassant/easyclassifier/actions/workflows/tests.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/easyclassifier.svg)](https://pypi.org/project/easyclassifier/)
5
+ [![Python](https://img.shields.io/badge/python-3.10%E2%80%933.14-blue.svg)](https://www.python.org/)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE)
7
+
8
+ **Machine Learning without Programming.**
9
+
10
+ EasyClassifier is a guided, menu-driven assistant that lets anyone build and
11
+ evaluate machine learning classification models from a CSV or Excel file —
12
+ no Python code, no scripting, and no machine learning jargon required
13
+ (unless you want it).
14
+
15
+ **New to Python or the terminal? Read the step-by-step
16
+ [User Guide](https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md).** It covers installing Python, installing
17
+ EasyClassifier, running it, what each question means, and what the results
18
+ files contain.
19
+
20
+ ## Quick start
21
+
22
+ **Requirements:** Python 3.10 or newer (3.12–3.14 recommended) on Windows,
23
+ macOS or Linux. About 400 MB of disk space for the scientific libraries.
24
+
25
+ | | Windows (Command Prompt) | macOS (Terminal) |
26
+ |---|---|---|
27
+ | Install | `py -m pip install easyclassifier` | `python3 -m pip install easyclassifier` |
28
+ | Check | `py -m easyclassifier --version` | `python3 -m easyclassifier --version` |
29
+ | Run | `py -m easyclassifier` | `python3 -m easyclassifier` |
30
+
31
+ On Linux (and with Homebrew Python on macOS), install into a separate
32
+ environment:
33
+
34
+ ```bash
35
+ python3 -m venv ~/easyclassifier-env
36
+ ~/easyclassifier-env/bin/python -m pip install easyclassifier
37
+ ~/easyclassifier-env/bin/python -m easyclassifier
38
+ ```
39
+
40
+ The shorter command `easyclassifier` also works when Python's scripts folder
41
+ is on your PATH.
42
+
43
+ **First run:** type `demo` when asked for a data file, then press ENTER at
44
+ every question. It takes under a minute and produces a complete example
45
+ report.
46
+
47
+ **What you get:** a new folder `Results/<file>_<date>_<time>/` for every run,
48
+ with `report.pdf` (plain-language report with a ready-to-adapt Methods
49
+ paragraph), `report.tex`, score tables (`summary.csv`, `results.xlsx`),
50
+ predictions, column importance, figures, the trained model, `citations.txt`
51
+ and a full `log.txt`. The PDF is made when a LaTeX program is installed
52
+ (MiKTeX, MacTeX, TeX Live); otherwise `report.tex` can be opened in Overleaf.
53
+ An example is in [`examples/iris/report.pdf`](https://github.com/ahmadhassant/easyclassifier/blob/main/examples/iris/report.pdf).
54
+
55
+ **Optional extras:** `pip install "easyclassifier[full]"` adds XGBoost and
56
+ LightGBM.
57
+
58
+ ## What the wizard does
59
+
60
+ One simple question at a time:
61
+
62
+ 1. Load your data file (`.csv` or `.xlsx`; you can drag the file into the
63
+ window)
64
+ 2. Inspect the data (rows, columns, missing values, duplicates, types)
65
+ 3. Choose the column to predict, with checks that it makes sense
66
+ 4. Prepare the data (missing values, duplicates, text columns, scaling)
67
+ 5. Choose classifiers — or try them all
68
+ 6. Review your choices, then train and evaluate with honest scores
69
+ 7. Save everything in a new results folder
70
+
71
+ ## Figures and colour themes
72
+
73
+ By default each run draws the five figures most used in classification
74
+ papers: class distribution, classifier comparison (with the score of every
75
+ test fold, so you can see whether the winner is clearly better), confusion
76
+ matrix (counts and percentages), ROC curves (one per class for three or more
77
+ classes) and feature importance. Precision-recall curves are added when
78
+ classes are imbalanced, and a missing-value map when the data have empty
79
+ cells. Advanced mode can add a correlation matrix, the most important
80
+ columns by class, and a learning curve ("would more data help?"). The report
81
+ explains how to read each figure, and states in words whether the winner is
82
+ clearly better and whether more data would help.
83
+
84
+ Figures are saved as PNG at 300 dpi, optionally also as PDF or SVG (vector).
85
+ Four colour themes: colour-blind safe (default, Okabe-Ito colours),
86
+ greyscale with patterns (for print), high contrast (slides) and soft. Each
87
+ class keeps the same colour in every figure.
88
+
89
+ ## Choosing what to predict
90
+
91
+ Every column is listed with a short plain-language description, for example
92
+ `Purchased - 2 categories (No, Yes)` or `income - numbers from 39000 to
93
+ 120000 (13 different)`. The most likely class column is marked as suggested
94
+ and chosen by pressing ENTER. Long lists are shown 20 at a time; typing part
95
+ of a name searches them.
96
+
97
+ The choice is checked before continuing:
98
+
99
+ * **One value only** - refused; there is nothing to predict.
100
+ * **Different in every row** (an ID or a name) - warning; choose another
101
+ column or continue anyway.
102
+ * **A measurement** (numbers with many different values) - EasyClassifier
103
+ predicts groups, not exact numbers (regression is not supported yet). The
104
+ user can split the numbers into 2, 3 or 4 equal-sized groups, split at a
105
+ value of their choice, or pick another column. The cut-points are shown and
106
+ recorded in the log and results.
107
+ * **Classes with a single row** - these can never be tested; the user can
108
+ remove those rows or pick another column. Classes with fewer than 5 rows
109
+ trigger a reliability note.
110
+
111
+ Predictor columns that cannot help (IDs, names, single-value columns) are
112
+ left out automatically, with a note; manual mode asks first.
113
+
114
+ ## The report (LaTeX and PDF)
115
+
116
+ Every run writes `Results/report.tex` and, when a LaTeX program is installed
117
+ (TeX Live, MiKTeX, MacTeX or tectonic), compiles it to `Results/report.pdf`.
118
+ Without LaTeX, upload `report.tex` together with the `figures` folder to an
119
+ online editor such as Overleaf. The report contains:
120
+
121
+ * a short summary of the question and the honest final result;
122
+ * the data: rows, class sizes, columns used and left out;
123
+ * a **Methods paragraph ready to adapt** for a paper or thesis, describing
124
+ exactly what was done (cleaning, encoding, scaling, classifiers with default
125
+ parameters, validation, final-score method, KNN distance);
126
+ * the results, with a plain-language explanation of every measure, the
127
+ comparison of classifiers, and the figures;
128
+ * **which columns mattered**: permutation importance, measured on rows the
129
+ model was not trained on (also saved as `feature_importance.csv`);
130
+ * references to cite.
131
+
132
+ Column names in non-Latin scripts (e.g. Arabic) are shown as `?` in the PDF,
133
+ because pdfLaTeX cannot typeset them.
134
+
135
+ ## Design choice: default parameters, no tuning
136
+
137
+ Classifiers use their default parameters, without hyperparameter tuning.
138
+ Default values were chosen by the methods' developers after evaluation across
139
+ many datasets, so they perform well on average; parameters tuned on a single
140
+ dataset tend to fit its particular characteristics and transfer poorly to new
141
+ data. Fixed defaults also make results exactly reproducible. For the same
142
+ reason, and to keep the tool simple, feature selection and extraction (e.g.
143
+ PCA) are not included; they are planned as future work.
144
+
145
+ ## Honest evaluation (no data leakage)
146
+
147
+ Filling missing values, scaling and encoding are learned from the training
148
+ part only, separately for every hold-out split and every cross-validation
149
+ fold, using a scikit-learn Pipeline. Only steps that learn nothing from the
150
+ data (removing duplicate rows, removing incomplete rows) run before splitting.
151
+ The saved `trained_model.pkl` is the complete pipeline, so it can be applied
152
+ directly to new raw data with the same columns. The tests in `tests/` check
153
+ this (`python -m pytest tests`).
154
+
155
+ ## An honest score for the "best" classifier
156
+
157
+ When several classifiers are compared, the winner's score is slightly too
158
+ optimistic: it partly won by luck on those particular splits. EasyClassifier
159
+ therefore reports a separate final score for the selected classifier:
160
+
161
+ * **Nested cross-validation** (datasets up to 2,000 rows): the whole
162
+ comparison is repeated inside each of 5 folds using only that fold's
163
+ training rows, and the winner is scored on the fold's test rows. In the
164
+ benchmarks it gave the most accurate final scores on small and medium data.
165
+ * **Final test set** (larger datasets): 20% of the rows are set aside before
166
+ anything else, the classifiers are compared on the remaining 80%, and the
167
+ winner is scored once on the untouched 20%. Accurate at this size and
168
+ about five times faster.
169
+
170
+ The choice is automatic; Advanced and Research modes can pick either one.
171
+ Classifiers are ranked by balanced accuracy. The comparison table is still
172
+ shown and saved, clearly marked as used only for choosing. The saved model is
173
+ the selected classifier refitted on all rows. The tests in
174
+ `tests/test_selection.py` check that no row is ever predicted by a model
175
+ trained on it, and that on random data the honest estimates stay near chance
176
+ while the winner's comparison score does not.
177
+
178
+ ## KNN and the Hassanat distance
179
+
180
+ When you choose KNN, the wizard asks which distance to use: Hassanat (the
181
+ default), Euclidean, Manhattan, Chebyshev, Canberra, or Cosine. For the
182
+ Hassanat distance, EasyClassifier scales the numeric columns to 0–1 (learned
183
+ on training data only), which improved it on 9 of 10 benchmark datasets and
184
+ keeps all values non-negative; the other distances use standardised columns.
185
+ The report says whether the normal form (all values >= 0) or the signed form
186
+ (negative values present) of the formula was applied, and the citation is
187
+ shown on screen and saved to `citations.txt`:
188
+
189
+ > Hassanat, A. B. (2014). Dimensionality Invariant Similarity Measure.
190
+ > Journal of American Science, 10(8). arXiv:1409.0923.
191
+
192
+ ## Benchmarks
193
+
194
+ On eleven public datasets from medicine, botany, chemistry, computer vision,
195
+ sociology, political science and psychology plus a random-label control
196
+ ([details](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/README.md), [results](https://github.com/ahmadhassant/easyclassifier/blob/main/benchmarks/results/RESULTS.md)):
197
+
198
+ * The common practice of reporting the best cross-validation score after
199
+ fitting preprocessing on all rows was optimistic by 2.9 percentage points
200
+ of balanced accuracy on average (up to about 10 on small datasets).
201
+ EasyClassifier's final score was within 0.7 points on average and had the
202
+ smallest average error (2.3 vs 3.0 points). On random labels it reported
203
+ 48.8% (truth: 50%), while the naive estimate said 52.9%.
204
+ * Among the KNN distances, with EasyClassifier's preprocessing, no distance
205
+ was significantly better than the others (Friedman p = 0.053); scaling to
206
+ 0–1 improved the Hassanat distance on 9 of 10 datasets.
207
+
208
+ ## Try it without your own data
209
+
210
+ At the first question, type `demo` to load a built-in sample dataset so you
211
+ can see the whole workflow immediately.
212
+
213
+ ## Philosophy
214
+
215
+ > The user should never write Python code, never edit scripts, and never
216
+ > understand machine learning terminology unless they want to.
217
+
218
+ ## Citing
219
+
220
+ If you use EasyClassifier in published work, please cite it (GitHub's
221
+ "Cite this repository" button uses [`CITATION.cff`](CITATION.cff)); each
222
+ run also writes the exact references to `citations.txt`.
223
+
224
+ ## Contributing
225
+
226
+ Problem reports and suggestions are very welcome – see
227
+ [CONTRIBUTING.md](https://github.com/ahmadhassant/easyclassifier/blob/main/CONTRIBUTING.md).
228
+
229
+ ## License
230
+
231
+ MIT – see [LICENSE](https://github.com/ahmadhassant/easyclassifier/blob/main/LICENSE).
@@ -0,0 +1,51 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "easyclassifier"
7
+ version = "0.8.1"
8
+ description = "Machine learning classification without programming - a guided, menu-driven wizard."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ authors = [{ name = "Ahmad Hassanat", email = "ahmad.hassanat@gmail.com" }]
13
+ keywords = ["machine learning", "classification", "wizard", "no-code", "scikit-learn"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Science/Research",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.10",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: Python :: 3.13",
22
+ "Programming Language :: Python :: 3.14",
23
+ "Operating System :: OS Independent",
24
+ ]
25
+ dependencies = [
26
+ "pandas>=1.5",
27
+ "numpy>=1.23",
28
+ "scikit-learn>=1.2",
29
+ "matplotlib>=3.6",
30
+ "joblib>=1.1",
31
+ "openpyxl>=3.0.10",
32
+ ]
33
+
34
+ [project.optional-dependencies]
35
+ full = ["xgboost>=1.5", "lightgbm>=3.3"]
36
+ test = ["pytest>=7"]
37
+
38
+ [project.scripts]
39
+ easyclassifier = "easyclassifier.__main__:main"
40
+
41
+ [project.urls]
42
+ Homepage = "https://github.com/ahmadhassant/easyclassifier"
43
+ Documentation = "https://github.com/ahmadhassant/easyclassifier/blob/main/docs/USER_GUIDE.md"
44
+ Issues = "https://github.com/ahmadhassant/easyclassifier/issues"
45
+ Changelog = "https://github.com/ahmadhassant/easyclassifier/blob/main/CHANGELOG.md"
46
+
47
+ [tool.setuptools.packages.find]
48
+ where = ["src"]
49
+
50
+ [tool.setuptools.package-data]
51
+ easyclassifier = ["data/*.csv"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,23 @@
1
+ """EasyClassifier - Machine Learning without Programming.
2
+
3
+ A guided, menu-driven wizard that lets non-programmers build and evaluate
4
+ classification models from a CSV file.
5
+ """
6
+
7
+ __version__ = "0.8.1"
8
+ __author__ = "Ahmad Hassanat"
9
+
10
+ CITATION = (
11
+ "Hassanat, A. (2026). EasyClassifier: Machine Learning without "
12
+ "Programming (Version {version}) [Software].".format(version=__version__)
13
+ )
14
+
15
+ from .wizard import Wizard # noqa: E402
16
+
17
+
18
+ def run():
19
+ """Launch the interactive wizard."""
20
+ Wizard().run()
21
+
22
+
23
+ __all__ = ["Wizard", "run", "__version__", "CITATION"]
@@ -0,0 +1,61 @@
1
+ """Console entry point for EasyClassifier.
2
+
3
+ Start the wizard with either of:
4
+
5
+ easyclassifier
6
+ python -m easyclassifier
7
+
8
+ Options:
9
+
10
+ --version show the installed version and exit
11
+ --help show this help and exit
12
+ """
13
+
14
+ import sys
15
+
16
+ HELP = """EasyClassifier - machine learning classification without programming.
17
+
18
+ Usage:
19
+ easyclassifier start the step-by-step wizard
20
+ python -m easyclassifier the same, if the 'easyclassifier' command is
21
+ not found
22
+
23
+ Options:
24
+ --version show the installed version
25
+ --help show this help
26
+
27
+ The wizard asks for a data file (.csv or .xlsx). Type 'demo' at that
28
+ question to try it with a built-in example. Results are saved in a new
29
+ folder inside 'Results', in the folder you started from.
30
+ """
31
+
32
+
33
+ def main(argv=None):
34
+ args = sys.argv[1:] if argv is None else list(argv)
35
+ if any(a in ("-h", "--help", "/?") for a in args):
36
+ print(HELP)
37
+ return 0
38
+ if any(a in ("-V", "--version") for a in args):
39
+ from . import __version__
40
+ print(f"EasyClassifier {__version__} "
41
+ f"(Python {sys.version.split()[0]})")
42
+ return 0
43
+ if args:
44
+ print(f"Unknown option: {' '.join(args)}\n")
45
+ print(HELP)
46
+ return 2
47
+
48
+ import warnings
49
+ # Keep the screen clean for non-programmers; details go to the log.
50
+ warnings.filterwarnings("ignore")
51
+ from . import run
52
+ try:
53
+ run()
54
+ except KeyboardInterrupt:
55
+ print("\n\nExiting EasyClassifier. Goodbye!")
56
+ return 130
57
+ return 0
58
+
59
+
60
+ if __name__ == "__main__":
61
+ sys.exit(main())