configurable-automl-engine 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- configurable_automl_engine-0.1.0/.github/workflows/run_test.yaml +44 -0
- configurable_automl_engine-0.1.0/.gitignore +130 -0
- configurable_automl_engine-0.1.0/LICENSE.md +28 -0
- configurable_automl_engine-0.1.0/PKG-INFO +273 -0
- configurable_automl_engine-0.1.0/README.md +242 -0
- configurable_automl_engine-0.1.0/config.schema.json +647 -0
- configurable_automl_engine-0.1.0/example.py +43 -0
- configurable_automl_engine-0.1.0/pyproject.toml +44 -0
- configurable_automl_engine-0.1.0/requirements.txt +12 -0
- configurable_automl_engine-0.1.0/schema_generator.py +37 -0
- configurable_automl_engine-0.1.0/setup.cfg +4 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/__init__.py +14 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/_version.py +24 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/common/definitions.py +27 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/common/dependency_utils.py +22 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/common/hyperopt_defaults.py +228 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/common/serialization_utils.py +31 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/common/validation_utils.py +38 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/models.py +146 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/oversampling.py +469 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/trainer.py +707 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/__init__.py +19 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/component.py +478 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/config_parser.py +461 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/logger.py +100 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/metrics.py +291 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/training_engine/thread_pool.py +512 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/tuner.py +467 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine/validation.py +198 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine.egg-info/PKG-INFO +273 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine.egg-info/SOURCES.txt +54 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine.egg-info/dependency_links.txt +1 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine.egg-info/requires.txt +22 -0
- configurable_automl_engine-0.1.0/src/configurable_automl_engine.egg-info/top_level.txt +1 -0
- configurable_automl_engine-0.1.0/tests/.gitkeep +0 -0
- configurable_automl_engine-0.1.0/tests/__init__.py +0 -0
- configurable_automl_engine-0.1.0/tests/conftest.py +44 -0
- configurable_automl_engine-0.1.0/tests/data_factory.py +46 -0
- configurable_automl_engine-0.1.0/tests/test_algorithms.py +45 -0
- configurable_automl_engine-0.1.0/tests/test_algorithms_extended.py +103 -0
- configurable_automl_engine-0.1.0/tests/test_component.py +724 -0
- configurable_automl_engine-0.1.0/tests/test_config_validation.py +395 -0
- configurable_automl_engine-0.1.0/tests/test_dependency_utils.py +30 -0
- configurable_automl_engine-0.1.0/tests/test_e2e_pipeline.py +50 -0
- configurable_automl_engine-0.1.0/tests/test_logger.py +143 -0
- configurable_automl_engine-0.1.0/tests/test_metrics.py +163 -0
- configurable_automl_engine-0.1.0/tests/test_models.py +49 -0
- configurable_automl_engine-0.1.0/tests/test_objective.py +80 -0
- configurable_automl_engine-0.1.0/tests/test_oversampling.py +633 -0
- configurable_automl_engine-0.1.0/tests/test_parallel.py +673 -0
- configurable_automl_engine-0.1.0/tests/test_pipeline_with_fractional_os.py +64 -0
- configurable_automl_engine-0.1.0/tests/test_serialization.py +80 -0
- configurable_automl_engine-0.1.0/tests/test_train_model.py +1113 -0
- configurable_automl_engine-0.1.0/tests/test_tuner.py +420 -0
- configurable_automl_engine-0.1.0/tests/test_validation.py +208 -0
- configurable_automl_engine-0.1.0/tests/test_validation_strategy.py +35 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
name: Run tests and upload coverage
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push
|
|
5
|
+
|
|
6
|
+
jobs:
|
|
7
|
+
test:
|
|
8
|
+
name: Run tests and collect coverage
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- name: Checkout
|
|
12
|
+
uses: actions/checkout@v4
|
|
13
|
+
with:
|
|
14
|
+
fetch-depth: 2
|
|
15
|
+
|
|
16
|
+
- name: Set up Python
|
|
17
|
+
uses: actions/setup-python@v4
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
- name: Install dependencies
|
|
21
|
+
run: |
|
|
22
|
+
python -m pip install --upgrade pip
|
|
23
|
+
pip install pytest pytest-cov pytest-mock ruff mypy types-PyYAML
|
|
24
|
+
# Устанавливаем сам проект в "редактируемом" режиме, чтобы он был виден как пакет
|
|
25
|
+
# Это сработает, если у вас в корне есть setup.py или pyproject.toml
|
|
26
|
+
if [ -f setup.py ] || [ -f pyproject.toml ]; then pip install -e .; fi
|
|
27
|
+
# Устанавливаем остальные зависимости
|
|
28
|
+
if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
|
29
|
+
|
|
30
|
+
- name: Lint with Ruff
|
|
31
|
+
run: ruff check src/
|
|
32
|
+
|
|
33
|
+
- name: Type check with Mypy
|
|
34
|
+
run: mypy src --ignore-missing-imports
|
|
35
|
+
|
|
36
|
+
- name: Run tests
|
|
37
|
+
env:
|
|
38
|
+
PYTHONPATH: .
|
|
39
|
+
run: pytest --cov=src --cov-report=xml
|
|
40
|
+
|
|
41
|
+
- name: Upload results to Codecov
|
|
42
|
+
uses: codecov/codecov-action@v5
|
|
43
|
+
with:
|
|
44
|
+
token: ${{ secrets.CODECOV_TOKEN }}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Byte-compiled / optimized / DLL files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# C extensions
|
|
7
|
+
*.so
|
|
8
|
+
|
|
9
|
+
# Distribution / packaging
|
|
10
|
+
.Python
|
|
11
|
+
build/
|
|
12
|
+
develop-eggs/
|
|
13
|
+
dist/
|
|
14
|
+
downloads/
|
|
15
|
+
eggs/
|
|
16
|
+
.eggs/
|
|
17
|
+
lib/
|
|
18
|
+
lib64/
|
|
19
|
+
parts/
|
|
20
|
+
sdist/
|
|
21
|
+
var/
|
|
22
|
+
wheels/
|
|
23
|
+
pip-wheel-metadata/
|
|
24
|
+
share/python-wheels/
|
|
25
|
+
*.egg-info/
|
|
26
|
+
.installed.cfg
|
|
27
|
+
*.egg
|
|
28
|
+
MANIFEST
|
|
29
|
+
|
|
30
|
+
# PyInstaller
|
|
31
|
+
# Usually these files are written by a python script from a template
|
|
32
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
33
|
+
*.manifest
|
|
34
|
+
*.spec
|
|
35
|
+
|
|
36
|
+
# Installer logs
|
|
37
|
+
pip-log.txt
|
|
38
|
+
pip-delete-this-directory.txt
|
|
39
|
+
|
|
40
|
+
# Unit test / coverage reports
|
|
41
|
+
htmlcov/
|
|
42
|
+
.tox/
|
|
43
|
+
.nox/
|
|
44
|
+
.coverage
|
|
45
|
+
.coverage.*
|
|
46
|
+
.cache
|
|
47
|
+
nosetests.xml
|
|
48
|
+
coverage.xml
|
|
49
|
+
*.cover
|
|
50
|
+
.hypothesis/
|
|
51
|
+
.pytest_cache/
|
|
52
|
+
|
|
53
|
+
# Translations
|
|
54
|
+
*.mo
|
|
55
|
+
*.pot
|
|
56
|
+
|
|
57
|
+
# Django stuff:
|
|
58
|
+
*.log
|
|
59
|
+
local_settings.py
|
|
60
|
+
db.sqlite3
|
|
61
|
+
db.sqlite3-journal
|
|
62
|
+
|
|
63
|
+
# Flask stuff:
|
|
64
|
+
instance/
|
|
65
|
+
.webassets-cache
|
|
66
|
+
|
|
67
|
+
# Scrapy stuff:
|
|
68
|
+
.scrapy
|
|
69
|
+
|
|
70
|
+
# Sphinx documentation
|
|
71
|
+
docs/_build/
|
|
72
|
+
|
|
73
|
+
# PyBuilder
|
|
74
|
+
target/
|
|
75
|
+
|
|
76
|
+
# Jupyter Notebook
|
|
77
|
+
.ipynb_checkpoints
|
|
78
|
+
|
|
79
|
+
# IPython
|
|
80
|
+
profile_default/
|
|
81
|
+
ipython_config.py
|
|
82
|
+
|
|
83
|
+
# pyenv
|
|
84
|
+
.python-version
|
|
85
|
+
|
|
86
|
+
# pipenv
|
|
87
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
88
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
89
|
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
|
90
|
+
# install all needed dependencies.
|
|
91
|
+
#Pipfile.lock
|
|
92
|
+
|
|
93
|
+
# celery beat schedule file
|
|
94
|
+
celerybeat-schedule
|
|
95
|
+
|
|
96
|
+
# SageMath parsed files
|
|
97
|
+
*.sage.py
|
|
98
|
+
|
|
99
|
+
# Environments
|
|
100
|
+
.env
|
|
101
|
+
.venv
|
|
102
|
+
env/
|
|
103
|
+
venv/
|
|
104
|
+
ENV/
|
|
105
|
+
env.bak/
|
|
106
|
+
venv.bak/
|
|
107
|
+
|
|
108
|
+
# Spyder project settings
|
|
109
|
+
.spyderproject
|
|
110
|
+
.spyproject
|
|
111
|
+
|
|
112
|
+
# Rope project settings
|
|
113
|
+
.ropeproject
|
|
114
|
+
|
|
115
|
+
# mkdocs documentation
|
|
116
|
+
/site
|
|
117
|
+
|
|
118
|
+
# mypy
|
|
119
|
+
.mypy_cache/
|
|
120
|
+
.dmypy.json
|
|
121
|
+
dmypy.json
|
|
122
|
+
|
|
123
|
+
# Pyre type checker
|
|
124
|
+
.pyre/
|
|
125
|
+
|
|
126
|
+
*.pkl
|
|
127
|
+
|
|
128
|
+
src/configurable_automl_engine/_version.py
|
|
129
|
+
|
|
130
|
+
*.joblib
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
|
|
2
|
+
Copyright (c) 2026, Леонид Лапшин
|
|
3
|
+
All rights reserved.
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
* Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
* Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
* Neither the name of [project] nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: configurable-automl-engine
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A configuration-driven AutoML orchestrator for automated model selection and two-stage hyperparameter optimization.
|
|
5
|
+
Author-email: Leonid Lapshin <leonid.lapshin.a@gmail.com>
|
|
6
|
+
License: BSD 3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/LapshinLeonid/configurable_automl_engine
|
|
8
|
+
Project-URL: Issues, https://github.com/LapshinLeonid/configurable_automl_engine/issues
|
|
9
|
+
Requires-Python: >=3.11
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE.md
|
|
12
|
+
Requires-Dist: numpy>=2.4.2
|
|
13
|
+
Requires-Dist: pandas>=3.0.0
|
|
14
|
+
Requires-Dist: imbalanced-learn>=0.14.1
|
|
15
|
+
Requires-Dist: scikit-learn>=1.8.0
|
|
16
|
+
Requires-Dist: joblib>=1.5.3
|
|
17
|
+
Requires-Dist: pyyaml>=6.0.3
|
|
18
|
+
Requires-Dist: scipy>=1.17.0
|
|
19
|
+
Requires-Dist: optuna>=4.7.0
|
|
20
|
+
Requires-Dist: pydantic>=2.12.5
|
|
21
|
+
Requires-Dist: pyarrow>23.0.0
|
|
22
|
+
Provides-Extra: xgboost
|
|
23
|
+
Requires-Dist: xgboost>=3.1.3; extra == "xgboost"
|
|
24
|
+
Provides-Extra: test
|
|
25
|
+
Requires-Dist: pytest>=9.0.2; extra == "test"
|
|
26
|
+
Provides-Extra: all
|
|
27
|
+
Requires-Dist: configurable-automl-engine[xgboost]; extra == "all"
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: automl-engine[all,test]; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
AutoML Engine is a configuration-driven automated machine learning library for Python.
|
|
33
|
+
It provides a high-performance ecosystem for model selection and hyperparameter optimization,
|
|
34
|
+
designed to scale from local experimentation to large-scale data processing.
|
|
35
|
+
|
|
36
|
+
# Features
|
|
37
|
+
|
|
38
|
+
* Configuration-Driven Architecture: Fully controlled via YAML schemas and Python configuration classes (Pydantic-based) for reproducible experiments.
|
|
39
|
+
* Flexible Validation Strategies: Supports various splitting techniques including KFold, StratifiedKFold, GroupKFold.
|
|
40
|
+
* Dynamic Hyperparameter Optimization: Integrated wrapper for Optuna to automate search space configuration and trial management.
|
|
41
|
+
* Extensible Model Factory: Built-in support for multiple regression algorithms.
|
|
42
|
+
* Robust Preprocessing Pipeline: Automated handling of scaling, encoding, and missing value imputation.
|
|
43
|
+
* Advanced Imbalance Handling: Built-in oversampling module supporting SMOTE, ADASYN, and BorderlineSMOTE.
|
|
44
|
+
* Nested Validation Support: Ability to perform complex nested cross-validation to ensure model generalizability.
|
|
45
|
+
* Parallel Execution: Utilizes threading and multi-processing for faster hyperparameter searches and cross-validation loops.
|
|
46
|
+
* Seamless Serialization: Robust I/O tools for saving and loading models, metadata, and preprocessing artifacts in joblib or pickle formats.
|
|
47
|
+
|
|
48
|
+
# Dependencies
|
|
49
|
+
|
|
50
|
+
We recommend using the latest version of Python. AutoML Engine supports Python 3.9 and newer.
|
|
51
|
+
|
|
52
|
+
These distributions are essential for the core functionality and will be installed automatically:
|
|
53
|
+
|
|
54
|
+
* **NumPy** (>=2.4.2): Base package for numerical computing and array manipulation.
|
|
55
|
+
* **Scipy** (>=1.17.0): Used for advanced scientific computing and statistical functions.
|
|
56
|
+
* **Pandas** (>=3.0.0): Used for data structures and high-level manipulation of tabular datasets before model ingestion.
|
|
57
|
+
* **PyArrow** (>=23.0.0): Provides a cross-language development platform for in-memory data, enabling efficient data exchange and high-performance integration with Pandas through the Arrow columnar format.
|
|
58
|
+
* **Scikit-learn** (>=1.8.0): The primary library for machine learning algorithms, preprocessing tools, and validation frameworks.
|
|
59
|
+
* **Imbalanced-learn** (>=0.14.1): Provides oversampling algorithms (like SMOTE) for handling datasets with skewed class distributions.
|
|
60
|
+
* **Optuna**(>=4.7.0): Powers the engine to perform automated hyperparameter optimization searches.
|
|
61
|
+
* **Pydantic** (>=2.12.5): Data validation and settings management using Python type annotations.
|
|
62
|
+
* **PyYAML**(>=6.0.3): Implements the standard configuration schema, allowing the system to parse YAML files for model parameters and training setups.
|
|
63
|
+
* **Joblib** (>=1.5.3): Provides lightweight pipelining and model serialization (saving/loading).
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
## Optional Dependencies
|
|
67
|
+
|
|
68
|
+
These distributions will not be installed automatically. Tou can install them using the bracket syntax (e.g., pip install "automl-engine[xgboost]").
|
|
69
|
+
|
|
70
|
+
* **XGBoost**: Adds support for high-performance gradient boosting models..
|
|
71
|
+
|
|
72
|
+
# Installation
|
|
73
|
+
|
|
74
|
+
To install, run:
|
|
75
|
+
|
|
76
|
+
pip install configurable-automl-engine
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
For the full suite including all supported gradient boosting backends:
|
|
80
|
+
|
|
81
|
+
pip install configurable-automl-engine[all]
|
|
82
|
+
|
|
83
|
+
Testing:
|
|
84
|
+
|
|
85
|
+
python -c "import configurable_automl_engine; print('Success!')"
|
|
86
|
+
|
|
87
|
+
# Quick Start
|
|
88
|
+
|
|
89
|
+
The example can be run from [example.py](example.py).
|
|
90
|
+
|
|
91
|
+
from sklearn.datasets import load_diabetes
|
|
92
|
+
|
|
93
|
+
import configurable_automl_engine as caml
|
|
94
|
+
|
|
95
|
+
data = load_diabetes(as_frame=True)
|
|
96
|
+
df = data.frame
|
|
97
|
+
|
|
98
|
+
config = {
|
|
99
|
+
"general": {
|
|
100
|
+
"comparison_metric": "r2",
|
|
101
|
+
"phases": [
|
|
102
|
+
{"n_trials": 100, "action": "all_algorithms"},
|
|
103
|
+
{"n_trials": 200, "action": "refine_winner"}
|
|
104
|
+
],
|
|
105
|
+
"path_to_model": "diabetes_model.joblib"
|
|
106
|
+
},
|
|
107
|
+
"algorithms": {
|
|
108
|
+
"random_forest": {
|
|
109
|
+
"enable": True,
|
|
110
|
+
"limit_hyperparameters": True,
|
|
111
|
+
"hyperparameters": {"n_estimators": [10, 500], "max_depth": [3, 20]}
|
|
112
|
+
},
|
|
113
|
+
"ridge": {
|
|
114
|
+
"enable": True,
|
|
115
|
+
"limit_hyperparameters": True,
|
|
116
|
+
"hyperparameters": {"alpha": [0.1, 1.0]}
|
|
117
|
+
},
|
|
118
|
+
"xgboosting": {
|
|
119
|
+
"enable": True,
|
|
120
|
+
"limit_hyperparameters": True,
|
|
121
|
+
"hyperparameters": {
|
|
122
|
+
"n_estimators": [100, 1000],
|
|
123
|
+
"max_depth": [3, 10],
|
|
124
|
+
"learning_rate": [0.01, 0.3],
|
|
125
|
+
"subsample": [0.5, 1.0]
|
|
126
|
+
}
|
|
127
|
+
},
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
results = caml.train_best_model(config=config, df=df, target='target')
|
|
132
|
+
|
|
133
|
+
print(f"Winner: {results['algorithm']}, Score: {results['score']:.4f}")
|
|
134
|
+
|
|
135
|
+
# Configuration File Structure
|
|
136
|
+
|
|
137
|
+
The system uses a typed YAML or JSON config based on Pydantic.
|
|
138
|
+
|
|
139
|
+
Scheme: [config.schema.json](config.schema.json).
|
|
140
|
+
|
|
141
|
+
Some Pydantic validation rules cannot be described in the schema. See [config_parser.py](/src/configurable_automl_engine/training_engine/config_parser.py) for details.
|
|
142
|
+
|
|
143
|
+
The file must contain two sections: "general" and "alghorithms". Additionally, it may include an optional "oversampling" section.
|
|
144
|
+
|
|
145
|
+
## General section
|
|
146
|
+
The "general" section may include the following attributes:
|
|
147
|
+
* "phases" - required section (array) of hyperparameter optimization phases
|
|
148
|
+
* "comparison_metric" - (optional) accuracy metric for model comparison. Defaults to "r2" if not specified
|
|
149
|
+
* "path_to_model" - (optional) path to save the best model
|
|
150
|
+
* "serialization_format" - (optional) format for saving the model
|
|
151
|
+
* "log_to_file" - (optional) path to the log file
|
|
152
|
+
* "validation_strategy" - (optional) strategy for evaluating model accuracy
|
|
153
|
+
* "n_folds" - (optional) number of folds for cross-validation; used only if "validation_strategy" = "k_fold"
|
|
154
|
+
* "max_workers" - (optional) maximum number of threads/processes. If not specified the number of CPU cores is used
|
|
155
|
+
|
|
156
|
+
Structure of an optimization phase ("phases"):
|
|
157
|
+
* "n_trials" - number of iterations within this phase
|
|
158
|
+
* "name" - (optional) user-defined name of the optimization phase
|
|
159
|
+
* "action" - (optional) action for the phase, default is "all_algorithms"
|
|
160
|
+
|
|
161
|
+
Allowed actions for an optimization phase:
|
|
162
|
+
* "all_algorithms" - for each algorithm, performs "n_trials" hyperparameter optimization attempts. The best algorithm is passed to the next phase.
|
|
163
|
+
* "refine_winner" - performs "n_trials" hyperparameter optimization attempts for the best algorithm from the previous phase.
|
|
164
|
+
|
|
165
|
+
Allowed values for "comparison_metric":
|
|
166
|
+
* "nrmse"
|
|
167
|
+
* "rmse"
|
|
168
|
+
* "mae"
|
|
169
|
+
* "mse"
|
|
170
|
+
* "r2"
|
|
171
|
+
|
|
172
|
+
Allowed values for "serialization_format":
|
|
173
|
+
* "pickle"
|
|
174
|
+
* "joblib"
|
|
175
|
+
|
|
176
|
+
Allowed values for "validation_strategy":
|
|
177
|
+
* "train_test_split"
|
|
178
|
+
* "k_fold"
|
|
179
|
+
* "loo"
|
|
180
|
+
|
|
181
|
+
## Alghorithms section
|
|
182
|
+
|
|
183
|
+
The "alghorithms" section is a dictionary where the key is the algorithm name, and the value is a set of configurations for that algorithm.
|
|
184
|
+
|
|
185
|
+
Supported algorithms:
|
|
186
|
+
* "elasticnet"
|
|
187
|
+
* "sgdregressor"
|
|
188
|
+
* "decision_tree"
|
|
189
|
+
* "random_forest"
|
|
190
|
+
* "extra_trees"
|
|
191
|
+
* "gradient_boosting"
|
|
192
|
+
* "adaboost"
|
|
193
|
+
* "poissonregressor"
|
|
194
|
+
* "gammaregressor"
|
|
195
|
+
* "tweedieregressor"
|
|
196
|
+
* "gaussian_process_regression"
|
|
197
|
+
* "isotonic_regression"
|
|
198
|
+
* "nearest_neighbors_regression"
|
|
199
|
+
* "svr"
|
|
200
|
+
* "ardregression"
|
|
201
|
+
* "glm"
|
|
202
|
+
* "ridge"
|
|
203
|
+
* "lasso"
|
|
204
|
+
* "xgboosting"
|
|
205
|
+
|
|
206
|
+
Algorithm configuration consists of:
|
|
207
|
+
* "enable" - boolean flag, whether hyperparameter search is performed for the algorithm
|
|
208
|
+
* "limit_hyperparameters" - (optional) boolean flag to set limits for hyperparameter search
|
|
209
|
+
* "hyperparameters" - (optional) hyperparameter value constraints, unique to each algorithm. See [ALGO_HYPERPARAMETER_REGISTRY](/src/configurable_automl_engine/common/hyperopt_defaults.py) for details
|
|
210
|
+
|
|
211
|
+
## Oversampling section
|
|
212
|
+
|
|
213
|
+
The optional "oversampling" section may include:
|
|
214
|
+
* "enable" - (optional) enable oversampling
|
|
215
|
+
* "multiplier" - (optional) factor to increase dataset size
|
|
216
|
+
* "algorithm" - (optional) oversampling algorithm
|
|
217
|
+
|
|
218
|
+
#### Supported oversampling algorithms:
|
|
219
|
+
* "random"
|
|
220
|
+
* "random_with_noise"
|
|
221
|
+
* "smote"
|
|
222
|
+
* "adasyn"
|
|
223
|
+
|
|
224
|
+
# Contributing
|
|
225
|
+
|
|
226
|
+
Small improvements, fixes, reporting issues, requesting features are always appreciated. Use [ GitHub issue tracker](https://github.com/LapshinLeonid/configurable_automl_engine/issues)
|
|
227
|
+
|
|
228
|
+
If you are considering larger contributions to the source code, please contact author first.
|
|
229
|
+
|
|
230
|
+
If you contribute, please ensure your code:
|
|
231
|
+
* 100% covered by tests
|
|
232
|
+
* has 0 errors when checked by the ruff linter
|
|
233
|
+
* has 0 errors when checked by the mypy static analyzer with the --strict key
|
|
234
|
+
* All docstrings written on English or Russian
|
|
235
|
+
|
|
236
|
+
# Project Structure
|
|
237
|
+
```
|
|
238
|
+
├── src # Project source code root
|
|
239
|
+
│ └── configurable_automl_engine # Main AutoML engine package
|
|
240
|
+
│ ├── common # Shared utilities and helper functions
|
|
241
|
+
│ │ ├── definitions.py # Constants, enums, and schema definitions
|
|
242
|
+
│ │ ├── dependency_utils.py # Optional library and dependency checks
|
|
243
|
+
│ │ ├── hyperopt_defaults.py # Default search spaces for tuning
|
|
244
|
+
│ │ ├── serialization_utils.py # Model/pipeline serialization logic
|
|
245
|
+
│ │ └── validation_utils.py # Low-level data validation helpers
|
|
246
|
+
│ ├── training_engine # Core orchestration and execution logic
|
|
247
|
+
│ │ ├── component.py # Pipeline building block base classes
|
|
248
|
+
│ │ ├── config_parser.py # Configuration parsing and validation
|
|
249
|
+
│ │ ├── logger.py # Centralized logging management
|
|
250
|
+
│ │ ├── metrics.py # Evaluation metrics implementation
|
|
251
|
+
│ │ └── thread_pool.py # Multi-threading and parallel execution
|
|
252
|
+
│ ├── models.py # Model factory and algorithm wrappers
|
|
253
|
+
│ ├── oversampling.py # Imbalance handling and resampling
|
|
254
|
+
│ ├── trainer.py # Training process orchestrator
|
|
255
|
+
│ ├── tuner.py # Hyperparameter optimization logic
|
|
256
|
+
│ └── validation.py # High-level cross-validation strategies
|
|
257
|
+
└── tests # Unit and integration test suites
|
|
258
|
+
```
|
|
259
|
+
# ⚠️ NeuroSlop Warning
|
|
260
|
+
|
|
261
|
+
This project utilizes Large Language Models (LLMs) to assist in development and maintenance. To ensure transparency regarding the origin of the codebase, please note the following:
|
|
262
|
+
|
|
263
|
+
* Original Core & Architecture: The fundamental architecture, core logic, and overall project conceptualization are 100% original and authored by the human creator.
|
|
264
|
+
* Automated Testing: The test suite is almost entirely LLM-generated. While these tests aim for high coverage and functional verification, they were synthesized based on the provided source code.
|
|
265
|
+
* Source Code Generation: Portions of non-critical source code were also LLM-generated. However, all generated code has undergone a manual code review by the author to the full extent of their technical expertise and competence to ensure quality and logic. Additionally, the code is fully compliant with modern development standards, showing no issues or warnings from ruff and mypy, and maintains 100% unit test coverage to ensure reliability and correctness.
|
|
266
|
+
* Commit History: All commit messages and titles have been generated by an LLM. This ensures a consistent (though automated) narrative of the project's evolution.
|
|
267
|
+
* Documentation: The project documentation, including parts of this README and inline comments, is partially LLM-generated. AI was used to expand on technical details and improve readability based on the original technical specifications.
|
|
268
|
+
|
|
269
|
+
While this project leverages neural synthesis for scaffolding, verification, and non-critical components, users and contributors should be aware that the intellectual core, architectural vision, and primary logic remain entirely human-made. The AI serves as an assistant, while the fundamental value and creative direction are the result of deliberate human engineering.
|
|
270
|
+
|
|
271
|
+
# Contact
|
|
272
|
+
|
|
273
|
+
If you'd like to contact the author, you can use Telegram @Lapshin_LA or email leonid.lapshin.a@gmail.com
|