autoforge-engine 0.1.2__tar.gz → 0.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.3/PKG-INFO +355 -0
- autoforge_engine-0.1.3/README.md +323 -0
- autoforge_engine-0.1.3/autoforge_engine.egg-info/PKG-INFO +355 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/autoforge_engine.egg-info/SOURCES.txt +2 -0
- autoforge_engine-0.1.3/modelforge/__init__.py +16 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/artifact_manager.py +484 -484
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/automl.py +1653 -1472
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/cli.py +1257 -1257
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/column_intelligence.py +403 -403
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/config.py +579 -579
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/cross_validation.py +748 -748
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/data_audit.py +391 -391
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/data_loader.py +83 -75
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/evaluation.py +396 -396
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/experiment_tracker.py +489 -489
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/explainability.py +345 -345
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/feature_engineering.py +392 -392
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/feature_selection.py +527 -527
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/hyperparameter_optimization.py +592 -592
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/model_registry.py +683 -683
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/model_screening.py +530 -530
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/persistence.py +455 -455
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/pipeline_generator.py +277 -277
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/prediction_validator.py +315 -315
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/preprocessing.py +178 -178
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/profiler.py +84 -84
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/ranking.py +350 -350
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/reproducibility.py +294 -294
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/reproducibility_integration.py +191 -191
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/run_manager.py +199 -199
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/modelforge/target_selector.py +107 -107
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/pyproject.toml +56 -56
- autoforge_engine-0.1.3/setup.cfg +7 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_artifact_manager.py +412 -412
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_automl.py +878 -851
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_automl_reproducibility.py +494 -494
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_automl_robustness.py +178 -178
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_cli.py +592 -592
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_cli_workflow.py +354 -354
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_column_intelligence.py +134 -134
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_config.py +385 -385
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_config_automl_integration.py +277 -277
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_cross_validation.py +430 -430
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_data_audit.py +152 -152
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_data_loader.py +55 -40
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_end_to_end.py +403 -403
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_evaluation.py +338 -338
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_experiment_artifacts.py +578 -578
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_experiment_tracker.py +310 -310
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_experiment_tracker_reproducibility.py +249 -249
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_explainability.py +423 -423
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_failure_isolation.py +251 -251
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_feature_engineering.py +216 -216
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_feature_selection.py +218 -218
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_full_system.py +756 -756
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_hyperparameter_optimization.py +359 -359
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_model_registry.py +231 -231
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_model_screening.py +616 -616
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_persistence.py +776 -776
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_pipeline_generator.py +385 -385
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_prediction_robustness.py +273 -273
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_prediction_validator.py +339 -339
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_preprocessing.py +192 -192
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_profiler.py +74 -74
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_ranking.py +445 -445
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_reproducibility.py +301 -301
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_reproducibility_integration.py +302 -302
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_robustness.py +132 -132
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_run_manager.py +226 -226
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/tests/test_target_selector.py +86 -86
- autoforge_engine-0.1.2/PKG-INFO +0 -109
- autoforge_engine-0.1.2/README.md +0 -78
- autoforge_engine-0.1.2/autoforge_engine.egg-info/PKG-INFO +0 -109
- autoforge_engine-0.1.2/setup.cfg +0 -4
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/LICENSE +0 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/autoforge_engine.egg-info/dependency_links.txt +0 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/autoforge_engine.egg-info/entry_points.txt +0 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/autoforge_engine.egg-info/requires.txt +0 -0
- {autoforge_engine-0.1.2 → autoforge_engine-0.1.3}/autoforge_engine.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: autoforge-engine
|
|
3
|
+
Version: 0.1.3
|
|
4
|
+
Summary: A transparent, local-first AutoML framework for automated ML pipeline discovery.
|
|
5
|
+
Author: Aditya Kumar Singh
|
|
6
|
+
Author-email: Aditya Kumar Singh <adityasingh45245@gmail.com>
|
|
7
|
+
License: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/Adityasinghrajput01/ModelForge
|
|
9
|
+
Project-URL: Repository, https://github.com/Adityasinghrajput01/ModelForge
|
|
10
|
+
Project-URL: Issues, https://github.com/Adityasinghrajput01/ModelForge/issues
|
|
11
|
+
Requires-Python: >=3.11
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Requires-Dist: numpy>=1.26
|
|
15
|
+
Requires-Dist: pandas>=2.1
|
|
16
|
+
Requires-Dist: scikit-learn>=1.4
|
|
17
|
+
Requires-Dist: rich>=13.7
|
|
18
|
+
Requires-Dist: typer>=0.12
|
|
19
|
+
Requires-Dist: pyyaml>=6.0
|
|
20
|
+
Provides-Extra: boosting
|
|
21
|
+
Requires-Dist: xgboost>=2.0; extra == "boosting"
|
|
22
|
+
Requires-Dist: lightgbm>=4.0; extra == "boosting"
|
|
23
|
+
Requires-Dist: catboost>=1.2; extra == "boosting"
|
|
24
|
+
Provides-Extra: optimization
|
|
25
|
+
Requires-Dist: optuna>=3.6; extra == "optimization"
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
28
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
29
|
+
Requires-Dist: black>=24.0; extra == "dev"
|
|
30
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
31
|
+
Dynamic: license-file
|
|
32
|
+
|
|
33
|
+
# ModelForge
|
|
34
|
+
|
|
35
|
+
> **Transparent, local-first AutoML experimentation for reproducible and inspectable machine learning.**
|
|
36
|
+
|
|
37
|
+
ModelForge is a Python AutoML library for preparing data, screening and ranking
|
|
38
|
+
scikit-learn pipelines, evaluating models, and saving models for later
|
|
39
|
+
predictions. Training runs also record experiment and reproducibility
|
|
40
|
+
information locally.
|
|
41
|
+
|
|
42
|
+
A typical ModelForge workflow profiles a dataset, selects a target, audits and
|
|
43
|
+
preprocesses its features, generates candidate pipelines, screens and ranks
|
|
44
|
+
models, and saves the chosen pipeline for prediction. The workflow stays
|
|
45
|
+
inspectable rather than hiding every stage behind a black-box call.
|
|
46
|
+
|
|
47
|
+
The package is published as **`autoforge-engine`** and imported in Python as
|
|
48
|
+
**`modelforge`**.
|
|
49
|
+
|
|
50
|
+
## Features
|
|
51
|
+
|
|
52
|
+
- Regression and classification workflows
|
|
53
|
+
- Data profiling, column intelligence, and data-quality auditing
|
|
54
|
+
- Preprocessing, feature engineering, and feature selection
|
|
55
|
+
- Model screening, cross-validation, and ranking
|
|
56
|
+
- Saved model pipelines and predictions from the command line or Python
|
|
57
|
+
- Local experiment tracking, run metadata, and reproducibility information
|
|
58
|
+
- Optional boosting models and hyperparameter optimization
|
|
59
|
+
|
|
60
|
+
## Requirements
|
|
61
|
+
|
|
62
|
+
- Python 3.11 or newer
|
|
63
|
+
- A local dataset in a supported format
|
|
64
|
+
|
|
65
|
+
The default installation includes NumPy, pandas, scikit-learn, Rich, Typer,
|
|
66
|
+
and PyYAML.
|
|
67
|
+
|
|
68
|
+
## Install
|
|
69
|
+
|
|
70
|
+
### Install from PyPI
|
|
71
|
+
|
|
72
|
+
Windows PowerShell:
|
|
73
|
+
|
|
74
|
+
```powershell
|
|
75
|
+
py -3.11 -m venv .venv
|
|
76
|
+
.\.venv\Scripts\Activate.ps1
|
|
77
|
+
python -m pip install --upgrade pip
|
|
78
|
+
python -m pip install autoforge-engine
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
If PowerShell does not allow virtual-environment activation, you can call its
|
|
82
|
+
Python executable directly:
|
|
83
|
+
|
|
84
|
+
```powershell
|
|
85
|
+
py -3.11 -m venv .venv
|
|
86
|
+
.\.venv\Scripts\python.exe -m pip install --upgrade pip
|
|
87
|
+
.\.venv\Scripts\python.exe -m pip install autoforge-engine
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Linux or macOS:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
python3 -m venv .venv
|
|
94
|
+
source .venv/bin/activate
|
|
95
|
+
python -m pip install --upgrade pip
|
|
96
|
+
python -m pip install autoforge-engine
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Optional integrations can be installed with extras:
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
python -m pip install "autoforge-engine[boosting,optimization]"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
The `boosting` extra installs XGBoost, LightGBM, and CatBoost. The
|
|
106
|
+
`optimization` extra installs Optuna.
|
|
107
|
+
|
|
108
|
+
### Install from source
|
|
109
|
+
|
|
110
|
+
Clone the repository, enter its directory, and install the development extras.
|
|
111
|
+
This makes the `modelforge` command and test tools available in the active
|
|
112
|
+
Python environment.
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
git clone https://github.com/Adityasinghrajput01/ModelForge.git
|
|
116
|
+
cd ModelForge
|
|
117
|
+
python -m venv .venv
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Windows PowerShell:
|
|
121
|
+
|
|
122
|
+
```powershell
|
|
123
|
+
.\.venv\Scripts\Activate.ps1
|
|
124
|
+
python -m pip install --upgrade pip
|
|
125
|
+
python -m pip install -e ".[dev]"
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Linux or macOS:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
source .venv/bin/activate
|
|
132
|
+
python -m pip install --upgrade pip
|
|
133
|
+
python -m pip install -e ".[dev]"
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
To install optional integrations from a source checkout, use
|
|
137
|
+
`python -m pip install -e ".[dev,boosting,optimization]"`.
|
|
138
|
+
|
|
139
|
+
## Dataset and file paths
|
|
140
|
+
|
|
141
|
+
ModelForge reads local files. Training through the Python API supports CSV,
|
|
142
|
+
Excel (`.xlsx` or `.xls`), Parquet, and JSON files. The CLI prediction command
|
|
143
|
+
expects a CSV file.
|
|
144
|
+
|
|
145
|
+
The training data must have a header row. Choose the column you want the model
|
|
146
|
+
to predict as the **target**; all other usable columns become input features.
|
|
147
|
+
For example, a regression CSV might look like this:
|
|
148
|
+
|
|
149
|
+
```csv
|
|
150
|
+
area,bedrooms,age,price
|
|
151
|
+
1200,2,15,250000
|
|
152
|
+
1850,3,8,385000
|
|
153
|
+
900,1,30,190000
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Here, `price` is the target. A prediction CSV should contain the feature
|
|
157
|
+
columns (`area`, `bedrooms`, and `age`) with compatible names, but should not
|
|
158
|
+
contain the target column.
|
|
159
|
+
|
|
160
|
+
Paths are interpreted from the directory where you run the command or Python
|
|
161
|
+
script:
|
|
162
|
+
|
|
163
|
+
- A **relative path** such as `data/train.csv` starts from the current working
|
|
164
|
+
directory. Run commands from the repository root when using paths such as
|
|
165
|
+
`examples/house_prices.csv`.
|
|
166
|
+
- An **absolute path** works from any directory. On Windows, quote paths that
|
|
167
|
+
contain spaces, for example `"C:\Users\Aditya Kumar\Datasets\train.csv"`.
|
|
168
|
+
- In Python, use a raw string for a Windows path, such as
|
|
169
|
+
`r"C:\Users\Aditya Kumar\Datasets\train.csv"`, or use forward slashes:
|
|
170
|
+
`"C:/Users/Aditya Kumar/Datasets/train.csv"`.
|
|
171
|
+
|
|
172
|
+
## Train and predict with the CLI
|
|
173
|
+
|
|
174
|
+
After installation, check that the command is available:
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
modelforge --help
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
From the repository root, train a regression model using the included housing
|
|
181
|
+
dataset. The target column is `price`:
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
modelforge train \
|
|
185
|
+
--data "examples/house_prices.csv" \
|
|
186
|
+
--target price \
|
|
187
|
+
--task-type regression \
|
|
188
|
+
--output "house_prices_model.joblib"
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
Run the saved model on new rows. The example prediction file contains features
|
|
192
|
+
without the `price` target:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
modelforge predict \
|
|
196
|
+
--model "house_prices_model.joblib" \
|
|
197
|
+
--data "examples/house_prices_new.csv" \
|
|
198
|
+
--output "predictions.csv"
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
On Windows PowerShell, the same commands can be written on one line, or use a
|
|
202
|
+
backtick at the end of each continued line. For a dataset outside the
|
|
203
|
+
repository, supply its absolute path to `--data`.
|
|
204
|
+
|
|
205
|
+
For classification, use a classification dataset, provide its label column,
|
|
206
|
+
and set `--task-type classification`. For example, the included Iris dataset
|
|
207
|
+
uses `species` as its label:
|
|
208
|
+
|
|
209
|
+
```bash
|
|
210
|
+
modelforge train --data "examples/iris_classification.csv" --target species --task-type classification --output "iris_model.joblib"
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
To return class probabilities instead of predicted class labels, add
|
|
214
|
+
`--proba` to `modelforge predict`. This option is only for classification
|
|
215
|
+
models.
|
|
216
|
+
|
|
217
|
+
Useful CLI commands:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
modelforge train --help
|
|
221
|
+
modelforge predict --help
|
|
222
|
+
modelforge models --task-type classification
|
|
223
|
+
modelforge experiments list --directory .modelforge/experiments
|
|
224
|
+
modelforge experiments get EXPERIMENT_ID
|
|
225
|
+
modelforge experiments compare ID_ONE ID_TWO
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
By default, experiment records are written to `.modelforge/experiments` in
|
|
229
|
+
the current working directory. The model is saved to the path passed to
|
|
230
|
+
`--output` (default: `model.joblib`). Use `--overwrite` to replace an existing
|
|
231
|
+
model file.
|
|
232
|
+
|
|
233
|
+
## Use ModelForge from Python
|
|
234
|
+
|
|
235
|
+
For a one-call workflow that prints a dataset, model-ranking, data-quality,
|
|
236
|
+
and run report, use the lowercase `automl` helper. It accepts a file path or a
|
|
237
|
+
pandas DataFrame and returns the fitted `AutoML` instance:
|
|
238
|
+
|
|
239
|
+
```python
|
|
240
|
+
from modelforge import automl
|
|
241
|
+
|
|
242
|
+
run = automl("data/heart_failure.csv", "DEATH_EVENT")
|
|
243
|
+
predictions = run.predict("data/new_patients.csv")
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
Pass options such as `task_type="classification"`, `cv=5`, or
|
|
247
|
+
`model_names=["logistic_regression"]` as keyword arguments when needed.
|
|
248
|
+
|
|
249
|
+
Pass a pandas DataFrame to `AutoML.fit`, name the target column, then save the
|
|
250
|
+
fitted pipeline. Replace the example path with the path to your own dataset.
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
from pathlib import Path
|
|
254
|
+
|
|
255
|
+
import pandas as pd
|
|
256
|
+
from modelforge import AutoML
|
|
257
|
+
|
|
258
|
+
data_path = Path("examples/house_prices.csv")
|
|
259
|
+
data = pd.read_csv(data_path)
|
|
260
|
+
|
|
261
|
+
automl = AutoML(cv=5, random_state=42)
|
|
262
|
+
result = automl.fit(
|
|
263
|
+
data,
|
|
264
|
+
target="price",
|
|
265
|
+
task_type="regression",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
model_path = Path("house_prices_model.joblib")
|
|
269
|
+
automl.save(str(model_path), overwrite=True)
|
|
270
|
+
print("Best model:", result["best_model"])
|
|
271
|
+
|
|
272
|
+
new_data = pd.read_csv("examples/house_prices_new.csv")
|
|
273
|
+
predictions = automl.predict(new_data)
|
|
274
|
+
print(predictions.head())
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
For a file path outside the working directory, set `data_path` to an absolute
|
|
278
|
+
path, for example `Path(r"C:\Users\Aditya Kumar\Datasets\train.csv")`.
|
|
279
|
+
|
|
280
|
+
## Configuration
|
|
281
|
+
|
|
282
|
+
You can put training settings in a YAML file and pass it to the CLI with
|
|
283
|
+
`--config`. For example:
|
|
284
|
+
|
|
285
|
+
```yaml
|
|
286
|
+
target: price
|
|
287
|
+
task_type: regression
|
|
288
|
+
objective: balanced
|
|
289
|
+
test_size: 0.2
|
|
290
|
+
cv: 5
|
|
291
|
+
random_state: 42
|
|
292
|
+
experiment_directory: .modelforge/experiments
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
Save this as `modelforge.yaml`, then run:
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
modelforge train --data "examples/house_prices.csv" --config "modelforge.yaml" --output "house_prices_model.joblib"
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
The CLI options `--target`, `--task-type`, `--objective`, `--cv`, and
|
|
302
|
+
`--test-size` can also be set directly on the command line. See
|
|
303
|
+
`examples/modelforge_config.yaml` for a configuration that includes feature
|
|
304
|
+
selection settings.
|
|
305
|
+
|
|
306
|
+
## Run tests and checks
|
|
307
|
+
|
|
308
|
+
Install the source checkout with the `dev` extra first, then run these from
|
|
309
|
+
the repository root with the virtual environment activated:
|
|
310
|
+
|
|
311
|
+
```bash
|
|
312
|
+
python -m pytest -q
|
|
313
|
+
ruff check .
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
To run a single test module while developing:
|
|
317
|
+
|
|
318
|
+
```bash
|
|
319
|
+
python -m pytest tests/test_cli_workflow.py -q
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
To exercise the included Python workflows and benchmark:
|
|
323
|
+
|
|
324
|
+
```bash
|
|
325
|
+
python examples/end_to_end_regression.py
|
|
326
|
+
python benchmarks/benchmark_baselines.py
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
To verify a classification train-and-predict workflow with the included Iris
|
|
330
|
+
files:
|
|
331
|
+
|
|
332
|
+
```bash
|
|
333
|
+
modelforge train --data "examples/iris_classification.csv" --target species --task-type classification --output "iris_model.joblib"
|
|
334
|
+
modelforge predict --model "iris_model.joblib" --data "examples/iris_new.csv" --output "iris_predictions.csv"
|
|
335
|
+
```
|
|
336
|
+
|
|
337
|
+
To check the installed CLI and the available model names:
|
|
338
|
+
|
|
339
|
+
```bash
|
|
340
|
+
modelforge --help
|
|
341
|
+
modelforge models --task-type regression
|
|
342
|
+
modelforge models --task-type classification
|
|
343
|
+
```
|
|
344
|
+
|
|
345
|
+
## Project contents
|
|
346
|
+
|
|
347
|
+
- `examples/` contains sample datasets, YAML configuration, and end-to-end
|
|
348
|
+
workflows.
|
|
349
|
+
- `tests/` contains the automated test suite.
|
|
350
|
+
- `benchmarks/` contains a baseline comparison script.
|
|
351
|
+
- `docs/ROADMAP.md` describes planned project work.
|
|
352
|
+
|
|
353
|
+
## License
|
|
354
|
+
|
|
355
|
+
ModelForge is distributed under the MIT License. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
# ModelForge
|
|
2
|
+
|
|
3
|
+
> **Transparent, local-first AutoML experimentation for reproducible and inspectable machine learning.**
|
|
4
|
+
|
|
5
|
+
ModelForge is a Python AutoML library for preparing data, screening and ranking
|
|
6
|
+
scikit-learn pipelines, evaluating models, and saving models for later
|
|
7
|
+
predictions. Training runs also record experiment and reproducibility
|
|
8
|
+
information locally.
|
|
9
|
+
|
|
10
|
+
A typical ModelForge workflow profiles a dataset, selects a target, audits and
|
|
11
|
+
preprocesses its features, generates candidate pipelines, screens and ranks
|
|
12
|
+
models, and saves the chosen pipeline for prediction. The workflow stays
|
|
13
|
+
inspectable rather than hiding every stage behind a black-box call.
|
|
14
|
+
|
|
15
|
+
The package is published as **`autoforge-engine`** and imported in Python as
|
|
16
|
+
**`modelforge`**.
|
|
17
|
+
|
|
18
|
+
## Features
|
|
19
|
+
|
|
20
|
+
- Regression and classification workflows
|
|
21
|
+
- Data profiling, column intelligence, and data-quality auditing
|
|
22
|
+
- Preprocessing, feature engineering, and feature selection
|
|
23
|
+
- Model screening, cross-validation, and ranking
|
|
24
|
+
- Saved model pipelines and predictions from the command line or Python
|
|
25
|
+
- Local experiment tracking, run metadata, and reproducibility information
|
|
26
|
+
- Optional boosting models and hyperparameter optimization
|
|
27
|
+
|
|
28
|
+
## Requirements
|
|
29
|
+
|
|
30
|
+
- Python 3.11 or newer
|
|
31
|
+
- A local dataset in a supported format
|
|
32
|
+
|
|
33
|
+
The default installation includes NumPy, pandas, scikit-learn, Rich, Typer,
|
|
34
|
+
and PyYAML.
|
|
35
|
+
|
|
36
|
+
## Install
|
|
37
|
+
|
|
38
|
+
### Install from PyPI
|
|
39
|
+
|
|
40
|
+
Windows PowerShell:
|
|
41
|
+
|
|
42
|
+
```powershell
|
|
43
|
+
py -3.11 -m venv .venv
|
|
44
|
+
.\.venv\Scripts\Activate.ps1
|
|
45
|
+
python -m pip install --upgrade pip
|
|
46
|
+
python -m pip install autoforge-engine
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
If PowerShell does not allow virtual-environment activation, you can call its
|
|
50
|
+
Python executable directly:
|
|
51
|
+
|
|
52
|
+
```powershell
|
|
53
|
+
py -3.11 -m venv .venv
|
|
54
|
+
.\.venv\Scripts\python.exe -m pip install --upgrade pip
|
|
55
|
+
.\.venv\Scripts\python.exe -m pip install autoforge-engine
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Linux or macOS:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
python3 -m venv .venv
|
|
62
|
+
source .venv/bin/activate
|
|
63
|
+
python -m pip install --upgrade pip
|
|
64
|
+
python -m pip install autoforge-engine
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Optional integrations can be installed with extras:
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
python -m pip install "autoforge-engine[boosting,optimization]"
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The `boosting` extra installs XGBoost, LightGBM, and CatBoost. The
|
|
74
|
+
`optimization` extra installs Optuna.
|
|
75
|
+
|
|
76
|
+
### Install from source
|
|
77
|
+
|
|
78
|
+
Clone the repository, enter its directory, and install the development extras.
|
|
79
|
+
This makes the `modelforge` command and test tools available in the active
|
|
80
|
+
Python environment.
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
git clone https://github.com/Adityasinghrajput01/ModelForge.git
|
|
84
|
+
cd ModelForge
|
|
85
|
+
python -m venv .venv
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Windows PowerShell:
|
|
89
|
+
|
|
90
|
+
```powershell
|
|
91
|
+
.\.venv\Scripts\Activate.ps1
|
|
92
|
+
python -m pip install --upgrade pip
|
|
93
|
+
python -m pip install -e ".[dev]"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Linux or macOS:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
source .venv/bin/activate
|
|
100
|
+
python -m pip install --upgrade pip
|
|
101
|
+
python -m pip install -e ".[dev]"
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
To install optional integrations from a source checkout, use
|
|
105
|
+
`python -m pip install -e ".[dev,boosting,optimization]"`.
|
|
106
|
+
|
|
107
|
+
## Dataset and file paths
|
|
108
|
+
|
|
109
|
+
ModelForge reads local files. Training through the Python API supports CSV,
|
|
110
|
+
Excel (`.xlsx` or `.xls`), Parquet, and JSON files. The CLI prediction command
|
|
111
|
+
expects a CSV file.
|
|
112
|
+
|
|
113
|
+
The training data must have a header row. Choose the column you want the model
|
|
114
|
+
to predict as the **target**; all other usable columns become input features.
|
|
115
|
+
For example, a regression CSV might look like this:
|
|
116
|
+
|
|
117
|
+
```csv
|
|
118
|
+
area,bedrooms,age,price
|
|
119
|
+
1200,2,15,250000
|
|
120
|
+
1850,3,8,385000
|
|
121
|
+
900,1,30,190000
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Here, `price` is the target. A prediction CSV should contain the feature
|
|
125
|
+
columns (`area`, `bedrooms`, and `age`) with compatible names, but should not
|
|
126
|
+
contain the target column.
|
|
127
|
+
|
|
128
|
+
Paths are interpreted from the directory where you run the command or Python
|
|
129
|
+
script:
|
|
130
|
+
|
|
131
|
+
- A **relative path** such as `data/train.csv` starts from the current working
|
|
132
|
+
directory. Run commands from the repository root when using paths such as
|
|
133
|
+
`examples/house_prices.csv`.
|
|
134
|
+
- An **absolute path** works from any directory. On Windows, quote paths that
|
|
135
|
+
contain spaces, for example `"C:\Users\Aditya Kumar\Datasets\train.csv"`.
|
|
136
|
+
- In Python, use a raw string for a Windows path, such as
|
|
137
|
+
`r"C:\Users\Aditya Kumar\Datasets\train.csv"`, or use forward slashes:
|
|
138
|
+
`"C:/Users/Aditya Kumar/Datasets/train.csv"`.
|
|
139
|
+
|
|
140
|
+
## Train and predict with the CLI
|
|
141
|
+
|
|
142
|
+
After installation, check that the command is available:
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
modelforge --help
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
From the repository root, train a regression model using the included housing
|
|
149
|
+
dataset. The target column is `price`:
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
modelforge train \
|
|
153
|
+
--data "examples/house_prices.csv" \
|
|
154
|
+
--target price \
|
|
155
|
+
--task-type regression \
|
|
156
|
+
--output "house_prices_model.joblib"
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Run the saved model on new rows. The example prediction file contains features
|
|
160
|
+
without the `price` target:
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
modelforge predict \
|
|
164
|
+
--model "house_prices_model.joblib" \
|
|
165
|
+
--data "examples/house_prices_new.csv" \
|
|
166
|
+
--output "predictions.csv"
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
On Windows PowerShell, the same commands can be written on one line, or use a
|
|
170
|
+
backtick at the end of each continued line. For a dataset outside the
|
|
171
|
+
repository, supply its absolute path to `--data`.
|
|
172
|
+
|
|
173
|
+
For classification, use a classification dataset, provide its label column,
|
|
174
|
+
and set `--task-type classification`. For example, the included Iris dataset
|
|
175
|
+
uses `species` as its label:
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
modelforge train --data "examples/iris_classification.csv" --target species --task-type classification --output "iris_model.joblib"
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
To return class probabilities instead of predicted class labels, add
|
|
182
|
+
`--proba` to `modelforge predict`. This option is only for classification
|
|
183
|
+
models.
|
|
184
|
+
|
|
185
|
+
Useful CLI commands:
|
|
186
|
+
|
|
187
|
+
```bash
|
|
188
|
+
modelforge train --help
|
|
189
|
+
modelforge predict --help
|
|
190
|
+
modelforge models --task-type classification
|
|
191
|
+
modelforge experiments list --directory .modelforge/experiments
|
|
192
|
+
modelforge experiments get EXPERIMENT_ID
|
|
193
|
+
modelforge experiments compare ID_ONE ID_TWO
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
By default, experiment records are written to `.modelforge/experiments` in
|
|
197
|
+
the current working directory. The model is saved to the path passed to
|
|
198
|
+
`--output` (default: `model.joblib`). Use `--overwrite` to replace an existing
|
|
199
|
+
model file.
|
|
200
|
+
|
|
201
|
+
## Use ModelForge from Python
|
|
202
|
+
|
|
203
|
+
For a one-call workflow that prints a dataset, model-ranking, data-quality,
|
|
204
|
+
and run report, use the lowercase `automl` helper. It accepts a file path or a
|
|
205
|
+
pandas DataFrame and returns the fitted `AutoML` instance:
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
from modelforge import automl
|
|
209
|
+
|
|
210
|
+
run = automl("data/heart_failure.csv", "DEATH_EVENT")
|
|
211
|
+
predictions = run.predict("data/new_patients.csv")
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
Pass options such as `task_type="classification"`, `cv=5`, or
|
|
215
|
+
`model_names=["logistic_regression"]` as keyword arguments when needed.
|
|
216
|
+
|
|
217
|
+
Pass a pandas DataFrame to `AutoML.fit`, name the target column, then save the
|
|
218
|
+
fitted pipeline. Replace the example path with the path to your own dataset.
|
|
219
|
+
|
|
220
|
+
```python
|
|
221
|
+
from pathlib import Path
|
|
222
|
+
|
|
223
|
+
import pandas as pd
|
|
224
|
+
from modelforge import AutoML
|
|
225
|
+
|
|
226
|
+
data_path = Path("examples/house_prices.csv")
|
|
227
|
+
data = pd.read_csv(data_path)
|
|
228
|
+
|
|
229
|
+
automl = AutoML(cv=5, random_state=42)
|
|
230
|
+
result = automl.fit(
|
|
231
|
+
data,
|
|
232
|
+
target="price",
|
|
233
|
+
task_type="regression",
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
model_path = Path("house_prices_model.joblib")
|
|
237
|
+
automl.save(str(model_path), overwrite=True)
|
|
238
|
+
print("Best model:", result["best_model"])
|
|
239
|
+
|
|
240
|
+
new_data = pd.read_csv("examples/house_prices_new.csv")
|
|
241
|
+
predictions = automl.predict(new_data)
|
|
242
|
+
print(predictions.head())
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
For a file path outside the working directory, set `data_path` to an absolute
|
|
246
|
+
path, for example `Path(r"C:\Users\Aditya Kumar\Datasets\train.csv")`.
|
|
247
|
+
|
|
248
|
+
## Configuration
|
|
249
|
+
|
|
250
|
+
You can put training settings in a YAML file and pass it to the CLI with
|
|
251
|
+
`--config`. For example:
|
|
252
|
+
|
|
253
|
+
```yaml
|
|
254
|
+
target: price
|
|
255
|
+
task_type: regression
|
|
256
|
+
objective: balanced
|
|
257
|
+
test_size: 0.2
|
|
258
|
+
cv: 5
|
|
259
|
+
random_state: 42
|
|
260
|
+
experiment_directory: .modelforge/experiments
|
|
261
|
+
```
|
|
262
|
+
|
|
263
|
+
Save this as `modelforge.yaml`, then run:
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
modelforge train --data "examples/house_prices.csv" --config "modelforge.yaml" --output "house_prices_model.joblib"
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
The CLI options `--target`, `--task-type`, `--objective`, `--cv`, and
|
|
270
|
+
`--test-size` can also be set directly on the command line. See
|
|
271
|
+
`examples/modelforge_config.yaml` for a configuration that includes feature
|
|
272
|
+
selection settings.
|
|
273
|
+
|
|
274
|
+
## Run tests and checks
|
|
275
|
+
|
|
276
|
+
Install the source checkout with the `dev` extra first, then run these from
|
|
277
|
+
the repository root with the virtual environment activated:
|
|
278
|
+
|
|
279
|
+
```bash
|
|
280
|
+
python -m pytest -q
|
|
281
|
+
ruff check .
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
To run a single test module while developing:
|
|
285
|
+
|
|
286
|
+
```bash
|
|
287
|
+
python -m pytest tests/test_cli_workflow.py -q
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
To exercise the included Python workflows and benchmark:
|
|
291
|
+
|
|
292
|
+
```bash
|
|
293
|
+
python examples/end_to_end_regression.py
|
|
294
|
+
python benchmarks/benchmark_baselines.py
|
|
295
|
+
```
|
|
296
|
+
|
|
297
|
+
To verify a classification train-and-predict workflow with the included Iris
|
|
298
|
+
files:
|
|
299
|
+
|
|
300
|
+
```bash
|
|
301
|
+
modelforge train --data "examples/iris_classification.csv" --target species --task-type classification --output "iris_model.joblib"
|
|
302
|
+
modelforge predict --model "iris_model.joblib" --data "examples/iris_new.csv" --output "iris_predictions.csv"
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
To check the installed CLI and the available model names:
|
|
306
|
+
|
|
307
|
+
```bash
|
|
308
|
+
modelforge --help
|
|
309
|
+
modelforge models --task-type regression
|
|
310
|
+
modelforge models --task-type classification
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
## Project contents
|
|
314
|
+
|
|
315
|
+
- `examples/` contains sample datasets, YAML configuration, and end-to-end
|
|
316
|
+
workflows.
|
|
317
|
+
- `tests/` contains the automated test suite.
|
|
318
|
+
- `benchmarks/` contains a baseline comparison script.
|
|
319
|
+
- `docs/ROADMAP.md` describes planned project work.
|
|
320
|
+
|
|
321
|
+
## License
|
|
322
|
+
|
|
323
|
+
ModelForge is distributed under the MIT License. See [LICENSE](LICENSE).
|