autoforge-engine 0.1.3__tar.gz → 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. autoforge_engine-1.0.0/PKG-INFO +425 -0
  2. autoforge_engine-1.0.0/README.md +393 -0
  3. autoforge_engine-1.0.0/autoforge_engine.egg-info/PKG-INFO +425 -0
  4. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/autoforge_engine.egg-info/SOURCES.txt +1 -0
  5. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/__init__.py +15 -15
  6. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/artifact_manager.py +484 -484
  7. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/automl.py +2092 -1652
  8. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/cli.py +1336 -1257
  9. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/column_intelligence.py +403 -403
  10. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/config.py +579 -579
  11. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/cross_validation.py +748 -748
  12. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/data_audit.py +391 -391
  13. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/data_loader.py +73 -73
  14. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/evaluation.py +396 -396
  15. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/experiment_tracker.py +489 -489
  16. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/explainability.py +345 -345
  17. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/feature_engineering.py +392 -392
  18. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/feature_selection.py +527 -527
  19. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/hyperparameter_optimization.py +592 -592
  20. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/model_registry.py +682 -683
  21. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/model_screening.py +530 -530
  22. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/persistence.py +455 -455
  23. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/pipeline_generator.py +277 -277
  24. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/prediction_validator.py +315 -315
  25. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/preprocessing.py +178 -178
  26. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/profiler.py +84 -84
  27. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/ranking.py +350 -350
  28. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/reproducibility.py +294 -294
  29. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/reproducibility_integration.py +191 -191
  30. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/run_manager.py +199 -199
  31. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/modelforge/target_selector.py +107 -107
  32. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/pyproject.toml +58 -57
  33. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/setup.cfg +7 -7
  34. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_artifact_manager.py +412 -412
  35. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_automl.py +962 -878
  36. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_automl_reproducibility.py +494 -494
  37. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_automl_robustness.py +178 -178
  38. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_cli.py +604 -592
  39. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_cli_workflow.py +354 -354
  40. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_column_intelligence.py +134 -134
  41. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_config.py +385 -385
  42. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_config_automl_integration.py +279 -277
  43. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_cross_validation.py +430 -430
  44. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_data_audit.py +152 -152
  45. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_data_loader.py +43 -43
  46. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_end_to_end.py +403 -403
  47. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_evaluation.py +338 -338
  48. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_experiment_artifacts.py +578 -578
  49. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_experiment_tracker.py +310 -310
  50. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_experiment_tracker_reproducibility.py +249 -249
  51. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_explainability.py +423 -423
  52. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_failure_isolation.py +251 -251
  53. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_feature_engineering.py +216 -216
  54. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_feature_selection.py +218 -218
  55. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_full_system.py +756 -756
  56. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_hyperparameter_optimization.py +359 -359
  57. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_model_registry.py +245 -231
  58. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_model_screening.py +616 -616
  59. autoforge_engine-1.0.0/tests/test_model_selection_export.py +600 -0
  60. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_persistence.py +776 -776
  61. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_pipeline_generator.py +385 -385
  62. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_prediction_robustness.py +273 -273
  63. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_prediction_validator.py +339 -339
  64. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_preprocessing.py +192 -192
  65. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_profiler.py +74 -74
  66. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_ranking.py +445 -445
  67. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_reproducibility.py +301 -301
  68. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_reproducibility_integration.py +302 -302
  69. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_robustness.py +132 -132
  70. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_run_manager.py +226 -226
  71. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/tests/test_target_selector.py +86 -86
  72. autoforge_engine-0.1.3/PKG-INFO +0 -355
  73. autoforge_engine-0.1.3/README.md +0 -323
  74. autoforge_engine-0.1.3/autoforge_engine.egg-info/PKG-INFO +0 -355
  75. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/LICENSE +0 -0
  76. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/autoforge_engine.egg-info/dependency_links.txt +0 -0
  77. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/autoforge_engine.egg-info/entry_points.txt +0 -0
  78. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/autoforge_engine.egg-info/requires.txt +0 -0
  79. {autoforge_engine-0.1.3 → autoforge_engine-1.0.0}/autoforge_engine.egg-info/top_level.txt +0 -0
@@ -0,0 +1,425 @@
1
+ Metadata-Version: 2.4
2
+ Name: autoforge-engine
3
+ Version: 1.0.0
4
+ Summary: A transparent, local-first AutoML framework for automated ML pipeline discovery.
5
+ Author: Aditya Kumar Singh
6
+ Author-email: Aditya Kumar Singh <adityasingh45245@gmail.com>
7
+ License: MIT
8
+ Project-URL: Homepage, https://github.com/Adityasinghrajput01/ModelForge
9
+ Project-URL: Repository, https://github.com/Adityasinghrajput01/ModelForge
10
+ Project-URL: Issues, https://github.com/Adityasinghrajput01/ModelForge/issues
11
+ Requires-Python: >=3.11
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE
14
+ Requires-Dist: numpy>=1.26
15
+ Requires-Dist: pandas>=2.1
16
+ Requires-Dist: scikit-learn>=1.4
17
+ Requires-Dist: rich>=13.7
18
+ Requires-Dist: typer>=0.12
19
+ Requires-Dist: pyyaml>=6.0
20
+ Provides-Extra: boosting
21
+ Requires-Dist: xgboost>=2.0; extra == "boosting"
22
+ Requires-Dist: lightgbm>=4.0; extra == "boosting"
23
+ Requires-Dist: catboost>=1.2; extra == "boosting"
24
+ Provides-Extra: optimization
25
+ Requires-Dist: optuna>=3.6; extra == "optimization"
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest>=8.0; extra == "dev"
28
+ Requires-Dist: ruff>=0.5; extra == "dev"
29
+ Requires-Dist: black>=24.0; extra == "dev"
30
+ Requires-Dist: mypy>=1.10; extra == "dev"
31
+ Dynamic: license-file
32
+
33
+ <div align="center">
34
+
35
+ # 🔨 ModelForge
36
+
37
+ **A transparent, local-first AutoML framework for Python**
38
+
39
+ [![PyPI version](https://img.shields.io/pypi/v/autoforge-engine.svg)](https://pypi.org/project/autoforge-engine/)
40
+ [![Python](https://img.shields.io/badge/python-3.8%2B-blue.svg)](https://pypi.org/project/autoforge-engine/)
41
+ [![License](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
42
+ [![Tests](https://img.shields.io/badge/tests-491%20passing-brightgreen.svg)](#testing--validation)
43
+ [![GitHub](https://img.shields.io/badge/GitHub-ModelForge-181717?logo=github)](https://github.com/Adityasinghrajput01/ModelForge)
44
+
45
+ *Give it a dataset and a target column. It handles the rest — transparently.*
46
+
47
+ [Installation](#installation) •
48
+ [Quick Start](#quick-start) •
49
+ [Configuration](#configuration) •
50
+ [Architecture](#architecture) •
51
+ [Stress Test Results](#stress-test-results) •
52
+ [Roadmap](#roadmap)
53
+
54
+ </div>
55
+
56
+ ---
57
+
58
+ ## What is ModelForge?
59
+
60
+ **ModelForge** is a local-first AutoML framework that automates the repetitive parts of a machine learning workflow — data profiling, quality auditing, preprocessing, pipeline construction, model evaluation, cross-validation, ranking, and reporting — while staying inspectable at every step.
61
+
62
+ Unlike black-box AutoML tools, ModelForge is built around one core principle:
63
+
64
+ > **You should always be able to see what it did and why.**
65
+
66
+ Give it a dataset and a target column:
67
+
68
+ ```python
69
+ import pandas as pd
70
+ from modelforge.automl import AutoML
71
+
72
+ df = pd.read_csv("dataset.csv")
73
+
74
+ automl = AutoML()
75
+ automl.fit(df, target="target_column")
76
+ ```
77
+
78
+ And it walks through the full pipeline automatically:
79
+
80
+ ```
81
+ Dataset → Loading → Profiling → Data Quality Audit → Target/Task Detection
82
+ → Column Intelligence → Feature Selection/Engineering → Preprocessing
83
+ → Pipeline Generation → Multi-Model Evaluation → Cross-Validation
84
+ → Metric Calculation → Model Ranking → Best Model Selection
85
+ → (Optional) Hyperparameter Optimization → Explainability
86
+ → Persistence/Artifacts → Experiment Tracking → Human-Readable Report
87
+ ```
88
+
89
+ ---
90
+
91
+ ## Installation
92
+
93
+ ```bash
94
+ pip install autoforge-engine
95
+ ```
96
+
97
+ Install a specific version:
98
+
99
+ ```bash
100
+ pip install autoforge-engine==0.1.4
101
+ ```
102
+
103
+ Verify the install:
104
+
105
+ ```bash
106
+ pip show autoforge-engine
107
+ python -c "import modelforge; print(modelforge.__file__)"
108
+ ```
109
+
110
+ > **Note:** the PyPI distribution name is `autoforge-engine`, but the importable package is `modelforge`.
111
+
112
+ ---
113
+
114
+ ## Quick Start
115
+
116
+ ### Using the `AutoML` class
117
+
118
+ ```python
119
+ import pandas as pd
120
+ from modelforge.automl import AutoML
121
+
122
+ df = pd.read_csv("autoforge_10000_test_dataset.csv")
123
+
124
+ automl = AutoML()
125
+ automl.fit(df, target="loan_default")
126
+ ```
127
+
128
+ ### Using the convenience function
129
+
130
+ ```python
131
+ from modelforge.automl import automl
132
+
133
+ result = automl(
134
+ data=df,
135
+ target="loan_default"
136
+ )
137
+ ```
138
+
139
+ ### Command line
140
+
141
+ ```bash
142
+ modelforge --help
143
+ ```
144
+
145
+ ---
146
+
147
+ ## Selecting and Exporting a Model
148
+
149
+ ModelForge evaluates and ranks candidate models automatically. By default,
150
+ `best_model` / `best_pipeline` hold the top-ranked result, and `save()` persists
151
+ that automatically recommended pipeline.
152
+
153
+ You can override the recommendation after `fit()` without re-running AutoML.
154
+ The selected candidate is refitted on the full training dataset before it is
155
+ used for prediction or export. The exported artifact is the complete
156
+ scikit-learn pipeline (preprocessing + estimator). Metadata records how the
157
+ model was selected.
158
+
159
+ ### Automatic behavior
160
+
161
+ ```python
162
+ automl.fit(df, target="target")
163
+ automl.save("best_model.joblib")
164
+ ```
165
+
166
+ ### Select by rank
167
+
168
+ ```python
169
+ automl.fit(df, target="target")
170
+
171
+ automl.select_model(rank=2)
172
+
173
+ automl.export_model(
174
+ "selected_model.joblib"
175
+ )
176
+ ```
177
+
178
+ ### Select by model name
179
+
180
+ ```python
181
+ automl.select_model(
182
+ model="random_forest_classifier"
183
+ )
184
+
185
+ automl.export_model(
186
+ "random_forest_model.joblib"
187
+ )
188
+ ```
189
+
190
+ Notes:
191
+
192
+ - `best_model` remains the automatically ranked best model.
193
+ - `selected_model` / `selected_pipeline` hold the user override.
194
+ - After `select_model()`, `predict()` and `predict_proba()` use the selected
195
+ pipeline. Before selection they continue to use the best pipeline.
196
+ - `save()` always saves the automatically ranked best model.
197
+ - `export_model()` saves the selected model when one was chosen; otherwise it
198
+ saves the best model.
199
+
200
+ ### CLI selection
201
+
202
+ Export a ranked model without changing the default train workflow:
203
+
204
+ ```bash
205
+ modelforge train \
206
+ --data data.csv \
207
+ --target target \
208
+ --model-rank 2 \
209
+ --output random_forest_model.joblib
210
+ ```
211
+
212
+ Or select by registry model name:
213
+
214
+ ```bash
215
+ modelforge train \
216
+ --data data.csv \
217
+ --target target \
218
+ --model random_forest_classifier \
219
+ --output random_forest_model.joblib
220
+ ```
221
+
222
+ Existing `modelforge train` commands without `--model` / `--model-rank`
223
+ continue to save the automatically recommended best model.
224
+
225
+ ## Configuration
226
+
227
+ The `AutoML` constructor supports the following parameters:
228
+
229
+ ```python
230
+ AutoML(
231
+ test_size=0.2,
232
+ cv=5,
233
+ random_state=42,
234
+ objective="balanced",
235
+ variance_threshold=None,
236
+ correlation_threshold=None,
237
+ enable_optimization=False,
238
+ optimization_models=3,
239
+ optimization_max_trials=10,
240
+ experiment_directory=".modelforge/experiments",
241
+ config=None
242
+ )
243
+ ```
244
+
245
+ | Parameter | Description |
246
+ |---|---|
247
+ | `test_size` | Holdout test fraction |
248
+ | `cv` | Number of cross-validation folds |
249
+ | `random_state` | Reproducibility seed |
250
+ | `objective` | Model ranking objective (default: `"balanced"`) |
251
+ | `variance_threshold` | Optional low-variance feature filtering |
252
+ | `correlation_threshold` | Optional high-correlation feature filtering |
253
+ | `enable_optimization` | Enables/disables hyperparameter optimization |
254
+ | `optimization_models` | Number of models considered for optimization |
255
+ | `optimization_max_trials` | Maximum optimization trials |
256
+ | `experiment_directory` | Where experiment metadata/artifacts are stored |
257
+
258
+ The automatic post-fit report can be suppressed with `print_report=False` (if supported by the installed `fit()` signature).
259
+
260
+ ---
261
+
262
+ ## What ModelForge Does
263
+
264
+ ### 🧹 Data Quality Auditing
265
+ Automatically flags:
266
+ - Missing values
267
+ - Duplicate rows
268
+ - Constant columns
269
+ - Possible identifier columns
270
+ - Other dataset-quality findings
271
+
272
+ ### 🎯 Target & Task Detection
273
+ Given a target column, ModelForge determines whether the problem is **classification** or **regression**.
274
+
275
+ ### ⚙️ Preprocessing
276
+ Builds a leakage-safe `scikit-learn` `Pipeline` / `ColumnTransformer`:
277
+
278
+ | Data type | Steps |
279
+ |---|---|
280
+ | Numerical | `SimpleImputer` → `StandardScaler` |
281
+ | Categorical | `SimpleImputer` → `OneHotEncoder` |
282
+
283
+ Preprocessing is fit independently inside each cross-validation fold to prevent data leakage.
284
+
285
+ ### 🤖 Model Evaluation
286
+ Evaluates multiple candidate models (verified against source for the current release), including — for classification tasks — Logistic Regression, Random Forest, Extra Trees, Gradient Boosting, K-Nearest Neighbors, Support Vector Classifier, and Decision Tree.
287
+
288
+ ### 🔁 Cross-Validation
289
+ Uses `StratifiedKFold` (`n_splits=5`, `shuffle=True`, `random_state=42`) for classification tasks where class counts permit. Each fold clones the pipeline fresh, fits on the training split, and evaluates on the validation split.
290
+
291
+ ### 📊 Metrics
292
+
293
+ **Classification:** Accuracy, Precision, Recall, F1 (`average="weighted"`, `zero_division=0`), ROC-AUC (via `predict_proba` or `decision_function`), Log Loss.
294
+
295
+ **Regression:** R², Adjusted R², MAE, MSE, RMSE, MAPE.
296
+
297
+ ### 🏆 Model Ranking
298
+ Candidate models are ranked according to the configured `objective` (default: `"balanced"`) rather than by a single metric alone.
299
+
300
+ ### 📄 Automatic Reporting
301
+ After `fit()`, ModelForge generates a human-readable report covering dataset shape, target/task, data-quality findings, evaluated models and metrics, the selected model, pipeline steps, run/experiment IDs, and run duration.
302
+
303
+ ---
304
+
305
+ ## Architecture
306
+
307
+ ModelForge is organized into focused modules, each with a single responsibility:
308
+
309
+ | Module | Responsibility |
310
+ |---|---|
311
+ | `data_loader.py` | Dataset loading |
312
+ | `profiler.py` | Dataset profiling |
313
+ | `data_audit.py` | Data quality / auditing |
314
+ | `target_selector.py` | Target and task detection |
315
+ | `column_intelligence.py` | Column-level analysis |
316
+ | `feature_engineering.py` | Feature engineering |
317
+ | `feature_selection.py` | Feature selection |
318
+ | `preprocessing.py` | Imputation, scaling, encoding |
319
+ | `pipeline_generator.py` | sklearn pipeline construction |
320
+ | `model_registry.py` | Candidate model catalog |
321
+ | `model_screening.py` | Fast holdout evaluation |
322
+ | `cross_validation.py` | K-fold cross-validation |
323
+ | `evaluation.py` | Metric computation |
324
+ | `ranking.py` | Model ranking logic |
325
+ | `hyperparameter_optimization.py` | Optional HPO |
326
+ | `explainability.py` | Model explainability |
327
+ | `persistence.py` | Model save/load |
328
+ | `artifact_manager.py` | Artifact management |
329
+ | `experiment_tracker.py` | Experiment tracking |
330
+ | `reproducibility.py` / `reproducibility_integration.py` | Reproducibility guarantees |
331
+ | `prediction_validator.py` | Prediction-time validation |
332
+ | `run_manager.py` | Run metadata |
333
+ | `cli.py` | Typer-based CLI |
334
+ | `automl.py` | Orchestration layer |
335
+ | `config.py` | Configuration handling |
336
+
337
+ > **Note on Model Screening vs. Cross-Validation:** these are distinct evaluation paths. `ModelScreeningEngine` performs a fast `train_test_split` (80/20, stratified where applicable) holdout evaluation, while `CrossValidationEngine` performs full K-fold evaluation. Holdout screening metrics and CV metrics should not be conflated.
338
+
339
+ ---
340
+
341
+ ## Stress Test Results
342
+
343
+ ModelForge 0.1.4 was validated against a synthetic 10,020-row, 18-column dataset (target: `loan_default`) intentionally containing missing values, duplicates, a constant feature, and an identifier-like column.
344
+
345
+ **Data quality detected:** 1,000 missing cells · 20 duplicate rows · 3 quality findings (all correctly identified)
346
+
347
+ **Cross-validation accuracy:**
348
+
349
+ | Model | CV Accuracy |
350
+ |---|---|
351
+ | Logistic Regression | 69.51% |
352
+ | Gradient Boosting | 69.51% |
353
+ | SVC | 69.12% |
354
+ | Random Forest | 69.02% |
355
+ | Extra Trees | 68.26% |
356
+ | KNN | 64.52% |
357
+ | Decision Tree | 58.59% |
358
+
359
+ **Selected model:** Logistic Regression · **Run time:** ~33.25 seconds
360
+
361
+ > These figures are a functional stress-test snapshot, not a claim that any single algorithm is universally best.
362
+
363
+ ### Independent Validation
364
+
365
+ Results were cross-checked against raw `scikit-learn` implementations. Six of seven models matched almost exactly; Decision Tree showed a ~0.13 percentage-point difference (58.59% vs. 58.72%), currently logged as an open, low-priority investigation rather than a confirmed defect.
366
+
367
+ ---
368
+
369
+ ## Testing & Validation
370
+
371
+ - ✅ 491 tests passing locally
372
+ - ✅ GitHub Actions CI passing
373
+ - ✅ Package built (`.tar.gz` + `.whl`) and validated with `twine check`
374
+ - ✅ Published to PyPI: [`autoforge-engine`](https://pypi.org/project/autoforge-engine/0.1.4/)
375
+
376
+ Test coverage includes unit, integration, CLI, preprocessing, model evaluation, cross-validation, ranking, persistence, experiment tracking, edge cases, and large mixed-type stress tests.
377
+
378
+ ---
379
+
380
+ ## Development Workflow
381
+
382
+ ```bash
383
+ # Run tests
384
+ pytest
385
+
386
+ # Build the package
387
+ python -m build
388
+
389
+ # Validate the build
390
+ python -m twine check dist/*
391
+
392
+ # Upload to PyPI
393
+ python -m twine upload dist/*
394
+ ```
395
+
396
+ ---
397
+
398
+ ## Roadmap
399
+
400
+ - [ ] Investigate the minor Decision Tree CV discrepancy (`model_registry` / `pipeline_generator` / `preprocessing`)
401
+ - [ ] Expand hyperparameter optimization coverage
402
+ - [ ] Broaden regression model support
403
+ - [ ] Deepen explainability outputs
404
+ - [ ] Continued CLI and documentation improvements
405
+
406
+ ---
407
+
408
+ ## Philosophy
409
+
410
+ ModelForge is built to demonstrate serious ML engineering practice — not to replace a data scientist's judgment. It automates repeatable, mechanical steps of an ML workflow (preprocessing, evaluation, cross-validation, ranking, reporting) while keeping every step reproducible and inspectable, so the framework never becomes a black box.
411
+
412
+ ---
413
+
414
+ ## Author
415
+
416
+ **Aditya Kumar Singh**
417
+
418
+ ## License
419
+
420
+ [MIT](LICENSE)
421
+
422
+ ## Links
423
+
424
+ - 📦 PyPI: [autoforge-engine](https://pypi.org/project/autoforge-engine/)
425
+ - 💻 GitHub: [ModelForge](https://github.com/Adityasinghrajput01/ModelForge)