autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,531 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from sklearn.metrics import (
|
|
10
|
+
accuracy_score,
|
|
11
|
+
f1_score,
|
|
12
|
+
mean_absolute_error,
|
|
13
|
+
mean_squared_error,
|
|
14
|
+
precision_score,
|
|
15
|
+
r2_score,
|
|
16
|
+
recall_score,
|
|
17
|
+
roc_auc_score,
|
|
18
|
+
)
|
|
19
|
+
from sklearn.model_selection import train_test_split
|
|
20
|
+
from sklearn.pipeline import Pipeline
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ModelScreeningEngine:
|
|
24
|
+
"""
|
|
25
|
+
Train and evaluate candidate ModelForge pipelines.
|
|
26
|
+
|
|
27
|
+
This component performs fast model screening only.
|
|
28
|
+
It does not rank or select the final best model.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__(
|
|
32
|
+
self,
|
|
33
|
+
test_size: float = 0.2,
|
|
34
|
+
random_state: int = 42,
|
|
35
|
+
):
|
|
36
|
+
self.test_size = test_size
|
|
37
|
+
self.random_state = random_state
|
|
38
|
+
|
|
39
|
+
self._validate_configuration()
|
|
40
|
+
|
|
41
|
+
def screen(
|
|
42
|
+
self,
|
|
43
|
+
data: pd.DataFrame,
|
|
44
|
+
target: str,
|
|
45
|
+
pipelines: dict[str, Pipeline],
|
|
46
|
+
task_type: str,
|
|
47
|
+
) -> pd.DataFrame:
|
|
48
|
+
"""
|
|
49
|
+
Train and evaluate multiple candidate pipelines.
|
|
50
|
+
|
|
51
|
+
Parameters
|
|
52
|
+
----------
|
|
53
|
+
data:
|
|
54
|
+
Dataset containing features and target.
|
|
55
|
+
|
|
56
|
+
target:
|
|
57
|
+
Target column.
|
|
58
|
+
|
|
59
|
+
pipelines:
|
|
60
|
+
Candidate pipelines keyed by model name.
|
|
61
|
+
|
|
62
|
+
task_type:
|
|
63
|
+
'regression' or 'classification'.
|
|
64
|
+
|
|
65
|
+
Returns
|
|
66
|
+
-------
|
|
67
|
+
pd.DataFrame
|
|
68
|
+
One result row per candidate model.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
self._validate_inputs(
|
|
72
|
+
data=data,
|
|
73
|
+
target=target,
|
|
74
|
+
pipelines=pipelines,
|
|
75
|
+
task_type=task_type,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
X = data.drop(
|
|
79
|
+
columns=[target]
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
y = data[target]
|
|
83
|
+
|
|
84
|
+
stratify = (
|
|
85
|
+
y
|
|
86
|
+
if (
|
|
87
|
+
task_type == "classification"
|
|
88
|
+
and self._can_stratify(y)
|
|
89
|
+
)
|
|
90
|
+
else None
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
X_train, X_test, y_train, y_test = (
|
|
94
|
+
train_test_split(
|
|
95
|
+
X,
|
|
96
|
+
y,
|
|
97
|
+
test_size=self.test_size,
|
|
98
|
+
random_state=self.random_state,
|
|
99
|
+
stratify=stratify,
|
|
100
|
+
)
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
results: list[dict[str, Any]] = []
|
|
104
|
+
|
|
105
|
+
for model_name, pipeline in pipelines.items():
|
|
106
|
+
result = self._evaluate_pipeline(
|
|
107
|
+
model_name=model_name,
|
|
108
|
+
pipeline=pipeline,
|
|
109
|
+
X_train=X_train,
|
|
110
|
+
X_test=X_test,
|
|
111
|
+
y_train=y_train,
|
|
112
|
+
y_test=y_test,
|
|
113
|
+
task_type=task_type,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
results.append(result)
|
|
117
|
+
|
|
118
|
+
return pd.DataFrame(results)
|
|
119
|
+
|
|
120
|
+
def _evaluate_pipeline(
|
|
121
|
+
self,
|
|
122
|
+
model_name: str,
|
|
123
|
+
pipeline: Pipeline,
|
|
124
|
+
X_train: pd.DataFrame,
|
|
125
|
+
X_test: pd.DataFrame,
|
|
126
|
+
y_train: pd.Series,
|
|
127
|
+
y_test: pd.Series,
|
|
128
|
+
task_type: str,
|
|
129
|
+
) -> dict[str, Any]:
|
|
130
|
+
"""Train and evaluate one pipeline safely."""
|
|
131
|
+
|
|
132
|
+
start_time = time.perf_counter()
|
|
133
|
+
|
|
134
|
+
try:
|
|
135
|
+
fit_start = time.perf_counter()
|
|
136
|
+
|
|
137
|
+
pipeline.fit(
|
|
138
|
+
X_train,
|
|
139
|
+
y_train,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
training_time = (
|
|
143
|
+
time.perf_counter()
|
|
144
|
+
- fit_start
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
prediction_start = time.perf_counter()
|
|
148
|
+
|
|
149
|
+
predictions = pipeline.predict(
|
|
150
|
+
X_test
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
prediction_time = (
|
|
154
|
+
time.perf_counter()
|
|
155
|
+
- prediction_start
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
if task_type == "regression":
|
|
159
|
+
metrics = self._regression_metrics(
|
|
160
|
+
y_test,
|
|
161
|
+
predictions,
|
|
162
|
+
)
|
|
163
|
+
else:
|
|
164
|
+
metrics = self._classification_metrics(
|
|
165
|
+
pipeline,
|
|
166
|
+
X_test,
|
|
167
|
+
y_test,
|
|
168
|
+
predictions,
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
total_time = (
|
|
172
|
+
time.perf_counter()
|
|
173
|
+
- start_time
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
return {
|
|
177
|
+
"model": model_name,
|
|
178
|
+
"status": "success",
|
|
179
|
+
"training_time_seconds": float(
|
|
180
|
+
training_time
|
|
181
|
+
),
|
|
182
|
+
"prediction_time_seconds": float(
|
|
183
|
+
prediction_time
|
|
184
|
+
),
|
|
185
|
+
"total_time_seconds": float(
|
|
186
|
+
total_time
|
|
187
|
+
),
|
|
188
|
+
**metrics,
|
|
189
|
+
"error": None,
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
except Exception as exc:
|
|
193
|
+
total_time = (
|
|
194
|
+
time.perf_counter()
|
|
195
|
+
- start_time
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
return {
|
|
199
|
+
"model": model_name,
|
|
200
|
+
"status": "failed",
|
|
201
|
+
"training_time_seconds": float(
|
|
202
|
+
total_time
|
|
203
|
+
),
|
|
204
|
+
"prediction_time_seconds": None,
|
|
205
|
+
"total_time_seconds": float(
|
|
206
|
+
total_time
|
|
207
|
+
),
|
|
208
|
+
**self._empty_metrics(
|
|
209
|
+
task_type
|
|
210
|
+
),
|
|
211
|
+
"error": str(exc),
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
@staticmethod
|
|
215
|
+
def _regression_metrics(
|
|
216
|
+
y_true,
|
|
217
|
+
predictions,
|
|
218
|
+
) -> dict[str, float]:
|
|
219
|
+
"""Calculate regression metrics."""
|
|
220
|
+
|
|
221
|
+
mse = mean_squared_error(
|
|
222
|
+
y_true,
|
|
223
|
+
predictions,
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
rmse = float(
|
|
227
|
+
np.sqrt(mse)
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
return {
|
|
231
|
+
"r2": float(
|
|
232
|
+
r2_score(
|
|
233
|
+
y_true,
|
|
234
|
+
predictions,
|
|
235
|
+
)
|
|
236
|
+
),
|
|
237
|
+
"mae": float(
|
|
238
|
+
mean_absolute_error(
|
|
239
|
+
y_true,
|
|
240
|
+
predictions,
|
|
241
|
+
)
|
|
242
|
+
),
|
|
243
|
+
"mse": float(mse),
|
|
244
|
+
"rmse": rmse,
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
@staticmethod
|
|
248
|
+
def _classification_metrics(
|
|
249
|
+
pipeline: Pipeline,
|
|
250
|
+
X_test: pd.DataFrame,
|
|
251
|
+
y_test: pd.Series,
|
|
252
|
+
predictions,
|
|
253
|
+
) -> dict[str, float | None]:
|
|
254
|
+
"""Calculate classification metrics."""
|
|
255
|
+
|
|
256
|
+
metrics: dict[str, float | None] = {
|
|
257
|
+
"accuracy": float(
|
|
258
|
+
accuracy_score(
|
|
259
|
+
y_test,
|
|
260
|
+
predictions,
|
|
261
|
+
)
|
|
262
|
+
),
|
|
263
|
+
"precision": float(
|
|
264
|
+
precision_score(
|
|
265
|
+
y_test,
|
|
266
|
+
predictions,
|
|
267
|
+
average="weighted",
|
|
268
|
+
zero_division=0,
|
|
269
|
+
)
|
|
270
|
+
),
|
|
271
|
+
"recall": float(
|
|
272
|
+
recall_score(
|
|
273
|
+
y_test,
|
|
274
|
+
predictions,
|
|
275
|
+
average="weighted",
|
|
276
|
+
zero_division=0,
|
|
277
|
+
)
|
|
278
|
+
),
|
|
279
|
+
"f1": float(
|
|
280
|
+
f1_score(
|
|
281
|
+
y_test,
|
|
282
|
+
predictions,
|
|
283
|
+
average="weighted",
|
|
284
|
+
zero_division=0,
|
|
285
|
+
)
|
|
286
|
+
),
|
|
287
|
+
"roc_auc": None,
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
metrics["roc_auc"] = (
|
|
291
|
+
ModelScreeningEngine._calculate_roc_auc(
|
|
292
|
+
pipeline,
|
|
293
|
+
X_test,
|
|
294
|
+
y_test,
|
|
295
|
+
)
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
return metrics
|
|
299
|
+
|
|
300
|
+
@staticmethod
|
|
301
|
+
def _calculate_roc_auc(
|
|
302
|
+
pipeline: Pipeline,
|
|
303
|
+
X_test: pd.DataFrame,
|
|
304
|
+
y_test: pd.Series,
|
|
305
|
+
) -> float | None:
|
|
306
|
+
"""Calculate ROC-AUC using probabilities or decision scores."""
|
|
307
|
+
|
|
308
|
+
if hasattr(
|
|
309
|
+
pipeline,
|
|
310
|
+
"predict_proba",
|
|
311
|
+
):
|
|
312
|
+
try:
|
|
313
|
+
probabilities = pipeline.predict_proba(
|
|
314
|
+
X_test
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
probabilities = np.asarray(
|
|
318
|
+
probabilities
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
if probabilities.ndim != 2:
|
|
322
|
+
return None
|
|
323
|
+
|
|
324
|
+
if probabilities.shape[1] == 2:
|
|
325
|
+
return float(
|
|
326
|
+
roc_auc_score(
|
|
327
|
+
y_test,
|
|
328
|
+
probabilities[:, 1],
|
|
329
|
+
)
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
if probabilities.shape[1] > 2:
|
|
333
|
+
return float(
|
|
334
|
+
roc_auc_score(
|
|
335
|
+
y_test,
|
|
336
|
+
probabilities,
|
|
337
|
+
multi_class="ovr",
|
|
338
|
+
average="weighted",
|
|
339
|
+
)
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
except (
|
|
343
|
+
ValueError,
|
|
344
|
+
TypeError,
|
|
345
|
+
AttributeError,
|
|
346
|
+
):
|
|
347
|
+
return None
|
|
348
|
+
|
|
349
|
+
if hasattr(
|
|
350
|
+
pipeline,
|
|
351
|
+
"decision_function",
|
|
352
|
+
):
|
|
353
|
+
try:
|
|
354
|
+
decision_scores = (
|
|
355
|
+
pipeline.decision_function(
|
|
356
|
+
X_test
|
|
357
|
+
)
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
unique_classes = np.unique(
|
|
361
|
+
y_test
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
if len(unique_classes) == 2:
|
|
365
|
+
return float(
|
|
366
|
+
roc_auc_score(
|
|
367
|
+
y_test,
|
|
368
|
+
decision_scores,
|
|
369
|
+
)
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
if (
|
|
373
|
+
np.asarray(
|
|
374
|
+
decision_scores
|
|
375
|
+
).ndim == 2
|
|
376
|
+
):
|
|
377
|
+
return float(
|
|
378
|
+
roc_auc_score(
|
|
379
|
+
y_test,
|
|
380
|
+
decision_scores,
|
|
381
|
+
multi_class="ovr",
|
|
382
|
+
average="weighted",
|
|
383
|
+
)
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
except (
|
|
387
|
+
ValueError,
|
|
388
|
+
TypeError,
|
|
389
|
+
AttributeError,
|
|
390
|
+
):
|
|
391
|
+
return None
|
|
392
|
+
|
|
393
|
+
return None
|
|
394
|
+
|
|
395
|
+
@staticmethod
|
|
396
|
+
def _empty_metrics(
|
|
397
|
+
task_type: str,
|
|
398
|
+
) -> dict[str, None]:
|
|
399
|
+
"""Return empty metrics for failed models."""
|
|
400
|
+
|
|
401
|
+
if task_type == "regression":
|
|
402
|
+
return {
|
|
403
|
+
"r2": None,
|
|
404
|
+
"mae": None,
|
|
405
|
+
"mse": None,
|
|
406
|
+
"rmse": None,
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
return {
|
|
410
|
+
"accuracy": None,
|
|
411
|
+
"precision": None,
|
|
412
|
+
"recall": None,
|
|
413
|
+
"f1": None,
|
|
414
|
+
"roc_auc": None,
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
@staticmethod
|
|
418
|
+
def _can_stratify(
|
|
419
|
+
target: pd.Series,
|
|
420
|
+
) -> bool:
|
|
421
|
+
"""
|
|
422
|
+
Determine whether classification data can safely use
|
|
423
|
+
stratified splitting.
|
|
424
|
+
"""
|
|
425
|
+
|
|
426
|
+
if target.empty:
|
|
427
|
+
return False
|
|
428
|
+
|
|
429
|
+
class_counts = target.value_counts()
|
|
430
|
+
|
|
431
|
+
return bool(
|
|
432
|
+
len(class_counts) >= 2
|
|
433
|
+
and class_counts.min() >= 2
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
def _validate_configuration(self) -> None:
|
|
437
|
+
"""Validate engine configuration."""
|
|
438
|
+
|
|
439
|
+
if not 0 < self.test_size < 1:
|
|
440
|
+
raise ValueError(
|
|
441
|
+
"test_size must be between 0 and 1."
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
if not isinstance(
|
|
445
|
+
self.random_state,
|
|
446
|
+
int,
|
|
447
|
+
):
|
|
448
|
+
raise TypeError(
|
|
449
|
+
"random_state must be an integer."
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
@staticmethod
|
|
453
|
+
def _validate_inputs(
|
|
454
|
+
data: pd.DataFrame,
|
|
455
|
+
target: str,
|
|
456
|
+
pipelines: dict[str, Pipeline],
|
|
457
|
+
task_type: str,
|
|
458
|
+
) -> None:
|
|
459
|
+
"""Validate screening inputs."""
|
|
460
|
+
|
|
461
|
+
if not isinstance(
|
|
462
|
+
data,
|
|
463
|
+
pd.DataFrame,
|
|
464
|
+
):
|
|
465
|
+
raise TypeError(
|
|
466
|
+
"data must be a pandas DataFrame."
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
if data.empty:
|
|
470
|
+
raise ValueError(
|
|
471
|
+
"Cannot screen models on an empty dataset."
|
|
472
|
+
)
|
|
473
|
+
|
|
474
|
+
if not isinstance(
|
|
475
|
+
target,
|
|
476
|
+
str,
|
|
477
|
+
):
|
|
478
|
+
raise TypeError(
|
|
479
|
+
"target must be a string."
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
if target not in data.columns:
|
|
483
|
+
raise ValueError(
|
|
484
|
+
f"Target column '{target}' "
|
|
485
|
+
"does not exist."
|
|
486
|
+
)
|
|
487
|
+
|
|
488
|
+
if not isinstance(
|
|
489
|
+
pipelines,
|
|
490
|
+
dict,
|
|
491
|
+
):
|
|
492
|
+
raise TypeError(
|
|
493
|
+
"pipelines must be a dictionary."
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
if not pipelines:
|
|
497
|
+
raise ValueError(
|
|
498
|
+
"At least one pipeline is required."
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
if task_type not in {
|
|
502
|
+
"regression",
|
|
503
|
+
"classification",
|
|
504
|
+
}:
|
|
505
|
+
raise ValueError(
|
|
506
|
+
"task_type must be 'regression' "
|
|
507
|
+
"or 'classification'."
|
|
508
|
+
)
|
|
509
|
+
|
|
510
|
+
for name, pipeline in pipelines.items():
|
|
511
|
+
if not isinstance(
|
|
512
|
+
name,
|
|
513
|
+
str,
|
|
514
|
+
):
|
|
515
|
+
raise TypeError(
|
|
516
|
+
"Pipeline names must be strings."
|
|
517
|
+
)
|
|
518
|
+
|
|
519
|
+
if not name.strip():
|
|
520
|
+
raise ValueError(
|
|
521
|
+
"Pipeline names cannot be empty."
|
|
522
|
+
)
|
|
523
|
+
|
|
524
|
+
if not isinstance(
|
|
525
|
+
pipeline,
|
|
526
|
+
Pipeline,
|
|
527
|
+
):
|
|
528
|
+
raise TypeError(
|
|
529
|
+
f"Pipeline '{name}' must be "
|
|
530
|
+
"a sklearn Pipeline."
|
|
531
|
+
)
|