autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
modelforge/evaluation.py
ADDED
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from sklearn.metrics import (
|
|
7
|
+
accuracy_score,
|
|
8
|
+
f1_score,
|
|
9
|
+
log_loss,
|
|
10
|
+
mean_absolute_error,
|
|
11
|
+
mean_absolute_percentage_error,
|
|
12
|
+
mean_squared_error,
|
|
13
|
+
precision_score,
|
|
14
|
+
r2_score,
|
|
15
|
+
recall_score,
|
|
16
|
+
roc_auc_score,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class EvaluationEngine:
|
|
21
|
+
"""
|
|
22
|
+
Centralized evaluation engine for ModelForge.
|
|
23
|
+
|
|
24
|
+
Provides standardized metrics for regression
|
|
25
|
+
and classification tasks.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
REGRESSION_METRICS = (
|
|
29
|
+
"r2",
|
|
30
|
+
"adjusted_r2",
|
|
31
|
+
"mae",
|
|
32
|
+
"mse",
|
|
33
|
+
"rmse",
|
|
34
|
+
"mape",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
CLASSIFICATION_METRICS = (
|
|
38
|
+
"accuracy",
|
|
39
|
+
"precision",
|
|
40
|
+
"recall",
|
|
41
|
+
"f1",
|
|
42
|
+
"roc_auc",
|
|
43
|
+
"log_loss",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
METRIC_DIRECTIONS = {
|
|
47
|
+
"r2": "maximize",
|
|
48
|
+
"adjusted_r2": "maximize",
|
|
49
|
+
"mae": "minimize",
|
|
50
|
+
"mse": "minimize",
|
|
51
|
+
"rmse": "minimize",
|
|
52
|
+
"mape": "minimize",
|
|
53
|
+
"accuracy": "maximize",
|
|
54
|
+
"precision": "maximize",
|
|
55
|
+
"recall": "maximize",
|
|
56
|
+
"f1": "maximize",
|
|
57
|
+
"roc_auc": "maximize",
|
|
58
|
+
"log_loss": "minimize",
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
def evaluate_regression(
|
|
62
|
+
self,
|
|
63
|
+
y_true,
|
|
64
|
+
predictions,
|
|
65
|
+
feature_count: int | None = None,
|
|
66
|
+
) -> dict[str, float]:
|
|
67
|
+
"""
|
|
68
|
+
Evaluate regression predictions.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
self._validate_regression_inputs(
|
|
72
|
+
y_true,
|
|
73
|
+
predictions,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
mse = mean_squared_error(
|
|
77
|
+
y_true,
|
|
78
|
+
predictions,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
rmse = float(
|
|
82
|
+
np.sqrt(mse)
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
r2 = float(
|
|
86
|
+
r2_score(
|
|
87
|
+
y_true,
|
|
88
|
+
predictions,
|
|
89
|
+
)
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
metrics = {
|
|
93
|
+
"r2": r2,
|
|
94
|
+
"mae": float(
|
|
95
|
+
mean_absolute_error(
|
|
96
|
+
y_true,
|
|
97
|
+
predictions,
|
|
98
|
+
)
|
|
99
|
+
),
|
|
100
|
+
"mse": float(mse),
|
|
101
|
+
"rmse": rmse,
|
|
102
|
+
"mape": float(
|
|
103
|
+
mean_absolute_percentage_error(
|
|
104
|
+
y_true,
|
|
105
|
+
predictions,
|
|
106
|
+
)
|
|
107
|
+
),
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
if feature_count is not None:
|
|
111
|
+
metrics["adjusted_r2"] = (
|
|
112
|
+
self.adjusted_r2(
|
|
113
|
+
r2=r2,
|
|
114
|
+
sample_count=len(y_true),
|
|
115
|
+
feature_count=feature_count,
|
|
116
|
+
)
|
|
117
|
+
)
|
|
118
|
+
else:
|
|
119
|
+
metrics["adjusted_r2"] = np.nan
|
|
120
|
+
|
|
121
|
+
return metrics
|
|
122
|
+
|
|
123
|
+
def evaluate_classification(
|
|
124
|
+
self,
|
|
125
|
+
y_true,
|
|
126
|
+
predictions,
|
|
127
|
+
probabilities=None,
|
|
128
|
+
decision_scores=None,
|
|
129
|
+
) -> dict[str, float | None]:
|
|
130
|
+
"""
|
|
131
|
+
Evaluate classification predictions.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
self._validate_classification_inputs(
|
|
135
|
+
y_true,
|
|
136
|
+
predictions,
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
metrics = {
|
|
140
|
+
"accuracy": float(
|
|
141
|
+
accuracy_score(
|
|
142
|
+
y_true,
|
|
143
|
+
predictions,
|
|
144
|
+
)
|
|
145
|
+
),
|
|
146
|
+
"precision": float(
|
|
147
|
+
precision_score(
|
|
148
|
+
y_true,
|
|
149
|
+
predictions,
|
|
150
|
+
average="weighted",
|
|
151
|
+
zero_division=0,
|
|
152
|
+
)
|
|
153
|
+
),
|
|
154
|
+
"recall": float(
|
|
155
|
+
recall_score(
|
|
156
|
+
y_true,
|
|
157
|
+
predictions,
|
|
158
|
+
average="weighted",
|
|
159
|
+
zero_division=0,
|
|
160
|
+
)
|
|
161
|
+
),
|
|
162
|
+
"f1": float(
|
|
163
|
+
f1_score(
|
|
164
|
+
y_true,
|
|
165
|
+
predictions,
|
|
166
|
+
average="weighted",
|
|
167
|
+
zero_division=0,
|
|
168
|
+
)
|
|
169
|
+
),
|
|
170
|
+
"roc_auc": self._calculate_roc_auc(
|
|
171
|
+
y_true=y_true,
|
|
172
|
+
probabilities=probabilities,
|
|
173
|
+
decision_scores=decision_scores,
|
|
174
|
+
),
|
|
175
|
+
"log_loss": self._calculate_log_loss(
|
|
176
|
+
y_true=y_true,
|
|
177
|
+
probabilities=probabilities,
|
|
178
|
+
),
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
return metrics
|
|
182
|
+
|
|
183
|
+
@staticmethod
|
|
184
|
+
def adjusted_r2(
|
|
185
|
+
r2: float,
|
|
186
|
+
sample_count: int,
|
|
187
|
+
feature_count: int,
|
|
188
|
+
) -> float:
|
|
189
|
+
"""
|
|
190
|
+
Calculate adjusted R².
|
|
191
|
+
"""
|
|
192
|
+
|
|
193
|
+
if sample_count <= feature_count + 1:
|
|
194
|
+
return float("nan")
|
|
195
|
+
|
|
196
|
+
return float(
|
|
197
|
+
1
|
|
198
|
+
- (
|
|
199
|
+
(1 - r2)
|
|
200
|
+
* (
|
|
201
|
+
(sample_count - 1)
|
|
202
|
+
/ (
|
|
203
|
+
sample_count
|
|
204
|
+
- feature_count
|
|
205
|
+
- 1
|
|
206
|
+
)
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
@classmethod
|
|
212
|
+
def metric_direction(
|
|
213
|
+
cls,
|
|
214
|
+
metric: str,
|
|
215
|
+
) -> str:
|
|
216
|
+
"""
|
|
217
|
+
Return whether a metric should be maximized
|
|
218
|
+
or minimized.
|
|
219
|
+
"""
|
|
220
|
+
|
|
221
|
+
if metric not in cls.METRIC_DIRECTIONS:
|
|
222
|
+
raise KeyError(
|
|
223
|
+
f"Unknown metric: {metric}"
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
return cls.METRIC_DIRECTIONS[
|
|
227
|
+
metric
|
|
228
|
+
]
|
|
229
|
+
|
|
230
|
+
@classmethod
|
|
231
|
+
def available_metrics(
|
|
232
|
+
cls,
|
|
233
|
+
task_type: str,
|
|
234
|
+
) -> list[str]:
|
|
235
|
+
"""
|
|
236
|
+
Return available metrics for a task.
|
|
237
|
+
"""
|
|
238
|
+
|
|
239
|
+
if task_type == "regression":
|
|
240
|
+
return list(
|
|
241
|
+
cls.REGRESSION_METRICS
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
if task_type == "classification":
|
|
245
|
+
return list(
|
|
246
|
+
cls.CLASSIFICATION_METRICS
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
raise ValueError(
|
|
250
|
+
"task_type must be 'regression' "
|
|
251
|
+
"or 'classification'."
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
@staticmethod
|
|
255
|
+
def _calculate_roc_auc(
|
|
256
|
+
y_true,
|
|
257
|
+
probabilities=None,
|
|
258
|
+
decision_scores=None,
|
|
259
|
+
) -> float | None:
|
|
260
|
+
"""Calculate ROC-AUC when possible."""
|
|
261
|
+
|
|
262
|
+
try:
|
|
263
|
+
unique_classes = np.unique(
|
|
264
|
+
y_true
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
if probabilities is not None:
|
|
268
|
+
probabilities = np.asarray(
|
|
269
|
+
probabilities
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
if len(unique_classes) == 2:
|
|
273
|
+
if (
|
|
274
|
+
probabilities.ndim == 2
|
|
275
|
+
and probabilities.shape[1] >= 2
|
|
276
|
+
):
|
|
277
|
+
return float(
|
|
278
|
+
roc_auc_score(
|
|
279
|
+
y_true,
|
|
280
|
+
probabilities[:, 1],
|
|
281
|
+
)
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
if probabilities.ndim == 1:
|
|
285
|
+
return float(
|
|
286
|
+
roc_auc_score(
|
|
287
|
+
y_true,
|
|
288
|
+
probabilities,
|
|
289
|
+
)
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
if (
|
|
293
|
+
len(unique_classes) > 2
|
|
294
|
+
and probabilities.ndim == 2
|
|
295
|
+
):
|
|
296
|
+
return float(
|
|
297
|
+
roc_auc_score(
|
|
298
|
+
y_true,
|
|
299
|
+
probabilities,
|
|
300
|
+
multi_class="ovr",
|
|
301
|
+
average="weighted",
|
|
302
|
+
)
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
if decision_scores is not None:
|
|
306
|
+
decision_scores = np.asarray(
|
|
307
|
+
decision_scores
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
if len(unique_classes) == 2:
|
|
311
|
+
return float(
|
|
312
|
+
roc_auc_score(
|
|
313
|
+
y_true,
|
|
314
|
+
decision_scores,
|
|
315
|
+
)
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
if (
|
|
319
|
+
len(unique_classes) > 2
|
|
320
|
+
and decision_scores.ndim == 2
|
|
321
|
+
):
|
|
322
|
+
return float(
|
|
323
|
+
roc_auc_score(
|
|
324
|
+
y_true,
|
|
325
|
+
decision_scores,
|
|
326
|
+
multi_class="ovr",
|
|
327
|
+
average="weighted",
|
|
328
|
+
)
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
except (
|
|
332
|
+
ValueError,
|
|
333
|
+
TypeError,
|
|
334
|
+
):
|
|
335
|
+
return None
|
|
336
|
+
|
|
337
|
+
return None
|
|
338
|
+
|
|
339
|
+
@staticmethod
|
|
340
|
+
def _calculate_log_loss(
|
|
341
|
+
y_true,
|
|
342
|
+
probabilities=None,
|
|
343
|
+
) -> float | None:
|
|
344
|
+
"""Calculate log loss when probabilities are available."""
|
|
345
|
+
|
|
346
|
+
if probabilities is None:
|
|
347
|
+
return None
|
|
348
|
+
|
|
349
|
+
try:
|
|
350
|
+
return float(
|
|
351
|
+
log_loss(
|
|
352
|
+
y_true,
|
|
353
|
+
probabilities,
|
|
354
|
+
)
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
except (
|
|
358
|
+
ValueError,
|
|
359
|
+
TypeError,
|
|
360
|
+
):
|
|
361
|
+
return None
|
|
362
|
+
|
|
363
|
+
@staticmethod
|
|
364
|
+
def _validate_regression_inputs(
|
|
365
|
+
y_true,
|
|
366
|
+
predictions,
|
|
367
|
+
) -> None:
|
|
368
|
+
"""Validate regression inputs."""
|
|
369
|
+
|
|
370
|
+
if len(y_true) != len(predictions):
|
|
371
|
+
raise ValueError(
|
|
372
|
+
"y_true and predictions must "
|
|
373
|
+
"have the same length."
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
if len(y_true) == 0:
|
|
377
|
+
raise ValueError(
|
|
378
|
+
"Cannot evaluate empty predictions."
|
|
379
|
+
)
|
|
380
|
+
|
|
381
|
+
@staticmethod
|
|
382
|
+
def _validate_classification_inputs(
|
|
383
|
+
y_true,
|
|
384
|
+
predictions,
|
|
385
|
+
) -> None:
|
|
386
|
+
"""Validate classification inputs."""
|
|
387
|
+
|
|
388
|
+
if len(y_true) != len(predictions):
|
|
389
|
+
raise ValueError(
|
|
390
|
+
"y_true and predictions must "
|
|
391
|
+
"have the same length."
|
|
392
|
+
)
|
|
393
|
+
|
|
394
|
+
if len(y_true) == 0:
|
|
395
|
+
raise ValueError(
|
|
396
|
+
"Cannot evaluate empty predictions."
|
|
397
|
+
)
|