autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,397 @@
1
+ from typing import Any
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ from sklearn.metrics import (
7
+ accuracy_score,
8
+ f1_score,
9
+ log_loss,
10
+ mean_absolute_error,
11
+ mean_absolute_percentage_error,
12
+ mean_squared_error,
13
+ precision_score,
14
+ r2_score,
15
+ recall_score,
16
+ roc_auc_score,
17
+ )
18
+
19
+
20
+ class EvaluationEngine:
21
+ """
22
+ Centralized evaluation engine for ModelForge.
23
+
24
+ Provides standardized metrics for regression
25
+ and classification tasks.
26
+ """
27
+
28
+ REGRESSION_METRICS = (
29
+ "r2",
30
+ "adjusted_r2",
31
+ "mae",
32
+ "mse",
33
+ "rmse",
34
+ "mape",
35
+ )
36
+
37
+ CLASSIFICATION_METRICS = (
38
+ "accuracy",
39
+ "precision",
40
+ "recall",
41
+ "f1",
42
+ "roc_auc",
43
+ "log_loss",
44
+ )
45
+
46
+ METRIC_DIRECTIONS = {
47
+ "r2": "maximize",
48
+ "adjusted_r2": "maximize",
49
+ "mae": "minimize",
50
+ "mse": "minimize",
51
+ "rmse": "minimize",
52
+ "mape": "minimize",
53
+ "accuracy": "maximize",
54
+ "precision": "maximize",
55
+ "recall": "maximize",
56
+ "f1": "maximize",
57
+ "roc_auc": "maximize",
58
+ "log_loss": "minimize",
59
+ }
60
+
61
+ def evaluate_regression(
62
+ self,
63
+ y_true,
64
+ predictions,
65
+ feature_count: int | None = None,
66
+ ) -> dict[str, float]:
67
+ """
68
+ Evaluate regression predictions.
69
+ """
70
+
71
+ self._validate_regression_inputs(
72
+ y_true,
73
+ predictions,
74
+ )
75
+
76
+ mse = mean_squared_error(
77
+ y_true,
78
+ predictions,
79
+ )
80
+
81
+ rmse = float(
82
+ np.sqrt(mse)
83
+ )
84
+
85
+ r2 = float(
86
+ r2_score(
87
+ y_true,
88
+ predictions,
89
+ )
90
+ )
91
+
92
+ metrics = {
93
+ "r2": r2,
94
+ "mae": float(
95
+ mean_absolute_error(
96
+ y_true,
97
+ predictions,
98
+ )
99
+ ),
100
+ "mse": float(mse),
101
+ "rmse": rmse,
102
+ "mape": float(
103
+ mean_absolute_percentage_error(
104
+ y_true,
105
+ predictions,
106
+ )
107
+ ),
108
+ }
109
+
110
+ if feature_count is not None:
111
+ metrics["adjusted_r2"] = (
112
+ self.adjusted_r2(
113
+ r2=r2,
114
+ sample_count=len(y_true),
115
+ feature_count=feature_count,
116
+ )
117
+ )
118
+ else:
119
+ metrics["adjusted_r2"] = np.nan
120
+
121
+ return metrics
122
+
123
+ def evaluate_classification(
124
+ self,
125
+ y_true,
126
+ predictions,
127
+ probabilities=None,
128
+ decision_scores=None,
129
+ ) -> dict[str, float | None]:
130
+ """
131
+ Evaluate classification predictions.
132
+ """
133
+
134
+ self._validate_classification_inputs(
135
+ y_true,
136
+ predictions,
137
+ )
138
+
139
+ metrics = {
140
+ "accuracy": float(
141
+ accuracy_score(
142
+ y_true,
143
+ predictions,
144
+ )
145
+ ),
146
+ "precision": float(
147
+ precision_score(
148
+ y_true,
149
+ predictions,
150
+ average="weighted",
151
+ zero_division=0,
152
+ )
153
+ ),
154
+ "recall": float(
155
+ recall_score(
156
+ y_true,
157
+ predictions,
158
+ average="weighted",
159
+ zero_division=0,
160
+ )
161
+ ),
162
+ "f1": float(
163
+ f1_score(
164
+ y_true,
165
+ predictions,
166
+ average="weighted",
167
+ zero_division=0,
168
+ )
169
+ ),
170
+ "roc_auc": self._calculate_roc_auc(
171
+ y_true=y_true,
172
+ probabilities=probabilities,
173
+ decision_scores=decision_scores,
174
+ ),
175
+ "log_loss": self._calculate_log_loss(
176
+ y_true=y_true,
177
+ probabilities=probabilities,
178
+ ),
179
+ }
180
+
181
+ return metrics
182
+
183
+ @staticmethod
184
+ def adjusted_r2(
185
+ r2: float,
186
+ sample_count: int,
187
+ feature_count: int,
188
+ ) -> float:
189
+ """
190
+ Calculate adjusted R².
191
+ """
192
+
193
+ if sample_count <= feature_count + 1:
194
+ return float("nan")
195
+
196
+ return float(
197
+ 1
198
+ - (
199
+ (1 - r2)
200
+ * (
201
+ (sample_count - 1)
202
+ / (
203
+ sample_count
204
+ - feature_count
205
+ - 1
206
+ )
207
+ )
208
+ )
209
+ )
210
+
211
+ @classmethod
212
+ def metric_direction(
213
+ cls,
214
+ metric: str,
215
+ ) -> str:
216
+ """
217
+ Return whether a metric should be maximized
218
+ or minimized.
219
+ """
220
+
221
+ if metric not in cls.METRIC_DIRECTIONS:
222
+ raise KeyError(
223
+ f"Unknown metric: {metric}"
224
+ )
225
+
226
+ return cls.METRIC_DIRECTIONS[
227
+ metric
228
+ ]
229
+
230
+ @classmethod
231
+ def available_metrics(
232
+ cls,
233
+ task_type: str,
234
+ ) -> list[str]:
235
+ """
236
+ Return available metrics for a task.
237
+ """
238
+
239
+ if task_type == "regression":
240
+ return list(
241
+ cls.REGRESSION_METRICS
242
+ )
243
+
244
+ if task_type == "classification":
245
+ return list(
246
+ cls.CLASSIFICATION_METRICS
247
+ )
248
+
249
+ raise ValueError(
250
+ "task_type must be 'regression' "
251
+ "or 'classification'."
252
+ )
253
+
254
+ @staticmethod
255
+ def _calculate_roc_auc(
256
+ y_true,
257
+ probabilities=None,
258
+ decision_scores=None,
259
+ ) -> float | None:
260
+ """Calculate ROC-AUC when possible."""
261
+
262
+ try:
263
+ unique_classes = np.unique(
264
+ y_true
265
+ )
266
+
267
+ if probabilities is not None:
268
+ probabilities = np.asarray(
269
+ probabilities
270
+ )
271
+
272
+ if len(unique_classes) == 2:
273
+ if (
274
+ probabilities.ndim == 2
275
+ and probabilities.shape[1] >= 2
276
+ ):
277
+ return float(
278
+ roc_auc_score(
279
+ y_true,
280
+ probabilities[:, 1],
281
+ )
282
+ )
283
+
284
+ if probabilities.ndim == 1:
285
+ return float(
286
+ roc_auc_score(
287
+ y_true,
288
+ probabilities,
289
+ )
290
+ )
291
+
292
+ if (
293
+ len(unique_classes) > 2
294
+ and probabilities.ndim == 2
295
+ ):
296
+ return float(
297
+ roc_auc_score(
298
+ y_true,
299
+ probabilities,
300
+ multi_class="ovr",
301
+ average="weighted",
302
+ )
303
+ )
304
+
305
+ if decision_scores is not None:
306
+ decision_scores = np.asarray(
307
+ decision_scores
308
+ )
309
+
310
+ if len(unique_classes) == 2:
311
+ return float(
312
+ roc_auc_score(
313
+ y_true,
314
+ decision_scores,
315
+ )
316
+ )
317
+
318
+ if (
319
+ len(unique_classes) > 2
320
+ and decision_scores.ndim == 2
321
+ ):
322
+ return float(
323
+ roc_auc_score(
324
+ y_true,
325
+ decision_scores,
326
+ multi_class="ovr",
327
+ average="weighted",
328
+ )
329
+ )
330
+
331
+ except (
332
+ ValueError,
333
+ TypeError,
334
+ ):
335
+ return None
336
+
337
+ return None
338
+
339
+ @staticmethod
340
+ def _calculate_log_loss(
341
+ y_true,
342
+ probabilities=None,
343
+ ) -> float | None:
344
+ """Calculate log loss when probabilities are available."""
345
+
346
+ if probabilities is None:
347
+ return None
348
+
349
+ try:
350
+ return float(
351
+ log_loss(
352
+ y_true,
353
+ probabilities,
354
+ )
355
+ )
356
+
357
+ except (
358
+ ValueError,
359
+ TypeError,
360
+ ):
361
+ return None
362
+
363
+ @staticmethod
364
+ def _validate_regression_inputs(
365
+ y_true,
366
+ predictions,
367
+ ) -> None:
368
+ """Validate regression inputs."""
369
+
370
+ if len(y_true) != len(predictions):
371
+ raise ValueError(
372
+ "y_true and predictions must "
373
+ "have the same length."
374
+ )
375
+
376
+ if len(y_true) == 0:
377
+ raise ValueError(
378
+ "Cannot evaluate empty predictions."
379
+ )
380
+
381
+ @staticmethod
382
+ def _validate_classification_inputs(
383
+ y_true,
384
+ predictions,
385
+ ) -> None:
386
+ """Validate classification inputs."""
387
+
388
+ if len(y_true) != len(predictions):
389
+ raise ValueError(
390
+ "y_true and predictions must "
391
+ "have the same length."
392
+ )
393
+
394
+ if len(y_true) == 0:
395
+ raise ValueError(
396
+ "Cannot evaluate empty predictions."
397
+ )