autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,749 @@
1
+ from __future__ import annotations
2
+
3
+ import time
4
+ from typing import Any
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+
9
+ from sklearn.base import clone
10
+ from sklearn.metrics import (
11
+ accuracy_score,
12
+ f1_score,
13
+ mean_absolute_error,
14
+ mean_squared_error,
15
+ precision_score,
16
+ r2_score,
17
+ recall_score,
18
+ roc_auc_score,
19
+ )
20
+ from sklearn.model_selection import (
21
+ KFold,
22
+ StratifiedKFold,
23
+ )
24
+ from sklearn.pipeline import Pipeline
25
+
26
+
27
+ class CrossValidationEngine:
28
+ """
29
+ Evaluate ModelForge pipelines using cross-validation.
30
+
31
+ Every fold receives a fresh cloned pipeline so that preprocessing
32
+ and model fitting happen independently inside each fold.
33
+ """
34
+
35
+ def __init__(
36
+ self,
37
+ cv: int = 5,
38
+ random_state: int = 42,
39
+ shuffle: bool = True,
40
+ ):
41
+ self.cv = cv
42
+ self.random_state = random_state
43
+ self.shuffle = shuffle
44
+
45
+ self._validate_configuration()
46
+
47
+ def evaluate(
48
+ self,
49
+ data: pd.DataFrame,
50
+ target: str,
51
+ pipelines: dict[str, Pipeline],
52
+ task_type: str,
53
+ ) -> pd.DataFrame:
54
+ """Evaluate pipelines using K-fold cross-validation."""
55
+
56
+ self._validate_inputs(
57
+ data=data,
58
+ target=target,
59
+ pipelines=pipelines,
60
+ task_type=task_type,
61
+ )
62
+
63
+ X = data.drop(
64
+ columns=[target]
65
+ )
66
+
67
+ y = data[target]
68
+
69
+ splitter = self._create_splitter(
70
+ y=y,
71
+ task_type=task_type,
72
+ )
73
+
74
+ results: list[dict[str, Any]] = []
75
+
76
+ for model_name, pipeline in pipelines.items():
77
+ result = self._evaluate_pipeline(
78
+ model_name=model_name,
79
+ pipeline=pipeline,
80
+ X=X,
81
+ y=y,
82
+ splitter=splitter,
83
+ task_type=task_type,
84
+ )
85
+
86
+ results.append(result)
87
+
88
+ return pd.DataFrame(results)
89
+
90
+ def _evaluate_pipeline(
91
+ self,
92
+ model_name: str,
93
+ pipeline: Pipeline,
94
+ X: pd.DataFrame,
95
+ y: pd.Series,
96
+ splitter,
97
+ task_type: str,
98
+ ) -> dict[str, Any]:
99
+ """Evaluate one pipeline across all folds."""
100
+
101
+ fold_results: list[dict[str, Any]] = []
102
+
103
+ total_start = time.perf_counter()
104
+
105
+ for fold_number, (
106
+ train_indices,
107
+ validation_indices,
108
+ ) in enumerate(
109
+ splitter.split(
110
+ X,
111
+ y if task_type == "classification" else None,
112
+ ),
113
+ start=1,
114
+ ):
115
+ fold_start = time.perf_counter()
116
+
117
+ try:
118
+ X_train = X.iloc[
119
+ train_indices
120
+ ]
121
+
122
+ X_validation = X.iloc[
123
+ validation_indices
124
+ ]
125
+
126
+ y_train = y.iloc[
127
+ train_indices
128
+ ]
129
+
130
+ y_validation = y.iloc[
131
+ validation_indices
132
+ ]
133
+
134
+ fold_pipeline = clone(
135
+ pipeline
136
+ )
137
+
138
+ training_start = time.perf_counter()
139
+
140
+ fold_pipeline.fit(
141
+ X_train,
142
+ y_train,
143
+ )
144
+
145
+ training_time = (
146
+ time.perf_counter()
147
+ - training_start
148
+ )
149
+
150
+ prediction_start = time.perf_counter()
151
+
152
+ predictions = (
153
+ fold_pipeline.predict(
154
+ X_validation
155
+ )
156
+ )
157
+
158
+ prediction_time = (
159
+ time.perf_counter()
160
+ - prediction_start
161
+ )
162
+
163
+ if task_type == "regression":
164
+ metrics = self._regression_metrics(
165
+ y_validation,
166
+ predictions,
167
+ )
168
+ else:
169
+ metrics = self._classification_metrics(
170
+ fold_pipeline,
171
+ X_validation,
172
+ y_validation,
173
+ predictions,
174
+ )
175
+
176
+ fold_time = (
177
+ time.perf_counter()
178
+ - fold_start
179
+ )
180
+
181
+ fold_results.append(
182
+ {
183
+ "fold": fold_number,
184
+ "status": "success",
185
+ "training_time_seconds": float(
186
+ training_time
187
+ ),
188
+ "prediction_time_seconds": float(
189
+ prediction_time
190
+ ),
191
+ "total_time_seconds": float(
192
+ fold_time
193
+ ),
194
+ **metrics,
195
+ "error": None,
196
+ }
197
+ )
198
+
199
+ except Exception as exc:
200
+ fold_time = (
201
+ time.perf_counter()
202
+ - fold_start
203
+ )
204
+
205
+ fold_results.append(
206
+ {
207
+ "fold": fold_number,
208
+ "status": "failed",
209
+ "training_time_seconds": float(
210
+ fold_time
211
+ ),
212
+ "prediction_time_seconds": None,
213
+ "total_time_seconds": float(
214
+ fold_time
215
+ ),
216
+ **self._empty_metrics(
217
+ task_type
218
+ ),
219
+ "error": str(exc),
220
+ }
221
+ )
222
+
223
+ total_time = (
224
+ time.perf_counter()
225
+ - total_start
226
+ )
227
+
228
+ return self._aggregate_results(
229
+ model_name=model_name,
230
+ fold_results=fold_results,
231
+ task_type=task_type,
232
+ total_time=total_time,
233
+ )
234
+
235
+ @staticmethod
236
+ def _regression_metrics(
237
+ y_true,
238
+ predictions,
239
+ ) -> dict[str, float]:
240
+ """Calculate regression metrics."""
241
+
242
+ mse = mean_squared_error(
243
+ y_true,
244
+ predictions,
245
+ )
246
+
247
+ return {
248
+ "r2": float(
249
+ r2_score(
250
+ y_true,
251
+ predictions,
252
+ )
253
+ ),
254
+ "mae": float(
255
+ mean_absolute_error(
256
+ y_true,
257
+ predictions,
258
+ )
259
+ ),
260
+ "mse": float(mse),
261
+ "rmse": float(
262
+ np.sqrt(mse)
263
+ ),
264
+ }
265
+
266
+ @staticmethod
267
+ def _classification_metrics(
268
+ pipeline: Pipeline,
269
+ X_validation: pd.DataFrame,
270
+ y_validation: pd.Series,
271
+ predictions,
272
+ ) -> dict[str, float | None]:
273
+ """Calculate classification metrics."""
274
+
275
+ metrics: dict[str, float | None] = {
276
+ "accuracy": float(
277
+ accuracy_score(
278
+ y_validation,
279
+ predictions,
280
+ )
281
+ ),
282
+ "precision": float(
283
+ precision_score(
284
+ y_validation,
285
+ predictions,
286
+ average="weighted",
287
+ zero_division=0,
288
+ )
289
+ ),
290
+ "recall": float(
291
+ recall_score(
292
+ y_validation,
293
+ predictions,
294
+ average="weighted",
295
+ zero_division=0,
296
+ )
297
+ ),
298
+ "f1": float(
299
+ f1_score(
300
+ y_validation,
301
+ predictions,
302
+ average="weighted",
303
+ zero_division=0,
304
+ )
305
+ ),
306
+ "roc_auc": None,
307
+ }
308
+
309
+ metrics["roc_auc"] = (
310
+ CrossValidationEngine._calculate_roc_auc(
311
+ pipeline,
312
+ X_validation,
313
+ y_validation,
314
+ )
315
+ )
316
+
317
+ return metrics
318
+
319
+ @staticmethod
320
+ def _calculate_roc_auc(
321
+ pipeline: Pipeline,
322
+ X_validation: pd.DataFrame,
323
+ y_validation: pd.Series,
324
+ ) -> float | None:
325
+ """Calculate ROC-AUC using probabilities or decision scores."""
326
+
327
+ if hasattr(
328
+ pipeline,
329
+ "predict_proba",
330
+ ):
331
+ try:
332
+ probabilities = pipeline.predict_proba(
333
+ X_validation
334
+ )
335
+
336
+ probabilities = np.asarray(
337
+ probabilities
338
+ )
339
+
340
+ if probabilities.ndim != 2:
341
+ return None
342
+
343
+ if probabilities.shape[1] == 2:
344
+ return float(
345
+ roc_auc_score(
346
+ y_validation,
347
+ probabilities[:, 1],
348
+ )
349
+ )
350
+
351
+ if probabilities.shape[1] > 2:
352
+ return float(
353
+ roc_auc_score(
354
+ y_validation,
355
+ probabilities,
356
+ multi_class="ovr",
357
+ average="weighted",
358
+ )
359
+ )
360
+
361
+ except (
362
+ ValueError,
363
+ TypeError,
364
+ AttributeError,
365
+ ):
366
+ return None
367
+
368
+ if hasattr(
369
+ pipeline,
370
+ "decision_function",
371
+ ):
372
+ try:
373
+ decision_scores = (
374
+ pipeline.decision_function(
375
+ X_validation
376
+ )
377
+ )
378
+
379
+ unique_classes = np.unique(
380
+ y_validation
381
+ )
382
+
383
+ if len(unique_classes) == 2:
384
+ return float(
385
+ roc_auc_score(
386
+ y_validation,
387
+ decision_scores,
388
+ )
389
+ )
390
+
391
+ decision_scores = np.asarray(
392
+ decision_scores
393
+ )
394
+
395
+ if decision_scores.ndim == 2:
396
+ return float(
397
+ roc_auc_score(
398
+ y_validation,
399
+ decision_scores,
400
+ multi_class="ovr",
401
+ average="weighted",
402
+ )
403
+ )
404
+
405
+ except (
406
+ ValueError,
407
+ TypeError,
408
+ AttributeError,
409
+ ):
410
+ return None
411
+
412
+ return None
413
+
414
+ @staticmethod
415
+ def _empty_metrics(
416
+ task_type: str,
417
+ ) -> dict[str, None]:
418
+ """Return empty metrics for failed folds."""
419
+
420
+ if task_type == "regression":
421
+ return {
422
+ "r2": None,
423
+ "mae": None,
424
+ "mse": None,
425
+ "rmse": None,
426
+ }
427
+
428
+ return {
429
+ "accuracy": None,
430
+ "precision": None,
431
+ "recall": None,
432
+ "f1": None,
433
+ "roc_auc": None,
434
+ }
435
+
436
+ def _aggregate_results(
437
+ self,
438
+ model_name: str,
439
+ fold_results: list[dict],
440
+ task_type: str,
441
+ total_time: float,
442
+ ) -> dict[str, Any]:
443
+ """Aggregate metrics across successful folds."""
444
+
445
+ successful_folds = [
446
+ result
447
+ for result in fold_results
448
+ if result["status"] == "success"
449
+ ]
450
+
451
+ failed_folds = [
452
+ result
453
+ for result in fold_results
454
+ if result["status"] == "failed"
455
+ ]
456
+
457
+ result: dict[str, Any] = {
458
+ "model": model_name,
459
+ "status": (
460
+ "success"
461
+ if successful_folds
462
+ else "failed"
463
+ ),
464
+ "cv_folds": len(
465
+ fold_results
466
+ ),
467
+ "successful_folds": len(
468
+ successful_folds
469
+ ),
470
+ "failed_folds": len(
471
+ failed_folds
472
+ ),
473
+ "total_time_seconds": float(
474
+ total_time
475
+ ),
476
+ }
477
+
478
+ metric_names = self._metric_names(
479
+ task_type
480
+ )
481
+
482
+ for metric_name in metric_names:
483
+ values = [
484
+ fold[metric_name]
485
+ for fold in successful_folds
486
+ if fold[metric_name] is not None
487
+ ]
488
+
489
+ if values:
490
+ result[
491
+ f"cv_mean_{metric_name}"
492
+ ] = float(
493
+ np.mean(values)
494
+ )
495
+
496
+ result[
497
+ f"cv_std_{metric_name}"
498
+ ] = float(
499
+ np.std(
500
+ values,
501
+ ddof=1,
502
+ )
503
+ if len(values) > 1
504
+ else 0.0
505
+ )
506
+ else:
507
+ result[
508
+ f"cv_mean_{metric_name}"
509
+ ] = None
510
+
511
+ result[
512
+ f"cv_std_{metric_name}"
513
+ ] = None
514
+
515
+ successful_training_times = [
516
+ fold["training_time_seconds"]
517
+ for fold in successful_folds
518
+ ]
519
+
520
+ successful_prediction_times = [
521
+ fold["prediction_time_seconds"]
522
+ for fold in successful_folds
523
+ if fold["prediction_time_seconds"] is not None
524
+ ]
525
+
526
+ result["cv_mean_training_time_seconds"] = (
527
+ float(
528
+ np.mean(
529
+ successful_training_times
530
+ )
531
+ )
532
+ if successful_training_times
533
+ else None
534
+ )
535
+
536
+ result["cv_mean_prediction_time_seconds"] = (
537
+ float(
538
+ np.mean(
539
+ successful_prediction_times
540
+ )
541
+ )
542
+ if successful_prediction_times
543
+ else None
544
+ )
545
+
546
+ result["errors"] = [
547
+ fold["error"]
548
+ for fold in failed_folds
549
+ if fold["error"] is not None
550
+ ]
551
+
552
+ return result
553
+
554
+ @staticmethod
555
+ def _metric_names(
556
+ task_type: str,
557
+ ) -> list[str]:
558
+ """Return metrics for a task."""
559
+
560
+ if task_type == "regression":
561
+ return [
562
+ "r2",
563
+ "mae",
564
+ "mse",
565
+ "rmse",
566
+ ]
567
+
568
+ return [
569
+ "accuracy",
570
+ "precision",
571
+ "recall",
572
+ "f1",
573
+ "roc_auc",
574
+ ]
575
+
576
+ def _create_splitter(
577
+ self,
578
+ y: pd.Series,
579
+ task_type: str,
580
+ ):
581
+ """Create the appropriate CV splitter."""
582
+
583
+ if len(y) < self.cv:
584
+ raise ValueError(
585
+ f"Dataset must contain at least {self.cv} "
586
+ "samples for cross-validation."
587
+ )
588
+
589
+ if task_type == "classification":
590
+ class_counts = y.value_counts()
591
+
592
+ if (
593
+ len(class_counts) >= 2
594
+ and class_counts.min() >= self.cv
595
+ ):
596
+ return StratifiedKFold(
597
+ n_splits=self.cv,
598
+ shuffle=self.shuffle,
599
+ random_state=(
600
+ self.random_state
601
+ if self.shuffle
602
+ else None
603
+ ),
604
+ )
605
+
606
+ raise ValueError(
607
+ "Each classification class must "
608
+ f"contain at least {self.cv} samples "
609
+ "for stratified cross-validation."
610
+ )
611
+
612
+ return KFold(
613
+ n_splits=self.cv,
614
+ shuffle=self.shuffle,
615
+ random_state=(
616
+ self.random_state
617
+ if self.shuffle
618
+ else None
619
+ ),
620
+ )
621
+
622
+ def _validate_configuration(self):
623
+ """Validate engine configuration."""
624
+
625
+ if not isinstance(
626
+ self.cv,
627
+ int,
628
+ ):
629
+ raise TypeError(
630
+ "cv must be an integer."
631
+ )
632
+
633
+ if isinstance(
634
+ self.cv,
635
+ bool,
636
+ ):
637
+ raise TypeError(
638
+ "cv must be an integer."
639
+ )
640
+
641
+ if self.cv < 2:
642
+ raise ValueError(
643
+ "cv must be at least 2."
644
+ )
645
+
646
+ if not isinstance(
647
+ self.random_state,
648
+ int,
649
+ ):
650
+ raise TypeError(
651
+ "random_state must be an integer."
652
+ )
653
+
654
+ if isinstance(
655
+ self.random_state,
656
+ bool,
657
+ ):
658
+ raise TypeError(
659
+ "random_state must be an integer."
660
+ )
661
+
662
+ if not isinstance(
663
+ self.shuffle,
664
+ bool,
665
+ ):
666
+ raise TypeError(
667
+ "shuffle must be a boolean."
668
+ )
669
+
670
+ @staticmethod
671
+ def _validate_inputs(
672
+ data: pd.DataFrame,
673
+ target: str,
674
+ pipelines: dict[str, Pipeline],
675
+ task_type: str,
676
+ ):
677
+ """Validate evaluation inputs."""
678
+
679
+ if not isinstance(
680
+ data,
681
+ pd.DataFrame,
682
+ ):
683
+ raise TypeError(
684
+ "data must be a pandas DataFrame."
685
+ )
686
+
687
+ if data.empty:
688
+ raise ValueError(
689
+ "Cannot evaluate an empty dataset."
690
+ )
691
+
692
+ if not isinstance(
693
+ target,
694
+ str,
695
+ ):
696
+ raise TypeError(
697
+ "target must be a string."
698
+ )
699
+
700
+ if target not in data.columns:
701
+ raise ValueError(
702
+ f"Target column '{target}' "
703
+ "does not exist."
704
+ )
705
+
706
+ if not isinstance(
707
+ pipelines,
708
+ dict,
709
+ ):
710
+ raise TypeError(
711
+ "pipelines must be a dictionary."
712
+ )
713
+
714
+ if not pipelines:
715
+ raise ValueError(
716
+ "At least one pipeline is required."
717
+ )
718
+
719
+ if task_type not in {
720
+ "regression",
721
+ "classification",
722
+ }:
723
+ raise ValueError(
724
+ "task_type must be 'regression' "
725
+ "or 'classification'."
726
+ )
727
+
728
+ for name, pipeline in pipelines.items():
729
+ if not isinstance(
730
+ name,
731
+ str,
732
+ ):
733
+ raise TypeError(
734
+ "Pipeline names must be strings."
735
+ )
736
+
737
+ if not name.strip():
738
+ raise ValueError(
739
+ "Pipeline names cannot be empty."
740
+ )
741
+
742
+ if not isinstance(
743
+ pipeline,
744
+ Pipeline,
745
+ ):
746
+ raise TypeError(
747
+ f"Pipeline '{name}' must be "
748
+ "a sklearn Pipeline."
749
+ )