autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
modelforge/automl.py ADDED
@@ -0,0 +1,1472 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Any
5
+
6
+ import pandas as pd
7
+
8
+ from modelforge.column_intelligence import ColumnIntelligence
9
+ from modelforge.config import ModelForgeConfig
10
+ from modelforge.cross_validation import CrossValidationEngine
11
+ from modelforge.data_audit import DataQualityAuditor
12
+ from modelforge.data_loader import DatasetLoader
13
+ from modelforge.explainability import ExplainabilityEngine
14
+ from modelforge.experiment_tracker import ExperimentTracker
15
+ from modelforge.hyperparameter_optimization import (
16
+ HyperparameterOptimizationEngine,
17
+ )
18
+ from modelforge.model_registry import ModelRegistry
19
+ from modelforge.model_screening import ModelScreeningEngine
20
+ from modelforge.persistence import ModelPersistence
21
+ from modelforge.pipeline_generator import PipelineGenerator
22
+ from modelforge.profiler import DatasetProfiler
23
+ from modelforge.ranking import RankingEngine
24
+ from modelforge.reproducibility_integration import (
25
+ ReproducibilityIntegration,
26
+ )
27
+ from modelforge.run_manager import RunManager
28
+ from modelforge.target_selector import TargetSelector
29
+
30
+
31
+ class AutoML:
32
+ """
33
+ Main ModelForge AutoML orchestration engine.
34
+ """
35
+
36
+ def __init__(
37
+ self,
38
+ test_size: float = 0.2,
39
+ cv: int = 5,
40
+ random_state: int = 42,
41
+ objective: str = "balanced",
42
+ variance_threshold: float | None = None,
43
+ correlation_threshold: float | None = None,
44
+ enable_optimization: bool = False,
45
+ optimization_models: int = 3,
46
+ optimization_max_trials: int = 10,
47
+ experiment_directory: str | Path = (
48
+ ".modelforge/experiments"
49
+ ),
50
+ config: ModelForgeConfig | None = None,
51
+ ):
52
+ if config is not None and not isinstance(
53
+ config,
54
+ ModelForgeConfig,
55
+ ):
56
+ raise TypeError(
57
+ "config must be a ModelForgeConfig"
58
+ )
59
+
60
+ self._config_provided = config is not None
61
+
62
+ self.config = (
63
+ config
64
+ if config is not None
65
+ else ModelForgeConfig()
66
+ )
67
+
68
+ self._configured_target = self._config_value(
69
+ "target",
70
+ None,
71
+ )
72
+
73
+ self._configured_task_type = self._config_value(
74
+ "task_type",
75
+ None,
76
+ )
77
+
78
+ self._configured_models = self._config_value(
79
+ "models",
80
+ None,
81
+ )
82
+
83
+ self._configured_excluded_columns = (
84
+ self._config_value(
85
+ "excluded_columns",
86
+ [],
87
+ )
88
+ )
89
+
90
+ self.test_size = self._config_value(
91
+ "test_size",
92
+ test_size,
93
+ )
94
+
95
+ self.cv = self._config_value(
96
+ "cv",
97
+ cv,
98
+ )
99
+
100
+ self.random_state = self._config_value(
101
+ "random_state",
102
+ random_state,
103
+ )
104
+
105
+ self.objective = self._config_value(
106
+ "objective",
107
+ objective,
108
+ )
109
+
110
+ self.variance_threshold = self._config_value(
111
+ "feature_selection.variance_threshold",
112
+ variance_threshold,
113
+ )
114
+
115
+ self.correlation_threshold = self._config_value(
116
+ "feature_selection.correlation_threshold",
117
+ correlation_threshold,
118
+ )
119
+
120
+ configured_optimization_enabled = (
121
+ self._config_value(
122
+ "optimization.enabled",
123
+ None,
124
+ )
125
+ )
126
+
127
+ self.enable_optimization = (
128
+ enable_optimization
129
+ if configured_optimization_enabled is None
130
+ else configured_optimization_enabled
131
+ )
132
+
133
+ configured_optimization_models = (
134
+ self._config_value(
135
+ "optimization.models",
136
+ None,
137
+ )
138
+ )
139
+
140
+ self.optimization_models = (
141
+ optimization_models
142
+ if configured_optimization_models is None
143
+ else configured_optimization_models
144
+ )
145
+
146
+ configured_optimization_max_trials = (
147
+ self._config_value(
148
+ "optimization.max_trials",
149
+ None,
150
+ )
151
+ )
152
+
153
+ self.optimization_max_trials = (
154
+ optimization_max_trials
155
+ if configured_optimization_max_trials is None
156
+ else configured_optimization_max_trials
157
+ )
158
+
159
+ self._configured_experiment_directory = (
160
+ self._config_value(
161
+ "experiment_directory",
162
+ experiment_directory,
163
+ )
164
+ )
165
+
166
+ self.experiment_directory = Path(
167
+ self._configured_experiment_directory
168
+ )
169
+
170
+ self._validate_configuration()
171
+
172
+ self.loader = DatasetLoader()
173
+ self.target_selector = TargetSelector()
174
+ self.profiler = DatasetProfiler()
175
+ self.column_intelligence = ColumnIntelligence()
176
+ self.data_audit = DataQualityAuditor()
177
+
178
+ self.registry = ModelRegistry()
179
+ self.pipeline_generator = PipelineGenerator()
180
+
181
+ self.screening = ModelScreeningEngine(
182
+ test_size=self.test_size,
183
+ random_state=self.random_state,
184
+ )
185
+
186
+ self.cross_validator = CrossValidationEngine(
187
+ cv=self.cv,
188
+ random_state=self.random_state,
189
+ )
190
+
191
+ self.optimizer = (
192
+ HyperparameterOptimizationEngine(
193
+ cv=self.cv,
194
+ random_state=self.random_state,
195
+ max_trials=self.optimization_max_trials,
196
+ )
197
+ )
198
+
199
+ self.ranking = RankingEngine()
200
+ self.explainability = ExplainabilityEngine()
201
+ self.persistence = ModelPersistence()
202
+
203
+ self.run_manager = RunManager()
204
+
205
+ self.experiment_tracker = ExperimentTracker(
206
+ self.experiment_directory
207
+ )
208
+
209
+ self.is_fitted = False
210
+ self.best_pipeline = None
211
+ self.best_model = None
212
+ self.result = None
213
+ self.target = None
214
+ self.task_type = None
215
+ self.run_id = None
216
+ self.experiment_id = None
217
+
218
+ self.reproducibility = None
219
+
220
+ self.reproducibility_integration = (
221
+ ReproducibilityIntegration(
222
+ random_state=self.random_state
223
+ )
224
+ )
225
+
226
+ def fit(
227
+ self,
228
+ data: Any,
229
+ target: str | None = None,
230
+ task_type: str | None = None,
231
+ model_names: list[str] | None = None,
232
+ excluded_columns: list[str] | None = None,
233
+ ) -> dict[str, Any]:
234
+ """
235
+ Run the complete ModelForge AutoML workflow.
236
+
237
+ A run is automatically started before training and
238
+ recorded after successful or failed execution.
239
+
240
+ Reproducibility metadata is captured for every run.
241
+ """
242
+
243
+ target = (
244
+ target
245
+ if target is not None
246
+ else self._configured_target
247
+ )
248
+
249
+ task_type = (
250
+ task_type
251
+ if task_type is not None
252
+ else self._configured_task_type
253
+ )
254
+
255
+ model_names = (
256
+ model_names
257
+ if model_names is not None
258
+ else self._configured_models
259
+ )
260
+
261
+ excluded_columns = (
262
+ excluded_columns
263
+ if excluded_columns is not None
264
+ else self._configured_excluded_columns
265
+ )
266
+
267
+ if target is None:
268
+ raise ValueError(
269
+ "target must be provided to fit() "
270
+ "or defined in ModelForgeConfig."
271
+ )
272
+
273
+ self.run_id = self.run_manager.start(
274
+ {
275
+ "target": target,
276
+ "requested_task_type": task_type,
277
+ "objective": self.objective,
278
+ "cv": self.cv,
279
+ "test_size": self.test_size,
280
+ "random_state": self.random_state,
281
+ "enable_optimization": (
282
+ self.enable_optimization
283
+ ),
284
+ }
285
+ )
286
+
287
+ self.reproducibility = None
288
+
289
+ try:
290
+ reproducibility_data = self._load_data(data)
291
+
292
+ result = self._fit_workflow(
293
+ data=reproducibility_data,
294
+ target=target,
295
+ task_type=task_type,
296
+ model_names=model_names,
297
+ excluded_columns=excluded_columns,
298
+ )
299
+
300
+ effective_configuration = (
301
+ self._configuration(
302
+ target=target,
303
+ task_type=self.task_type,
304
+ models=model_names,
305
+ excluded_columns=excluded_columns,
306
+ )
307
+ )
308
+
309
+ self.reproducibility = (
310
+ self.reproducibility_integration.create_run_snapshot(
311
+ data=reproducibility_data,
312
+ configuration=effective_configuration,
313
+ target=self.target,
314
+ task_type=self.task_type,
315
+ run_id=self.run_id,
316
+ extra_metadata={
317
+ "model_names": model_names,
318
+ "excluded_columns": excluded_columns,
319
+ },
320
+ )
321
+ )
322
+
323
+ result["reproducibility"] = (
324
+ self.reproducibility
325
+ )
326
+
327
+ self.run_manager.update(
328
+ {
329
+ "best_model": self.best_model,
330
+ "task_type": self.task_type,
331
+ "models_evaluated": result.get(
332
+ "models_evaluated"
333
+ ),
334
+ }
335
+ )
336
+
337
+ run_summary = self.run_manager.complete()
338
+
339
+ self.run_id = run_summary["run_id"]
340
+
341
+ result["run_id"] = self.run_id
342
+ result["run_summary"] = run_summary
343
+
344
+ self.experiment_id = (
345
+ self.experiment_tracker.record(
346
+ result=result,
347
+ configuration=effective_configuration,
348
+ )
349
+ )
350
+
351
+ result["experiment_id"] = (
352
+ self.experiment_id
353
+ )
354
+
355
+ self.result = result
356
+
357
+ return result
358
+
359
+ except Exception as exc:
360
+ failure_summary = self.run_manager.fail(exc)
361
+
362
+ self.run_id = failure_summary["run_id"]
363
+
364
+ if self.reproducibility is None:
365
+ try:
366
+ reproducibility_data = (
367
+ self._load_data(data)
368
+ )
369
+
370
+ effective_configuration = (
371
+ self._configuration(
372
+ target=target,
373
+ task_type=self.task_type,
374
+ models=model_names,
375
+ excluded_columns=excluded_columns,
376
+ )
377
+ )
378
+
379
+ self.reproducibility = (
380
+ self.reproducibility_integration.create_run_snapshot(
381
+ data=reproducibility_data,
382
+ configuration=effective_configuration,
383
+ target=target,
384
+ task_type=self.task_type,
385
+ run_id=self.run_id,
386
+ extra_metadata={
387
+ "model_names": model_names,
388
+ "excluded_columns": (
389
+ excluded_columns
390
+ ),
391
+ },
392
+ )
393
+ )
394
+ except Exception:
395
+ self.reproducibility = None
396
+
397
+ failure_configuration = (
398
+ self._configuration(
399
+ target=target,
400
+ task_type=self.task_type,
401
+ models=model_names,
402
+ excluded_columns=excluded_columns,
403
+ )
404
+ )
405
+
406
+ failure_result = {
407
+ "run_id": self.run_id,
408
+ "status": "failed",
409
+ "error": str(exc),
410
+ "run_summary": failure_summary,
411
+ "target": self.target,
412
+ "task_type": self.task_type,
413
+ "configuration": failure_configuration,
414
+ "reproducibility": (
415
+ self.reproducibility
416
+ ),
417
+ }
418
+
419
+ self.experiment_id = (
420
+ self.experiment_tracker.record(
421
+ result=failure_result,
422
+ configuration=failure_configuration,
423
+ )
424
+ )
425
+
426
+ self.result = failure_result
427
+
428
+ raise
429
+
430
+ def _fit_workflow(
431
+ self,
432
+ data: Any,
433
+ target: str,
434
+ task_type: str | None,
435
+ model_names: list[str] | None,
436
+ excluded_columns: list[str] | None,
437
+ ) -> dict[str, Any]:
438
+ """
439
+ Execute the core AutoML workflow.
440
+
441
+ Run lifecycle and experiment persistence are handled
442
+ by fit().
443
+ """
444
+
445
+ dataframe = self._load_data(data)
446
+
447
+ target_info = self.target_selector.select(
448
+ dataframe,
449
+ target=target,
450
+ task_type=task_type,
451
+ )
452
+
453
+ self.target = target_info["target"]
454
+ self.task_type = target_info["task_type"]
455
+
456
+ profile = self.profiler.profile(dataframe)
457
+
458
+ column_info = self.column_intelligence.analyze(
459
+ dataframe
460
+ )
461
+
462
+ audit = self.data_audit.audit(
463
+ data=dataframe,
464
+ target=self.target,
465
+ column_intelligence=column_info,
466
+ )
467
+
468
+ selected_models = self._select_models(
469
+ model_names=model_names,
470
+ task_type=self.task_type,
471
+ )
472
+
473
+ pipelines = self._build_pipelines(
474
+ dataframe=dataframe,
475
+ target=self.target,
476
+ task_type=self.task_type,
477
+ model_names=selected_models,
478
+ excluded_columns=excluded_columns,
479
+ )
480
+
481
+ screening_results = self.screening.screen(
482
+ data=dataframe,
483
+ target=self.target,
484
+ pipelines=pipelines,
485
+ task_type=self.task_type,
486
+ )
487
+
488
+ cv_results = self.cross_validator.evaluate(
489
+ data=dataframe,
490
+ target=self.target,
491
+ pipelines=pipelines,
492
+ task_type=self.task_type,
493
+ )
494
+
495
+ combined_results = self._combine_results(
496
+ screening_results,
497
+ cv_results,
498
+ )
499
+
500
+ initial_ranking = self.ranking.rank(
501
+ results=combined_results,
502
+ task_type=self.task_type,
503
+ objective=self.objective,
504
+ )
505
+
506
+ optimization_results = None
507
+ optimized_pipelines = {}
508
+
509
+ if self.enable_optimization:
510
+ optimization_results = (
511
+ self._optimize_top_models(
512
+ data=dataframe,
513
+ target=self.target,
514
+ task_type=self.task_type,
515
+ ranking=initial_ranking,
516
+ pipelines=pipelines,
517
+ )
518
+ )
519
+
520
+ optimized_pipelines = {
521
+ name: result["best_pipeline"]
522
+ for name, result in (
523
+ optimization_results.items()
524
+ )
525
+ if result.get("best_pipeline")
526
+ is not None
527
+ }
528
+
529
+ if optimized_pipelines:
530
+ final_candidates = optimized_pipelines
531
+
532
+ final_cv_results = (
533
+ self.cross_validator.evaluate(
534
+ data=dataframe,
535
+ target=self.target,
536
+ pipelines=final_candidates,
537
+ task_type=self.task_type,
538
+ )
539
+ )
540
+
541
+ final_screening_results = (
542
+ self.screening.screen(
543
+ data=dataframe,
544
+ target=self.target,
545
+ pipelines=final_candidates,
546
+ task_type=self.task_type,
547
+ )
548
+ )
549
+
550
+ final_combined_results = (
551
+ self._combine_results(
552
+ final_screening_results,
553
+ final_cv_results,
554
+ )
555
+ )
556
+
557
+ final_ranking = self.ranking.rank(
558
+ results=final_combined_results,
559
+ task_type=self.task_type,
560
+ objective=self.objective,
561
+ )
562
+
563
+ best_model = self._select_best_model(
564
+ final_ranking
565
+ )
566
+
567
+ candidate_pipeline = (
568
+ final_candidates.get(best_model)
569
+ )
570
+
571
+ else:
572
+ final_ranking = initial_ranking
573
+
574
+ best_model = self._select_best_model(
575
+ final_ranking
576
+ )
577
+
578
+ candidate_pipeline = pipelines.get(
579
+ best_model
580
+ )
581
+
582
+ if candidate_pipeline is None:
583
+ raise RuntimeError(
584
+ "Unable to locate the best pipeline."
585
+ )
586
+
587
+ final_pipeline = self._fit_final_pipeline(
588
+ pipeline=candidate_pipeline,
589
+ data=dataframe,
590
+ target=self.target,
591
+ )
592
+
593
+ self.best_pipeline = final_pipeline
594
+ self.best_model = best_model
595
+ self.is_fitted = True
596
+
597
+ return {
598
+ "target": target_info,
599
+ "task_type": self.task_type,
600
+ "profile": profile,
601
+ "column_intelligence": column_info,
602
+ "audit": audit,
603
+ "models_evaluated": len(selected_models),
604
+ "screening_results": screening_results,
605
+ "cv_results": cv_results,
606
+ "model_results": cv_results,
607
+ "initial_ranking": initial_ranking,
608
+ "optimization_enabled": (
609
+ self.enable_optimization
610
+ ),
611
+ "optimization_results": (
612
+ optimization_results
613
+ ),
614
+ "ranking": final_ranking,
615
+ "rankings": final_ranking,
616
+ "best_model": best_model,
617
+ "best_pipeline": final_pipeline,
618
+ "feature_selection": {
619
+ "variance_threshold": (
620
+ self.variance_threshold
621
+ ),
622
+ "correlation_threshold": (
623
+ self.correlation_threshold
624
+ ),
625
+ },
626
+ }
627
+
628
+ def predict(
629
+ self,
630
+ data: Any,
631
+ ) -> pd.Series:
632
+ """
633
+ Generate predictions using the fitted pipeline.
634
+ """
635
+
636
+ self._require_fitted()
637
+
638
+ dataframe = self._load_data(data)
639
+
640
+ return self.persistence.predict(
641
+ self.best_pipeline,
642
+ dataframe,
643
+ )
644
+
645
+ def predict_proba(
646
+ self,
647
+ data: Any,
648
+ ) -> pd.DataFrame:
649
+ """
650
+ Generate class probabilities.
651
+ """
652
+
653
+ self._require_fitted()
654
+
655
+ if self.task_type != "classification":
656
+ raise RuntimeError(
657
+ "predict_proba is only available "
658
+ "for classification tasks."
659
+ )
660
+
661
+ dataframe = self._load_data(data)
662
+
663
+ return self.persistence.predict_proba(
664
+ self.best_pipeline,
665
+ dataframe,
666
+ )
667
+
668
+ def save(
669
+ self,
670
+ path: str,
671
+ overwrite: bool = False,
672
+ ) -> str:
673
+ """
674
+ Save the fitted pipeline and metadata.
675
+ """
676
+
677
+ self._require_fitted()
678
+
679
+ metadata = {
680
+ "target": self.target,
681
+ "task_type": self.task_type,
682
+ "best_model": self.best_model,
683
+ "objective": self.objective,
684
+ "test_size": self.test_size,
685
+ "cv": self.cv,
686
+ "random_state": self.random_state,
687
+ "variance_threshold": (
688
+ self.variance_threshold
689
+ ),
690
+ "correlation_threshold": (
691
+ self.correlation_threshold
692
+ ),
693
+ "enable_optimization": (
694
+ self.enable_optimization
695
+ ),
696
+ "optimization_models": (
697
+ self.optimization_models
698
+ ),
699
+ "optimization_max_trials": (
700
+ self.optimization_max_trials
701
+ ),
702
+ "run_id": self.run_id,
703
+ "experiment_id": self.experiment_id,
704
+ "reproducibility": self.reproducibility,
705
+ }
706
+
707
+ return self.persistence.save(
708
+ pipeline=self.best_pipeline,
709
+ path=path,
710
+ metadata=metadata,
711
+ overwrite=overwrite,
712
+ )
713
+
714
+ def load(
715
+ self,
716
+ path: str,
717
+ ) -> "AutoML":
718
+ """
719
+ Load a previously saved ModelForge pipeline.
720
+ """
721
+
722
+ self.best_pipeline = self.persistence.load(path)
723
+
724
+ metadata = self.persistence.load_metadata(path)
725
+
726
+ self.best_model = metadata.get(
727
+ "best_model"
728
+ )
729
+
730
+ self.target = metadata.get(
731
+ "target"
732
+ )
733
+
734
+ self.task_type = metadata.get(
735
+ "task_type"
736
+ )
737
+
738
+ self.objective = metadata.get(
739
+ "objective",
740
+ self.objective,
741
+ )
742
+
743
+ self.test_size = metadata.get(
744
+ "test_size",
745
+ self.test_size,
746
+ )
747
+
748
+ self.cv = metadata.get(
749
+ "cv",
750
+ self.cv,
751
+ )
752
+
753
+ self.random_state = metadata.get(
754
+ "random_state",
755
+ self.random_state,
756
+ )
757
+
758
+ self.variance_threshold = metadata.get(
759
+ "variance_threshold",
760
+ self.variance_threshold,
761
+ )
762
+
763
+ self.correlation_threshold = metadata.get(
764
+ "correlation_threshold",
765
+ self.correlation_threshold,
766
+ )
767
+
768
+ self.enable_optimization = metadata.get(
769
+ "enable_optimization",
770
+ self.enable_optimization,
771
+ )
772
+
773
+ self.optimization_models = metadata.get(
774
+ "optimization_models",
775
+ self.optimization_models,
776
+ )
777
+
778
+ self.optimization_max_trials = metadata.get(
779
+ "optimization_max_trials",
780
+ self.optimization_max_trials,
781
+ )
782
+
783
+ self.run_id = metadata.get(
784
+ "run_id"
785
+ )
786
+
787
+ self.experiment_id = metadata.get(
788
+ "experiment_id"
789
+ )
790
+
791
+ self.reproducibility = metadata.get(
792
+ "reproducibility"
793
+ )
794
+
795
+ self.reproducibility_integration = (
796
+ ReproducibilityIntegration(
797
+ random_state=self.random_state
798
+ )
799
+ )
800
+
801
+ self.is_fitted = True
802
+
803
+ return self
804
+
805
+ def explain(
806
+ self,
807
+ top_n: int = 10,
808
+ ) -> pd.DataFrame:
809
+ """
810
+ Return the most important features.
811
+ """
812
+
813
+ self._require_fitted()
814
+
815
+ importance = (
816
+ self.explainability.feature_importance(
817
+ self.best_pipeline
818
+ )
819
+ )
820
+
821
+ return self.explainability.top_features(
822
+ importance,
823
+ n=top_n,
824
+ )
825
+
826
+ def summary(self) -> dict[str, Any]:
827
+ """
828
+ Return a compact AutoML summary.
829
+ """
830
+
831
+ self._require_fitted()
832
+
833
+ return {
834
+ "run_id": self.run_id,
835
+ "experiment_id": self.experiment_id,
836
+ "target": self.target,
837
+ "task_type": self.task_type,
838
+ "best_model": self.best_model,
839
+ "objective": self.objective,
840
+ "test_size": self.test_size,
841
+ "cv": self.cv,
842
+ "random_state": self.random_state,
843
+ "variance_threshold": (
844
+ self.variance_threshold
845
+ ),
846
+ "correlation_threshold": (
847
+ self.correlation_threshold
848
+ ),
849
+ "optimization_enabled": (
850
+ self.enable_optimization
851
+ ),
852
+ "optimization_models": (
853
+ self.optimization_models
854
+ ),
855
+ "optimization_max_trials": (
856
+ self.optimization_max_trials
857
+ ),
858
+ }
859
+
860
+ def list_experiments(
861
+ self,
862
+ ) -> list[dict[str, Any]]:
863
+ """
864
+ Return all locally tracked experiments.
865
+ """
866
+
867
+ return self.experiment_tracker.list_experiments()
868
+
869
+ def get_experiment(
870
+ self,
871
+ experiment_id: str,
872
+ ) -> dict[str, Any]:
873
+ """
874
+ Retrieve one tracked experiment.
875
+ """
876
+
877
+ return self.experiment_tracker.get(
878
+ experiment_id
879
+ )
880
+
881
+ def _optimize_top_models(
882
+ self,
883
+ data: pd.DataFrame,
884
+ target: str,
885
+ task_type: str,
886
+ ranking: pd.DataFrame,
887
+ pipelines: dict[str, Any],
888
+ ) -> dict[str, dict[str, Any]]:
889
+ """
890
+ Optimize the top-ranked candidate models.
891
+ """
892
+
893
+ if ranking.empty:
894
+ return {}
895
+
896
+ if "model" not in ranking.columns:
897
+ raise RuntimeError(
898
+ "Ranking results must contain "
899
+ "a 'model' column."
900
+ )
901
+
902
+ successful = ranking
903
+
904
+ if "status" in ranking.columns:
905
+ successful = ranking[
906
+ ranking["status"] == "success"
907
+ ]
908
+
909
+ top_models = successful.head(
910
+ self.optimization_models
911
+ )
912
+
913
+ results: dict[str, dict[str, Any]] = {}
914
+
915
+ for model_name in top_models[
916
+ "model"
917
+ ].tolist():
918
+ pipeline = pipelines.get(
919
+ model_name
920
+ )
921
+
922
+ if pipeline is None:
923
+ continue
924
+
925
+ parameter_space = (
926
+ self.registry.get_hyperparameter_space(
927
+ model_name
928
+ )
929
+ )
930
+
931
+ if not parameter_space:
932
+ continue
933
+
934
+ parameter_space = (
935
+ self._pipeline_parameter_space(
936
+ parameter_space
937
+ )
938
+ )
939
+
940
+ try:
941
+ results[model_name] = (
942
+ self.optimizer.optimize(
943
+ data=data,
944
+ target=target,
945
+ pipeline=pipeline,
946
+ parameter_space=parameter_space,
947
+ task_type=task_type,
948
+ )
949
+ )
950
+ except Exception as exc:
951
+ results[model_name] = {
952
+ "best_pipeline": None,
953
+ "best_params": {},
954
+ "best_score": None,
955
+ "trials": [],
956
+ "successful_trials": 0,
957
+ "failed_trials": 0,
958
+ "total_time_seconds": 0.0,
959
+ "error": str(exc),
960
+ }
961
+
962
+ return results
963
+
964
+ @staticmethod
965
+ def _pipeline_parameter_space(
966
+ parameter_space: dict[str, Any],
967
+ ) -> dict[str, Any]:
968
+ """
969
+ Convert model parameters into sklearn pipeline
970
+ parameter names.
971
+ """
972
+
973
+ return {
974
+ (
975
+ parameter_name
976
+ if "__" in parameter_name
977
+ else f"model__{parameter_name}"
978
+ ): values
979
+ for parameter_name, values
980
+ in parameter_space.items()
981
+ }
982
+
983
+ def _fit_final_pipeline(
984
+ self,
985
+ pipeline,
986
+ data: pd.DataFrame,
987
+ target: str,
988
+ ):
989
+ """
990
+ Fit the winning pipeline on the complete dataset.
991
+ """
992
+
993
+ if not isinstance(
994
+ data,
995
+ pd.DataFrame,
996
+ ):
997
+ raise TypeError(
998
+ "data must be a pandas DataFrame."
999
+ )
1000
+
1001
+ if target not in data.columns:
1002
+ raise ValueError(
1003
+ f"Target column '{target}' "
1004
+ "does not exist."
1005
+ )
1006
+
1007
+ X = data.drop(
1008
+ columns=[target]
1009
+ )
1010
+
1011
+ y = data[target]
1012
+
1013
+ try:
1014
+ pipeline.fit(
1015
+ X,
1016
+ y,
1017
+ )
1018
+ except Exception as exc:
1019
+ raise RuntimeError(
1020
+ "Failed to fit the final selected "
1021
+ f"pipeline: {exc}"
1022
+ ) from exc
1023
+
1024
+ return pipeline
1025
+
1026
+ def _build_pipelines(
1027
+ self,
1028
+ dataframe: pd.DataFrame,
1029
+ target: str,
1030
+ task_type: str,
1031
+ model_names: list[str],
1032
+ excluded_columns: list[str] | None,
1033
+ ) -> dict:
1034
+ """
1035
+ Generate candidate pipelines.
1036
+ """
1037
+
1038
+ pipelines = {}
1039
+
1040
+ for model_name in model_names:
1041
+ pipelines[model_name] = (
1042
+ self.pipeline_generator.build(
1043
+ data=dataframe,
1044
+ target=target,
1045
+ model_name=model_name,
1046
+ task_type=task_type,
1047
+ excluded_columns=excluded_columns,
1048
+ variance_threshold=(
1049
+ self.variance_threshold
1050
+ ),
1051
+ correlation_threshold=(
1052
+ self.correlation_threshold
1053
+ ),
1054
+ )
1055
+ )
1056
+
1057
+ return pipelines
1058
+
1059
+ def _select_models(
1060
+ self,
1061
+ model_names: list[str] | None,
1062
+ task_type: str,
1063
+ ) -> list[str]:
1064
+ """
1065
+ Resolve and validate model names.
1066
+ """
1067
+
1068
+ available = self.registry.list_models(
1069
+ task_type=task_type
1070
+ )
1071
+
1072
+ if model_names is None:
1073
+ return available
1074
+
1075
+ invalid = [
1076
+ name
1077
+ for name in model_names
1078
+ if name not in available
1079
+ ]
1080
+
1081
+ if invalid:
1082
+ raise ValueError(
1083
+ "Unknown or incompatible models: "
1084
+ + ", ".join(invalid)
1085
+ )
1086
+
1087
+ if not model_names:
1088
+ raise ValueError(
1089
+ "At least one model must be selected."
1090
+ )
1091
+
1092
+ return model_names
1093
+
1094
+ def _combine_results(
1095
+ self,
1096
+ screening_results: pd.DataFrame,
1097
+ cv_results: pd.DataFrame,
1098
+ ) -> pd.DataFrame:
1099
+ """
1100
+ Combine holdout and cross-validation results.
1101
+ """
1102
+
1103
+ screening = pd.DataFrame(
1104
+ screening_results
1105
+ )
1106
+
1107
+ cross_validation = pd.DataFrame(
1108
+ cv_results
1109
+ )
1110
+
1111
+ if screening.empty:
1112
+ raise RuntimeError(
1113
+ "No model screening results available."
1114
+ )
1115
+
1116
+ if cross_validation.empty:
1117
+ return screening
1118
+
1119
+ if "model" not in screening.columns:
1120
+ raise RuntimeError(
1121
+ "Screening results must contain "
1122
+ "a 'model' column."
1123
+ )
1124
+
1125
+ if "model" not in cross_validation.columns:
1126
+ raise RuntimeError(
1127
+ "Cross-validation results must "
1128
+ "contain a 'model' column."
1129
+ )
1130
+
1131
+ return screening.merge(
1132
+ cross_validation,
1133
+ on="model",
1134
+ how="left",
1135
+ suffixes=(
1136
+ "",
1137
+ "_cv",
1138
+ ),
1139
+ )
1140
+
1141
+ def _select_best_model(
1142
+ self,
1143
+ ranking: pd.DataFrame,
1144
+ ) -> str:
1145
+ """
1146
+ Select the top successful model from ranking.
1147
+ """
1148
+
1149
+ if ranking.empty:
1150
+ raise RuntimeError(
1151
+ "Ranking produced no models."
1152
+ )
1153
+
1154
+ if "status" in ranking.columns:
1155
+ successful = ranking[
1156
+ ranking["status"] == "success"
1157
+ ]
1158
+ else:
1159
+ successful = ranking
1160
+
1161
+ if successful.empty:
1162
+ raise RuntimeError(
1163
+ "No successful model was found."
1164
+ )
1165
+
1166
+ return str(
1167
+ successful.iloc[0]["model"]
1168
+ )
1169
+
1170
+ def _load_data(
1171
+ self,
1172
+ data: Any,
1173
+ ) -> pd.DataFrame:
1174
+ """
1175
+ Load a dataset through DatasetLoader.
1176
+ """
1177
+
1178
+ if isinstance(
1179
+ data,
1180
+ pd.DataFrame,
1181
+ ):
1182
+ return data.copy()
1183
+
1184
+ if isinstance(
1185
+ data,
1186
+ (str, Path),
1187
+ ):
1188
+ return self.loader.load(
1189
+ str(data)
1190
+ )
1191
+
1192
+ raise TypeError(
1193
+ "data must be a pandas DataFrame "
1194
+ "or a supported dataset path."
1195
+ )
1196
+
1197
+ def _configuration(
1198
+ self,
1199
+ target: str | None = None,
1200
+ task_type: str | None = None,
1201
+ models: list[str] | None = None,
1202
+ excluded_columns: list[str] | None = None,
1203
+ ) -> dict[str, Any]:
1204
+ """
1205
+ Return the effective configuration used by
1206
+ the current AutoML run.
1207
+
1208
+ Explicit runtime values are included so the
1209
+ reproducibility snapshot and experiment tracker
1210
+ persist exactly the same configuration.
1211
+ """
1212
+
1213
+ effective_target = (
1214
+ target
1215
+ if target is not None
1216
+ else self._configured_target
1217
+ )
1218
+
1219
+ effective_task_type = (
1220
+ task_type
1221
+ if task_type is not None
1222
+ else self._configured_task_type
1223
+ )
1224
+
1225
+ effective_models = (
1226
+ models
1227
+ if models is not None
1228
+ else self._configured_models
1229
+ )
1230
+
1231
+ effective_excluded_columns = (
1232
+ excluded_columns
1233
+ if excluded_columns is not None
1234
+ else self._configured_excluded_columns
1235
+ )
1236
+
1237
+ return {
1238
+ "target": effective_target,
1239
+ "task_type": effective_task_type,
1240
+ "models": effective_models,
1241
+ "excluded_columns": (
1242
+ effective_excluded_columns
1243
+ ),
1244
+ "test_size": self.test_size,
1245
+ "cv": self.cv,
1246
+ "random_state": self.random_state,
1247
+ "objective": self.objective,
1248
+ "variance_threshold": (
1249
+ self.variance_threshold
1250
+ ),
1251
+ "correlation_threshold": (
1252
+ self.correlation_threshold
1253
+ ),
1254
+ "enable_optimization": (
1255
+ self.enable_optimization
1256
+ ),
1257
+ "optimization_models": (
1258
+ self.optimization_models
1259
+ ),
1260
+ "optimization_max_trials": (
1261
+ self.optimization_max_trials
1262
+ ),
1263
+ "experiment_directory": (
1264
+ str(self._configured_experiment_directory)
1265
+ ),
1266
+ }
1267
+
1268
+ def _config_value(
1269
+ self,
1270
+ key: str,
1271
+ default: Any = None,
1272
+ ) -> Any:
1273
+ """
1274
+ Read a configuration value, including nested keys.
1275
+ """
1276
+
1277
+ if not self._config_provided:
1278
+ return default
1279
+
1280
+ values = self.config.to_dict()
1281
+ current: Any = values
1282
+
1283
+ for part in key.split("."):
1284
+ if not isinstance(
1285
+ current,
1286
+ dict,
1287
+ ):
1288
+ return default
1289
+
1290
+ if part not in current:
1291
+ return default
1292
+
1293
+ current = current[part]
1294
+
1295
+ return current
1296
+
1297
+ def _require_fitted(self) -> None:
1298
+ if not self.is_fitted:
1299
+ raise RuntimeError(
1300
+ "AutoML has not been fitted yet."
1301
+ )
1302
+
1303
+ if self.best_pipeline is None:
1304
+ raise RuntimeError(
1305
+ "AutoML is marked as fitted but "
1306
+ "no fitted pipeline is available."
1307
+ )
1308
+
1309
+ def _validate_configuration(self) -> None:
1310
+ """
1311
+ Validate AutoML constructor configuration.
1312
+ """
1313
+
1314
+ if not isinstance(
1315
+ self.test_size,
1316
+ (int, float),
1317
+ ):
1318
+ raise TypeError(
1319
+ "test_size must be numeric."
1320
+ )
1321
+
1322
+ if isinstance(
1323
+ self.test_size,
1324
+ bool,
1325
+ ):
1326
+ raise TypeError(
1327
+ "test_size must be numeric."
1328
+ )
1329
+
1330
+ if not 0 < self.test_size < 1:
1331
+ raise ValueError(
1332
+ "test_size must be between 0 and 1."
1333
+ )
1334
+
1335
+ if not isinstance(
1336
+ self.cv,
1337
+ int,
1338
+ ):
1339
+ raise TypeError(
1340
+ "cv must be an integer."
1341
+ )
1342
+
1343
+ if isinstance(
1344
+ self.cv,
1345
+ bool,
1346
+ ):
1347
+ raise TypeError(
1348
+ "cv must be an integer."
1349
+ )
1350
+
1351
+ if self.cv < 2:
1352
+ raise ValueError(
1353
+ "cv must be at least 2."
1354
+ )
1355
+
1356
+ if not isinstance(
1357
+ self.random_state,
1358
+ int,
1359
+ ):
1360
+ raise TypeError(
1361
+ "random_state must be an integer."
1362
+ )
1363
+
1364
+ if isinstance(
1365
+ self.random_state,
1366
+ bool,
1367
+ ):
1368
+ raise TypeError(
1369
+ "random_state must be an integer."
1370
+ )
1371
+
1372
+ if self.objective not in {
1373
+ "balanced",
1374
+ "performance",
1375
+ "error",
1376
+ "speed",
1377
+ }:
1378
+ raise ValueError(
1379
+ "Invalid ranking objective."
1380
+ )
1381
+
1382
+ if self.variance_threshold is not None:
1383
+ if isinstance(
1384
+ self.variance_threshold,
1385
+ bool,
1386
+ ):
1387
+ raise TypeError(
1388
+ "variance_threshold must be numeric."
1389
+ )
1390
+
1391
+ if not isinstance(
1392
+ self.variance_threshold,
1393
+ (int, float),
1394
+ ):
1395
+ raise TypeError(
1396
+ "variance_threshold must be numeric."
1397
+ )
1398
+
1399
+ if self.variance_threshold < 0:
1400
+ raise ValueError(
1401
+ "variance_threshold cannot "
1402
+ "be negative."
1403
+ )
1404
+
1405
+ if self.correlation_threshold is not None:
1406
+ if isinstance(
1407
+ self.correlation_threshold,
1408
+ bool,
1409
+ ):
1410
+ raise TypeError(
1411
+ "correlation_threshold must "
1412
+ "be numeric."
1413
+ )
1414
+
1415
+ if not isinstance(
1416
+ self.correlation_threshold,
1417
+ (int, float),
1418
+ ):
1419
+ raise TypeError(
1420
+ "correlation_threshold must "
1421
+ "be numeric."
1422
+ )
1423
+
1424
+ if not (
1425
+ 0
1426
+ < self.correlation_threshold
1427
+ <= 1
1428
+ ):
1429
+ raise ValueError(
1430
+ "correlation_threshold must "
1431
+ "be between 0 and 1."
1432
+ )
1433
+
1434
+ if not isinstance(
1435
+ self.enable_optimization,
1436
+ bool,
1437
+ ):
1438
+ raise TypeError(
1439
+ "enable_optimization must be a boolean."
1440
+ )
1441
+
1442
+ if not isinstance(
1443
+ self.optimization_models,
1444
+ int,
1445
+ ) or isinstance(
1446
+ self.optimization_models,
1447
+ bool,
1448
+ ):
1449
+ raise TypeError(
1450
+ "optimization_models must be an integer."
1451
+ )
1452
+
1453
+ if self.optimization_models < 1:
1454
+ raise ValueError(
1455
+ "optimization_models must be at least 1."
1456
+ )
1457
+
1458
+ if not isinstance(
1459
+ self.optimization_max_trials,
1460
+ int,
1461
+ ) or isinstance(
1462
+ self.optimization_max_trials,
1463
+ bool,
1464
+ ):
1465
+ raise TypeError(
1466
+ "optimization_max_trials must be an integer."
1467
+ )
1468
+
1469
+ if self.optimization_max_trials < 1:
1470
+ raise ValueError(
1471
+ "optimization_max_trials must be at least 1."
1472
+ )