autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
modelforge/automl.py
ADDED
|
@@ -0,0 +1,1472 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from modelforge.column_intelligence import ColumnIntelligence
|
|
9
|
+
from modelforge.config import ModelForgeConfig
|
|
10
|
+
from modelforge.cross_validation import CrossValidationEngine
|
|
11
|
+
from modelforge.data_audit import DataQualityAuditor
|
|
12
|
+
from modelforge.data_loader import DatasetLoader
|
|
13
|
+
from modelforge.explainability import ExplainabilityEngine
|
|
14
|
+
from modelforge.experiment_tracker import ExperimentTracker
|
|
15
|
+
from modelforge.hyperparameter_optimization import (
|
|
16
|
+
HyperparameterOptimizationEngine,
|
|
17
|
+
)
|
|
18
|
+
from modelforge.model_registry import ModelRegistry
|
|
19
|
+
from modelforge.model_screening import ModelScreeningEngine
|
|
20
|
+
from modelforge.persistence import ModelPersistence
|
|
21
|
+
from modelforge.pipeline_generator import PipelineGenerator
|
|
22
|
+
from modelforge.profiler import DatasetProfiler
|
|
23
|
+
from modelforge.ranking import RankingEngine
|
|
24
|
+
from modelforge.reproducibility_integration import (
|
|
25
|
+
ReproducibilityIntegration,
|
|
26
|
+
)
|
|
27
|
+
from modelforge.run_manager import RunManager
|
|
28
|
+
from modelforge.target_selector import TargetSelector
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class AutoML:
|
|
32
|
+
"""
|
|
33
|
+
Main ModelForge AutoML orchestration engine.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
test_size: float = 0.2,
|
|
39
|
+
cv: int = 5,
|
|
40
|
+
random_state: int = 42,
|
|
41
|
+
objective: str = "balanced",
|
|
42
|
+
variance_threshold: float | None = None,
|
|
43
|
+
correlation_threshold: float | None = None,
|
|
44
|
+
enable_optimization: bool = False,
|
|
45
|
+
optimization_models: int = 3,
|
|
46
|
+
optimization_max_trials: int = 10,
|
|
47
|
+
experiment_directory: str | Path = (
|
|
48
|
+
".modelforge/experiments"
|
|
49
|
+
),
|
|
50
|
+
config: ModelForgeConfig | None = None,
|
|
51
|
+
):
|
|
52
|
+
if config is not None and not isinstance(
|
|
53
|
+
config,
|
|
54
|
+
ModelForgeConfig,
|
|
55
|
+
):
|
|
56
|
+
raise TypeError(
|
|
57
|
+
"config must be a ModelForgeConfig"
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
self._config_provided = config is not None
|
|
61
|
+
|
|
62
|
+
self.config = (
|
|
63
|
+
config
|
|
64
|
+
if config is not None
|
|
65
|
+
else ModelForgeConfig()
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
self._configured_target = self._config_value(
|
|
69
|
+
"target",
|
|
70
|
+
None,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
self._configured_task_type = self._config_value(
|
|
74
|
+
"task_type",
|
|
75
|
+
None,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
self._configured_models = self._config_value(
|
|
79
|
+
"models",
|
|
80
|
+
None,
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
self._configured_excluded_columns = (
|
|
84
|
+
self._config_value(
|
|
85
|
+
"excluded_columns",
|
|
86
|
+
[],
|
|
87
|
+
)
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
self.test_size = self._config_value(
|
|
91
|
+
"test_size",
|
|
92
|
+
test_size,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
self.cv = self._config_value(
|
|
96
|
+
"cv",
|
|
97
|
+
cv,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
self.random_state = self._config_value(
|
|
101
|
+
"random_state",
|
|
102
|
+
random_state,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
self.objective = self._config_value(
|
|
106
|
+
"objective",
|
|
107
|
+
objective,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
self.variance_threshold = self._config_value(
|
|
111
|
+
"feature_selection.variance_threshold",
|
|
112
|
+
variance_threshold,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
self.correlation_threshold = self._config_value(
|
|
116
|
+
"feature_selection.correlation_threshold",
|
|
117
|
+
correlation_threshold,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
configured_optimization_enabled = (
|
|
121
|
+
self._config_value(
|
|
122
|
+
"optimization.enabled",
|
|
123
|
+
None,
|
|
124
|
+
)
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
self.enable_optimization = (
|
|
128
|
+
enable_optimization
|
|
129
|
+
if configured_optimization_enabled is None
|
|
130
|
+
else configured_optimization_enabled
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
configured_optimization_models = (
|
|
134
|
+
self._config_value(
|
|
135
|
+
"optimization.models",
|
|
136
|
+
None,
|
|
137
|
+
)
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
self.optimization_models = (
|
|
141
|
+
optimization_models
|
|
142
|
+
if configured_optimization_models is None
|
|
143
|
+
else configured_optimization_models
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
configured_optimization_max_trials = (
|
|
147
|
+
self._config_value(
|
|
148
|
+
"optimization.max_trials",
|
|
149
|
+
None,
|
|
150
|
+
)
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
self.optimization_max_trials = (
|
|
154
|
+
optimization_max_trials
|
|
155
|
+
if configured_optimization_max_trials is None
|
|
156
|
+
else configured_optimization_max_trials
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
self._configured_experiment_directory = (
|
|
160
|
+
self._config_value(
|
|
161
|
+
"experiment_directory",
|
|
162
|
+
experiment_directory,
|
|
163
|
+
)
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
self.experiment_directory = Path(
|
|
167
|
+
self._configured_experiment_directory
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
self._validate_configuration()
|
|
171
|
+
|
|
172
|
+
self.loader = DatasetLoader()
|
|
173
|
+
self.target_selector = TargetSelector()
|
|
174
|
+
self.profiler = DatasetProfiler()
|
|
175
|
+
self.column_intelligence = ColumnIntelligence()
|
|
176
|
+
self.data_audit = DataQualityAuditor()
|
|
177
|
+
|
|
178
|
+
self.registry = ModelRegistry()
|
|
179
|
+
self.pipeline_generator = PipelineGenerator()
|
|
180
|
+
|
|
181
|
+
self.screening = ModelScreeningEngine(
|
|
182
|
+
test_size=self.test_size,
|
|
183
|
+
random_state=self.random_state,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
self.cross_validator = CrossValidationEngine(
|
|
187
|
+
cv=self.cv,
|
|
188
|
+
random_state=self.random_state,
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
self.optimizer = (
|
|
192
|
+
HyperparameterOptimizationEngine(
|
|
193
|
+
cv=self.cv,
|
|
194
|
+
random_state=self.random_state,
|
|
195
|
+
max_trials=self.optimization_max_trials,
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
self.ranking = RankingEngine()
|
|
200
|
+
self.explainability = ExplainabilityEngine()
|
|
201
|
+
self.persistence = ModelPersistence()
|
|
202
|
+
|
|
203
|
+
self.run_manager = RunManager()
|
|
204
|
+
|
|
205
|
+
self.experiment_tracker = ExperimentTracker(
|
|
206
|
+
self.experiment_directory
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
self.is_fitted = False
|
|
210
|
+
self.best_pipeline = None
|
|
211
|
+
self.best_model = None
|
|
212
|
+
self.result = None
|
|
213
|
+
self.target = None
|
|
214
|
+
self.task_type = None
|
|
215
|
+
self.run_id = None
|
|
216
|
+
self.experiment_id = None
|
|
217
|
+
|
|
218
|
+
self.reproducibility = None
|
|
219
|
+
|
|
220
|
+
self.reproducibility_integration = (
|
|
221
|
+
ReproducibilityIntegration(
|
|
222
|
+
random_state=self.random_state
|
|
223
|
+
)
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
def fit(
|
|
227
|
+
self,
|
|
228
|
+
data: Any,
|
|
229
|
+
target: str | None = None,
|
|
230
|
+
task_type: str | None = None,
|
|
231
|
+
model_names: list[str] | None = None,
|
|
232
|
+
excluded_columns: list[str] | None = None,
|
|
233
|
+
) -> dict[str, Any]:
|
|
234
|
+
"""
|
|
235
|
+
Run the complete ModelForge AutoML workflow.
|
|
236
|
+
|
|
237
|
+
A run is automatically started before training and
|
|
238
|
+
recorded after successful or failed execution.
|
|
239
|
+
|
|
240
|
+
Reproducibility metadata is captured for every run.
|
|
241
|
+
"""
|
|
242
|
+
|
|
243
|
+
target = (
|
|
244
|
+
target
|
|
245
|
+
if target is not None
|
|
246
|
+
else self._configured_target
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
task_type = (
|
|
250
|
+
task_type
|
|
251
|
+
if task_type is not None
|
|
252
|
+
else self._configured_task_type
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
model_names = (
|
|
256
|
+
model_names
|
|
257
|
+
if model_names is not None
|
|
258
|
+
else self._configured_models
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
excluded_columns = (
|
|
262
|
+
excluded_columns
|
|
263
|
+
if excluded_columns is not None
|
|
264
|
+
else self._configured_excluded_columns
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
if target is None:
|
|
268
|
+
raise ValueError(
|
|
269
|
+
"target must be provided to fit() "
|
|
270
|
+
"or defined in ModelForgeConfig."
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
self.run_id = self.run_manager.start(
|
|
274
|
+
{
|
|
275
|
+
"target": target,
|
|
276
|
+
"requested_task_type": task_type,
|
|
277
|
+
"objective": self.objective,
|
|
278
|
+
"cv": self.cv,
|
|
279
|
+
"test_size": self.test_size,
|
|
280
|
+
"random_state": self.random_state,
|
|
281
|
+
"enable_optimization": (
|
|
282
|
+
self.enable_optimization
|
|
283
|
+
),
|
|
284
|
+
}
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
self.reproducibility = None
|
|
288
|
+
|
|
289
|
+
try:
|
|
290
|
+
reproducibility_data = self._load_data(data)
|
|
291
|
+
|
|
292
|
+
result = self._fit_workflow(
|
|
293
|
+
data=reproducibility_data,
|
|
294
|
+
target=target,
|
|
295
|
+
task_type=task_type,
|
|
296
|
+
model_names=model_names,
|
|
297
|
+
excluded_columns=excluded_columns,
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
effective_configuration = (
|
|
301
|
+
self._configuration(
|
|
302
|
+
target=target,
|
|
303
|
+
task_type=self.task_type,
|
|
304
|
+
models=model_names,
|
|
305
|
+
excluded_columns=excluded_columns,
|
|
306
|
+
)
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
self.reproducibility = (
|
|
310
|
+
self.reproducibility_integration.create_run_snapshot(
|
|
311
|
+
data=reproducibility_data,
|
|
312
|
+
configuration=effective_configuration,
|
|
313
|
+
target=self.target,
|
|
314
|
+
task_type=self.task_type,
|
|
315
|
+
run_id=self.run_id,
|
|
316
|
+
extra_metadata={
|
|
317
|
+
"model_names": model_names,
|
|
318
|
+
"excluded_columns": excluded_columns,
|
|
319
|
+
},
|
|
320
|
+
)
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
result["reproducibility"] = (
|
|
324
|
+
self.reproducibility
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
self.run_manager.update(
|
|
328
|
+
{
|
|
329
|
+
"best_model": self.best_model,
|
|
330
|
+
"task_type": self.task_type,
|
|
331
|
+
"models_evaluated": result.get(
|
|
332
|
+
"models_evaluated"
|
|
333
|
+
),
|
|
334
|
+
}
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
run_summary = self.run_manager.complete()
|
|
338
|
+
|
|
339
|
+
self.run_id = run_summary["run_id"]
|
|
340
|
+
|
|
341
|
+
result["run_id"] = self.run_id
|
|
342
|
+
result["run_summary"] = run_summary
|
|
343
|
+
|
|
344
|
+
self.experiment_id = (
|
|
345
|
+
self.experiment_tracker.record(
|
|
346
|
+
result=result,
|
|
347
|
+
configuration=effective_configuration,
|
|
348
|
+
)
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
result["experiment_id"] = (
|
|
352
|
+
self.experiment_id
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
self.result = result
|
|
356
|
+
|
|
357
|
+
return result
|
|
358
|
+
|
|
359
|
+
except Exception as exc:
|
|
360
|
+
failure_summary = self.run_manager.fail(exc)
|
|
361
|
+
|
|
362
|
+
self.run_id = failure_summary["run_id"]
|
|
363
|
+
|
|
364
|
+
if self.reproducibility is None:
|
|
365
|
+
try:
|
|
366
|
+
reproducibility_data = (
|
|
367
|
+
self._load_data(data)
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
effective_configuration = (
|
|
371
|
+
self._configuration(
|
|
372
|
+
target=target,
|
|
373
|
+
task_type=self.task_type,
|
|
374
|
+
models=model_names,
|
|
375
|
+
excluded_columns=excluded_columns,
|
|
376
|
+
)
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
self.reproducibility = (
|
|
380
|
+
self.reproducibility_integration.create_run_snapshot(
|
|
381
|
+
data=reproducibility_data,
|
|
382
|
+
configuration=effective_configuration,
|
|
383
|
+
target=target,
|
|
384
|
+
task_type=self.task_type,
|
|
385
|
+
run_id=self.run_id,
|
|
386
|
+
extra_metadata={
|
|
387
|
+
"model_names": model_names,
|
|
388
|
+
"excluded_columns": (
|
|
389
|
+
excluded_columns
|
|
390
|
+
),
|
|
391
|
+
},
|
|
392
|
+
)
|
|
393
|
+
)
|
|
394
|
+
except Exception:
|
|
395
|
+
self.reproducibility = None
|
|
396
|
+
|
|
397
|
+
failure_configuration = (
|
|
398
|
+
self._configuration(
|
|
399
|
+
target=target,
|
|
400
|
+
task_type=self.task_type,
|
|
401
|
+
models=model_names,
|
|
402
|
+
excluded_columns=excluded_columns,
|
|
403
|
+
)
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
failure_result = {
|
|
407
|
+
"run_id": self.run_id,
|
|
408
|
+
"status": "failed",
|
|
409
|
+
"error": str(exc),
|
|
410
|
+
"run_summary": failure_summary,
|
|
411
|
+
"target": self.target,
|
|
412
|
+
"task_type": self.task_type,
|
|
413
|
+
"configuration": failure_configuration,
|
|
414
|
+
"reproducibility": (
|
|
415
|
+
self.reproducibility
|
|
416
|
+
),
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
self.experiment_id = (
|
|
420
|
+
self.experiment_tracker.record(
|
|
421
|
+
result=failure_result,
|
|
422
|
+
configuration=failure_configuration,
|
|
423
|
+
)
|
|
424
|
+
)
|
|
425
|
+
|
|
426
|
+
self.result = failure_result
|
|
427
|
+
|
|
428
|
+
raise
|
|
429
|
+
|
|
430
|
+
def _fit_workflow(
|
|
431
|
+
self,
|
|
432
|
+
data: Any,
|
|
433
|
+
target: str,
|
|
434
|
+
task_type: str | None,
|
|
435
|
+
model_names: list[str] | None,
|
|
436
|
+
excluded_columns: list[str] | None,
|
|
437
|
+
) -> dict[str, Any]:
|
|
438
|
+
"""
|
|
439
|
+
Execute the core AutoML workflow.
|
|
440
|
+
|
|
441
|
+
Run lifecycle and experiment persistence are handled
|
|
442
|
+
by fit().
|
|
443
|
+
"""
|
|
444
|
+
|
|
445
|
+
dataframe = self._load_data(data)
|
|
446
|
+
|
|
447
|
+
target_info = self.target_selector.select(
|
|
448
|
+
dataframe,
|
|
449
|
+
target=target,
|
|
450
|
+
task_type=task_type,
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
self.target = target_info["target"]
|
|
454
|
+
self.task_type = target_info["task_type"]
|
|
455
|
+
|
|
456
|
+
profile = self.profiler.profile(dataframe)
|
|
457
|
+
|
|
458
|
+
column_info = self.column_intelligence.analyze(
|
|
459
|
+
dataframe
|
|
460
|
+
)
|
|
461
|
+
|
|
462
|
+
audit = self.data_audit.audit(
|
|
463
|
+
data=dataframe,
|
|
464
|
+
target=self.target,
|
|
465
|
+
column_intelligence=column_info,
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
selected_models = self._select_models(
|
|
469
|
+
model_names=model_names,
|
|
470
|
+
task_type=self.task_type,
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
pipelines = self._build_pipelines(
|
|
474
|
+
dataframe=dataframe,
|
|
475
|
+
target=self.target,
|
|
476
|
+
task_type=self.task_type,
|
|
477
|
+
model_names=selected_models,
|
|
478
|
+
excluded_columns=excluded_columns,
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
screening_results = self.screening.screen(
|
|
482
|
+
data=dataframe,
|
|
483
|
+
target=self.target,
|
|
484
|
+
pipelines=pipelines,
|
|
485
|
+
task_type=self.task_type,
|
|
486
|
+
)
|
|
487
|
+
|
|
488
|
+
cv_results = self.cross_validator.evaluate(
|
|
489
|
+
data=dataframe,
|
|
490
|
+
target=self.target,
|
|
491
|
+
pipelines=pipelines,
|
|
492
|
+
task_type=self.task_type,
|
|
493
|
+
)
|
|
494
|
+
|
|
495
|
+
combined_results = self._combine_results(
|
|
496
|
+
screening_results,
|
|
497
|
+
cv_results,
|
|
498
|
+
)
|
|
499
|
+
|
|
500
|
+
initial_ranking = self.ranking.rank(
|
|
501
|
+
results=combined_results,
|
|
502
|
+
task_type=self.task_type,
|
|
503
|
+
objective=self.objective,
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
optimization_results = None
|
|
507
|
+
optimized_pipelines = {}
|
|
508
|
+
|
|
509
|
+
if self.enable_optimization:
|
|
510
|
+
optimization_results = (
|
|
511
|
+
self._optimize_top_models(
|
|
512
|
+
data=dataframe,
|
|
513
|
+
target=self.target,
|
|
514
|
+
task_type=self.task_type,
|
|
515
|
+
ranking=initial_ranking,
|
|
516
|
+
pipelines=pipelines,
|
|
517
|
+
)
|
|
518
|
+
)
|
|
519
|
+
|
|
520
|
+
optimized_pipelines = {
|
|
521
|
+
name: result["best_pipeline"]
|
|
522
|
+
for name, result in (
|
|
523
|
+
optimization_results.items()
|
|
524
|
+
)
|
|
525
|
+
if result.get("best_pipeline")
|
|
526
|
+
is not None
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
if optimized_pipelines:
|
|
530
|
+
final_candidates = optimized_pipelines
|
|
531
|
+
|
|
532
|
+
final_cv_results = (
|
|
533
|
+
self.cross_validator.evaluate(
|
|
534
|
+
data=dataframe,
|
|
535
|
+
target=self.target,
|
|
536
|
+
pipelines=final_candidates,
|
|
537
|
+
task_type=self.task_type,
|
|
538
|
+
)
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
final_screening_results = (
|
|
542
|
+
self.screening.screen(
|
|
543
|
+
data=dataframe,
|
|
544
|
+
target=self.target,
|
|
545
|
+
pipelines=final_candidates,
|
|
546
|
+
task_type=self.task_type,
|
|
547
|
+
)
|
|
548
|
+
)
|
|
549
|
+
|
|
550
|
+
final_combined_results = (
|
|
551
|
+
self._combine_results(
|
|
552
|
+
final_screening_results,
|
|
553
|
+
final_cv_results,
|
|
554
|
+
)
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
final_ranking = self.ranking.rank(
|
|
558
|
+
results=final_combined_results,
|
|
559
|
+
task_type=self.task_type,
|
|
560
|
+
objective=self.objective,
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
best_model = self._select_best_model(
|
|
564
|
+
final_ranking
|
|
565
|
+
)
|
|
566
|
+
|
|
567
|
+
candidate_pipeline = (
|
|
568
|
+
final_candidates.get(best_model)
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
else:
|
|
572
|
+
final_ranking = initial_ranking
|
|
573
|
+
|
|
574
|
+
best_model = self._select_best_model(
|
|
575
|
+
final_ranking
|
|
576
|
+
)
|
|
577
|
+
|
|
578
|
+
candidate_pipeline = pipelines.get(
|
|
579
|
+
best_model
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
if candidate_pipeline is None:
|
|
583
|
+
raise RuntimeError(
|
|
584
|
+
"Unable to locate the best pipeline."
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
final_pipeline = self._fit_final_pipeline(
|
|
588
|
+
pipeline=candidate_pipeline,
|
|
589
|
+
data=dataframe,
|
|
590
|
+
target=self.target,
|
|
591
|
+
)
|
|
592
|
+
|
|
593
|
+
self.best_pipeline = final_pipeline
|
|
594
|
+
self.best_model = best_model
|
|
595
|
+
self.is_fitted = True
|
|
596
|
+
|
|
597
|
+
return {
|
|
598
|
+
"target": target_info,
|
|
599
|
+
"task_type": self.task_type,
|
|
600
|
+
"profile": profile,
|
|
601
|
+
"column_intelligence": column_info,
|
|
602
|
+
"audit": audit,
|
|
603
|
+
"models_evaluated": len(selected_models),
|
|
604
|
+
"screening_results": screening_results,
|
|
605
|
+
"cv_results": cv_results,
|
|
606
|
+
"model_results": cv_results,
|
|
607
|
+
"initial_ranking": initial_ranking,
|
|
608
|
+
"optimization_enabled": (
|
|
609
|
+
self.enable_optimization
|
|
610
|
+
),
|
|
611
|
+
"optimization_results": (
|
|
612
|
+
optimization_results
|
|
613
|
+
),
|
|
614
|
+
"ranking": final_ranking,
|
|
615
|
+
"rankings": final_ranking,
|
|
616
|
+
"best_model": best_model,
|
|
617
|
+
"best_pipeline": final_pipeline,
|
|
618
|
+
"feature_selection": {
|
|
619
|
+
"variance_threshold": (
|
|
620
|
+
self.variance_threshold
|
|
621
|
+
),
|
|
622
|
+
"correlation_threshold": (
|
|
623
|
+
self.correlation_threshold
|
|
624
|
+
),
|
|
625
|
+
},
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
def predict(
|
|
629
|
+
self,
|
|
630
|
+
data: Any,
|
|
631
|
+
) -> pd.Series:
|
|
632
|
+
"""
|
|
633
|
+
Generate predictions using the fitted pipeline.
|
|
634
|
+
"""
|
|
635
|
+
|
|
636
|
+
self._require_fitted()
|
|
637
|
+
|
|
638
|
+
dataframe = self._load_data(data)
|
|
639
|
+
|
|
640
|
+
return self.persistence.predict(
|
|
641
|
+
self.best_pipeline,
|
|
642
|
+
dataframe,
|
|
643
|
+
)
|
|
644
|
+
|
|
645
|
+
def predict_proba(
|
|
646
|
+
self,
|
|
647
|
+
data: Any,
|
|
648
|
+
) -> pd.DataFrame:
|
|
649
|
+
"""
|
|
650
|
+
Generate class probabilities.
|
|
651
|
+
"""
|
|
652
|
+
|
|
653
|
+
self._require_fitted()
|
|
654
|
+
|
|
655
|
+
if self.task_type != "classification":
|
|
656
|
+
raise RuntimeError(
|
|
657
|
+
"predict_proba is only available "
|
|
658
|
+
"for classification tasks."
|
|
659
|
+
)
|
|
660
|
+
|
|
661
|
+
dataframe = self._load_data(data)
|
|
662
|
+
|
|
663
|
+
return self.persistence.predict_proba(
|
|
664
|
+
self.best_pipeline,
|
|
665
|
+
dataframe,
|
|
666
|
+
)
|
|
667
|
+
|
|
668
|
+
def save(
|
|
669
|
+
self,
|
|
670
|
+
path: str,
|
|
671
|
+
overwrite: bool = False,
|
|
672
|
+
) -> str:
|
|
673
|
+
"""
|
|
674
|
+
Save the fitted pipeline and metadata.
|
|
675
|
+
"""
|
|
676
|
+
|
|
677
|
+
self._require_fitted()
|
|
678
|
+
|
|
679
|
+
metadata = {
|
|
680
|
+
"target": self.target,
|
|
681
|
+
"task_type": self.task_type,
|
|
682
|
+
"best_model": self.best_model,
|
|
683
|
+
"objective": self.objective,
|
|
684
|
+
"test_size": self.test_size,
|
|
685
|
+
"cv": self.cv,
|
|
686
|
+
"random_state": self.random_state,
|
|
687
|
+
"variance_threshold": (
|
|
688
|
+
self.variance_threshold
|
|
689
|
+
),
|
|
690
|
+
"correlation_threshold": (
|
|
691
|
+
self.correlation_threshold
|
|
692
|
+
),
|
|
693
|
+
"enable_optimization": (
|
|
694
|
+
self.enable_optimization
|
|
695
|
+
),
|
|
696
|
+
"optimization_models": (
|
|
697
|
+
self.optimization_models
|
|
698
|
+
),
|
|
699
|
+
"optimization_max_trials": (
|
|
700
|
+
self.optimization_max_trials
|
|
701
|
+
),
|
|
702
|
+
"run_id": self.run_id,
|
|
703
|
+
"experiment_id": self.experiment_id,
|
|
704
|
+
"reproducibility": self.reproducibility,
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
return self.persistence.save(
|
|
708
|
+
pipeline=self.best_pipeline,
|
|
709
|
+
path=path,
|
|
710
|
+
metadata=metadata,
|
|
711
|
+
overwrite=overwrite,
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
def load(
|
|
715
|
+
self,
|
|
716
|
+
path: str,
|
|
717
|
+
) -> "AutoML":
|
|
718
|
+
"""
|
|
719
|
+
Load a previously saved ModelForge pipeline.
|
|
720
|
+
"""
|
|
721
|
+
|
|
722
|
+
self.best_pipeline = self.persistence.load(path)
|
|
723
|
+
|
|
724
|
+
metadata = self.persistence.load_metadata(path)
|
|
725
|
+
|
|
726
|
+
self.best_model = metadata.get(
|
|
727
|
+
"best_model"
|
|
728
|
+
)
|
|
729
|
+
|
|
730
|
+
self.target = metadata.get(
|
|
731
|
+
"target"
|
|
732
|
+
)
|
|
733
|
+
|
|
734
|
+
self.task_type = metadata.get(
|
|
735
|
+
"task_type"
|
|
736
|
+
)
|
|
737
|
+
|
|
738
|
+
self.objective = metadata.get(
|
|
739
|
+
"objective",
|
|
740
|
+
self.objective,
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
self.test_size = metadata.get(
|
|
744
|
+
"test_size",
|
|
745
|
+
self.test_size,
|
|
746
|
+
)
|
|
747
|
+
|
|
748
|
+
self.cv = metadata.get(
|
|
749
|
+
"cv",
|
|
750
|
+
self.cv,
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
self.random_state = metadata.get(
|
|
754
|
+
"random_state",
|
|
755
|
+
self.random_state,
|
|
756
|
+
)
|
|
757
|
+
|
|
758
|
+
self.variance_threshold = metadata.get(
|
|
759
|
+
"variance_threshold",
|
|
760
|
+
self.variance_threshold,
|
|
761
|
+
)
|
|
762
|
+
|
|
763
|
+
self.correlation_threshold = metadata.get(
|
|
764
|
+
"correlation_threshold",
|
|
765
|
+
self.correlation_threshold,
|
|
766
|
+
)
|
|
767
|
+
|
|
768
|
+
self.enable_optimization = metadata.get(
|
|
769
|
+
"enable_optimization",
|
|
770
|
+
self.enable_optimization,
|
|
771
|
+
)
|
|
772
|
+
|
|
773
|
+
self.optimization_models = metadata.get(
|
|
774
|
+
"optimization_models",
|
|
775
|
+
self.optimization_models,
|
|
776
|
+
)
|
|
777
|
+
|
|
778
|
+
self.optimization_max_trials = metadata.get(
|
|
779
|
+
"optimization_max_trials",
|
|
780
|
+
self.optimization_max_trials,
|
|
781
|
+
)
|
|
782
|
+
|
|
783
|
+
self.run_id = metadata.get(
|
|
784
|
+
"run_id"
|
|
785
|
+
)
|
|
786
|
+
|
|
787
|
+
self.experiment_id = metadata.get(
|
|
788
|
+
"experiment_id"
|
|
789
|
+
)
|
|
790
|
+
|
|
791
|
+
self.reproducibility = metadata.get(
|
|
792
|
+
"reproducibility"
|
|
793
|
+
)
|
|
794
|
+
|
|
795
|
+
self.reproducibility_integration = (
|
|
796
|
+
ReproducibilityIntegration(
|
|
797
|
+
random_state=self.random_state
|
|
798
|
+
)
|
|
799
|
+
)
|
|
800
|
+
|
|
801
|
+
self.is_fitted = True
|
|
802
|
+
|
|
803
|
+
return self
|
|
804
|
+
|
|
805
|
+
def explain(
|
|
806
|
+
self,
|
|
807
|
+
top_n: int = 10,
|
|
808
|
+
) -> pd.DataFrame:
|
|
809
|
+
"""
|
|
810
|
+
Return the most important features.
|
|
811
|
+
"""
|
|
812
|
+
|
|
813
|
+
self._require_fitted()
|
|
814
|
+
|
|
815
|
+
importance = (
|
|
816
|
+
self.explainability.feature_importance(
|
|
817
|
+
self.best_pipeline
|
|
818
|
+
)
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
return self.explainability.top_features(
|
|
822
|
+
importance,
|
|
823
|
+
n=top_n,
|
|
824
|
+
)
|
|
825
|
+
|
|
826
|
+
def summary(self) -> dict[str, Any]:
|
|
827
|
+
"""
|
|
828
|
+
Return a compact AutoML summary.
|
|
829
|
+
"""
|
|
830
|
+
|
|
831
|
+
self._require_fitted()
|
|
832
|
+
|
|
833
|
+
return {
|
|
834
|
+
"run_id": self.run_id,
|
|
835
|
+
"experiment_id": self.experiment_id,
|
|
836
|
+
"target": self.target,
|
|
837
|
+
"task_type": self.task_type,
|
|
838
|
+
"best_model": self.best_model,
|
|
839
|
+
"objective": self.objective,
|
|
840
|
+
"test_size": self.test_size,
|
|
841
|
+
"cv": self.cv,
|
|
842
|
+
"random_state": self.random_state,
|
|
843
|
+
"variance_threshold": (
|
|
844
|
+
self.variance_threshold
|
|
845
|
+
),
|
|
846
|
+
"correlation_threshold": (
|
|
847
|
+
self.correlation_threshold
|
|
848
|
+
),
|
|
849
|
+
"optimization_enabled": (
|
|
850
|
+
self.enable_optimization
|
|
851
|
+
),
|
|
852
|
+
"optimization_models": (
|
|
853
|
+
self.optimization_models
|
|
854
|
+
),
|
|
855
|
+
"optimization_max_trials": (
|
|
856
|
+
self.optimization_max_trials
|
|
857
|
+
),
|
|
858
|
+
}
|
|
859
|
+
|
|
860
|
+
def list_experiments(
|
|
861
|
+
self,
|
|
862
|
+
) -> list[dict[str, Any]]:
|
|
863
|
+
"""
|
|
864
|
+
Return all locally tracked experiments.
|
|
865
|
+
"""
|
|
866
|
+
|
|
867
|
+
return self.experiment_tracker.list_experiments()
|
|
868
|
+
|
|
869
|
+
def get_experiment(
|
|
870
|
+
self,
|
|
871
|
+
experiment_id: str,
|
|
872
|
+
) -> dict[str, Any]:
|
|
873
|
+
"""
|
|
874
|
+
Retrieve one tracked experiment.
|
|
875
|
+
"""
|
|
876
|
+
|
|
877
|
+
return self.experiment_tracker.get(
|
|
878
|
+
experiment_id
|
|
879
|
+
)
|
|
880
|
+
|
|
881
|
+
def _optimize_top_models(
|
|
882
|
+
self,
|
|
883
|
+
data: pd.DataFrame,
|
|
884
|
+
target: str,
|
|
885
|
+
task_type: str,
|
|
886
|
+
ranking: pd.DataFrame,
|
|
887
|
+
pipelines: dict[str, Any],
|
|
888
|
+
) -> dict[str, dict[str, Any]]:
|
|
889
|
+
"""
|
|
890
|
+
Optimize the top-ranked candidate models.
|
|
891
|
+
"""
|
|
892
|
+
|
|
893
|
+
if ranking.empty:
|
|
894
|
+
return {}
|
|
895
|
+
|
|
896
|
+
if "model" not in ranking.columns:
|
|
897
|
+
raise RuntimeError(
|
|
898
|
+
"Ranking results must contain "
|
|
899
|
+
"a 'model' column."
|
|
900
|
+
)
|
|
901
|
+
|
|
902
|
+
successful = ranking
|
|
903
|
+
|
|
904
|
+
if "status" in ranking.columns:
|
|
905
|
+
successful = ranking[
|
|
906
|
+
ranking["status"] == "success"
|
|
907
|
+
]
|
|
908
|
+
|
|
909
|
+
top_models = successful.head(
|
|
910
|
+
self.optimization_models
|
|
911
|
+
)
|
|
912
|
+
|
|
913
|
+
results: dict[str, dict[str, Any]] = {}
|
|
914
|
+
|
|
915
|
+
for model_name in top_models[
|
|
916
|
+
"model"
|
|
917
|
+
].tolist():
|
|
918
|
+
pipeline = pipelines.get(
|
|
919
|
+
model_name
|
|
920
|
+
)
|
|
921
|
+
|
|
922
|
+
if pipeline is None:
|
|
923
|
+
continue
|
|
924
|
+
|
|
925
|
+
parameter_space = (
|
|
926
|
+
self.registry.get_hyperparameter_space(
|
|
927
|
+
model_name
|
|
928
|
+
)
|
|
929
|
+
)
|
|
930
|
+
|
|
931
|
+
if not parameter_space:
|
|
932
|
+
continue
|
|
933
|
+
|
|
934
|
+
parameter_space = (
|
|
935
|
+
self._pipeline_parameter_space(
|
|
936
|
+
parameter_space
|
|
937
|
+
)
|
|
938
|
+
)
|
|
939
|
+
|
|
940
|
+
try:
|
|
941
|
+
results[model_name] = (
|
|
942
|
+
self.optimizer.optimize(
|
|
943
|
+
data=data,
|
|
944
|
+
target=target,
|
|
945
|
+
pipeline=pipeline,
|
|
946
|
+
parameter_space=parameter_space,
|
|
947
|
+
task_type=task_type,
|
|
948
|
+
)
|
|
949
|
+
)
|
|
950
|
+
except Exception as exc:
|
|
951
|
+
results[model_name] = {
|
|
952
|
+
"best_pipeline": None,
|
|
953
|
+
"best_params": {},
|
|
954
|
+
"best_score": None,
|
|
955
|
+
"trials": [],
|
|
956
|
+
"successful_trials": 0,
|
|
957
|
+
"failed_trials": 0,
|
|
958
|
+
"total_time_seconds": 0.0,
|
|
959
|
+
"error": str(exc),
|
|
960
|
+
}
|
|
961
|
+
|
|
962
|
+
return results
|
|
963
|
+
|
|
964
|
+
@staticmethod
|
|
965
|
+
def _pipeline_parameter_space(
|
|
966
|
+
parameter_space: dict[str, Any],
|
|
967
|
+
) -> dict[str, Any]:
|
|
968
|
+
"""
|
|
969
|
+
Convert model parameters into sklearn pipeline
|
|
970
|
+
parameter names.
|
|
971
|
+
"""
|
|
972
|
+
|
|
973
|
+
return {
|
|
974
|
+
(
|
|
975
|
+
parameter_name
|
|
976
|
+
if "__" in parameter_name
|
|
977
|
+
else f"model__{parameter_name}"
|
|
978
|
+
): values
|
|
979
|
+
for parameter_name, values
|
|
980
|
+
in parameter_space.items()
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
def _fit_final_pipeline(
|
|
984
|
+
self,
|
|
985
|
+
pipeline,
|
|
986
|
+
data: pd.DataFrame,
|
|
987
|
+
target: str,
|
|
988
|
+
):
|
|
989
|
+
"""
|
|
990
|
+
Fit the winning pipeline on the complete dataset.
|
|
991
|
+
"""
|
|
992
|
+
|
|
993
|
+
if not isinstance(
|
|
994
|
+
data,
|
|
995
|
+
pd.DataFrame,
|
|
996
|
+
):
|
|
997
|
+
raise TypeError(
|
|
998
|
+
"data must be a pandas DataFrame."
|
|
999
|
+
)
|
|
1000
|
+
|
|
1001
|
+
if target not in data.columns:
|
|
1002
|
+
raise ValueError(
|
|
1003
|
+
f"Target column '{target}' "
|
|
1004
|
+
"does not exist."
|
|
1005
|
+
)
|
|
1006
|
+
|
|
1007
|
+
X = data.drop(
|
|
1008
|
+
columns=[target]
|
|
1009
|
+
)
|
|
1010
|
+
|
|
1011
|
+
y = data[target]
|
|
1012
|
+
|
|
1013
|
+
try:
|
|
1014
|
+
pipeline.fit(
|
|
1015
|
+
X,
|
|
1016
|
+
y,
|
|
1017
|
+
)
|
|
1018
|
+
except Exception as exc:
|
|
1019
|
+
raise RuntimeError(
|
|
1020
|
+
"Failed to fit the final selected "
|
|
1021
|
+
f"pipeline: {exc}"
|
|
1022
|
+
) from exc
|
|
1023
|
+
|
|
1024
|
+
return pipeline
|
|
1025
|
+
|
|
1026
|
+
def _build_pipelines(
|
|
1027
|
+
self,
|
|
1028
|
+
dataframe: pd.DataFrame,
|
|
1029
|
+
target: str,
|
|
1030
|
+
task_type: str,
|
|
1031
|
+
model_names: list[str],
|
|
1032
|
+
excluded_columns: list[str] | None,
|
|
1033
|
+
) -> dict:
|
|
1034
|
+
"""
|
|
1035
|
+
Generate candidate pipelines.
|
|
1036
|
+
"""
|
|
1037
|
+
|
|
1038
|
+
pipelines = {}
|
|
1039
|
+
|
|
1040
|
+
for model_name in model_names:
|
|
1041
|
+
pipelines[model_name] = (
|
|
1042
|
+
self.pipeline_generator.build(
|
|
1043
|
+
data=dataframe,
|
|
1044
|
+
target=target,
|
|
1045
|
+
model_name=model_name,
|
|
1046
|
+
task_type=task_type,
|
|
1047
|
+
excluded_columns=excluded_columns,
|
|
1048
|
+
variance_threshold=(
|
|
1049
|
+
self.variance_threshold
|
|
1050
|
+
),
|
|
1051
|
+
correlation_threshold=(
|
|
1052
|
+
self.correlation_threshold
|
|
1053
|
+
),
|
|
1054
|
+
)
|
|
1055
|
+
)
|
|
1056
|
+
|
|
1057
|
+
return pipelines
|
|
1058
|
+
|
|
1059
|
+
def _select_models(
|
|
1060
|
+
self,
|
|
1061
|
+
model_names: list[str] | None,
|
|
1062
|
+
task_type: str,
|
|
1063
|
+
) -> list[str]:
|
|
1064
|
+
"""
|
|
1065
|
+
Resolve and validate model names.
|
|
1066
|
+
"""
|
|
1067
|
+
|
|
1068
|
+
available = self.registry.list_models(
|
|
1069
|
+
task_type=task_type
|
|
1070
|
+
)
|
|
1071
|
+
|
|
1072
|
+
if model_names is None:
|
|
1073
|
+
return available
|
|
1074
|
+
|
|
1075
|
+
invalid = [
|
|
1076
|
+
name
|
|
1077
|
+
for name in model_names
|
|
1078
|
+
if name not in available
|
|
1079
|
+
]
|
|
1080
|
+
|
|
1081
|
+
if invalid:
|
|
1082
|
+
raise ValueError(
|
|
1083
|
+
"Unknown or incompatible models: "
|
|
1084
|
+
+ ", ".join(invalid)
|
|
1085
|
+
)
|
|
1086
|
+
|
|
1087
|
+
if not model_names:
|
|
1088
|
+
raise ValueError(
|
|
1089
|
+
"At least one model must be selected."
|
|
1090
|
+
)
|
|
1091
|
+
|
|
1092
|
+
return model_names
|
|
1093
|
+
|
|
1094
|
+
def _combine_results(
|
|
1095
|
+
self,
|
|
1096
|
+
screening_results: pd.DataFrame,
|
|
1097
|
+
cv_results: pd.DataFrame,
|
|
1098
|
+
) -> pd.DataFrame:
|
|
1099
|
+
"""
|
|
1100
|
+
Combine holdout and cross-validation results.
|
|
1101
|
+
"""
|
|
1102
|
+
|
|
1103
|
+
screening = pd.DataFrame(
|
|
1104
|
+
screening_results
|
|
1105
|
+
)
|
|
1106
|
+
|
|
1107
|
+
cross_validation = pd.DataFrame(
|
|
1108
|
+
cv_results
|
|
1109
|
+
)
|
|
1110
|
+
|
|
1111
|
+
if screening.empty:
|
|
1112
|
+
raise RuntimeError(
|
|
1113
|
+
"No model screening results available."
|
|
1114
|
+
)
|
|
1115
|
+
|
|
1116
|
+
if cross_validation.empty:
|
|
1117
|
+
return screening
|
|
1118
|
+
|
|
1119
|
+
if "model" not in screening.columns:
|
|
1120
|
+
raise RuntimeError(
|
|
1121
|
+
"Screening results must contain "
|
|
1122
|
+
"a 'model' column."
|
|
1123
|
+
)
|
|
1124
|
+
|
|
1125
|
+
if "model" not in cross_validation.columns:
|
|
1126
|
+
raise RuntimeError(
|
|
1127
|
+
"Cross-validation results must "
|
|
1128
|
+
"contain a 'model' column."
|
|
1129
|
+
)
|
|
1130
|
+
|
|
1131
|
+
return screening.merge(
|
|
1132
|
+
cross_validation,
|
|
1133
|
+
on="model",
|
|
1134
|
+
how="left",
|
|
1135
|
+
suffixes=(
|
|
1136
|
+
"",
|
|
1137
|
+
"_cv",
|
|
1138
|
+
),
|
|
1139
|
+
)
|
|
1140
|
+
|
|
1141
|
+
def _select_best_model(
|
|
1142
|
+
self,
|
|
1143
|
+
ranking: pd.DataFrame,
|
|
1144
|
+
) -> str:
|
|
1145
|
+
"""
|
|
1146
|
+
Select the top successful model from ranking.
|
|
1147
|
+
"""
|
|
1148
|
+
|
|
1149
|
+
if ranking.empty:
|
|
1150
|
+
raise RuntimeError(
|
|
1151
|
+
"Ranking produced no models."
|
|
1152
|
+
)
|
|
1153
|
+
|
|
1154
|
+
if "status" in ranking.columns:
|
|
1155
|
+
successful = ranking[
|
|
1156
|
+
ranking["status"] == "success"
|
|
1157
|
+
]
|
|
1158
|
+
else:
|
|
1159
|
+
successful = ranking
|
|
1160
|
+
|
|
1161
|
+
if successful.empty:
|
|
1162
|
+
raise RuntimeError(
|
|
1163
|
+
"No successful model was found."
|
|
1164
|
+
)
|
|
1165
|
+
|
|
1166
|
+
return str(
|
|
1167
|
+
successful.iloc[0]["model"]
|
|
1168
|
+
)
|
|
1169
|
+
|
|
1170
|
+
def _load_data(
|
|
1171
|
+
self,
|
|
1172
|
+
data: Any,
|
|
1173
|
+
) -> pd.DataFrame:
|
|
1174
|
+
"""
|
|
1175
|
+
Load a dataset through DatasetLoader.
|
|
1176
|
+
"""
|
|
1177
|
+
|
|
1178
|
+
if isinstance(
|
|
1179
|
+
data,
|
|
1180
|
+
pd.DataFrame,
|
|
1181
|
+
):
|
|
1182
|
+
return data.copy()
|
|
1183
|
+
|
|
1184
|
+
if isinstance(
|
|
1185
|
+
data,
|
|
1186
|
+
(str, Path),
|
|
1187
|
+
):
|
|
1188
|
+
return self.loader.load(
|
|
1189
|
+
str(data)
|
|
1190
|
+
)
|
|
1191
|
+
|
|
1192
|
+
raise TypeError(
|
|
1193
|
+
"data must be a pandas DataFrame "
|
|
1194
|
+
"or a supported dataset path."
|
|
1195
|
+
)
|
|
1196
|
+
|
|
1197
|
+
def _configuration(
|
|
1198
|
+
self,
|
|
1199
|
+
target: str | None = None,
|
|
1200
|
+
task_type: str | None = None,
|
|
1201
|
+
models: list[str] | None = None,
|
|
1202
|
+
excluded_columns: list[str] | None = None,
|
|
1203
|
+
) -> dict[str, Any]:
|
|
1204
|
+
"""
|
|
1205
|
+
Return the effective configuration used by
|
|
1206
|
+
the current AutoML run.
|
|
1207
|
+
|
|
1208
|
+
Explicit runtime values are included so the
|
|
1209
|
+
reproducibility snapshot and experiment tracker
|
|
1210
|
+
persist exactly the same configuration.
|
|
1211
|
+
"""
|
|
1212
|
+
|
|
1213
|
+
effective_target = (
|
|
1214
|
+
target
|
|
1215
|
+
if target is not None
|
|
1216
|
+
else self._configured_target
|
|
1217
|
+
)
|
|
1218
|
+
|
|
1219
|
+
effective_task_type = (
|
|
1220
|
+
task_type
|
|
1221
|
+
if task_type is not None
|
|
1222
|
+
else self._configured_task_type
|
|
1223
|
+
)
|
|
1224
|
+
|
|
1225
|
+
effective_models = (
|
|
1226
|
+
models
|
|
1227
|
+
if models is not None
|
|
1228
|
+
else self._configured_models
|
|
1229
|
+
)
|
|
1230
|
+
|
|
1231
|
+
effective_excluded_columns = (
|
|
1232
|
+
excluded_columns
|
|
1233
|
+
if excluded_columns is not None
|
|
1234
|
+
else self._configured_excluded_columns
|
|
1235
|
+
)
|
|
1236
|
+
|
|
1237
|
+
return {
|
|
1238
|
+
"target": effective_target,
|
|
1239
|
+
"task_type": effective_task_type,
|
|
1240
|
+
"models": effective_models,
|
|
1241
|
+
"excluded_columns": (
|
|
1242
|
+
effective_excluded_columns
|
|
1243
|
+
),
|
|
1244
|
+
"test_size": self.test_size,
|
|
1245
|
+
"cv": self.cv,
|
|
1246
|
+
"random_state": self.random_state,
|
|
1247
|
+
"objective": self.objective,
|
|
1248
|
+
"variance_threshold": (
|
|
1249
|
+
self.variance_threshold
|
|
1250
|
+
),
|
|
1251
|
+
"correlation_threshold": (
|
|
1252
|
+
self.correlation_threshold
|
|
1253
|
+
),
|
|
1254
|
+
"enable_optimization": (
|
|
1255
|
+
self.enable_optimization
|
|
1256
|
+
),
|
|
1257
|
+
"optimization_models": (
|
|
1258
|
+
self.optimization_models
|
|
1259
|
+
),
|
|
1260
|
+
"optimization_max_trials": (
|
|
1261
|
+
self.optimization_max_trials
|
|
1262
|
+
),
|
|
1263
|
+
"experiment_directory": (
|
|
1264
|
+
str(self._configured_experiment_directory)
|
|
1265
|
+
),
|
|
1266
|
+
}
|
|
1267
|
+
|
|
1268
|
+
def _config_value(
|
|
1269
|
+
self,
|
|
1270
|
+
key: str,
|
|
1271
|
+
default: Any = None,
|
|
1272
|
+
) -> Any:
|
|
1273
|
+
"""
|
|
1274
|
+
Read a configuration value, including nested keys.
|
|
1275
|
+
"""
|
|
1276
|
+
|
|
1277
|
+
if not self._config_provided:
|
|
1278
|
+
return default
|
|
1279
|
+
|
|
1280
|
+
values = self.config.to_dict()
|
|
1281
|
+
current: Any = values
|
|
1282
|
+
|
|
1283
|
+
for part in key.split("."):
|
|
1284
|
+
if not isinstance(
|
|
1285
|
+
current,
|
|
1286
|
+
dict,
|
|
1287
|
+
):
|
|
1288
|
+
return default
|
|
1289
|
+
|
|
1290
|
+
if part not in current:
|
|
1291
|
+
return default
|
|
1292
|
+
|
|
1293
|
+
current = current[part]
|
|
1294
|
+
|
|
1295
|
+
return current
|
|
1296
|
+
|
|
1297
|
+
def _require_fitted(self) -> None:
|
|
1298
|
+
if not self.is_fitted:
|
|
1299
|
+
raise RuntimeError(
|
|
1300
|
+
"AutoML has not been fitted yet."
|
|
1301
|
+
)
|
|
1302
|
+
|
|
1303
|
+
if self.best_pipeline is None:
|
|
1304
|
+
raise RuntimeError(
|
|
1305
|
+
"AutoML is marked as fitted but "
|
|
1306
|
+
"no fitted pipeline is available."
|
|
1307
|
+
)
|
|
1308
|
+
|
|
1309
|
+
def _validate_configuration(self) -> None:
|
|
1310
|
+
"""
|
|
1311
|
+
Validate AutoML constructor configuration.
|
|
1312
|
+
"""
|
|
1313
|
+
|
|
1314
|
+
if not isinstance(
|
|
1315
|
+
self.test_size,
|
|
1316
|
+
(int, float),
|
|
1317
|
+
):
|
|
1318
|
+
raise TypeError(
|
|
1319
|
+
"test_size must be numeric."
|
|
1320
|
+
)
|
|
1321
|
+
|
|
1322
|
+
if isinstance(
|
|
1323
|
+
self.test_size,
|
|
1324
|
+
bool,
|
|
1325
|
+
):
|
|
1326
|
+
raise TypeError(
|
|
1327
|
+
"test_size must be numeric."
|
|
1328
|
+
)
|
|
1329
|
+
|
|
1330
|
+
if not 0 < self.test_size < 1:
|
|
1331
|
+
raise ValueError(
|
|
1332
|
+
"test_size must be between 0 and 1."
|
|
1333
|
+
)
|
|
1334
|
+
|
|
1335
|
+
if not isinstance(
|
|
1336
|
+
self.cv,
|
|
1337
|
+
int,
|
|
1338
|
+
):
|
|
1339
|
+
raise TypeError(
|
|
1340
|
+
"cv must be an integer."
|
|
1341
|
+
)
|
|
1342
|
+
|
|
1343
|
+
if isinstance(
|
|
1344
|
+
self.cv,
|
|
1345
|
+
bool,
|
|
1346
|
+
):
|
|
1347
|
+
raise TypeError(
|
|
1348
|
+
"cv must be an integer."
|
|
1349
|
+
)
|
|
1350
|
+
|
|
1351
|
+
if self.cv < 2:
|
|
1352
|
+
raise ValueError(
|
|
1353
|
+
"cv must be at least 2."
|
|
1354
|
+
)
|
|
1355
|
+
|
|
1356
|
+
if not isinstance(
|
|
1357
|
+
self.random_state,
|
|
1358
|
+
int,
|
|
1359
|
+
):
|
|
1360
|
+
raise TypeError(
|
|
1361
|
+
"random_state must be an integer."
|
|
1362
|
+
)
|
|
1363
|
+
|
|
1364
|
+
if isinstance(
|
|
1365
|
+
self.random_state,
|
|
1366
|
+
bool,
|
|
1367
|
+
):
|
|
1368
|
+
raise TypeError(
|
|
1369
|
+
"random_state must be an integer."
|
|
1370
|
+
)
|
|
1371
|
+
|
|
1372
|
+
if self.objective not in {
|
|
1373
|
+
"balanced",
|
|
1374
|
+
"performance",
|
|
1375
|
+
"error",
|
|
1376
|
+
"speed",
|
|
1377
|
+
}:
|
|
1378
|
+
raise ValueError(
|
|
1379
|
+
"Invalid ranking objective."
|
|
1380
|
+
)
|
|
1381
|
+
|
|
1382
|
+
if self.variance_threshold is not None:
|
|
1383
|
+
if isinstance(
|
|
1384
|
+
self.variance_threshold,
|
|
1385
|
+
bool,
|
|
1386
|
+
):
|
|
1387
|
+
raise TypeError(
|
|
1388
|
+
"variance_threshold must be numeric."
|
|
1389
|
+
)
|
|
1390
|
+
|
|
1391
|
+
if not isinstance(
|
|
1392
|
+
self.variance_threshold,
|
|
1393
|
+
(int, float),
|
|
1394
|
+
):
|
|
1395
|
+
raise TypeError(
|
|
1396
|
+
"variance_threshold must be numeric."
|
|
1397
|
+
)
|
|
1398
|
+
|
|
1399
|
+
if self.variance_threshold < 0:
|
|
1400
|
+
raise ValueError(
|
|
1401
|
+
"variance_threshold cannot "
|
|
1402
|
+
"be negative."
|
|
1403
|
+
)
|
|
1404
|
+
|
|
1405
|
+
if self.correlation_threshold is not None:
|
|
1406
|
+
if isinstance(
|
|
1407
|
+
self.correlation_threshold,
|
|
1408
|
+
bool,
|
|
1409
|
+
):
|
|
1410
|
+
raise TypeError(
|
|
1411
|
+
"correlation_threshold must "
|
|
1412
|
+
"be numeric."
|
|
1413
|
+
)
|
|
1414
|
+
|
|
1415
|
+
if not isinstance(
|
|
1416
|
+
self.correlation_threshold,
|
|
1417
|
+
(int, float),
|
|
1418
|
+
):
|
|
1419
|
+
raise TypeError(
|
|
1420
|
+
"correlation_threshold must "
|
|
1421
|
+
"be numeric."
|
|
1422
|
+
)
|
|
1423
|
+
|
|
1424
|
+
if not (
|
|
1425
|
+
0
|
|
1426
|
+
< self.correlation_threshold
|
|
1427
|
+
<= 1
|
|
1428
|
+
):
|
|
1429
|
+
raise ValueError(
|
|
1430
|
+
"correlation_threshold must "
|
|
1431
|
+
"be between 0 and 1."
|
|
1432
|
+
)
|
|
1433
|
+
|
|
1434
|
+
if not isinstance(
|
|
1435
|
+
self.enable_optimization,
|
|
1436
|
+
bool,
|
|
1437
|
+
):
|
|
1438
|
+
raise TypeError(
|
|
1439
|
+
"enable_optimization must be a boolean."
|
|
1440
|
+
)
|
|
1441
|
+
|
|
1442
|
+
if not isinstance(
|
|
1443
|
+
self.optimization_models,
|
|
1444
|
+
int,
|
|
1445
|
+
) or isinstance(
|
|
1446
|
+
self.optimization_models,
|
|
1447
|
+
bool,
|
|
1448
|
+
):
|
|
1449
|
+
raise TypeError(
|
|
1450
|
+
"optimization_models must be an integer."
|
|
1451
|
+
)
|
|
1452
|
+
|
|
1453
|
+
if self.optimization_models < 1:
|
|
1454
|
+
raise ValueError(
|
|
1455
|
+
"optimization_models must be at least 1."
|
|
1456
|
+
)
|
|
1457
|
+
|
|
1458
|
+
if not isinstance(
|
|
1459
|
+
self.optimization_max_trials,
|
|
1460
|
+
int,
|
|
1461
|
+
) or isinstance(
|
|
1462
|
+
self.optimization_max_trials,
|
|
1463
|
+
bool,
|
|
1464
|
+
):
|
|
1465
|
+
raise TypeError(
|
|
1466
|
+
"optimization_max_trials must be an integer."
|
|
1467
|
+
)
|
|
1468
|
+
|
|
1469
|
+
if self.optimization_max_trials < 1:
|
|
1470
|
+
raise ValueError(
|
|
1471
|
+
"optimization_max_trials must be at least 1."
|
|
1472
|
+
)
|