autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,593 @@
1
+ from __future__ import annotations
2
+
3
+ import time
4
+ from typing import Any
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+
9
+ from sklearn.base import clone
10
+ from sklearn.metrics import (
11
+ accuracy_score,
12
+ f1_score,
13
+ mean_absolute_error,
14
+ mean_squared_error,
15
+ r2_score,
16
+ )
17
+ from sklearn.model_selection import (
18
+ KFold,
19
+ ParameterGrid,
20
+ StratifiedKFold,
21
+ )
22
+ from sklearn.pipeline import Pipeline
23
+
24
+
25
+ class HyperparameterOptimizationEngine:
26
+ """
27
+ Search model hyperparameters using cross-validation.
28
+
29
+ The engine operates on complete sklearn pipelines so that
30
+ preprocessing remains part of every optimization trial.
31
+ """
32
+
33
+ def __init__(
34
+ self,
35
+ cv: int = 5,
36
+ random_state: int = 42,
37
+ scoring: str | None = None,
38
+ max_trials: int | None = None,
39
+ ):
40
+ self.cv = cv
41
+ self.random_state = random_state
42
+ self.scoring = scoring
43
+ self.max_trials = max_trials
44
+
45
+ self._validate_configuration()
46
+
47
+ def optimize(
48
+ self,
49
+ data: pd.DataFrame,
50
+ target: str,
51
+ pipeline: Pipeline,
52
+ parameter_space: dict[str, Any],
53
+ task_type: str,
54
+ ) -> dict[str, Any]:
55
+ """
56
+ Optimize a pipeline's hyperparameters.
57
+
58
+ Returns a dictionary containing:
59
+ - best_pipeline
60
+ - best_params
61
+ - best_score
62
+ - trials
63
+ - successful_trials
64
+ - failed_trials
65
+ - total_time_seconds
66
+ """
67
+
68
+ self._validate_inputs(
69
+ data=data,
70
+ target=target,
71
+ pipeline=pipeline,
72
+ parameter_space=parameter_space,
73
+ task_type=task_type,
74
+ )
75
+
76
+ X = data.drop(
77
+ columns=[target]
78
+ )
79
+
80
+ y = data[target]
81
+
82
+ parameter_combinations = list(
83
+ ParameterGrid(parameter_space)
84
+ )
85
+
86
+ if self.max_trials is not None:
87
+ parameter_combinations = (
88
+ parameter_combinations[
89
+ : self.max_trials
90
+ ]
91
+ )
92
+
93
+ if not parameter_combinations:
94
+ fitted_pipeline = clone(
95
+ pipeline
96
+ )
97
+
98
+ start = time.perf_counter()
99
+
100
+ fitted_pipeline.fit(
101
+ X,
102
+ y,
103
+ )
104
+
105
+ elapsed = (
106
+ time.perf_counter()
107
+ - start
108
+ )
109
+
110
+ return {
111
+ "best_pipeline": fitted_pipeline,
112
+ "best_params": {},
113
+ "best_score": None,
114
+ "trials": [],
115
+ "successful_trials": 1,
116
+ "failed_trials": 0,
117
+ "total_time_seconds": float(
118
+ elapsed
119
+ ),
120
+ }
121
+
122
+ splitter = self._create_splitter(
123
+ y=y,
124
+ task_type=task_type,
125
+ )
126
+
127
+ trials: list[dict[str, Any]] = []
128
+
129
+ total_start = time.perf_counter()
130
+
131
+ for trial_number, params in enumerate(
132
+ parameter_combinations,
133
+ start=1,
134
+ ):
135
+ trial = self._evaluate_trial(
136
+ trial_number=trial_number,
137
+ pipeline=pipeline,
138
+ params=params,
139
+ X=X,
140
+ y=y,
141
+ splitter=splitter,
142
+ task_type=task_type,
143
+ )
144
+
145
+ trials.append(trial)
146
+
147
+ successful_trials = [
148
+ trial
149
+ for trial in trials
150
+ if trial["status"] == "success"
151
+ ]
152
+
153
+ failed_trials = [
154
+ trial
155
+ for trial in trials
156
+ if trial["status"] == "failed"
157
+ ]
158
+
159
+ if not successful_trials:
160
+ raise RuntimeError(
161
+ "All hyperparameter optimization "
162
+ "trials failed."
163
+ )
164
+
165
+ best_trial = self._select_best_trial(
166
+ successful_trials,
167
+ task_type=task_type,
168
+ )
169
+
170
+ best_pipeline = clone(
171
+ pipeline
172
+ )
173
+
174
+ best_pipeline.set_params(
175
+ **best_trial["params"]
176
+ )
177
+
178
+ total_time = (
179
+ time.perf_counter()
180
+ - total_start
181
+ )
182
+
183
+ return {
184
+ "best_pipeline": best_pipeline,
185
+ "best_params": best_trial["params"],
186
+ "best_score": best_trial["score"],
187
+ "trials": trials,
188
+ "successful_trials": len(
189
+ successful_trials
190
+ ),
191
+ "failed_trials": len(
192
+ failed_trials
193
+ ),
194
+ "total_time_seconds": float(
195
+ total_time
196
+ ),
197
+ }
198
+
199
+ def _evaluate_trial(
200
+ self,
201
+ trial_number: int,
202
+ pipeline: Pipeline,
203
+ params: dict[str, Any],
204
+ X: pd.DataFrame,
205
+ y: pd.Series,
206
+ splitter,
207
+ task_type: str,
208
+ ) -> dict[str, Any]:
209
+ """Evaluate one parameter combination."""
210
+
211
+ start = time.perf_counter()
212
+
213
+ fold_scores: list[float] = []
214
+
215
+ try:
216
+ for train_indices, validation_indices in splitter.split(
217
+ X,
218
+ y if task_type == "classification" else None,
219
+ ):
220
+ X_train = X.iloc[
221
+ train_indices
222
+ ]
223
+
224
+ X_validation = X.iloc[
225
+ validation_indices
226
+ ]
227
+
228
+ y_train = y.iloc[
229
+ train_indices
230
+ ]
231
+
232
+ y_validation = y.iloc[
233
+ validation_indices
234
+ ]
235
+
236
+ fold_pipeline = clone(
237
+ pipeline
238
+ )
239
+
240
+ fold_pipeline.set_params(
241
+ **params
242
+ )
243
+
244
+ fold_pipeline.fit(
245
+ X_train,
246
+ y_train,
247
+ )
248
+
249
+ predictions = (
250
+ fold_pipeline.predict(
251
+ X_validation
252
+ )
253
+ )
254
+
255
+ score = self._calculate_score(
256
+ y_true=y_validation,
257
+ predictions=predictions,
258
+ task_type=task_type,
259
+ )
260
+
261
+ fold_scores.append(
262
+ float(score)
263
+ )
264
+
265
+ mean_score = float(
266
+ np.mean(fold_scores)
267
+ )
268
+
269
+ elapsed = (
270
+ time.perf_counter()
271
+ - start
272
+ )
273
+
274
+ return {
275
+ "trial": trial_number,
276
+ "params": params.copy(),
277
+ "score": mean_score,
278
+ "fold_scores": fold_scores,
279
+ "status": "success",
280
+ "time_seconds": float(
281
+ elapsed
282
+ ),
283
+ "error": None,
284
+ }
285
+
286
+ except Exception as exc:
287
+ elapsed = (
288
+ time.perf_counter()
289
+ - start
290
+ )
291
+
292
+ return {
293
+ "trial": trial_number,
294
+ "params": params.copy(),
295
+ "score": None,
296
+ "fold_scores": fold_scores,
297
+ "status": "failed",
298
+ "time_seconds": float(
299
+ elapsed
300
+ ),
301
+ "error": str(exc),
302
+ }
303
+
304
+ def _calculate_score(
305
+ self,
306
+ y_true,
307
+ predictions,
308
+ task_type: str,
309
+ ) -> float:
310
+ """Calculate the optimization score."""
311
+
312
+ scoring = self.scoring
313
+
314
+ if task_type == "regression":
315
+ if scoring in {
316
+ None,
317
+ "r2",
318
+ }:
319
+ return float(
320
+ r2_score(
321
+ y_true,
322
+ predictions,
323
+ )
324
+ )
325
+
326
+ if scoring == "neg_mae":
327
+ return float(
328
+ -mean_absolute_error(
329
+ y_true,
330
+ predictions,
331
+ )
332
+ )
333
+
334
+ if scoring == "neg_mse":
335
+ return float(
336
+ -mean_squared_error(
337
+ y_true,
338
+ predictions,
339
+ )
340
+ )
341
+
342
+ if scoring == "neg_rmse":
343
+ mse = mean_squared_error(
344
+ y_true,
345
+ predictions,
346
+ )
347
+
348
+ return float(
349
+ -np.sqrt(mse)
350
+ )
351
+
352
+ raise ValueError(
353
+ "Unsupported regression scoring: "
354
+ f"{scoring}"
355
+ )
356
+
357
+ if scoring in {
358
+ None,
359
+ "f1",
360
+ "f1_weighted",
361
+ }:
362
+ return float(
363
+ f1_score(
364
+ y_true,
365
+ predictions,
366
+ average="weighted",
367
+ zero_division=0,
368
+ )
369
+ )
370
+
371
+ if scoring == "accuracy":
372
+ return float(
373
+ accuracy_score(
374
+ y_true,
375
+ predictions,
376
+ )
377
+ )
378
+
379
+ if scoring == "precision":
380
+ from sklearn.metrics import (
381
+ precision_score,
382
+ )
383
+
384
+ return float(
385
+ precision_score(
386
+ y_true,
387
+ predictions,
388
+ average="weighted",
389
+ zero_division=0,
390
+ )
391
+ )
392
+
393
+ if scoring == "recall":
394
+ from sklearn.metrics import (
395
+ recall_score,
396
+ )
397
+
398
+ return float(
399
+ recall_score(
400
+ y_true,
401
+ predictions,
402
+ average="weighted",
403
+ zero_division=0,
404
+ )
405
+ )
406
+
407
+ raise ValueError(
408
+ "Unsupported classification scoring: "
409
+ f"{scoring}"
410
+ )
411
+
412
+ @staticmethod
413
+ def _select_best_trial(
414
+ trials: list[dict[str, Any]],
415
+ task_type: str,
416
+ ) -> dict[str, Any]:
417
+ """Select the highest-scoring successful trial."""
418
+
419
+ if not trials:
420
+ raise ValueError(
421
+ "No successful trials available."
422
+ )
423
+
424
+ return max(
425
+ trials,
426
+ key=lambda trial: trial["score"],
427
+ )
428
+
429
+ def _create_splitter(
430
+ self,
431
+ y: pd.Series,
432
+ task_type: str,
433
+ ):
434
+ """Create the appropriate CV splitter."""
435
+
436
+ if len(y) < self.cv:
437
+ raise ValueError(
438
+ f"Dataset must contain at least "
439
+ f"{self.cv} samples."
440
+ )
441
+
442
+ if task_type == "classification":
443
+ class_counts = y.value_counts()
444
+
445
+ if (
446
+ len(class_counts) >= 2
447
+ and class_counts.min() >= self.cv
448
+ ):
449
+ return StratifiedKFold(
450
+ n_splits=self.cv,
451
+ shuffle=True,
452
+ random_state=self.random_state,
453
+ )
454
+
455
+ raise ValueError(
456
+ "Each classification class must "
457
+ f"contain at least {self.cv} "
458
+ "samples for optimization."
459
+ )
460
+
461
+ return KFold(
462
+ n_splits=self.cv,
463
+ shuffle=True,
464
+ random_state=self.random_state,
465
+ )
466
+
467
+ def _validate_configuration(self):
468
+ """Validate optimizer configuration."""
469
+
470
+ if not isinstance(
471
+ self.cv,
472
+ int,
473
+ ) or isinstance(
474
+ self.cv,
475
+ bool,
476
+ ):
477
+ raise TypeError(
478
+ "cv must be an integer."
479
+ )
480
+
481
+ if self.cv < 2:
482
+ raise ValueError(
483
+ "cv must be at least 2."
484
+ )
485
+
486
+ if not isinstance(
487
+ self.random_state,
488
+ int,
489
+ ) or isinstance(
490
+ self.random_state,
491
+ bool,
492
+ ):
493
+ raise TypeError(
494
+ "random_state must be an integer."
495
+ )
496
+
497
+ if self.scoring is not None:
498
+ if not isinstance(
499
+ self.scoring,
500
+ str,
501
+ ):
502
+ raise TypeError(
503
+ "scoring must be a string or None."
504
+ )
505
+
506
+ if self.max_trials is not None:
507
+ if not isinstance(
508
+ self.max_trials,
509
+ int,
510
+ ) or isinstance(
511
+ self.max_trials,
512
+ bool,
513
+ ):
514
+ raise TypeError(
515
+ "max_trials must be an integer "
516
+ "or None."
517
+ )
518
+
519
+ if self.max_trials < 1:
520
+ raise ValueError(
521
+ "max_trials must be at least 1."
522
+ )
523
+
524
+ @staticmethod
525
+ def _validate_inputs(
526
+ data: pd.DataFrame,
527
+ target: str,
528
+ pipeline: Pipeline,
529
+ parameter_space: dict[str, Any],
530
+ task_type: str,
531
+ ):
532
+ """Validate optimization inputs."""
533
+
534
+ if not isinstance(
535
+ data,
536
+ pd.DataFrame,
537
+ ):
538
+ raise TypeError(
539
+ "data must be a pandas DataFrame."
540
+ )
541
+
542
+ if data.empty:
543
+ raise ValueError(
544
+ "Cannot optimize on an empty dataset."
545
+ )
546
+
547
+ if not isinstance(
548
+ target,
549
+ str,
550
+ ):
551
+ raise TypeError(
552
+ "target must be a string."
553
+ )
554
+
555
+ if target not in data.columns:
556
+ raise ValueError(
557
+ f"Target column '{target}' "
558
+ "does not exist."
559
+ )
560
+
561
+ if not isinstance(
562
+ pipeline,
563
+ Pipeline,
564
+ ):
565
+ raise TypeError(
566
+ "pipeline must be a sklearn Pipeline."
567
+ )
568
+
569
+ if not isinstance(
570
+ parameter_space,
571
+ dict,
572
+ ):
573
+ raise TypeError(
574
+ "parameter_space must be a dictionary."
575
+ )
576
+
577
+ if task_type not in {
578
+ "regression",
579
+ "classification",
580
+ }:
581
+ raise ValueError(
582
+ "task_type must be 'regression' "
583
+ "or 'classification'."
584
+ )
585
+
586
+ for parameter_name in parameter_space:
587
+ if not isinstance(
588
+ parameter_name,
589
+ str,
590
+ ):
591
+ raise TypeError(
592
+ "Parameter names must be strings."
593
+ )