autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,456 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
import joblib
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from sklearn.pipeline import Pipeline
|
|
8
|
+
|
|
9
|
+
from modelforge.prediction_validator import (
|
|
10
|
+
PredictionSchemaError,
|
|
11
|
+
PredictionSchemaValidator,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ModelPersistence:
|
|
16
|
+
"""
|
|
17
|
+
Save, load, validate, and use trained ModelForge pipelines.
|
|
18
|
+
|
|
19
|
+
ModelForge persists the complete sklearn pipeline so that
|
|
20
|
+
preprocessing and model transformations remain identical
|
|
21
|
+
during future prediction.
|
|
22
|
+
|
|
23
|
+
Prediction schema validation is delegated to the dedicated
|
|
24
|
+
PredictionSchemaValidator component.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
MODEL_FILENAME = "model.joblib"
|
|
28
|
+
METADATA_FILENAME = "metadata.joblib"
|
|
29
|
+
|
|
30
|
+
def save(
|
|
31
|
+
self,
|
|
32
|
+
pipeline: Pipeline,
|
|
33
|
+
path: str,
|
|
34
|
+
metadata: dict[str, Any] | None = None,
|
|
35
|
+
overwrite: bool = False,
|
|
36
|
+
) -> str:
|
|
37
|
+
"""
|
|
38
|
+
Save a trained pipeline to disk.
|
|
39
|
+
|
|
40
|
+
Parameters
|
|
41
|
+
----------
|
|
42
|
+
pipeline:
|
|
43
|
+
Fitted sklearn Pipeline.
|
|
44
|
+
|
|
45
|
+
path:
|
|
46
|
+
Destination file path.
|
|
47
|
+
|
|
48
|
+
metadata:
|
|
49
|
+
Optional metadata associated with the model.
|
|
50
|
+
|
|
51
|
+
overwrite:
|
|
52
|
+
Whether an existing file may be replaced.
|
|
53
|
+
|
|
54
|
+
Returns
|
|
55
|
+
-------
|
|
56
|
+
str
|
|
57
|
+
Absolute path of the saved model.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
self._validate_pipeline(
|
|
61
|
+
pipeline
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
destination = Path(path)
|
|
65
|
+
|
|
66
|
+
if (
|
|
67
|
+
destination.exists()
|
|
68
|
+
and not overwrite
|
|
69
|
+
):
|
|
70
|
+
raise FileExistsError(
|
|
71
|
+
f"Model already exists: {destination}. "
|
|
72
|
+
"Set overwrite=True to replace it."
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
destination.parent.mkdir(
|
|
76
|
+
parents=True,
|
|
77
|
+
exist_ok=True,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
joblib.dump(
|
|
81
|
+
pipeline,
|
|
82
|
+
destination,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
if metadata is not None:
|
|
86
|
+
metadata_path = self._metadata_path(
|
|
87
|
+
destination
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
joblib.dump(
|
|
91
|
+
metadata,
|
|
92
|
+
metadata_path,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
return str(
|
|
96
|
+
destination.resolve()
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
def load(
|
|
100
|
+
self,
|
|
101
|
+
path: str,
|
|
102
|
+
) -> Pipeline:
|
|
103
|
+
"""
|
|
104
|
+
Load a saved ModelForge pipeline.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
model_path = Path(path)
|
|
108
|
+
|
|
109
|
+
if not model_path.exists():
|
|
110
|
+
raise FileNotFoundError(
|
|
111
|
+
f"Model not found: {model_path}"
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
if not model_path.is_file():
|
|
115
|
+
raise ValueError(
|
|
116
|
+
f"Model path is not a file: "
|
|
117
|
+
f"{model_path}"
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
pipeline = joblib.load(
|
|
121
|
+
model_path
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
self._validate_pipeline(
|
|
125
|
+
pipeline
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
return pipeline
|
|
129
|
+
|
|
130
|
+
def load_metadata(
|
|
131
|
+
self,
|
|
132
|
+
path: str,
|
|
133
|
+
) -> dict[str, Any]:
|
|
134
|
+
"""
|
|
135
|
+
Load metadata associated with a saved model.
|
|
136
|
+
"""
|
|
137
|
+
|
|
138
|
+
model_path = Path(path)
|
|
139
|
+
|
|
140
|
+
if not model_path.exists():
|
|
141
|
+
raise FileNotFoundError(
|
|
142
|
+
f"Model not found: {model_path}"
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
metadata_path = self._metadata_path(
|
|
146
|
+
model_path
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
if not metadata_path.exists():
|
|
150
|
+
return {}
|
|
151
|
+
|
|
152
|
+
metadata = joblib.load(
|
|
153
|
+
metadata_path
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
if not isinstance(
|
|
157
|
+
metadata,
|
|
158
|
+
dict,
|
|
159
|
+
):
|
|
160
|
+
raise ValueError(
|
|
161
|
+
"Saved metadata must be a dictionary."
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
return metadata
|
|
165
|
+
|
|
166
|
+
def predict(
|
|
167
|
+
self,
|
|
168
|
+
pipeline: Pipeline,
|
|
169
|
+
data: pd.DataFrame,
|
|
170
|
+
) -> pd.Series:
|
|
171
|
+
"""
|
|
172
|
+
Generate predictions using a fitted pipeline.
|
|
173
|
+
|
|
174
|
+
Prediction data is validated and aligned using
|
|
175
|
+
PredictionSchemaValidator before inference.
|
|
176
|
+
"""
|
|
177
|
+
|
|
178
|
+
self._validate_pipeline(
|
|
179
|
+
pipeline
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
validated_data = (
|
|
183
|
+
self._validate_prediction_data(
|
|
184
|
+
pipeline,
|
|
185
|
+
data,
|
|
186
|
+
)
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
predictions = pipeline.predict(
|
|
190
|
+
validated_data
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
return pd.Series(
|
|
194
|
+
predictions,
|
|
195
|
+
index=validated_data.index,
|
|
196
|
+
name="prediction",
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
def predict_from_file(
|
|
200
|
+
self,
|
|
201
|
+
pipeline: Pipeline,
|
|
202
|
+
data_path: str,
|
|
203
|
+
) -> pd.Series:
|
|
204
|
+
"""
|
|
205
|
+
Load a CSV dataset and generate predictions.
|
|
206
|
+
"""
|
|
207
|
+
|
|
208
|
+
path = Path(data_path)
|
|
209
|
+
|
|
210
|
+
if not path.exists():
|
|
211
|
+
raise FileNotFoundError(
|
|
212
|
+
f"Prediction dataset not found: {path}"
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
if not path.is_file():
|
|
216
|
+
raise ValueError(
|
|
217
|
+
f"Prediction dataset is not a file: "
|
|
218
|
+
f"{path}"
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
if path.suffix.lower() != ".csv":
|
|
222
|
+
raise ValueError(
|
|
223
|
+
"predict_from_file currently "
|
|
224
|
+
"supports CSV files only."
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
data = pd.read_csv(
|
|
228
|
+
path
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
return self.predict(
|
|
232
|
+
pipeline,
|
|
233
|
+
data,
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
@staticmethod
|
|
237
|
+
def predict_proba(
|
|
238
|
+
pipeline: Pipeline,
|
|
239
|
+
data: pd.DataFrame,
|
|
240
|
+
) -> pd.DataFrame:
|
|
241
|
+
"""
|
|
242
|
+
Generate class probabilities when supported.
|
|
243
|
+
|
|
244
|
+
Prediction data is validated and aligned using
|
|
245
|
+
PredictionSchemaValidator before inference.
|
|
246
|
+
"""
|
|
247
|
+
|
|
248
|
+
if not isinstance(
|
|
249
|
+
pipeline,
|
|
250
|
+
Pipeline,
|
|
251
|
+
):
|
|
252
|
+
raise TypeError(
|
|
253
|
+
"pipeline must be a sklearn Pipeline."
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
if not hasattr(
|
|
257
|
+
pipeline,
|
|
258
|
+
"predict_proba",
|
|
259
|
+
):
|
|
260
|
+
raise ValueError(
|
|
261
|
+
"This pipeline does not support "
|
|
262
|
+
"probability predictions."
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
validated_data = (
|
|
266
|
+
ModelPersistence._validate_prediction_data(
|
|
267
|
+
pipeline,
|
|
268
|
+
data,
|
|
269
|
+
)
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
probabilities = (
|
|
273
|
+
pipeline.predict_proba(
|
|
274
|
+
validated_data
|
|
275
|
+
)
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
model = pipeline.named_steps.get(
|
|
279
|
+
"model"
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
if (
|
|
283
|
+
model is not None
|
|
284
|
+
and hasattr(
|
|
285
|
+
model,
|
|
286
|
+
"classes_",
|
|
287
|
+
)
|
|
288
|
+
):
|
|
289
|
+
columns = [
|
|
290
|
+
f"probability_{label}"
|
|
291
|
+
for label in model.classes_
|
|
292
|
+
]
|
|
293
|
+
else:
|
|
294
|
+
columns = [
|
|
295
|
+
f"probability_{index}"
|
|
296
|
+
for index in range(
|
|
297
|
+
probabilities.shape[1]
|
|
298
|
+
)
|
|
299
|
+
]
|
|
300
|
+
|
|
301
|
+
return pd.DataFrame(
|
|
302
|
+
probabilities,
|
|
303
|
+
index=validated_data.index,
|
|
304
|
+
columns=columns,
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
@staticmethod
|
|
308
|
+
def predict_proba_from_file(
|
|
309
|
+
pipeline: Pipeline,
|
|
310
|
+
data_path: str,
|
|
311
|
+
) -> pd.DataFrame:
|
|
312
|
+
"""
|
|
313
|
+
Load a CSV dataset and generate class probabilities.
|
|
314
|
+
"""
|
|
315
|
+
|
|
316
|
+
path = Path(data_path)
|
|
317
|
+
|
|
318
|
+
if not path.exists():
|
|
319
|
+
raise FileNotFoundError(
|
|
320
|
+
f"Prediction dataset not found: {path}"
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
if not path.is_file():
|
|
324
|
+
raise ValueError(
|
|
325
|
+
f"Prediction dataset is not a file: "
|
|
326
|
+
f"{path}"
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
if path.suffix.lower() != ".csv":
|
|
330
|
+
raise ValueError(
|
|
331
|
+
"predict_proba_from_file currently "
|
|
332
|
+
"supports CSV files only."
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
data = pd.read_csv(
|
|
336
|
+
path
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
return ModelPersistence.predict_proba(
|
|
340
|
+
pipeline,
|
|
341
|
+
data,
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
@staticmethod
|
|
345
|
+
def _validate_prediction_data(
|
|
346
|
+
pipeline: Pipeline,
|
|
347
|
+
data: pd.DataFrame,
|
|
348
|
+
) -> pd.DataFrame:
|
|
349
|
+
"""
|
|
350
|
+
Validate prediction data through the dedicated
|
|
351
|
+
PredictionSchemaValidator.
|
|
352
|
+
|
|
353
|
+
The fitted sklearn pipeline provides the expected
|
|
354
|
+
training feature schema.
|
|
355
|
+
"""
|
|
356
|
+
|
|
357
|
+
if not isinstance(
|
|
358
|
+
data,
|
|
359
|
+
pd.DataFrame,
|
|
360
|
+
):
|
|
361
|
+
raise TypeError(
|
|
362
|
+
"data must be a pandas DataFrame."
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
if data.empty:
|
|
366
|
+
raise ValueError(
|
|
367
|
+
"Cannot predict on an empty dataset."
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
expected_columns = (
|
|
371
|
+
ModelPersistence._get_expected_columns(
|
|
372
|
+
pipeline
|
|
373
|
+
)
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
if expected_columns is None:
|
|
377
|
+
return data
|
|
378
|
+
|
|
379
|
+
validator = (
|
|
380
|
+
PredictionSchemaValidator(
|
|
381
|
+
allow_extra_columns=False,
|
|
382
|
+
enforce_column_order=False,
|
|
383
|
+
enforce_dtypes=False,
|
|
384
|
+
)
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
try:
|
|
388
|
+
return validator.validate_and_align(
|
|
389
|
+
data=data,
|
|
390
|
+
expected_columns=expected_columns,
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
except PredictionSchemaError as exc:
|
|
394
|
+
raise ValueError(
|
|
395
|
+
str(exc)
|
|
396
|
+
) from exc
|
|
397
|
+
|
|
398
|
+
@staticmethod
|
|
399
|
+
def _get_expected_columns(
|
|
400
|
+
pipeline: Pipeline,
|
|
401
|
+
) -> list[str] | None:
|
|
402
|
+
"""
|
|
403
|
+
Extract the feature schema recorded by sklearn.
|
|
404
|
+
|
|
405
|
+
sklearn pipelines fitted with a pandas DataFrame
|
|
406
|
+
expose feature names through feature_names_in_.
|
|
407
|
+
"""
|
|
408
|
+
|
|
409
|
+
feature_names = getattr(
|
|
410
|
+
pipeline,
|
|
411
|
+
"feature_names_in_",
|
|
412
|
+
None,
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
if feature_names is None:
|
|
416
|
+
return None
|
|
417
|
+
|
|
418
|
+
return [
|
|
419
|
+
str(column)
|
|
420
|
+
for column in feature_names
|
|
421
|
+
]
|
|
422
|
+
|
|
423
|
+
@staticmethod
|
|
424
|
+
def _metadata_path(
|
|
425
|
+
model_path: Path,
|
|
426
|
+
) -> Path:
|
|
427
|
+
"""
|
|
428
|
+
Generate the metadata path corresponding
|
|
429
|
+
to a model file.
|
|
430
|
+
"""
|
|
431
|
+
|
|
432
|
+
return model_path.with_name(
|
|
433
|
+
f"{model_path.stem}_metadata.joblib"
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
@staticmethod
|
|
437
|
+
def _validate_pipeline(
|
|
438
|
+
pipeline: Pipeline,
|
|
439
|
+
) -> None:
|
|
440
|
+
"""
|
|
441
|
+
Validate that the supplied object is a
|
|
442
|
+
usable sklearn Pipeline.
|
|
443
|
+
"""
|
|
444
|
+
|
|
445
|
+
if not isinstance(
|
|
446
|
+
pipeline,
|
|
447
|
+
Pipeline,
|
|
448
|
+
):
|
|
449
|
+
raise TypeError(
|
|
450
|
+
"pipeline must be a sklearn Pipeline."
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
if "model" not in pipeline.named_steps:
|
|
454
|
+
raise ValueError(
|
|
455
|
+
"Pipeline must contain a 'model' step."
|
|
456
|
+
)
|
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from sklearn.pipeline import Pipeline
|
|
6
|
+
|
|
7
|
+
from modelforge.feature_selection import (
|
|
8
|
+
CorrelationFeatureSelector,
|
|
9
|
+
VarianceFeatureSelector,
|
|
10
|
+
)
|
|
11
|
+
from modelforge.model_registry import (
|
|
12
|
+
ModelRegistry,
|
|
13
|
+
)
|
|
14
|
+
from modelforge.preprocessing import (
|
|
15
|
+
PreprocessingEngine,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class PipelineGenerator:
|
|
20
|
+
"""
|
|
21
|
+
Generate complete machine-learning pipelines
|
|
22
|
+
for ModelForge.
|
|
23
|
+
|
|
24
|
+
The generated pipeline connects:
|
|
25
|
+
|
|
26
|
+
preprocessing
|
|
27
|
+
↓
|
|
28
|
+
feature selection
|
|
29
|
+
↓
|
|
30
|
+
model
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __init__(
|
|
34
|
+
self,
|
|
35
|
+
model_registry: ModelRegistry | None = None,
|
|
36
|
+
preprocessing_engine: PreprocessingEngine | None = None,
|
|
37
|
+
):
|
|
38
|
+
self.model_registry = (
|
|
39
|
+
model_registry
|
|
40
|
+
if model_registry is not None
|
|
41
|
+
else ModelRegistry()
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
self.preprocessing_engine = (
|
|
45
|
+
preprocessing_engine
|
|
46
|
+
if preprocessing_engine is not None
|
|
47
|
+
else PreprocessingEngine()
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
def build(
|
|
51
|
+
self,
|
|
52
|
+
data: pd.DataFrame,
|
|
53
|
+
target: str,
|
|
54
|
+
model_name: str,
|
|
55
|
+
task_type: str,
|
|
56
|
+
excluded_columns: list[str] | None = None,
|
|
57
|
+
variance_threshold: float | None = None,
|
|
58
|
+
correlation_threshold: float | None = None,
|
|
59
|
+
model_params: dict[str, Any] | None = None,
|
|
60
|
+
) -> Pipeline:
|
|
61
|
+
"""
|
|
62
|
+
Build a complete ModelForge pipeline.
|
|
63
|
+
|
|
64
|
+
Parameters
|
|
65
|
+
----------
|
|
66
|
+
data : pd.DataFrame
|
|
67
|
+
Input dataset.
|
|
68
|
+
|
|
69
|
+
target : str
|
|
70
|
+
Target column.
|
|
71
|
+
|
|
72
|
+
model_name : str
|
|
73
|
+
Registered model identifier.
|
|
74
|
+
|
|
75
|
+
task_type : str
|
|
76
|
+
'regression' or 'classification'.
|
|
77
|
+
|
|
78
|
+
excluded_columns : list[str] | None
|
|
79
|
+
Columns that should not be used as features.
|
|
80
|
+
|
|
81
|
+
variance_threshold : float | None
|
|
82
|
+
Optional variance filtering threshold.
|
|
83
|
+
|
|
84
|
+
correlation_threshold : float | None
|
|
85
|
+
Optional correlation filtering threshold.
|
|
86
|
+
|
|
87
|
+
model_params : dict | None
|
|
88
|
+
Parameters overriding model defaults.
|
|
89
|
+
|
|
90
|
+
Returns
|
|
91
|
+
-------
|
|
92
|
+
sklearn.pipeline.Pipeline
|
|
93
|
+
Complete ML pipeline.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
self._validate_inputs(
|
|
97
|
+
data=data,
|
|
98
|
+
target=target,
|
|
99
|
+
task_type=task_type,
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
model_spec = self.model_registry.get(
|
|
103
|
+
model_name
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
if model_spec.task_type != task_type:
|
|
107
|
+
raise ValueError(
|
|
108
|
+
f"Model '{model_name}' is a "
|
|
109
|
+
f"{model_spec.task_type} model and "
|
|
110
|
+
f"cannot be used for {task_type}."
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
excluded = list(
|
|
114
|
+
excluded_columns or []
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
preprocessing = (
|
|
118
|
+
self.preprocessing_engine.build(
|
|
119
|
+
data=data,
|
|
120
|
+
target=target,
|
|
121
|
+
model_type=self._get_model_type(
|
|
122
|
+
model_spec
|
|
123
|
+
),
|
|
124
|
+
excluded_columns=excluded,
|
|
125
|
+
)
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# Feature-selection transformers in ModelForge
|
|
129
|
+
# explicitly require pandas DataFrames.
|
|
130
|
+
#
|
|
131
|
+
# ColumnTransformer normally produces a NumPy array.
|
|
132
|
+
# When feature selection is enabled, configure the
|
|
133
|
+
# preprocessing transformer to return pandas output.
|
|
134
|
+
if (
|
|
135
|
+
variance_threshold is not None
|
|
136
|
+
or correlation_threshold is not None
|
|
137
|
+
):
|
|
138
|
+
if not hasattr(
|
|
139
|
+
preprocessing,
|
|
140
|
+
"set_output",
|
|
141
|
+
):
|
|
142
|
+
raise RuntimeError(
|
|
143
|
+
"The preprocessing transformer does not "
|
|
144
|
+
"support pandas output. A compatible "
|
|
145
|
+
"scikit-learn version is required."
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
preprocessing.set_output(
|
|
149
|
+
transform="pandas"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
steps = []
|
|
153
|
+
|
|
154
|
+
steps.append(
|
|
155
|
+
(
|
|
156
|
+
"preprocessing",
|
|
157
|
+
preprocessing,
|
|
158
|
+
)
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
if variance_threshold is not None:
|
|
162
|
+
steps.append(
|
|
163
|
+
(
|
|
164
|
+
"variance_selection",
|
|
165
|
+
VarianceFeatureSelector(
|
|
166
|
+
threshold=variance_threshold
|
|
167
|
+
),
|
|
168
|
+
)
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
if correlation_threshold is not None:
|
|
172
|
+
steps.append(
|
|
173
|
+
(
|
|
174
|
+
"correlation_selection",
|
|
175
|
+
CorrelationFeatureSelector(
|
|
176
|
+
threshold=correlation_threshold
|
|
177
|
+
),
|
|
178
|
+
)
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
model = self.model_registry.create(
|
|
182
|
+
model_name,
|
|
183
|
+
**(model_params or {}),
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
steps.append(
|
|
187
|
+
(
|
|
188
|
+
"model",
|
|
189
|
+
model,
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
return Pipeline(
|
|
194
|
+
steps=steps
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
@staticmethod
|
|
198
|
+
def _validate_inputs(
|
|
199
|
+
data: pd.DataFrame,
|
|
200
|
+
target: str,
|
|
201
|
+
task_type: str,
|
|
202
|
+
) -> None:
|
|
203
|
+
"""
|
|
204
|
+
Validate pipeline construction inputs.
|
|
205
|
+
"""
|
|
206
|
+
|
|
207
|
+
if not isinstance(
|
|
208
|
+
data,
|
|
209
|
+
pd.DataFrame,
|
|
210
|
+
):
|
|
211
|
+
raise TypeError(
|
|
212
|
+
"data must be a pandas DataFrame."
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
if data.empty:
|
|
216
|
+
raise ValueError(
|
|
217
|
+
"Cannot build a pipeline from "
|
|
218
|
+
"an empty dataset."
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
if target not in data.columns:
|
|
222
|
+
raise ValueError(
|
|
223
|
+
f"Target column '{target}' "
|
|
224
|
+
"does not exist."
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
if task_type not in {
|
|
228
|
+
"regression",
|
|
229
|
+
"classification",
|
|
230
|
+
}:
|
|
231
|
+
raise ValueError(
|
|
232
|
+
"task_type must be 'regression' "
|
|
233
|
+
"or 'classification'."
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
@staticmethod
|
|
237
|
+
def _get_model_type(
|
|
238
|
+
model_spec,
|
|
239
|
+
) -> str:
|
|
240
|
+
"""
|
|
241
|
+
Convert model metadata into a preprocessing
|
|
242
|
+
model category.
|
|
243
|
+
"""
|
|
244
|
+
|
|
245
|
+
if model_spec.requires_scaling:
|
|
246
|
+
if model_spec.category == "svm":
|
|
247
|
+
return "svm"
|
|
248
|
+
|
|
249
|
+
if model_spec.category == "distance_based":
|
|
250
|
+
return "knn"
|
|
251
|
+
|
|
252
|
+
if model_spec.category == "linear":
|
|
253
|
+
if (
|
|
254
|
+
"classification"
|
|
255
|
+
== model_spec.task_type
|
|
256
|
+
):
|
|
257
|
+
return "logistic_regression"
|
|
258
|
+
|
|
259
|
+
if model_spec.name.lower().startswith(
|
|
260
|
+
"ridge"
|
|
261
|
+
):
|
|
262
|
+
return "ridge"
|
|
263
|
+
|
|
264
|
+
if model_spec.name.lower().startswith(
|
|
265
|
+
"lasso"
|
|
266
|
+
):
|
|
267
|
+
return "lasso"
|
|
268
|
+
|
|
269
|
+
if model_spec.name.lower().startswith(
|
|
270
|
+
"elastic"
|
|
271
|
+
):
|
|
272
|
+
return "elasticnet"
|
|
273
|
+
|
|
274
|
+
return "linear_regression"
|
|
275
|
+
|
|
276
|
+
return "mlp"
|
|
277
|
+
|
|
278
|
+
return "tree"
|