autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,456 @@
1
+ from pathlib import Path
2
+ from typing import Any
3
+
4
+ import joblib
5
+ import pandas as pd
6
+
7
+ from sklearn.pipeline import Pipeline
8
+
9
+ from modelforge.prediction_validator import (
10
+ PredictionSchemaError,
11
+ PredictionSchemaValidator,
12
+ )
13
+
14
+
15
+ class ModelPersistence:
16
+ """
17
+ Save, load, validate, and use trained ModelForge pipelines.
18
+
19
+ ModelForge persists the complete sklearn pipeline so that
20
+ preprocessing and model transformations remain identical
21
+ during future prediction.
22
+
23
+ Prediction schema validation is delegated to the dedicated
24
+ PredictionSchemaValidator component.
25
+ """
26
+
27
+ MODEL_FILENAME = "model.joblib"
28
+ METADATA_FILENAME = "metadata.joblib"
29
+
30
+ def save(
31
+ self,
32
+ pipeline: Pipeline,
33
+ path: str,
34
+ metadata: dict[str, Any] | None = None,
35
+ overwrite: bool = False,
36
+ ) -> str:
37
+ """
38
+ Save a trained pipeline to disk.
39
+
40
+ Parameters
41
+ ----------
42
+ pipeline:
43
+ Fitted sklearn Pipeline.
44
+
45
+ path:
46
+ Destination file path.
47
+
48
+ metadata:
49
+ Optional metadata associated with the model.
50
+
51
+ overwrite:
52
+ Whether an existing file may be replaced.
53
+
54
+ Returns
55
+ -------
56
+ str
57
+ Absolute path of the saved model.
58
+ """
59
+
60
+ self._validate_pipeline(
61
+ pipeline
62
+ )
63
+
64
+ destination = Path(path)
65
+
66
+ if (
67
+ destination.exists()
68
+ and not overwrite
69
+ ):
70
+ raise FileExistsError(
71
+ f"Model already exists: {destination}. "
72
+ "Set overwrite=True to replace it."
73
+ )
74
+
75
+ destination.parent.mkdir(
76
+ parents=True,
77
+ exist_ok=True,
78
+ )
79
+
80
+ joblib.dump(
81
+ pipeline,
82
+ destination,
83
+ )
84
+
85
+ if metadata is not None:
86
+ metadata_path = self._metadata_path(
87
+ destination
88
+ )
89
+
90
+ joblib.dump(
91
+ metadata,
92
+ metadata_path,
93
+ )
94
+
95
+ return str(
96
+ destination.resolve()
97
+ )
98
+
99
+ def load(
100
+ self,
101
+ path: str,
102
+ ) -> Pipeline:
103
+ """
104
+ Load a saved ModelForge pipeline.
105
+ """
106
+
107
+ model_path = Path(path)
108
+
109
+ if not model_path.exists():
110
+ raise FileNotFoundError(
111
+ f"Model not found: {model_path}"
112
+ )
113
+
114
+ if not model_path.is_file():
115
+ raise ValueError(
116
+ f"Model path is not a file: "
117
+ f"{model_path}"
118
+ )
119
+
120
+ pipeline = joblib.load(
121
+ model_path
122
+ )
123
+
124
+ self._validate_pipeline(
125
+ pipeline
126
+ )
127
+
128
+ return pipeline
129
+
130
+ def load_metadata(
131
+ self,
132
+ path: str,
133
+ ) -> dict[str, Any]:
134
+ """
135
+ Load metadata associated with a saved model.
136
+ """
137
+
138
+ model_path = Path(path)
139
+
140
+ if not model_path.exists():
141
+ raise FileNotFoundError(
142
+ f"Model not found: {model_path}"
143
+ )
144
+
145
+ metadata_path = self._metadata_path(
146
+ model_path
147
+ )
148
+
149
+ if not metadata_path.exists():
150
+ return {}
151
+
152
+ metadata = joblib.load(
153
+ metadata_path
154
+ )
155
+
156
+ if not isinstance(
157
+ metadata,
158
+ dict,
159
+ ):
160
+ raise ValueError(
161
+ "Saved metadata must be a dictionary."
162
+ )
163
+
164
+ return metadata
165
+
166
+ def predict(
167
+ self,
168
+ pipeline: Pipeline,
169
+ data: pd.DataFrame,
170
+ ) -> pd.Series:
171
+ """
172
+ Generate predictions using a fitted pipeline.
173
+
174
+ Prediction data is validated and aligned using
175
+ PredictionSchemaValidator before inference.
176
+ """
177
+
178
+ self._validate_pipeline(
179
+ pipeline
180
+ )
181
+
182
+ validated_data = (
183
+ self._validate_prediction_data(
184
+ pipeline,
185
+ data,
186
+ )
187
+ )
188
+
189
+ predictions = pipeline.predict(
190
+ validated_data
191
+ )
192
+
193
+ return pd.Series(
194
+ predictions,
195
+ index=validated_data.index,
196
+ name="prediction",
197
+ )
198
+
199
+ def predict_from_file(
200
+ self,
201
+ pipeline: Pipeline,
202
+ data_path: str,
203
+ ) -> pd.Series:
204
+ """
205
+ Load a CSV dataset and generate predictions.
206
+ """
207
+
208
+ path = Path(data_path)
209
+
210
+ if not path.exists():
211
+ raise FileNotFoundError(
212
+ f"Prediction dataset not found: {path}"
213
+ )
214
+
215
+ if not path.is_file():
216
+ raise ValueError(
217
+ f"Prediction dataset is not a file: "
218
+ f"{path}"
219
+ )
220
+
221
+ if path.suffix.lower() != ".csv":
222
+ raise ValueError(
223
+ "predict_from_file currently "
224
+ "supports CSV files only."
225
+ )
226
+
227
+ data = pd.read_csv(
228
+ path
229
+ )
230
+
231
+ return self.predict(
232
+ pipeline,
233
+ data,
234
+ )
235
+
236
+ @staticmethod
237
+ def predict_proba(
238
+ pipeline: Pipeline,
239
+ data: pd.DataFrame,
240
+ ) -> pd.DataFrame:
241
+ """
242
+ Generate class probabilities when supported.
243
+
244
+ Prediction data is validated and aligned using
245
+ PredictionSchemaValidator before inference.
246
+ """
247
+
248
+ if not isinstance(
249
+ pipeline,
250
+ Pipeline,
251
+ ):
252
+ raise TypeError(
253
+ "pipeline must be a sklearn Pipeline."
254
+ )
255
+
256
+ if not hasattr(
257
+ pipeline,
258
+ "predict_proba",
259
+ ):
260
+ raise ValueError(
261
+ "This pipeline does not support "
262
+ "probability predictions."
263
+ )
264
+
265
+ validated_data = (
266
+ ModelPersistence._validate_prediction_data(
267
+ pipeline,
268
+ data,
269
+ )
270
+ )
271
+
272
+ probabilities = (
273
+ pipeline.predict_proba(
274
+ validated_data
275
+ )
276
+ )
277
+
278
+ model = pipeline.named_steps.get(
279
+ "model"
280
+ )
281
+
282
+ if (
283
+ model is not None
284
+ and hasattr(
285
+ model,
286
+ "classes_",
287
+ )
288
+ ):
289
+ columns = [
290
+ f"probability_{label}"
291
+ for label in model.classes_
292
+ ]
293
+ else:
294
+ columns = [
295
+ f"probability_{index}"
296
+ for index in range(
297
+ probabilities.shape[1]
298
+ )
299
+ ]
300
+
301
+ return pd.DataFrame(
302
+ probabilities,
303
+ index=validated_data.index,
304
+ columns=columns,
305
+ )
306
+
307
+ @staticmethod
308
+ def predict_proba_from_file(
309
+ pipeline: Pipeline,
310
+ data_path: str,
311
+ ) -> pd.DataFrame:
312
+ """
313
+ Load a CSV dataset and generate class probabilities.
314
+ """
315
+
316
+ path = Path(data_path)
317
+
318
+ if not path.exists():
319
+ raise FileNotFoundError(
320
+ f"Prediction dataset not found: {path}"
321
+ )
322
+
323
+ if not path.is_file():
324
+ raise ValueError(
325
+ f"Prediction dataset is not a file: "
326
+ f"{path}"
327
+ )
328
+
329
+ if path.suffix.lower() != ".csv":
330
+ raise ValueError(
331
+ "predict_proba_from_file currently "
332
+ "supports CSV files only."
333
+ )
334
+
335
+ data = pd.read_csv(
336
+ path
337
+ )
338
+
339
+ return ModelPersistence.predict_proba(
340
+ pipeline,
341
+ data,
342
+ )
343
+
344
+ @staticmethod
345
+ def _validate_prediction_data(
346
+ pipeline: Pipeline,
347
+ data: pd.DataFrame,
348
+ ) -> pd.DataFrame:
349
+ """
350
+ Validate prediction data through the dedicated
351
+ PredictionSchemaValidator.
352
+
353
+ The fitted sklearn pipeline provides the expected
354
+ training feature schema.
355
+ """
356
+
357
+ if not isinstance(
358
+ data,
359
+ pd.DataFrame,
360
+ ):
361
+ raise TypeError(
362
+ "data must be a pandas DataFrame."
363
+ )
364
+
365
+ if data.empty:
366
+ raise ValueError(
367
+ "Cannot predict on an empty dataset."
368
+ )
369
+
370
+ expected_columns = (
371
+ ModelPersistence._get_expected_columns(
372
+ pipeline
373
+ )
374
+ )
375
+
376
+ if expected_columns is None:
377
+ return data
378
+
379
+ validator = (
380
+ PredictionSchemaValidator(
381
+ allow_extra_columns=False,
382
+ enforce_column_order=False,
383
+ enforce_dtypes=False,
384
+ )
385
+ )
386
+
387
+ try:
388
+ return validator.validate_and_align(
389
+ data=data,
390
+ expected_columns=expected_columns,
391
+ )
392
+
393
+ except PredictionSchemaError as exc:
394
+ raise ValueError(
395
+ str(exc)
396
+ ) from exc
397
+
398
+ @staticmethod
399
+ def _get_expected_columns(
400
+ pipeline: Pipeline,
401
+ ) -> list[str] | None:
402
+ """
403
+ Extract the feature schema recorded by sklearn.
404
+
405
+ sklearn pipelines fitted with a pandas DataFrame
406
+ expose feature names through feature_names_in_.
407
+ """
408
+
409
+ feature_names = getattr(
410
+ pipeline,
411
+ "feature_names_in_",
412
+ None,
413
+ )
414
+
415
+ if feature_names is None:
416
+ return None
417
+
418
+ return [
419
+ str(column)
420
+ for column in feature_names
421
+ ]
422
+
423
+ @staticmethod
424
+ def _metadata_path(
425
+ model_path: Path,
426
+ ) -> Path:
427
+ """
428
+ Generate the metadata path corresponding
429
+ to a model file.
430
+ """
431
+
432
+ return model_path.with_name(
433
+ f"{model_path.stem}_metadata.joblib"
434
+ )
435
+
436
+ @staticmethod
437
+ def _validate_pipeline(
438
+ pipeline: Pipeline,
439
+ ) -> None:
440
+ """
441
+ Validate that the supplied object is a
442
+ usable sklearn Pipeline.
443
+ """
444
+
445
+ if not isinstance(
446
+ pipeline,
447
+ Pipeline,
448
+ ):
449
+ raise TypeError(
450
+ "pipeline must be a sklearn Pipeline."
451
+ )
452
+
453
+ if "model" not in pipeline.named_steps:
454
+ raise ValueError(
455
+ "Pipeline must contain a 'model' step."
456
+ )
@@ -0,0 +1,278 @@
1
+ from typing import Any
2
+
3
+ import pandas as pd
4
+
5
+ from sklearn.pipeline import Pipeline
6
+
7
+ from modelforge.feature_selection import (
8
+ CorrelationFeatureSelector,
9
+ VarianceFeatureSelector,
10
+ )
11
+ from modelforge.model_registry import (
12
+ ModelRegistry,
13
+ )
14
+ from modelforge.preprocessing import (
15
+ PreprocessingEngine,
16
+ )
17
+
18
+
19
+ class PipelineGenerator:
20
+ """
21
+ Generate complete machine-learning pipelines
22
+ for ModelForge.
23
+
24
+ The generated pipeline connects:
25
+
26
+ preprocessing
27
+ ↓
28
+ feature selection
29
+ ↓
30
+ model
31
+ """
32
+
33
+ def __init__(
34
+ self,
35
+ model_registry: ModelRegistry | None = None,
36
+ preprocessing_engine: PreprocessingEngine | None = None,
37
+ ):
38
+ self.model_registry = (
39
+ model_registry
40
+ if model_registry is not None
41
+ else ModelRegistry()
42
+ )
43
+
44
+ self.preprocessing_engine = (
45
+ preprocessing_engine
46
+ if preprocessing_engine is not None
47
+ else PreprocessingEngine()
48
+ )
49
+
50
+ def build(
51
+ self,
52
+ data: pd.DataFrame,
53
+ target: str,
54
+ model_name: str,
55
+ task_type: str,
56
+ excluded_columns: list[str] | None = None,
57
+ variance_threshold: float | None = None,
58
+ correlation_threshold: float | None = None,
59
+ model_params: dict[str, Any] | None = None,
60
+ ) -> Pipeline:
61
+ """
62
+ Build a complete ModelForge pipeline.
63
+
64
+ Parameters
65
+ ----------
66
+ data : pd.DataFrame
67
+ Input dataset.
68
+
69
+ target : str
70
+ Target column.
71
+
72
+ model_name : str
73
+ Registered model identifier.
74
+
75
+ task_type : str
76
+ 'regression' or 'classification'.
77
+
78
+ excluded_columns : list[str] | None
79
+ Columns that should not be used as features.
80
+
81
+ variance_threshold : float | None
82
+ Optional variance filtering threshold.
83
+
84
+ correlation_threshold : float | None
85
+ Optional correlation filtering threshold.
86
+
87
+ model_params : dict | None
88
+ Parameters overriding model defaults.
89
+
90
+ Returns
91
+ -------
92
+ sklearn.pipeline.Pipeline
93
+ Complete ML pipeline.
94
+ """
95
+
96
+ self._validate_inputs(
97
+ data=data,
98
+ target=target,
99
+ task_type=task_type,
100
+ )
101
+
102
+ model_spec = self.model_registry.get(
103
+ model_name
104
+ )
105
+
106
+ if model_spec.task_type != task_type:
107
+ raise ValueError(
108
+ f"Model '{model_name}' is a "
109
+ f"{model_spec.task_type} model and "
110
+ f"cannot be used for {task_type}."
111
+ )
112
+
113
+ excluded = list(
114
+ excluded_columns or []
115
+ )
116
+
117
+ preprocessing = (
118
+ self.preprocessing_engine.build(
119
+ data=data,
120
+ target=target,
121
+ model_type=self._get_model_type(
122
+ model_spec
123
+ ),
124
+ excluded_columns=excluded,
125
+ )
126
+ )
127
+
128
+ # Feature-selection transformers in ModelForge
129
+ # explicitly require pandas DataFrames.
130
+ #
131
+ # ColumnTransformer normally produces a NumPy array.
132
+ # When feature selection is enabled, configure the
133
+ # preprocessing transformer to return pandas output.
134
+ if (
135
+ variance_threshold is not None
136
+ or correlation_threshold is not None
137
+ ):
138
+ if not hasattr(
139
+ preprocessing,
140
+ "set_output",
141
+ ):
142
+ raise RuntimeError(
143
+ "The preprocessing transformer does not "
144
+ "support pandas output. A compatible "
145
+ "scikit-learn version is required."
146
+ )
147
+
148
+ preprocessing.set_output(
149
+ transform="pandas"
150
+ )
151
+
152
+ steps = []
153
+
154
+ steps.append(
155
+ (
156
+ "preprocessing",
157
+ preprocessing,
158
+ )
159
+ )
160
+
161
+ if variance_threshold is not None:
162
+ steps.append(
163
+ (
164
+ "variance_selection",
165
+ VarianceFeatureSelector(
166
+ threshold=variance_threshold
167
+ ),
168
+ )
169
+ )
170
+
171
+ if correlation_threshold is not None:
172
+ steps.append(
173
+ (
174
+ "correlation_selection",
175
+ CorrelationFeatureSelector(
176
+ threshold=correlation_threshold
177
+ ),
178
+ )
179
+ )
180
+
181
+ model = self.model_registry.create(
182
+ model_name,
183
+ **(model_params or {}),
184
+ )
185
+
186
+ steps.append(
187
+ (
188
+ "model",
189
+ model,
190
+ )
191
+ )
192
+
193
+ return Pipeline(
194
+ steps=steps
195
+ )
196
+
197
+ @staticmethod
198
+ def _validate_inputs(
199
+ data: pd.DataFrame,
200
+ target: str,
201
+ task_type: str,
202
+ ) -> None:
203
+ """
204
+ Validate pipeline construction inputs.
205
+ """
206
+
207
+ if not isinstance(
208
+ data,
209
+ pd.DataFrame,
210
+ ):
211
+ raise TypeError(
212
+ "data must be a pandas DataFrame."
213
+ )
214
+
215
+ if data.empty:
216
+ raise ValueError(
217
+ "Cannot build a pipeline from "
218
+ "an empty dataset."
219
+ )
220
+
221
+ if target not in data.columns:
222
+ raise ValueError(
223
+ f"Target column '{target}' "
224
+ "does not exist."
225
+ )
226
+
227
+ if task_type not in {
228
+ "regression",
229
+ "classification",
230
+ }:
231
+ raise ValueError(
232
+ "task_type must be 'regression' "
233
+ "or 'classification'."
234
+ )
235
+
236
+ @staticmethod
237
+ def _get_model_type(
238
+ model_spec,
239
+ ) -> str:
240
+ """
241
+ Convert model metadata into a preprocessing
242
+ model category.
243
+ """
244
+
245
+ if model_spec.requires_scaling:
246
+ if model_spec.category == "svm":
247
+ return "svm"
248
+
249
+ if model_spec.category == "distance_based":
250
+ return "knn"
251
+
252
+ if model_spec.category == "linear":
253
+ if (
254
+ "classification"
255
+ == model_spec.task_type
256
+ ):
257
+ return "logistic_regression"
258
+
259
+ if model_spec.name.lower().startswith(
260
+ "ridge"
261
+ ):
262
+ return "ridge"
263
+
264
+ if model_spec.name.lower().startswith(
265
+ "lasso"
266
+ ):
267
+ return "lasso"
268
+
269
+ if model_spec.name.lower().startswith(
270
+ "elastic"
271
+ ):
272
+ return "elasticnet"
273
+
274
+ return "linear_regression"
275
+
276
+ return "mlp"
277
+
278
+ return "tree"