autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,316 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+
7
+
8
+ class PredictionSchemaError(ValueError):
9
+ """Raised when prediction data does not match the expected schema."""
10
+
11
+
12
+ class PredictionSchemaValidator:
13
+ """
14
+ Validate prediction data against a training feature schema.
15
+
16
+ The validator checks:
17
+ - required columns
18
+ - unexpected columns
19
+ - column order
20
+ - empty datasets
21
+ - basic data types
22
+ """
23
+
24
+ def __init__(
25
+ self,
26
+ allow_extra_columns: bool = True,
27
+ enforce_column_order: bool = False,
28
+ enforce_dtypes: bool = False,
29
+ ):
30
+ if not isinstance(
31
+ allow_extra_columns,
32
+ bool,
33
+ ):
34
+ raise TypeError(
35
+ "allow_extra_columns must be a boolean."
36
+ )
37
+
38
+ if not isinstance(
39
+ enforce_column_order,
40
+ bool,
41
+ ):
42
+ raise TypeError(
43
+ "enforce_column_order must be a boolean."
44
+ )
45
+
46
+ if not isinstance(
47
+ enforce_dtypes,
48
+ bool,
49
+ ):
50
+ raise TypeError(
51
+ "enforce_dtypes must be a boolean."
52
+ )
53
+
54
+ self.allow_extra_columns = (
55
+ allow_extra_columns
56
+ )
57
+
58
+ self.enforce_column_order = (
59
+ enforce_column_order
60
+ )
61
+
62
+ self.enforce_dtypes = (
63
+ enforce_dtypes
64
+ )
65
+
66
+ def validate(
67
+ self,
68
+ data: pd.DataFrame,
69
+ expected_columns: list[str],
70
+ expected_dtypes: dict[str, str] | None = None,
71
+ ) -> dict[str, Any]:
72
+ """
73
+ Validate prediction data.
74
+
75
+ Returns a validation report.
76
+
77
+ Raises:
78
+ TypeError:
79
+ If inputs have invalid types.
80
+
81
+ PredictionSchemaError:
82
+ If the prediction schema is invalid.
83
+ """
84
+
85
+ if not isinstance(
86
+ data,
87
+ pd.DataFrame,
88
+ ):
89
+ raise TypeError(
90
+ "data must be a pandas DataFrame."
91
+ )
92
+
93
+ if not isinstance(
94
+ expected_columns,
95
+ list,
96
+ ):
97
+ raise TypeError(
98
+ "expected_columns must be a list."
99
+ )
100
+
101
+ if not all(
102
+ isinstance(column, str)
103
+ for column in expected_columns
104
+ ):
105
+ raise TypeError(
106
+ "expected_columns must contain strings."
107
+ )
108
+
109
+ if expected_dtypes is not None:
110
+ if not isinstance(
111
+ expected_dtypes,
112
+ dict,
113
+ ):
114
+ raise TypeError(
115
+ "expected_dtypes must be a dictionary."
116
+ )
117
+
118
+ if data.empty:
119
+ raise PredictionSchemaError(
120
+ "Prediction data is empty."
121
+ )
122
+
123
+ actual_columns = list(
124
+ data.columns
125
+ )
126
+
127
+ missing_columns = [
128
+ column
129
+ for column in expected_columns
130
+ if column not in actual_columns
131
+ ]
132
+
133
+ unexpected_columns = [
134
+ column
135
+ for column in actual_columns
136
+ if column not in expected_columns
137
+ ]
138
+
139
+ order_matches = (
140
+ actual_columns
141
+ == expected_columns
142
+ )
143
+
144
+ dtype_mismatches = {}
145
+
146
+ if (
147
+ self.enforce_dtypes
148
+ and expected_dtypes is not None
149
+ ):
150
+ for column in expected_columns:
151
+ if column not in data.columns:
152
+ continue
153
+
154
+ expected_dtype = str(
155
+ expected_dtypes.get(
156
+ column,
157
+ "",
158
+ )
159
+ )
160
+
161
+ actual_dtype = str(
162
+ data[column].dtype
163
+ )
164
+
165
+ if (
166
+ expected_dtype
167
+ and actual_dtype
168
+ != expected_dtype
169
+ ):
170
+ dtype_mismatches[column] = {
171
+ "expected": expected_dtype,
172
+ "actual": actual_dtype,
173
+ }
174
+
175
+ errors = []
176
+
177
+ if missing_columns:
178
+ errors.append(
179
+ "Missing required columns: "
180
+ + ", ".join(
181
+ missing_columns
182
+ )
183
+ )
184
+
185
+ if (
186
+ unexpected_columns
187
+ and not self.allow_extra_columns
188
+ ):
189
+ errors.append(
190
+ "Unexpected columns: "
191
+ + ", ".join(
192
+ unexpected_columns
193
+ )
194
+ )
195
+
196
+ if (
197
+ self.enforce_column_order
198
+ and not order_matches
199
+ ):
200
+ errors.append(
201
+ "Column order does not match "
202
+ "the expected feature order."
203
+ )
204
+
205
+ if dtype_mismatches:
206
+ errors.append(
207
+ "Data type mismatch for columns: "
208
+ + ", ".join(
209
+ dtype_mismatches.keys()
210
+ )
211
+ )
212
+
213
+ valid = not errors
214
+
215
+ report = {
216
+ "valid": valid,
217
+ "expected_columns": expected_columns,
218
+ "actual_columns": actual_columns,
219
+ "missing_columns": missing_columns,
220
+ "unexpected_columns": unexpected_columns,
221
+ "order_matches": order_matches,
222
+ "dtype_mismatches": dtype_mismatches,
223
+ "row_count": len(data),
224
+ }
225
+
226
+ if not valid:
227
+ raise PredictionSchemaError(
228
+ "Prediction data failed schema "
229
+ "validation. "
230
+ + " ".join(errors)
231
+ )
232
+
233
+ return report
234
+
235
+ def validate_and_align(
236
+ self,
237
+ data: pd.DataFrame,
238
+ expected_columns: list[str],
239
+ expected_dtypes: dict[str, str] | None = None,
240
+ ) -> pd.DataFrame:
241
+ """
242
+ Validate prediction data and return it aligned
243
+ to the expected feature order.
244
+
245
+ Extra columns are removed when they are allowed.
246
+ """
247
+
248
+ self.validate(
249
+ data=data,
250
+ expected_columns=expected_columns,
251
+ expected_dtypes=expected_dtypes,
252
+ )
253
+
254
+ aligned = data.copy()
255
+
256
+ aligned = aligned[
257
+ expected_columns
258
+ ]
259
+
260
+ return aligned
261
+
262
+ @staticmethod
263
+ def infer_schema(
264
+ data: pd.DataFrame,
265
+ ) -> dict[str, Any]:
266
+ """
267
+ Infer a prediction schema from a DataFrame.
268
+ """
269
+
270
+ if not isinstance(
271
+ data,
272
+ pd.DataFrame,
273
+ ):
274
+ raise TypeError(
275
+ "data must be a pandas DataFrame."
276
+ )
277
+
278
+ if data.empty:
279
+ raise PredictionSchemaError(
280
+ "Cannot infer schema from "
281
+ "an empty DataFrame."
282
+ )
283
+
284
+ return {
285
+ "columns": list(
286
+ data.columns
287
+ ),
288
+ "dtypes": {
289
+ column: str(
290
+ data[column].dtype
291
+ )
292
+ for column in data.columns
293
+ },
294
+ "n_features": len(
295
+ data.columns
296
+ ),
297
+ }
298
+
299
+ @staticmethod
300
+ def format_error(
301
+ error: Exception,
302
+ ) -> str:
303
+ """
304
+ Return a clean user-facing error message.
305
+ """
306
+
307
+ if isinstance(
308
+ error,
309
+ PredictionSchemaError,
310
+ ):
311
+ return str(error)
312
+
313
+ return (
314
+ "Prediction validation failed: "
315
+ f"{error}"
316
+ )
@@ -0,0 +1,179 @@
1
+ from typing import Iterable
2
+
3
+ import pandas as pd
4
+ from sklearn.compose import ColumnTransformer
5
+ from sklearn.impute import SimpleImputer
6
+ from sklearn.pipeline import Pipeline
7
+ from sklearn.preprocessing import (
8
+ OneHotEncoder,
9
+ StandardScaler,
10
+ )
11
+
12
+
13
+ class PreprocessingEngine:
14
+ """
15
+ Build leakage-safe preprocessing pipelines for ModelForge.
16
+
17
+ All preprocessing operations are placed inside a scikit-learn
18
+ Pipeline / ColumnTransformer so that transformations are fitted
19
+ only on training data during cross-validation.
20
+ """
21
+
22
+ SCALE_MODELS = {
23
+ "knn",
24
+ "svm",
25
+ "svr",
26
+ "linear_regression",
27
+ "ridge",
28
+ "lasso",
29
+ "elasticnet",
30
+ "logistic_regression",
31
+ "mlp",
32
+ }
33
+
34
+ def build(
35
+ self,
36
+ data: pd.DataFrame,
37
+ target: str,
38
+ model_type: str = "tree",
39
+ excluded_columns: Iterable[str] | None = None,
40
+ ) -> ColumnTransformer:
41
+ """
42
+ Build a preprocessing transformer.
43
+
44
+ Parameters
45
+ ----------
46
+ data : pd.DataFrame
47
+ Input dataset.
48
+
49
+ target : str
50
+ Target column.
51
+
52
+ model_type : str
53
+ Model family. Determines whether numerical features
54
+ should be scaled.
55
+
56
+ excluded_columns : Iterable[str] | None
57
+ Columns to exclude from preprocessing.
58
+
59
+ Returns
60
+ -------
61
+ ColumnTransformer
62
+ Leakage-safe preprocessing transformer.
63
+ """
64
+
65
+ if not isinstance(data, pd.DataFrame):
66
+ raise TypeError("data must be a pandas DataFrame.")
67
+
68
+ if data.empty:
69
+ raise ValueError("Cannot preprocess an empty dataset.")
70
+
71
+ if target not in data.columns:
72
+ raise ValueError(
73
+ f"Target column '{target}' does not exist."
74
+ )
75
+
76
+ excluded = set(excluded_columns or [])
77
+
78
+ feature_data = data.drop(
79
+ columns=[target],
80
+ errors="ignore",
81
+ )
82
+
83
+ feature_data = feature_data.drop(
84
+ columns=[
85
+ column
86
+ for column in excluded
87
+ if column in feature_data.columns
88
+ ],
89
+ errors="ignore",
90
+ )
91
+
92
+ numerical_columns = feature_data.select_dtypes(
93
+ include="number"
94
+ ).columns.tolist()
95
+
96
+ categorical_columns = feature_data.select_dtypes(
97
+ include=[
98
+ "object",
99
+ "category",
100
+ "bool",
101
+ "string",
102
+ ]
103
+ ).columns.tolist()
104
+
105
+ if not numerical_columns and not categorical_columns:
106
+ raise ValueError(
107
+ "No supported feature columns were found."
108
+ )
109
+
110
+ scale_numeric = self._should_scale(model_type)
111
+
112
+ numeric_steps = [
113
+ (
114
+ "imputer",
115
+ SimpleImputer(strategy="median"),
116
+ )
117
+ ]
118
+
119
+ if scale_numeric:
120
+ numeric_steps.append(
121
+ (
122
+ "scaler",
123
+ StandardScaler(),
124
+ )
125
+ )
126
+
127
+ numeric_pipeline = Pipeline(
128
+ steps=numeric_steps
129
+ )
130
+
131
+ categorical_pipeline = Pipeline(
132
+ steps=[
133
+ (
134
+ "imputer",
135
+ SimpleImputer(
136
+ strategy="most_frequent"
137
+ ),
138
+ ),
139
+ (
140
+ "encoder",
141
+ OneHotEncoder(
142
+ handle_unknown="ignore",
143
+ sparse_output=True,
144
+ ),
145
+ ),
146
+ ]
147
+ )
148
+
149
+ transformers = []
150
+
151
+ if numerical_columns:
152
+ transformers.append(
153
+ (
154
+ "numerical",
155
+ numeric_pipeline,
156
+ numerical_columns,
157
+ )
158
+ )
159
+
160
+ if categorical_columns:
161
+ transformers.append(
162
+ (
163
+ "categorical",
164
+ categorical_pipeline,
165
+ categorical_columns,
166
+ )
167
+ )
168
+
169
+ return ColumnTransformer(
170
+ transformers=transformers,
171
+ remainder="drop",
172
+ )
173
+
174
+ def _should_scale(self, model_type: str) -> bool:
175
+ """Determine whether numerical features should be scaled."""
176
+
177
+ normalized_model = model_type.lower().strip()
178
+
179
+ return normalized_model in self.SCALE_MODELS
modelforge/profiler.py ADDED
@@ -0,0 +1,85 @@
1
+ import pandas as pd
2
+
3
+
4
+ class DatasetProfiler:
5
+ """Profile a dataset and generate structural and statistical information."""
6
+
7
+ def profile(self, data: pd.DataFrame) -> dict:
8
+ """
9
+ Generate a complete dataset profile.
10
+
11
+ Parameters
12
+ ----------
13
+ data : pd.DataFrame
14
+ Dataset to analyze.
15
+
16
+ Returns
17
+ -------
18
+ dict
19
+ Dataset profiling information.
20
+
21
+ Raises
22
+ ------
23
+ TypeError
24
+ If data is not a pandas DataFrame.
25
+
26
+ ValueError
27
+ If the dataset is empty.
28
+ """
29
+
30
+ if not isinstance(data, pd.DataFrame):
31
+ raise TypeError("data must be a pandas DataFrame.")
32
+
33
+ if data.empty:
34
+ raise ValueError("Cannot profile an empty dataset.")
35
+
36
+ column_info = {}
37
+
38
+ for column in data.columns:
39
+ series = data[column]
40
+
41
+ column_info[column] = {
42
+ "dtype": str(series.dtype),
43
+ "missing_values": int(series.isna().sum()),
44
+ "missing_percentage": float(
45
+ series.isna().mean() * 100
46
+ ),
47
+ "unique_values": int(
48
+ series.nunique(dropna=True)
49
+ ),
50
+ "is_numeric": bool(
51
+ pd.api.types.is_numeric_dtype(series)
52
+ ),
53
+ "is_categorical": bool(
54
+ pd.api.types.is_object_dtype(series)
55
+ or pd.api.types.is_string_dtype(series)
56
+ or isinstance(
57
+ series.dtype,
58
+ pd.CategoricalDtype,
59
+ )
60
+ or pd.api.types.is_bool_dtype(series)
61
+ ),
62
+ }
63
+
64
+ numeric_columns = data.select_dtypes(
65
+ include="number"
66
+ ).columns.tolist()
67
+
68
+ categorical_columns = data.select_dtypes(
69
+ include=["object", "category", "bool", "string"]
70
+ ).columns.tolist()
71
+
72
+ return {
73
+ "rows": int(data.shape[0]),
74
+ "columns": int(data.shape[1]),
75
+ "column_names": data.columns.tolist(),
76
+ "duplicate_rows": int(
77
+ data.duplicated().sum()
78
+ ),
79
+ "memory_usage_bytes": int(
80
+ data.memory_usage(deep=True).sum()
81
+ ),
82
+ "numeric_columns": numeric_columns,
83
+ "categorical_columns": categorical_columns,
84
+ "column_info": column_info,
85
+ }