autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,528 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from sklearn.base import BaseEstimator, TransformerMixin
|
|
9
|
+
from sklearn.feature_selection import (
|
|
10
|
+
SelectKBest,
|
|
11
|
+
f_classif,
|
|
12
|
+
f_regression,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class VarianceFeatureSelector(
|
|
17
|
+
BaseEstimator,
|
|
18
|
+
TransformerMixin,
|
|
19
|
+
):
|
|
20
|
+
"""Remove features whose variance is below a configured threshold."""
|
|
21
|
+
|
|
22
|
+
def __init__(self, threshold: float = 0.0):
|
|
23
|
+
self.threshold = threshold
|
|
24
|
+
|
|
25
|
+
def fit(self, X, y=None):
|
|
26
|
+
if not isinstance(X, pd.DataFrame):
|
|
27
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
28
|
+
|
|
29
|
+
if self.threshold < 0:
|
|
30
|
+
raise ValueError(
|
|
31
|
+
"threshold must be greater than or equal to 0."
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
self.feature_names_in_ = X.columns.tolist()
|
|
35
|
+
|
|
36
|
+
numeric = X.select_dtypes(include=np.number)
|
|
37
|
+
|
|
38
|
+
self.numeric_columns_ = numeric.columns.tolist()
|
|
39
|
+
|
|
40
|
+
if not self.numeric_columns_:
|
|
41
|
+
self.selected_columns_ = self.feature_names_in_.copy()
|
|
42
|
+
self.removed_columns_ = []
|
|
43
|
+
self.variances_ = pd.Series(dtype=float)
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
self.variances_ = numeric.var(
|
|
47
|
+
ddof=0,
|
|
48
|
+
numeric_only=True,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
selected_numeric = self.variances_[
|
|
52
|
+
self.variances_ > self.threshold
|
|
53
|
+
].index.tolist()
|
|
54
|
+
|
|
55
|
+
self.selected_columns_ = [
|
|
56
|
+
column
|
|
57
|
+
for column in self.feature_names_in_
|
|
58
|
+
if (
|
|
59
|
+
column not in self.numeric_columns_
|
|
60
|
+
or column in selected_numeric
|
|
61
|
+
)
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
self.removed_columns_ = [
|
|
65
|
+
column
|
|
66
|
+
for column in self.feature_names_in_
|
|
67
|
+
if column not in self.selected_columns_
|
|
68
|
+
]
|
|
69
|
+
|
|
70
|
+
return self
|
|
71
|
+
|
|
72
|
+
def transform(self, X):
|
|
73
|
+
if not isinstance(X, pd.DataFrame):
|
|
74
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
75
|
+
|
|
76
|
+
if not hasattr(self, "selected_columns_"):
|
|
77
|
+
raise RuntimeError(
|
|
78
|
+
"VarianceFeatureSelector must be fitted before transform."
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
missing = [
|
|
82
|
+
column
|
|
83
|
+
for column in self.selected_columns_
|
|
84
|
+
if column not in X.columns
|
|
85
|
+
]
|
|
86
|
+
|
|
87
|
+
if missing:
|
|
88
|
+
raise ValueError(
|
|
89
|
+
"Input data is missing selected features: "
|
|
90
|
+
f"{missing}"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
return X.loc[:, self.selected_columns_].copy()
|
|
94
|
+
|
|
95
|
+
def get_support(self) -> np.ndarray:
|
|
96
|
+
if not hasattr(self, "selected_columns_"):
|
|
97
|
+
raise RuntimeError(
|
|
98
|
+
"VarianceFeatureSelector must be fitted before "
|
|
99
|
+
"get_support()."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
return np.array(
|
|
103
|
+
[
|
|
104
|
+
column in self.selected_columns_
|
|
105
|
+
for column in self.feature_names_in_
|
|
106
|
+
],
|
|
107
|
+
dtype=bool,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class CorrelationFeatureSelector(
|
|
112
|
+
BaseEstimator,
|
|
113
|
+
TransformerMixin,
|
|
114
|
+
):
|
|
115
|
+
"""Remove highly correlated numerical features."""
|
|
116
|
+
|
|
117
|
+
def __init__(self, threshold: float = 0.95):
|
|
118
|
+
self.threshold = threshold
|
|
119
|
+
|
|
120
|
+
def fit(self, X, y=None):
|
|
121
|
+
if not isinstance(X, pd.DataFrame):
|
|
122
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
123
|
+
|
|
124
|
+
if not 0 < self.threshold <= 1:
|
|
125
|
+
raise ValueError(
|
|
126
|
+
"threshold must be greater than 0 and less than or equal "
|
|
127
|
+
"to 1."
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
self.feature_names_in_ = X.columns.tolist()
|
|
131
|
+
|
|
132
|
+
numeric = X.select_dtypes(include=np.number)
|
|
133
|
+
|
|
134
|
+
self.numeric_columns_ = numeric.columns.tolist()
|
|
135
|
+
|
|
136
|
+
if len(self.numeric_columns_) < 2:
|
|
137
|
+
self.selected_columns_ = self.feature_names_in_.copy()
|
|
138
|
+
self.removed_columns_ = []
|
|
139
|
+
self.correlation_matrix_ = pd.DataFrame()
|
|
140
|
+
return self
|
|
141
|
+
|
|
142
|
+
self.correlation_matrix_ = numeric.corr(
|
|
143
|
+
method="pearson"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
upper = self.correlation_matrix_.where(
|
|
147
|
+
np.triu(
|
|
148
|
+
np.ones(
|
|
149
|
+
self.correlation_matrix_.shape,
|
|
150
|
+
dtype=bool,
|
|
151
|
+
),
|
|
152
|
+
k=1,
|
|
153
|
+
)
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
correlated_columns = [
|
|
157
|
+
column
|
|
158
|
+
for column in upper.columns
|
|
159
|
+
if any(
|
|
160
|
+
upper[column].abs() > self.threshold
|
|
161
|
+
)
|
|
162
|
+
]
|
|
163
|
+
|
|
164
|
+
self.removed_columns_ = correlated_columns
|
|
165
|
+
|
|
166
|
+
self.selected_columns_ = [
|
|
167
|
+
column
|
|
168
|
+
for column in self.feature_names_in_
|
|
169
|
+
if column not in self.removed_columns_
|
|
170
|
+
]
|
|
171
|
+
|
|
172
|
+
return self
|
|
173
|
+
|
|
174
|
+
def transform(self, X):
|
|
175
|
+
if not isinstance(X, pd.DataFrame):
|
|
176
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
177
|
+
|
|
178
|
+
if not hasattr(self, "selected_columns_"):
|
|
179
|
+
raise RuntimeError(
|
|
180
|
+
"CorrelationFeatureSelector must be fitted before "
|
|
181
|
+
"transform."
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
missing = [
|
|
185
|
+
column
|
|
186
|
+
for column in self.selected_columns_
|
|
187
|
+
if column not in X.columns
|
|
188
|
+
]
|
|
189
|
+
|
|
190
|
+
if missing:
|
|
191
|
+
raise ValueError(
|
|
192
|
+
"Input data is missing selected features: "
|
|
193
|
+
f"{missing}"
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
return X.loc[:, self.selected_columns_].copy()
|
|
197
|
+
|
|
198
|
+
def get_support(self) -> np.ndarray:
|
|
199
|
+
if not hasattr(self, "selected_columns_"):
|
|
200
|
+
raise RuntimeError(
|
|
201
|
+
"CorrelationFeatureSelector must be fitted before "
|
|
202
|
+
"get_support()."
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
return np.array(
|
|
206
|
+
[
|
|
207
|
+
column in self.selected_columns_
|
|
208
|
+
for column in self.feature_names_in_
|
|
209
|
+
],
|
|
210
|
+
dtype=bool,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class SelectKBestFeatureSelector(
|
|
215
|
+
BaseEstimator,
|
|
216
|
+
TransformerMixin,
|
|
217
|
+
):
|
|
218
|
+
"""
|
|
219
|
+
Select the k statistically strongest features.
|
|
220
|
+
|
|
221
|
+
For classification, ANOVA F-test is used.
|
|
222
|
+
|
|
223
|
+
For regression, linear regression F-test is used.
|
|
224
|
+
|
|
225
|
+
If k is larger than the number of available features, k is
|
|
226
|
+
automatically capped to the feature count.
|
|
227
|
+
"""
|
|
228
|
+
|
|
229
|
+
def __init__(
|
|
230
|
+
self,
|
|
231
|
+
k: int | str = 10,
|
|
232
|
+
task_type: str = "classification",
|
|
233
|
+
):
|
|
234
|
+
self.k = k
|
|
235
|
+
self.task_type = task_type
|
|
236
|
+
|
|
237
|
+
def fit(self, X, y):
|
|
238
|
+
if not isinstance(X, pd.DataFrame):
|
|
239
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
240
|
+
|
|
241
|
+
if self.task_type not in {
|
|
242
|
+
"classification",
|
|
243
|
+
"regression",
|
|
244
|
+
}:
|
|
245
|
+
raise ValueError(
|
|
246
|
+
"task_type must be 'classification' or 'regression'."
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
if isinstance(self.k, int):
|
|
250
|
+
if self.k <= 0:
|
|
251
|
+
raise ValueError(
|
|
252
|
+
"k must be greater than 0."
|
|
253
|
+
)
|
|
254
|
+
elif self.k != "all":
|
|
255
|
+
raise ValueError(
|
|
256
|
+
"k must be a positive integer or 'all'."
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
self.feature_names_in_ = X.columns.tolist()
|
|
260
|
+
|
|
261
|
+
numeric = X.select_dtypes(
|
|
262
|
+
include=np.number
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
self.numeric_columns_ = numeric.columns.tolist()
|
|
266
|
+
|
|
267
|
+
non_numeric = [
|
|
268
|
+
column
|
|
269
|
+
for column in self.feature_names_in_
|
|
270
|
+
if column not in self.numeric_columns_
|
|
271
|
+
]
|
|
272
|
+
|
|
273
|
+
self.non_numeric_columns_ = non_numeric
|
|
274
|
+
|
|
275
|
+
if not self.numeric_columns_:
|
|
276
|
+
self.selected_columns_ = self.feature_names_in_.copy()
|
|
277
|
+
self.removed_columns_ = []
|
|
278
|
+
self.scores_ = pd.Series(dtype=float)
|
|
279
|
+
self.pvalues_ = pd.Series(dtype=float)
|
|
280
|
+
self.effective_k_ = "all"
|
|
281
|
+
return self
|
|
282
|
+
|
|
283
|
+
if self.k == "all":
|
|
284
|
+
effective_k = "all"
|
|
285
|
+
else:
|
|
286
|
+
effective_k = min(
|
|
287
|
+
self.k,
|
|
288
|
+
len(self.numeric_columns_),
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
self.effective_k_ = effective_k
|
|
292
|
+
|
|
293
|
+
if self.task_type == "classification":
|
|
294
|
+
score_func = f_classif
|
|
295
|
+
else:
|
|
296
|
+
score_func = f_regression
|
|
297
|
+
|
|
298
|
+
numeric_data = numeric.copy()
|
|
299
|
+
|
|
300
|
+
numeric_data = numeric_data.replace(
|
|
301
|
+
[np.inf, -np.inf],
|
|
302
|
+
np.nan,
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
for column in numeric_data.columns:
|
|
306
|
+
if numeric_data[column].isna().any():
|
|
307
|
+
median = numeric_data[column].median()
|
|
308
|
+
|
|
309
|
+
if pd.isna(median):
|
|
310
|
+
median = 0.0
|
|
311
|
+
|
|
312
|
+
numeric_data[column] = (
|
|
313
|
+
numeric_data[column].fillna(median)
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
y_series = pd.Series(y).reset_index(drop=True)
|
|
317
|
+
|
|
318
|
+
numeric_data = numeric_data.reset_index(
|
|
319
|
+
drop=True
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
selector = SelectKBest(
|
|
323
|
+
score_func=score_func,
|
|
324
|
+
k=effective_k,
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
selector.fit(
|
|
328
|
+
numeric_data,
|
|
329
|
+
y_series,
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
self.selector_ = selector
|
|
333
|
+
|
|
334
|
+
self.scores_ = pd.Series(
|
|
335
|
+
selector.scores_,
|
|
336
|
+
index=self.numeric_columns_,
|
|
337
|
+
dtype=float,
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
self.pvalues_ = pd.Series(
|
|
341
|
+
selector.pvalues_,
|
|
342
|
+
index=self.numeric_columns_,
|
|
343
|
+
dtype=float,
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
support = selector.get_support()
|
|
347
|
+
|
|
348
|
+
selected_numeric = [
|
|
349
|
+
column
|
|
350
|
+
for column, selected in zip(
|
|
351
|
+
self.numeric_columns_,
|
|
352
|
+
support,
|
|
353
|
+
)
|
|
354
|
+
if selected
|
|
355
|
+
]
|
|
356
|
+
|
|
357
|
+
self.selected_columns_ = (
|
|
358
|
+
non_numeric + selected_numeric
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
self.selected_columns_ = [
|
|
362
|
+
column
|
|
363
|
+
for column in self.feature_names_in_
|
|
364
|
+
if column in self.selected_columns_
|
|
365
|
+
]
|
|
366
|
+
|
|
367
|
+
self.removed_columns_ = [
|
|
368
|
+
column
|
|
369
|
+
for column in self.feature_names_in_
|
|
370
|
+
if column not in self.selected_columns_
|
|
371
|
+
]
|
|
372
|
+
|
|
373
|
+
return self
|
|
374
|
+
|
|
375
|
+
def transform(self, X):
|
|
376
|
+
if not isinstance(X, pd.DataFrame):
|
|
377
|
+
raise TypeError("X must be a pandas DataFrame.")
|
|
378
|
+
|
|
379
|
+
if not hasattr(self, "selected_columns_"):
|
|
380
|
+
raise RuntimeError(
|
|
381
|
+
"SelectKBestFeatureSelector must be fitted before "
|
|
382
|
+
"transform."
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
missing = [
|
|
386
|
+
column
|
|
387
|
+
for column in self.selected_columns_
|
|
388
|
+
if column not in X.columns
|
|
389
|
+
]
|
|
390
|
+
|
|
391
|
+
if missing:
|
|
392
|
+
raise ValueError(
|
|
393
|
+
"Input data is missing selected features: "
|
|
394
|
+
f"{missing}"
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
return X.loc[:, self.selected_columns_].copy()
|
|
398
|
+
|
|
399
|
+
def get_support(self) -> np.ndarray:
|
|
400
|
+
if not hasattr(self, "selected_columns_"):
|
|
401
|
+
raise RuntimeError(
|
|
402
|
+
"SelectKBestFeatureSelector must be fitted before "
|
|
403
|
+
"get_support()."
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
return np.array(
|
|
407
|
+
[
|
|
408
|
+
column in self.selected_columns_
|
|
409
|
+
for column in self.feature_names_in_
|
|
410
|
+
],
|
|
411
|
+
dtype=bool,
|
|
412
|
+
)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
class FeatureSelectionEngine:
|
|
416
|
+
"""Coordinate ModelForge feature-selection strategies."""
|
|
417
|
+
|
|
418
|
+
def variance_filter(
|
|
419
|
+
self,
|
|
420
|
+
data: pd.DataFrame,
|
|
421
|
+
threshold: float = 0.0,
|
|
422
|
+
) -> pd.DataFrame:
|
|
423
|
+
"""Apply variance filtering."""
|
|
424
|
+
|
|
425
|
+
selector = VarianceFeatureSelector(
|
|
426
|
+
threshold=threshold
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
return selector.fit_transform(data)
|
|
430
|
+
|
|
431
|
+
def correlation_filter(
|
|
432
|
+
self,
|
|
433
|
+
data: pd.DataFrame,
|
|
434
|
+
threshold: float = 0.95,
|
|
435
|
+
) -> pd.DataFrame:
|
|
436
|
+
"""Apply correlation filtering."""
|
|
437
|
+
|
|
438
|
+
selector = CorrelationFeatureSelector(
|
|
439
|
+
threshold=threshold
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
return selector.fit_transform(data)
|
|
443
|
+
|
|
444
|
+
def select_k_best(
|
|
445
|
+
self,
|
|
446
|
+
data: pd.DataFrame,
|
|
447
|
+
target,
|
|
448
|
+
k: int | str = 10,
|
|
449
|
+
task_type: str = "classification",
|
|
450
|
+
) -> pd.DataFrame:
|
|
451
|
+
"""Apply SelectKBest feature selection."""
|
|
452
|
+
|
|
453
|
+
selector = SelectKBestFeatureSelector(
|
|
454
|
+
k=k,
|
|
455
|
+
task_type=task_type,
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
return selector.fit_transform(
|
|
459
|
+
data,
|
|
460
|
+
target,
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
def select(
|
|
464
|
+
self,
|
|
465
|
+
data: pd.DataFrame,
|
|
466
|
+
target=None,
|
|
467
|
+
task_type: str | None = None,
|
|
468
|
+
variance_threshold: float | None = None,
|
|
469
|
+
correlation_threshold: float | None = None,
|
|
470
|
+
k: int | str | None = None,
|
|
471
|
+
) -> pd.DataFrame:
|
|
472
|
+
"""
|
|
473
|
+
Apply a configurable feature-selection sequence.
|
|
474
|
+
|
|
475
|
+
Order:
|
|
476
|
+
|
|
477
|
+
1. Variance filtering
|
|
478
|
+
2. Correlation filtering
|
|
479
|
+
3. SelectKBest
|
|
480
|
+
"""
|
|
481
|
+
|
|
482
|
+
if not isinstance(data, pd.DataFrame):
|
|
483
|
+
raise TypeError(
|
|
484
|
+
"data must be a pandas DataFrame."
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
result = data.copy()
|
|
488
|
+
|
|
489
|
+
if variance_threshold is not None:
|
|
490
|
+
variance_selector = VarianceFeatureSelector(
|
|
491
|
+
threshold=variance_threshold
|
|
492
|
+
)
|
|
493
|
+
|
|
494
|
+
result = variance_selector.fit_transform(
|
|
495
|
+
result
|
|
496
|
+
)
|
|
497
|
+
|
|
498
|
+
if correlation_threshold is not None:
|
|
499
|
+
correlation_selector = CorrelationFeatureSelector(
|
|
500
|
+
threshold=correlation_threshold
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
result = correlation_selector.fit_transform(
|
|
504
|
+
result
|
|
505
|
+
)
|
|
506
|
+
|
|
507
|
+
if k is not None:
|
|
508
|
+
if target is None:
|
|
509
|
+
raise ValueError(
|
|
510
|
+
"target is required when k is provided."
|
|
511
|
+
)
|
|
512
|
+
|
|
513
|
+
if task_type is None:
|
|
514
|
+
raise ValueError(
|
|
515
|
+
"task_type is required when k is provided."
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
k_selector = SelectKBestFeatureSelector(
|
|
519
|
+
k=k,
|
|
520
|
+
task_type=task_type,
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
result = k_selector.fit_transform(
|
|
524
|
+
result,
|
|
525
|
+
target,
|
|
526
|
+
)
|
|
527
|
+
|
|
528
|
+
return result
|