autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from sklearn.inspection import permutation_importance
|
|
7
|
+
from sklearn.pipeline import Pipeline
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ExplainabilityEngine:
|
|
11
|
+
"""
|
|
12
|
+
Generate model explanations for ModelForge pipelines.
|
|
13
|
+
|
|
14
|
+
Supports:
|
|
15
|
+
- native feature importance
|
|
16
|
+
- linear model coefficients
|
|
17
|
+
- permutation importance
|
|
18
|
+
- prediction-level summaries
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def feature_importance(
|
|
22
|
+
self,
|
|
23
|
+
pipeline: Pipeline,
|
|
24
|
+
) -> pd.DataFrame:
|
|
25
|
+
"""
|
|
26
|
+
Extract native feature importance or coefficients
|
|
27
|
+
from a fitted pipeline.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
self._validate_pipeline(pipeline)
|
|
31
|
+
|
|
32
|
+
model = self._get_model(pipeline)
|
|
33
|
+
|
|
34
|
+
feature_names = self._get_feature_names(
|
|
35
|
+
pipeline
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
if hasattr(model, "feature_importances_"):
|
|
39
|
+
importance = np.asarray(
|
|
40
|
+
model.feature_importances_
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
return self._build_importance_dataframe(
|
|
44
|
+
feature_names,
|
|
45
|
+
importance,
|
|
46
|
+
source="feature_importance",
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
if hasattr(model, "coef_"):
|
|
50
|
+
coefficients = np.asarray(
|
|
51
|
+
model.coef_
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
if coefficients.ndim == 1:
|
|
55
|
+
importance = np.abs(
|
|
56
|
+
coefficients
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
elif coefficients.ndim == 2:
|
|
60
|
+
importance = np.mean(
|
|
61
|
+
np.abs(coefficients),
|
|
62
|
+
axis=0,
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
else:
|
|
66
|
+
raise ValueError(
|
|
67
|
+
"Unsupported coefficient shape."
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
return self._build_importance_dataframe(
|
|
71
|
+
feature_names,
|
|
72
|
+
importance,
|
|
73
|
+
source="coefficient",
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"Model '{type(model).__name__}' "
|
|
78
|
+
"does not provide native feature importance "
|
|
79
|
+
"or coefficients."
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
def permutation_importance(
|
|
83
|
+
self,
|
|
84
|
+
pipeline: Pipeline,
|
|
85
|
+
X: pd.DataFrame,
|
|
86
|
+
y,
|
|
87
|
+
scoring: str | None = None,
|
|
88
|
+
n_repeats: int = 10,
|
|
89
|
+
random_state: int = 42,
|
|
90
|
+
) -> pd.DataFrame:
|
|
91
|
+
"""
|
|
92
|
+
Calculate permutation importance on the complete pipeline.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
self._validate_pipeline(pipeline)
|
|
96
|
+
|
|
97
|
+
if not isinstance(X, pd.DataFrame):
|
|
98
|
+
raise TypeError(
|
|
99
|
+
"X must be a pandas DataFrame."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
if X.empty:
|
|
103
|
+
raise ValueError(
|
|
104
|
+
"X cannot be empty."
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
if len(X) != len(y):
|
|
108
|
+
raise ValueError(
|
|
109
|
+
"X and y must contain the same number "
|
|
110
|
+
"of samples."
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
if n_repeats <= 0:
|
|
114
|
+
raise ValueError(
|
|
115
|
+
"n_repeats must be greater than 0."
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
result = permutation_importance(
|
|
119
|
+
estimator=pipeline,
|
|
120
|
+
X=X,
|
|
121
|
+
y=y,
|
|
122
|
+
scoring=scoring,
|
|
123
|
+
n_repeats=n_repeats,
|
|
124
|
+
random_state=random_state,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
feature_names = X.columns.tolist()
|
|
128
|
+
|
|
129
|
+
return self._build_importance_dataframe(
|
|
130
|
+
feature_names,
|
|
131
|
+
result.importances_mean,
|
|
132
|
+
source="permutation",
|
|
133
|
+
std=result.importances_std,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
def explain_prediction(
|
|
137
|
+
self,
|
|
138
|
+
pipeline: Pipeline,
|
|
139
|
+
X: pd.DataFrame,
|
|
140
|
+
) -> dict[str, Any]:
|
|
141
|
+
"""
|
|
142
|
+
Generate a simple prediction-level explanation.
|
|
143
|
+
|
|
144
|
+
This does not attempt to calculate SHAP-style local
|
|
145
|
+
contributions. Instead, it combines the prediction
|
|
146
|
+
with global feature importance information.
|
|
147
|
+
"""
|
|
148
|
+
|
|
149
|
+
self._validate_pipeline(pipeline)
|
|
150
|
+
|
|
151
|
+
if not isinstance(X, pd.DataFrame):
|
|
152
|
+
raise TypeError(
|
|
153
|
+
"X must be a pandas DataFrame."
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
if X.empty:
|
|
157
|
+
raise ValueError(
|
|
158
|
+
"X cannot be empty."
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
predictions = pipeline.predict(X)
|
|
162
|
+
|
|
163
|
+
try:
|
|
164
|
+
importance = self.feature_importance(
|
|
165
|
+
pipeline
|
|
166
|
+
)
|
|
167
|
+
except ValueError:
|
|
168
|
+
importance = None
|
|
169
|
+
|
|
170
|
+
return {
|
|
171
|
+
"predictions": predictions.tolist(),
|
|
172
|
+
"prediction_count": int(
|
|
173
|
+
len(predictions)
|
|
174
|
+
),
|
|
175
|
+
"feature_importance": importance,
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
@staticmethod
|
|
179
|
+
def top_features(
|
|
180
|
+
importance: pd.DataFrame,
|
|
181
|
+
n: int = 10,
|
|
182
|
+
) -> pd.DataFrame:
|
|
183
|
+
"""
|
|
184
|
+
Return the top N most important features.
|
|
185
|
+
"""
|
|
186
|
+
|
|
187
|
+
if not isinstance(
|
|
188
|
+
importance,
|
|
189
|
+
pd.DataFrame,
|
|
190
|
+
):
|
|
191
|
+
raise TypeError(
|
|
192
|
+
"importance must be a pandas DataFrame."
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
if importance.empty:
|
|
196
|
+
raise ValueError(
|
|
197
|
+
"importance cannot be empty."
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
if n <= 0:
|
|
201
|
+
raise ValueError(
|
|
202
|
+
"n must be greater than 0."
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
return (
|
|
206
|
+
importance
|
|
207
|
+
.sort_values(
|
|
208
|
+
by="importance",
|
|
209
|
+
ascending=False,
|
|
210
|
+
)
|
|
211
|
+
.head(n)
|
|
212
|
+
.reset_index(drop=True)
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
@staticmethod
|
|
216
|
+
def _get_model(
|
|
217
|
+
pipeline: Pipeline,
|
|
218
|
+
):
|
|
219
|
+
if not hasattr(
|
|
220
|
+
pipeline,
|
|
221
|
+
"named_steps",
|
|
222
|
+
):
|
|
223
|
+
raise TypeError(
|
|
224
|
+
"pipeline must contain named steps."
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
if "model" not in pipeline.named_steps:
|
|
228
|
+
raise ValueError(
|
|
229
|
+
"Pipeline must contain a 'model' step."
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
return pipeline.named_steps["model"]
|
|
233
|
+
|
|
234
|
+
@staticmethod
|
|
235
|
+
def _get_feature_names(
|
|
236
|
+
pipeline: Pipeline,
|
|
237
|
+
) -> list[str]:
|
|
238
|
+
"""
|
|
239
|
+
Extract transformed feature names from the
|
|
240
|
+
preprocessing step.
|
|
241
|
+
"""
|
|
242
|
+
|
|
243
|
+
if "preprocessing" not in pipeline.named_steps:
|
|
244
|
+
raise ValueError(
|
|
245
|
+
"Pipeline must contain a "
|
|
246
|
+
"'preprocessing' step."
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
preprocessing = pipeline.named_steps[
|
|
250
|
+
"preprocessing"
|
|
251
|
+
]
|
|
252
|
+
|
|
253
|
+
try:
|
|
254
|
+
feature_names = (
|
|
255
|
+
preprocessing.get_feature_names_out()
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
return [
|
|
259
|
+
str(name)
|
|
260
|
+
for name in feature_names
|
|
261
|
+
]
|
|
262
|
+
|
|
263
|
+
except (AttributeError, ValueError):
|
|
264
|
+
pass
|
|
265
|
+
|
|
266
|
+
if hasattr(
|
|
267
|
+
preprocessing,
|
|
268
|
+
"feature_names_in_",
|
|
269
|
+
):
|
|
270
|
+
return [
|
|
271
|
+
str(name)
|
|
272
|
+
for name in preprocessing.feature_names_in_
|
|
273
|
+
]
|
|
274
|
+
|
|
275
|
+
raise ValueError(
|
|
276
|
+
"Unable to determine transformed "
|
|
277
|
+
"feature names."
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
@staticmethod
|
|
281
|
+
def _build_importance_dataframe(
|
|
282
|
+
feature_names,
|
|
283
|
+
importance,
|
|
284
|
+
source: str,
|
|
285
|
+
std=None,
|
|
286
|
+
) -> pd.DataFrame:
|
|
287
|
+
importance = np.asarray(
|
|
288
|
+
importance,
|
|
289
|
+
dtype=float,
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
feature_names = list(
|
|
293
|
+
feature_names
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
if len(feature_names) != len(
|
|
297
|
+
importance
|
|
298
|
+
):
|
|
299
|
+
raise ValueError(
|
|
300
|
+
"Number of feature names does not "
|
|
301
|
+
"match number of importance values."
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
result = pd.DataFrame(
|
|
305
|
+
{
|
|
306
|
+
"feature": feature_names,
|
|
307
|
+
"importance": importance,
|
|
308
|
+
"source": source,
|
|
309
|
+
}
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
if std is not None:
|
|
313
|
+
result["std"] = np.asarray(
|
|
314
|
+
std,
|
|
315
|
+
dtype=float,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
result["absolute_importance"] = (
|
|
319
|
+
result["importance"].abs()
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
return (
|
|
323
|
+
result
|
|
324
|
+
.sort_values(
|
|
325
|
+
by="absolute_importance",
|
|
326
|
+
ascending=False,
|
|
327
|
+
)
|
|
328
|
+
.reset_index(drop=True)
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
@staticmethod
|
|
332
|
+
def _validate_pipeline(
|
|
333
|
+
pipeline: Pipeline,
|
|
334
|
+
) -> None:
|
|
335
|
+
if not isinstance(
|
|
336
|
+
pipeline,
|
|
337
|
+
Pipeline,
|
|
338
|
+
):
|
|
339
|
+
raise TypeError(
|
|
340
|
+
"pipeline must be a sklearn Pipeline."
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
if "model" not in pipeline.named_steps:
|
|
344
|
+
raise ValueError(
|
|
345
|
+
"Pipeline must contain a 'model' step."
|
|
346
|
+
)
|