autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,346 @@
1
+ from typing import Any
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ from sklearn.inspection import permutation_importance
7
+ from sklearn.pipeline import Pipeline
8
+
9
+
10
+ class ExplainabilityEngine:
11
+ """
12
+ Generate model explanations for ModelForge pipelines.
13
+
14
+ Supports:
15
+ - native feature importance
16
+ - linear model coefficients
17
+ - permutation importance
18
+ - prediction-level summaries
19
+ """
20
+
21
+ def feature_importance(
22
+ self,
23
+ pipeline: Pipeline,
24
+ ) -> pd.DataFrame:
25
+ """
26
+ Extract native feature importance or coefficients
27
+ from a fitted pipeline.
28
+ """
29
+
30
+ self._validate_pipeline(pipeline)
31
+
32
+ model = self._get_model(pipeline)
33
+
34
+ feature_names = self._get_feature_names(
35
+ pipeline
36
+ )
37
+
38
+ if hasattr(model, "feature_importances_"):
39
+ importance = np.asarray(
40
+ model.feature_importances_
41
+ )
42
+
43
+ return self._build_importance_dataframe(
44
+ feature_names,
45
+ importance,
46
+ source="feature_importance",
47
+ )
48
+
49
+ if hasattr(model, "coef_"):
50
+ coefficients = np.asarray(
51
+ model.coef_
52
+ )
53
+
54
+ if coefficients.ndim == 1:
55
+ importance = np.abs(
56
+ coefficients
57
+ )
58
+
59
+ elif coefficients.ndim == 2:
60
+ importance = np.mean(
61
+ np.abs(coefficients),
62
+ axis=0,
63
+ )
64
+
65
+ else:
66
+ raise ValueError(
67
+ "Unsupported coefficient shape."
68
+ )
69
+
70
+ return self._build_importance_dataframe(
71
+ feature_names,
72
+ importance,
73
+ source="coefficient",
74
+ )
75
+
76
+ raise ValueError(
77
+ f"Model '{type(model).__name__}' "
78
+ "does not provide native feature importance "
79
+ "or coefficients."
80
+ )
81
+
82
+ def permutation_importance(
83
+ self,
84
+ pipeline: Pipeline,
85
+ X: pd.DataFrame,
86
+ y,
87
+ scoring: str | None = None,
88
+ n_repeats: int = 10,
89
+ random_state: int = 42,
90
+ ) -> pd.DataFrame:
91
+ """
92
+ Calculate permutation importance on the complete pipeline.
93
+ """
94
+
95
+ self._validate_pipeline(pipeline)
96
+
97
+ if not isinstance(X, pd.DataFrame):
98
+ raise TypeError(
99
+ "X must be a pandas DataFrame."
100
+ )
101
+
102
+ if X.empty:
103
+ raise ValueError(
104
+ "X cannot be empty."
105
+ )
106
+
107
+ if len(X) != len(y):
108
+ raise ValueError(
109
+ "X and y must contain the same number "
110
+ "of samples."
111
+ )
112
+
113
+ if n_repeats <= 0:
114
+ raise ValueError(
115
+ "n_repeats must be greater than 0."
116
+ )
117
+
118
+ result = permutation_importance(
119
+ estimator=pipeline,
120
+ X=X,
121
+ y=y,
122
+ scoring=scoring,
123
+ n_repeats=n_repeats,
124
+ random_state=random_state,
125
+ )
126
+
127
+ feature_names = X.columns.tolist()
128
+
129
+ return self._build_importance_dataframe(
130
+ feature_names,
131
+ result.importances_mean,
132
+ source="permutation",
133
+ std=result.importances_std,
134
+ )
135
+
136
+ def explain_prediction(
137
+ self,
138
+ pipeline: Pipeline,
139
+ X: pd.DataFrame,
140
+ ) -> dict[str, Any]:
141
+ """
142
+ Generate a simple prediction-level explanation.
143
+
144
+ This does not attempt to calculate SHAP-style local
145
+ contributions. Instead, it combines the prediction
146
+ with global feature importance information.
147
+ """
148
+
149
+ self._validate_pipeline(pipeline)
150
+
151
+ if not isinstance(X, pd.DataFrame):
152
+ raise TypeError(
153
+ "X must be a pandas DataFrame."
154
+ )
155
+
156
+ if X.empty:
157
+ raise ValueError(
158
+ "X cannot be empty."
159
+ )
160
+
161
+ predictions = pipeline.predict(X)
162
+
163
+ try:
164
+ importance = self.feature_importance(
165
+ pipeline
166
+ )
167
+ except ValueError:
168
+ importance = None
169
+
170
+ return {
171
+ "predictions": predictions.tolist(),
172
+ "prediction_count": int(
173
+ len(predictions)
174
+ ),
175
+ "feature_importance": importance,
176
+ }
177
+
178
+ @staticmethod
179
+ def top_features(
180
+ importance: pd.DataFrame,
181
+ n: int = 10,
182
+ ) -> pd.DataFrame:
183
+ """
184
+ Return the top N most important features.
185
+ """
186
+
187
+ if not isinstance(
188
+ importance,
189
+ pd.DataFrame,
190
+ ):
191
+ raise TypeError(
192
+ "importance must be a pandas DataFrame."
193
+ )
194
+
195
+ if importance.empty:
196
+ raise ValueError(
197
+ "importance cannot be empty."
198
+ )
199
+
200
+ if n <= 0:
201
+ raise ValueError(
202
+ "n must be greater than 0."
203
+ )
204
+
205
+ return (
206
+ importance
207
+ .sort_values(
208
+ by="importance",
209
+ ascending=False,
210
+ )
211
+ .head(n)
212
+ .reset_index(drop=True)
213
+ )
214
+
215
+ @staticmethod
216
+ def _get_model(
217
+ pipeline: Pipeline,
218
+ ):
219
+ if not hasattr(
220
+ pipeline,
221
+ "named_steps",
222
+ ):
223
+ raise TypeError(
224
+ "pipeline must contain named steps."
225
+ )
226
+
227
+ if "model" not in pipeline.named_steps:
228
+ raise ValueError(
229
+ "Pipeline must contain a 'model' step."
230
+ )
231
+
232
+ return pipeline.named_steps["model"]
233
+
234
+ @staticmethod
235
+ def _get_feature_names(
236
+ pipeline: Pipeline,
237
+ ) -> list[str]:
238
+ """
239
+ Extract transformed feature names from the
240
+ preprocessing step.
241
+ """
242
+
243
+ if "preprocessing" not in pipeline.named_steps:
244
+ raise ValueError(
245
+ "Pipeline must contain a "
246
+ "'preprocessing' step."
247
+ )
248
+
249
+ preprocessing = pipeline.named_steps[
250
+ "preprocessing"
251
+ ]
252
+
253
+ try:
254
+ feature_names = (
255
+ preprocessing.get_feature_names_out()
256
+ )
257
+
258
+ return [
259
+ str(name)
260
+ for name in feature_names
261
+ ]
262
+
263
+ except (AttributeError, ValueError):
264
+ pass
265
+
266
+ if hasattr(
267
+ preprocessing,
268
+ "feature_names_in_",
269
+ ):
270
+ return [
271
+ str(name)
272
+ for name in preprocessing.feature_names_in_
273
+ ]
274
+
275
+ raise ValueError(
276
+ "Unable to determine transformed "
277
+ "feature names."
278
+ )
279
+
280
+ @staticmethod
281
+ def _build_importance_dataframe(
282
+ feature_names,
283
+ importance,
284
+ source: str,
285
+ std=None,
286
+ ) -> pd.DataFrame:
287
+ importance = np.asarray(
288
+ importance,
289
+ dtype=float,
290
+ )
291
+
292
+ feature_names = list(
293
+ feature_names
294
+ )
295
+
296
+ if len(feature_names) != len(
297
+ importance
298
+ ):
299
+ raise ValueError(
300
+ "Number of feature names does not "
301
+ "match number of importance values."
302
+ )
303
+
304
+ result = pd.DataFrame(
305
+ {
306
+ "feature": feature_names,
307
+ "importance": importance,
308
+ "source": source,
309
+ }
310
+ )
311
+
312
+ if std is not None:
313
+ result["std"] = np.asarray(
314
+ std,
315
+ dtype=float,
316
+ )
317
+
318
+ result["absolute_importance"] = (
319
+ result["importance"].abs()
320
+ )
321
+
322
+ return (
323
+ result
324
+ .sort_values(
325
+ by="absolute_importance",
326
+ ascending=False,
327
+ )
328
+ .reset_index(drop=True)
329
+ )
330
+
331
+ @staticmethod
332
+ def _validate_pipeline(
333
+ pipeline: Pipeline,
334
+ ) -> None:
335
+ if not isinstance(
336
+ pipeline,
337
+ Pipeline,
338
+ ):
339
+ raise TypeError(
340
+ "pipeline must be a sklearn Pipeline."
341
+ )
342
+
343
+ if "model" not in pipeline.named_steps:
344
+ raise ValueError(
345
+ "Pipeline must contain a 'model' step."
346
+ )