sofi-tabular 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sofi/__init__.py +51 -0
- sofi/encoding.py +235 -0
- sofi/explainer.py +492 -0
- sofi/explanation.py +310 -0
- sofi/marginalization.py +200 -0
- sofi/model.py +153 -0
- sofi/objective.py +307 -0
- sofi/plotting.py +255 -0
- sofi/search.py +184 -0
- sofi/utils.py +88 -0
- sofi_tabular-1.0.0.dist-info/METADATA +465 -0
- sofi_tabular-1.0.0.dist-info/RECORD +15 -0
- sofi_tabular-1.0.0.dist-info/WHEEL +5 -0
- sofi_tabular-1.0.0.dist-info/licenses/LICENSE +21 -0
- sofi_tabular-1.0.0.dist-info/top_level.txt +1 -0
sofi/__init__.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""SOFI, Sparseness Optimized Feature Importance.
|
|
2
|
+
|
|
3
|
+
A model agnostic post hoc explainer that searches for a ranking of features
|
|
4
|
+
whose cumulative marginalization degrades the response of the model as fast as
|
|
5
|
+
possible. The objective is the degradation score, namely the area between the
|
|
6
|
+
LeRF and the MoRF perturbation curves, which rewards sparsity and correctness
|
|
7
|
+
at once. The search works on the raw response of the model, and normalization
|
|
8
|
+
is applied afterwards, for reporting and for drawing alone.
|
|
9
|
+
Classification and regression are both supported, and the search is a hill
|
|
10
|
+
climbing procedure whose operator swaps two randomly selected ranking
|
|
11
|
+
positions.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from .encoding import FeatureSpace, feature_groups_from_encoder
|
|
15
|
+
from .explainer import SOFIExplainer
|
|
16
|
+
from .explanation import SOFIExplanation, aggregate_explanations, noise_onset
|
|
17
|
+
from .marginalization import Marginalizer
|
|
18
|
+
from .objective import Objective, RankingEvaluation, curve_auc, degradation_score
|
|
19
|
+
from .plotting import (
|
|
20
|
+
LERF_COLOR,
|
|
21
|
+
MORF_COLOR,
|
|
22
|
+
plot_degradation_curve,
|
|
23
|
+
plot_explanation_grid,
|
|
24
|
+
set_curve_colors,
|
|
25
|
+
set_plot_style,
|
|
26
|
+
)
|
|
27
|
+
from .search import hill_climbing
|
|
28
|
+
from .utils import select_reliable_instances
|
|
29
|
+
|
|
30
|
+
__version__ = "1.0.0"
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"SOFIExplainer",
|
|
34
|
+
"SOFIExplanation",
|
|
35
|
+
"FeatureSpace",
|
|
36
|
+
"Marginalizer",
|
|
37
|
+
"Objective",
|
|
38
|
+
"aggregate_explanations",
|
|
39
|
+
"RankingEvaluation",
|
|
40
|
+
"curve_auc",
|
|
41
|
+
"degradation_score",
|
|
42
|
+
"feature_groups_from_encoder",
|
|
43
|
+
"hill_climbing",
|
|
44
|
+
"noise_onset",
|
|
45
|
+
"plot_degradation_curve",
|
|
46
|
+
"set_curve_colors",
|
|
47
|
+
"plot_explanation_grid",
|
|
48
|
+
"set_plot_style",
|
|
49
|
+
"select_reliable_instances",
|
|
50
|
+
"__version__",
|
|
51
|
+
]
|
sofi/encoding.py
ADDED
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Resolution of the space in which features are marginalized.
|
|
2
|
+
|
|
3
|
+
SOFI perturbs whole features, never isolated columns of a one-hot block. The
|
|
4
|
+
classes below make that guarantee explicit through a mapping from a logical
|
|
5
|
+
feature to the set of columns that represent it, together with the
|
|
6
|
+
transformation that turns a perturbed frame into model input.
|
|
7
|
+
|
|
8
|
+
Three usage paths are supported.
|
|
9
|
+
|
|
10
|
+
1. A ``Pipeline`` that carries the encoder and the estimator. Perturbation
|
|
11
|
+
happens in the original feature space and the encoding never leaks into the
|
|
12
|
+
explainer. This path is the recommended one.
|
|
13
|
+
2. A fitted encoder passed apart from the estimator. Perturbation still happens
|
|
14
|
+
in the original space, and SOFI applies the encoder before each prediction.
|
|
15
|
+
3. Data that is already encoded, together with a ``feature_groups`` mapping
|
|
16
|
+
that states which columns belong to the same logical feature.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import warnings
|
|
22
|
+
from typing import Callable, Dict, List, Optional, Sequence
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
import pandas as pd
|
|
26
|
+
|
|
27
|
+
__all__ = ["FeatureSpace", "build_feature_space", "feature_groups_from_encoder"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class FeatureSpace:
|
|
31
|
+
"""Mapping between logical features and the columns SOFI perturbs.
|
|
32
|
+
|
|
33
|
+
Attributes
|
|
34
|
+
----------
|
|
35
|
+
feature_names : list of str
|
|
36
|
+
Names of the logical features, in the order used internally.
|
|
37
|
+
groups : dict
|
|
38
|
+
Maps every logical feature onto the list of columns that represent it.
|
|
39
|
+
transform : callable
|
|
40
|
+
Turns a perturbed frame into the container consumed by the estimator.
|
|
41
|
+
output_feature_names : list of str or None
|
|
42
|
+
Column names produced by ``transform`` when it returns a bare array.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
groups: Dict[str, List],
|
|
48
|
+
transform: Optional[Callable] = None,
|
|
49
|
+
output_feature_names: Optional[Sequence] = None,
|
|
50
|
+
path: str = "identity",
|
|
51
|
+
):
|
|
52
|
+
self.groups = {str(k): list(v) for k, v in groups.items()}
|
|
53
|
+
self.feature_names = list(self.groups.keys())
|
|
54
|
+
self._transform = transform
|
|
55
|
+
self.output_feature_names = (
|
|
56
|
+
None if output_feature_names is None else list(output_feature_names)
|
|
57
|
+
)
|
|
58
|
+
self.path = path
|
|
59
|
+
|
|
60
|
+
def __len__(self) -> int:
|
|
61
|
+
return len(self.feature_names)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def n_features(self) -> int:
|
|
65
|
+
return len(self.feature_names)
|
|
66
|
+
|
|
67
|
+
def columns_of(self, feature) -> List:
|
|
68
|
+
if isinstance(feature, (int, np.integer)):
|
|
69
|
+
feature = self.feature_names[int(feature)]
|
|
70
|
+
return self.groups[feature]
|
|
71
|
+
|
|
72
|
+
def transform(self, X: pd.DataFrame):
|
|
73
|
+
if self._transform is None:
|
|
74
|
+
return X
|
|
75
|
+
return self._transform(X)
|
|
76
|
+
|
|
77
|
+
def describe(self) -> pd.DataFrame:
|
|
78
|
+
"""Return one row per logical feature with its block width."""
|
|
79
|
+
return pd.DataFrame(
|
|
80
|
+
{
|
|
81
|
+
"feature": self.feature_names,
|
|
82
|
+
"n_columns": [len(self.groups[f]) for f in self.feature_names],
|
|
83
|
+
"columns": [list(self.groups[f]) for f in self.feature_names],
|
|
84
|
+
}
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _is_pipeline(model) -> bool:
|
|
89
|
+
try:
|
|
90
|
+
from sklearn.pipeline import Pipeline
|
|
91
|
+
except ImportError: # pragma: no cover
|
|
92
|
+
return False
|
|
93
|
+
return isinstance(model, Pipeline)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _one_hot_widths(transformer, columns) -> Optional[List[int]]:
|
|
97
|
+
"""Number of output columns produced per input column by a one-hot encoder."""
|
|
98
|
+
widths = getattr(transformer, "_n_features_outs", None)
|
|
99
|
+
if isinstance(widths, list) and len(widths) == len(columns):
|
|
100
|
+
return [int(w) for w in widths]
|
|
101
|
+
|
|
102
|
+
categories = getattr(transformer, "categories_", None)
|
|
103
|
+
if categories is None or len(categories) != len(columns):
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
drop_idx = getattr(transformer, "drop_idx_", None)
|
|
107
|
+
widths = []
|
|
108
|
+
for position, values in enumerate(categories):
|
|
109
|
+
width = len(values)
|
|
110
|
+
if drop_idx is not None and drop_idx[position] is not None:
|
|
111
|
+
width -= 1
|
|
112
|
+
widths.append(int(width))
|
|
113
|
+
return widths
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def feature_groups_from_encoder(encoder) -> Dict[str, List[str]]:
|
|
117
|
+
"""Recover logical feature blocks from a fitted ``ColumnTransformer``.
|
|
118
|
+
|
|
119
|
+
The mapping is derived from the transformer structure rather than from
|
|
120
|
+
column name patterns, so renaming conventions such as
|
|
121
|
+
``verbose_feature_names_out`` do not affect the result. Use it when the
|
|
122
|
+
estimator was fitted on already encoded data.
|
|
123
|
+
"""
|
|
124
|
+
if not hasattr(encoder, "transformers_"):
|
|
125
|
+
raise TypeError(
|
|
126
|
+
"feature_groups_from_encoder expects a fitted ColumnTransformer."
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
output_names = list(encoder.get_feature_names_out())
|
|
130
|
+
groups: Dict[str, List[str]] = {}
|
|
131
|
+
cursor = 0
|
|
132
|
+
|
|
133
|
+
for name, transformer, columns in encoder.transformers_:
|
|
134
|
+
if transformer == "drop" or transformer is None:
|
|
135
|
+
continue
|
|
136
|
+
if isinstance(columns, str):
|
|
137
|
+
columns = [columns]
|
|
138
|
+
columns = list(columns)
|
|
139
|
+
|
|
140
|
+
if transformer == "passthrough":
|
|
141
|
+
widths = [1] * len(columns)
|
|
142
|
+
else:
|
|
143
|
+
widths = _one_hot_widths(transformer, columns)
|
|
144
|
+
if widths is None:
|
|
145
|
+
try:
|
|
146
|
+
produced = list(transformer.get_feature_names_out(columns))
|
|
147
|
+
except Exception: # pragma: no cover - permissive fallback
|
|
148
|
+
produced = []
|
|
149
|
+
if len(produced) == len(columns):
|
|
150
|
+
widths = [1] * len(columns)
|
|
151
|
+
else:
|
|
152
|
+
warnings.warn(
|
|
153
|
+
f"The outputs of transformer '{name}' cannot be attributed to "
|
|
154
|
+
"individual input columns, so the whole block is treated as a "
|
|
155
|
+
"single logical feature.",
|
|
156
|
+
stacklevel=2,
|
|
157
|
+
)
|
|
158
|
+
total = len(produced) if produced else len(output_names) - cursor
|
|
159
|
+
groups[name] = output_names[cursor : cursor + total]
|
|
160
|
+
cursor += total
|
|
161
|
+
continue
|
|
162
|
+
|
|
163
|
+
for column, width in zip(columns, widths):
|
|
164
|
+
groups[str(column)] = output_names[cursor : cursor + width]
|
|
165
|
+
cursor += width
|
|
166
|
+
|
|
167
|
+
if cursor != len(output_names):
|
|
168
|
+
warnings.warn(
|
|
169
|
+
"Some encoded columns were not attributed to a logical feature. "
|
|
170
|
+
"Pass feature_groups explicitly if the mapping matters.",
|
|
171
|
+
stacklevel=2,
|
|
172
|
+
)
|
|
173
|
+
return groups
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def build_feature_space(
|
|
177
|
+
model,
|
|
178
|
+
X: pd.DataFrame,
|
|
179
|
+
encoder=None,
|
|
180
|
+
feature_groups: Optional[Dict] = None,
|
|
181
|
+
) -> FeatureSpace:
|
|
182
|
+
"""Select the marginalization space that matches the supplied objects."""
|
|
183
|
+
columns = list(X.columns)
|
|
184
|
+
|
|
185
|
+
if feature_groups is not None:
|
|
186
|
+
if feature_groups == "auto":
|
|
187
|
+
if encoder is None:
|
|
188
|
+
raise ValueError(
|
|
189
|
+
"feature_groups='auto' requires the fitted encoder that produced "
|
|
190
|
+
"the columns of X."
|
|
191
|
+
)
|
|
192
|
+
feature_groups = feature_groups_from_encoder(encoder)
|
|
193
|
+
|
|
194
|
+
groups: Dict[str, List] = {}
|
|
195
|
+
assigned = set()
|
|
196
|
+
for feature, block in feature_groups.items():
|
|
197
|
+
block = [block] if isinstance(block, str) else list(block)
|
|
198
|
+
unknown = [c for c in block if c not in X.columns]
|
|
199
|
+
if unknown:
|
|
200
|
+
raise ValueError(
|
|
201
|
+
f"Feature group '{feature}' refers to columns that are absent "
|
|
202
|
+
f"from X. First unknown column, {unknown[0]}."
|
|
203
|
+
)
|
|
204
|
+
overlap = assigned.intersection(block)
|
|
205
|
+
if overlap:
|
|
206
|
+
raise ValueError(
|
|
207
|
+
f"Column '{sorted(overlap)[0]}' belongs to more than one feature "
|
|
208
|
+
"group, which makes the marginalization order ambiguous."
|
|
209
|
+
)
|
|
210
|
+
groups[str(feature)] = block
|
|
211
|
+
assigned.update(block)
|
|
212
|
+
|
|
213
|
+
for column in columns:
|
|
214
|
+
if column not in assigned:
|
|
215
|
+
groups[str(column)] = [column]
|
|
216
|
+
return FeatureSpace(groups, transform=None, path="feature_groups")
|
|
217
|
+
|
|
218
|
+
if encoder is not None:
|
|
219
|
+
if not hasattr(encoder, "transform"):
|
|
220
|
+
raise TypeError("The encoder must be fitted and expose a transform method.")
|
|
221
|
+
try:
|
|
222
|
+
output_names = list(encoder.get_feature_names_out())
|
|
223
|
+
except Exception: # pragma: no cover - encoders without name support
|
|
224
|
+
output_names = None
|
|
225
|
+
groups = {str(c): [c] for c in columns}
|
|
226
|
+
return FeatureSpace(
|
|
227
|
+
groups,
|
|
228
|
+
transform=encoder.transform,
|
|
229
|
+
output_feature_names=output_names,
|
|
230
|
+
path="encoder",
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
groups = {str(c): [c] for c in columns}
|
|
234
|
+
path = "pipeline" if _is_pipeline(model) else "identity"
|
|
235
|
+
return FeatureSpace(groups, transform=None, path=path)
|