sofi-tabular 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sofi/__init__.py ADDED
@@ -0,0 +1,51 @@
1
+ """SOFI, Sparseness Optimized Feature Importance.
2
+
3
+ A model agnostic post hoc explainer that searches for a ranking of features
4
+ whose cumulative marginalization degrades the response of the model as fast as
5
+ possible. The objective is the degradation score, namely the area between the
6
+ LeRF and the MoRF perturbation curves, which rewards sparsity and correctness
7
+ at once. The search works on the raw response of the model, and normalization
8
+ is applied afterwards, for reporting and for drawing alone.
9
+ Classification and regression are both supported, and the search is a hill
10
+ climbing procedure whose operator swaps two randomly selected ranking
11
+ positions.
12
+ """
13
+
14
+ from .encoding import FeatureSpace, feature_groups_from_encoder
15
+ from .explainer import SOFIExplainer
16
+ from .explanation import SOFIExplanation, aggregate_explanations, noise_onset
17
+ from .marginalization import Marginalizer
18
+ from .objective import Objective, RankingEvaluation, curve_auc, degradation_score
19
+ from .plotting import (
20
+ LERF_COLOR,
21
+ MORF_COLOR,
22
+ plot_degradation_curve,
23
+ plot_explanation_grid,
24
+ set_curve_colors,
25
+ set_plot_style,
26
+ )
27
+ from .search import hill_climbing
28
+ from .utils import select_reliable_instances
29
+
30
+ __version__ = "1.0.0"
31
+
32
+ __all__ = [
33
+ "SOFIExplainer",
34
+ "SOFIExplanation",
35
+ "FeatureSpace",
36
+ "Marginalizer",
37
+ "Objective",
38
+ "aggregate_explanations",
39
+ "RankingEvaluation",
40
+ "curve_auc",
41
+ "degradation_score",
42
+ "feature_groups_from_encoder",
43
+ "hill_climbing",
44
+ "noise_onset",
45
+ "plot_degradation_curve",
46
+ "set_curve_colors",
47
+ "plot_explanation_grid",
48
+ "set_plot_style",
49
+ "select_reliable_instances",
50
+ "__version__",
51
+ ]
sofi/encoding.py ADDED
@@ -0,0 +1,235 @@
1
+ """Resolution of the space in which features are marginalized.
2
+
3
+ SOFI perturbs whole features, never isolated columns of a one-hot block. The
4
+ classes below make that guarantee explicit through a mapping from a logical
5
+ feature to the set of columns that represent it, together with the
6
+ transformation that turns a perturbed frame into model input.
7
+
8
+ Three usage paths are supported.
9
+
10
+ 1. A ``Pipeline`` that carries the encoder and the estimator. Perturbation
11
+ happens in the original feature space and the encoding never leaks into the
12
+ explainer. This path is the recommended one.
13
+ 2. A fitted encoder passed apart from the estimator. Perturbation still happens
14
+ in the original space, and SOFI applies the encoder before each prediction.
15
+ 3. Data that is already encoded, together with a ``feature_groups`` mapping
16
+ that states which columns belong to the same logical feature.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import warnings
22
+ from typing import Callable, Dict, List, Optional, Sequence
23
+
24
+ import numpy as np
25
+ import pandas as pd
26
+
27
+ __all__ = ["FeatureSpace", "build_feature_space", "feature_groups_from_encoder"]
28
+
29
+
30
+ class FeatureSpace:
31
+ """Mapping between logical features and the columns SOFI perturbs.
32
+
33
+ Attributes
34
+ ----------
35
+ feature_names : list of str
36
+ Names of the logical features, in the order used internally.
37
+ groups : dict
38
+ Maps every logical feature onto the list of columns that represent it.
39
+ transform : callable
40
+ Turns a perturbed frame into the container consumed by the estimator.
41
+ output_feature_names : list of str or None
42
+ Column names produced by ``transform`` when it returns a bare array.
43
+ """
44
+
45
+ def __init__(
46
+ self,
47
+ groups: Dict[str, List],
48
+ transform: Optional[Callable] = None,
49
+ output_feature_names: Optional[Sequence] = None,
50
+ path: str = "identity",
51
+ ):
52
+ self.groups = {str(k): list(v) for k, v in groups.items()}
53
+ self.feature_names = list(self.groups.keys())
54
+ self._transform = transform
55
+ self.output_feature_names = (
56
+ None if output_feature_names is None else list(output_feature_names)
57
+ )
58
+ self.path = path
59
+
60
+ def __len__(self) -> int:
61
+ return len(self.feature_names)
62
+
63
+ @property
64
+ def n_features(self) -> int:
65
+ return len(self.feature_names)
66
+
67
+ def columns_of(self, feature) -> List:
68
+ if isinstance(feature, (int, np.integer)):
69
+ feature = self.feature_names[int(feature)]
70
+ return self.groups[feature]
71
+
72
+ def transform(self, X: pd.DataFrame):
73
+ if self._transform is None:
74
+ return X
75
+ return self._transform(X)
76
+
77
+ def describe(self) -> pd.DataFrame:
78
+ """Return one row per logical feature with its block width."""
79
+ return pd.DataFrame(
80
+ {
81
+ "feature": self.feature_names,
82
+ "n_columns": [len(self.groups[f]) for f in self.feature_names],
83
+ "columns": [list(self.groups[f]) for f in self.feature_names],
84
+ }
85
+ )
86
+
87
+
88
+ def _is_pipeline(model) -> bool:
89
+ try:
90
+ from sklearn.pipeline import Pipeline
91
+ except ImportError: # pragma: no cover
92
+ return False
93
+ return isinstance(model, Pipeline)
94
+
95
+
96
+ def _one_hot_widths(transformer, columns) -> Optional[List[int]]:
97
+ """Number of output columns produced per input column by a one-hot encoder."""
98
+ widths = getattr(transformer, "_n_features_outs", None)
99
+ if isinstance(widths, list) and len(widths) == len(columns):
100
+ return [int(w) for w in widths]
101
+
102
+ categories = getattr(transformer, "categories_", None)
103
+ if categories is None or len(categories) != len(columns):
104
+ return None
105
+
106
+ drop_idx = getattr(transformer, "drop_idx_", None)
107
+ widths = []
108
+ for position, values in enumerate(categories):
109
+ width = len(values)
110
+ if drop_idx is not None and drop_idx[position] is not None:
111
+ width -= 1
112
+ widths.append(int(width))
113
+ return widths
114
+
115
+
116
+ def feature_groups_from_encoder(encoder) -> Dict[str, List[str]]:
117
+ """Recover logical feature blocks from a fitted ``ColumnTransformer``.
118
+
119
+ The mapping is derived from the transformer structure rather than from
120
+ column name patterns, so renaming conventions such as
121
+ ``verbose_feature_names_out`` do not affect the result. Use it when the
122
+ estimator was fitted on already encoded data.
123
+ """
124
+ if not hasattr(encoder, "transformers_"):
125
+ raise TypeError(
126
+ "feature_groups_from_encoder expects a fitted ColumnTransformer."
127
+ )
128
+
129
+ output_names = list(encoder.get_feature_names_out())
130
+ groups: Dict[str, List[str]] = {}
131
+ cursor = 0
132
+
133
+ for name, transformer, columns in encoder.transformers_:
134
+ if transformer == "drop" or transformer is None:
135
+ continue
136
+ if isinstance(columns, str):
137
+ columns = [columns]
138
+ columns = list(columns)
139
+
140
+ if transformer == "passthrough":
141
+ widths = [1] * len(columns)
142
+ else:
143
+ widths = _one_hot_widths(transformer, columns)
144
+ if widths is None:
145
+ try:
146
+ produced = list(transformer.get_feature_names_out(columns))
147
+ except Exception: # pragma: no cover - permissive fallback
148
+ produced = []
149
+ if len(produced) == len(columns):
150
+ widths = [1] * len(columns)
151
+ else:
152
+ warnings.warn(
153
+ f"The outputs of transformer '{name}' cannot be attributed to "
154
+ "individual input columns, so the whole block is treated as a "
155
+ "single logical feature.",
156
+ stacklevel=2,
157
+ )
158
+ total = len(produced) if produced else len(output_names) - cursor
159
+ groups[name] = output_names[cursor : cursor + total]
160
+ cursor += total
161
+ continue
162
+
163
+ for column, width in zip(columns, widths):
164
+ groups[str(column)] = output_names[cursor : cursor + width]
165
+ cursor += width
166
+
167
+ if cursor != len(output_names):
168
+ warnings.warn(
169
+ "Some encoded columns were not attributed to a logical feature. "
170
+ "Pass feature_groups explicitly if the mapping matters.",
171
+ stacklevel=2,
172
+ )
173
+ return groups
174
+
175
+
176
+ def build_feature_space(
177
+ model,
178
+ X: pd.DataFrame,
179
+ encoder=None,
180
+ feature_groups: Optional[Dict] = None,
181
+ ) -> FeatureSpace:
182
+ """Select the marginalization space that matches the supplied objects."""
183
+ columns = list(X.columns)
184
+
185
+ if feature_groups is not None:
186
+ if feature_groups == "auto":
187
+ if encoder is None:
188
+ raise ValueError(
189
+ "feature_groups='auto' requires the fitted encoder that produced "
190
+ "the columns of X."
191
+ )
192
+ feature_groups = feature_groups_from_encoder(encoder)
193
+
194
+ groups: Dict[str, List] = {}
195
+ assigned = set()
196
+ for feature, block in feature_groups.items():
197
+ block = [block] if isinstance(block, str) else list(block)
198
+ unknown = [c for c in block if c not in X.columns]
199
+ if unknown:
200
+ raise ValueError(
201
+ f"Feature group '{feature}' refers to columns that are absent "
202
+ f"from X. First unknown column, {unknown[0]}."
203
+ )
204
+ overlap = assigned.intersection(block)
205
+ if overlap:
206
+ raise ValueError(
207
+ f"Column '{sorted(overlap)[0]}' belongs to more than one feature "
208
+ "group, which makes the marginalization order ambiguous."
209
+ )
210
+ groups[str(feature)] = block
211
+ assigned.update(block)
212
+
213
+ for column in columns:
214
+ if column not in assigned:
215
+ groups[str(column)] = [column]
216
+ return FeatureSpace(groups, transform=None, path="feature_groups")
217
+
218
+ if encoder is not None:
219
+ if not hasattr(encoder, "transform"):
220
+ raise TypeError("The encoder must be fitted and expose a transform method.")
221
+ try:
222
+ output_names = list(encoder.get_feature_names_out())
223
+ except Exception: # pragma: no cover - encoders without name support
224
+ output_names = None
225
+ groups = {str(c): [c] for c in columns}
226
+ return FeatureSpace(
227
+ groups,
228
+ transform=encoder.transform,
229
+ output_feature_names=output_names,
230
+ path="encoder",
231
+ )
232
+
233
+ groups = {str(c): [c] for c in columns}
234
+ path = "pipeline" if _is_pipeline(model) else "identity"
235
+ return FeatureSpace(groups, transform=None, path=path)