subspaceknn 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- subspaceknn/__init__.py +15 -0
- subspaceknn/_classifier.py +468 -0
- subspaceknn/_explanation.py +92 -0
- subspaceknn/plotting.py +275 -0
- subspaceknn/py.typed +0 -0
- subspaceknn-0.1.0.dist-info/METADATA +161 -0
- subspaceknn-0.1.0.dist-info/RECORD +9 -0
- subspaceknn-0.1.0.dist-info/WHEEL +4 -0
- subspaceknn-0.1.0.dist-info/licenses/LICENSE +21 -0
subspaceknn/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Interpretable k-nearest-neighbour classification on low-dimensional feature subspaces.
|
|
2
|
+
|
|
3
|
+
The package provides :class:`SubspaceKNNClassifier`, a scikit-learn compatible
|
|
4
|
+
classifier that fits one k-nearest-neighbour model per small subset of features,
|
|
5
|
+
ranks those subspaces by cross-validated performance, and combines the best of
|
|
6
|
+
them by weighted voting. Because every member of the ensemble lives in a space of
|
|
7
|
+
one, two or three features, each prediction can be explained by looking at the
|
|
8
|
+
neighbourhoods that produced it.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from subspaceknn._classifier import SubspaceKNNClassifier
|
|
12
|
+
from subspaceknn._explanation import Explanation, SubspaceVote
|
|
13
|
+
|
|
14
|
+
__all__ = ["Explanation", "SubspaceKNNClassifier", "SubspaceVote"]
|
|
15
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
"""Ensemble of k-nearest-neighbour classifiers on low-dimensional feature subspaces."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from itertools import combinations
|
|
7
|
+
from math import ceil, comb
|
|
8
|
+
from numbers import Integral
|
|
9
|
+
from typing import TYPE_CHECKING, Any, Literal, TypeGuard
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
from sklearn.base import BaseEstimator, ClassifierMixin
|
|
13
|
+
from sklearn.metrics import get_scorer
|
|
14
|
+
from sklearn.model_selection import StratifiedKFold, cross_val_score
|
|
15
|
+
from sklearn.neighbors import KNeighborsClassifier
|
|
16
|
+
from sklearn.utils.multiclass import check_classification_targets, unique_labels
|
|
17
|
+
from sklearn.utils.validation import check_is_fitted, validate_data
|
|
18
|
+
|
|
19
|
+
from subspaceknn._explanation import Explanation, SubspaceVote
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
from collections.abc import Callable, Sequence
|
|
23
|
+
|
|
24
|
+
from numpy.typing import ArrayLike, NDArray
|
|
25
|
+
from sklearn.model_selection import BaseCrossValidator
|
|
26
|
+
|
|
27
|
+
Voting = Literal["soft", "hard"]
|
|
28
|
+
Weighting = Literal["score", "uniform"]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class SubspaceKNNClassifier(ClassifierMixin, BaseEstimator): # type: ignore[misc]
|
|
32
|
+
"""Weighted vote of k-nearest-neighbour classifiers fitted on small feature subspaces.
|
|
33
|
+
|
|
34
|
+
The estimator enumerates every subset of ``subspace_size`` features (or every
|
|
35
|
+
subset of each size when a sequence of sizes is given), fits a
|
|
36
|
+
:class:`~sklearn.neighbors.KNeighborsClassifier` on each subset, and scores
|
|
37
|
+
it by cross-validation on the training data. The ``n_subspaces`` best
|
|
38
|
+
subspaces then vote on new samples, each weighted by its cross-validated
|
|
39
|
+
score. Because the members of the ensemble live in spaces of one, two or
|
|
40
|
+
three features, every prediction can be explained by looking at the
|
|
41
|
+
neighbourhoods that produced it; see :meth:`explain`.
|
|
42
|
+
|
|
43
|
+
The method generalises the interpretable kNN (ikNN) idea of Brett Kennedy,
|
|
44
|
+
which uses pairs of features only, to subspaces of any small size. This is
|
|
45
|
+
an independent implementation that shares no code with the original.
|
|
46
|
+
|
|
47
|
+
Parameters
|
|
48
|
+
----------
|
|
49
|
+
n_neighbors : int, default=5
|
|
50
|
+
Number of neighbours used by every subspace model.
|
|
51
|
+
subspace_size : int or sequence of int, default=2
|
|
52
|
+
Number of features in each subspace. A sequence enumerates subspaces of
|
|
53
|
+
every listed size. Sizes larger than the number of features are ignored.
|
|
54
|
+
n_subspaces : int or None, default=5
|
|
55
|
+
Number of best-scoring subspaces used for prediction. ``None`` uses every
|
|
56
|
+
candidate subspace.
|
|
57
|
+
max_candidates : int or None, default=100
|
|
58
|
+
Soft cap on the number of candidate subspaces that are cross-validated.
|
|
59
|
+
When the number of subsets exceeds it, features are first screened by the
|
|
60
|
+
cross-validated score of their one-dimensional model and only the
|
|
61
|
+
best-scoring features are combined, as many as keep the candidate count
|
|
62
|
+
within the cap. ``None`` evaluates every subset, which grows
|
|
63
|
+
combinatorially with the number of features.
|
|
64
|
+
voting : {"soft", "hard"}, default="soft"
|
|
65
|
+
``"soft"`` averages the class probabilities of the subspace models,
|
|
66
|
+
``"hard"`` averages their one-hot predictions.
|
|
67
|
+
weighting : {"score", "uniform"}, default="score"
|
|
68
|
+
``"score"`` weights each subspace by its cross-validated score (negative
|
|
69
|
+
scores are clipped to zero), ``"uniform"`` gives every subspace the same
|
|
70
|
+
weight.
|
|
71
|
+
cv : int or cross-validation generator, default=5
|
|
72
|
+
Cross-validation used to score subspaces. An integer selects stratified
|
|
73
|
+
k-fold with that many splits, reduced automatically when a class has fewer
|
|
74
|
+
samples than splits. When the training set is too small to
|
|
75
|
+
cross-validate at all, the resubstitution score on the training data is
|
|
76
|
+
used instead.
|
|
77
|
+
scoring : str or callable, default="f1_macro"
|
|
78
|
+
Any scikit-learn scorer. Score-based weighting assumes higher is better
|
|
79
|
+
and scores are non-negative.
|
|
80
|
+
knn_weights : {"uniform", "distance"}, default="uniform"
|
|
81
|
+
Neighbour weighting passed to the subspace models.
|
|
82
|
+
metric : str, default="minkowski"
|
|
83
|
+
Distance metric passed to the subspace models.
|
|
84
|
+
n_jobs : int or None, default=None
|
|
85
|
+
Parallelism for cross-validation, passed to
|
|
86
|
+
:func:`~sklearn.model_selection.cross_val_score`.
|
|
87
|
+
|
|
88
|
+
Attributes
|
|
89
|
+
----------
|
|
90
|
+
classes_ : ndarray of shape (n_classes,)
|
|
91
|
+
Class labels.
|
|
92
|
+
n_features_in_ : int
|
|
93
|
+
Number of features seen during :meth:`fit`.
|
|
94
|
+
feature_names_in_ : ndarray of shape (n_features_in_,)
|
|
95
|
+
Feature names, only when ``X`` had string column names.
|
|
96
|
+
screened_features_ : ndarray of shape (n_screened,)
|
|
97
|
+
Indices of the features retained after screening, all features when no
|
|
98
|
+
screening was necessary.
|
|
99
|
+
feature_screening_scores_ : ndarray of shape (n_features_in_,) or None
|
|
100
|
+
Cross-validated score of each feature's one-dimensional model, only when
|
|
101
|
+
screening took place.
|
|
102
|
+
candidate_subspaces_ : list of tuple of int
|
|
103
|
+
Every subspace that was cross-validated, in enumeration order.
|
|
104
|
+
candidate_scores_ : ndarray of shape (n_candidates,)
|
|
105
|
+
Cross-validated score of each candidate subspace.
|
|
106
|
+
subspaces_ : list of tuple of int
|
|
107
|
+
Subspaces used for prediction, from the best to the worst score.
|
|
108
|
+
subspace_scores_ : ndarray of shape (n_selected,)
|
|
109
|
+
Scores of the selected subspaces.
|
|
110
|
+
subspace_weights_ : ndarray of shape (n_selected,)
|
|
111
|
+
Normalised voting weights of the selected subspaces; they sum to one.
|
|
112
|
+
estimators_ : list of KNeighborsClassifier
|
|
113
|
+
Fitted subspace models, aligned with ``subspaces_``.
|
|
114
|
+
feature_scores_ : ndarray of shape (n_features_in_,)
|
|
115
|
+
Mean score of the selected subspaces containing each feature, zero for
|
|
116
|
+
features that appear in none. A coarse measure of feature relevance.
|
|
117
|
+
|
|
118
|
+
Examples
|
|
119
|
+
--------
|
|
120
|
+
>>> from sklearn.datasets import load_iris
|
|
121
|
+
>>> from subspaceknn import SubspaceKNNClassifier
|
|
122
|
+
>>> X, y = load_iris(return_X_y=True)
|
|
123
|
+
>>> clf = SubspaceKNNClassifier(subspace_size=(1, 2)).fit(X, y)
|
|
124
|
+
>>> clf.subspaces_[0]
|
|
125
|
+
(2, 3)
|
|
126
|
+
>>> clf.explain(X[:1])[0].votes[0].feature_names
|
|
127
|
+
('x2', 'x3')
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
def __init__(
|
|
131
|
+
self,
|
|
132
|
+
*,
|
|
133
|
+
n_neighbors: int = 5,
|
|
134
|
+
subspace_size: int | Sequence[int] = 2,
|
|
135
|
+
n_subspaces: int | None = 5,
|
|
136
|
+
max_candidates: int | None = 100,
|
|
137
|
+
voting: Voting = "soft",
|
|
138
|
+
weighting: Weighting = "score",
|
|
139
|
+
cv: int | BaseCrossValidator = 5,
|
|
140
|
+
scoring: str | Callable[..., float] = "f1_macro",
|
|
141
|
+
knn_weights: Literal["uniform", "distance"] = "uniform",
|
|
142
|
+
metric: str = "minkowski",
|
|
143
|
+
n_jobs: int | None = None,
|
|
144
|
+
) -> None:
|
|
145
|
+
self.n_neighbors = n_neighbors
|
|
146
|
+
self.subspace_size = subspace_size
|
|
147
|
+
self.n_subspaces = n_subspaces
|
|
148
|
+
self.max_candidates = max_candidates
|
|
149
|
+
self.voting = voting
|
|
150
|
+
self.weighting = weighting
|
|
151
|
+
self.cv = cv
|
|
152
|
+
self.scoring = scoring
|
|
153
|
+
self.knn_weights = knn_weights
|
|
154
|
+
self.metric = metric
|
|
155
|
+
self.n_jobs = n_jobs
|
|
156
|
+
|
|
157
|
+
# ------------------------------------------------------------------ fitting
|
|
158
|
+
|
|
159
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> SubspaceKNNClassifier:
|
|
160
|
+
"""Fit one subspace model per candidate subspace and select the best.
|
|
161
|
+
|
|
162
|
+
Parameters
|
|
163
|
+
----------
|
|
164
|
+
X : array-like of shape (n_samples, n_features)
|
|
165
|
+
Training data.
|
|
166
|
+
y : array-like of shape (n_samples,)
|
|
167
|
+
Class labels.
|
|
168
|
+
|
|
169
|
+
Returns
|
|
170
|
+
-------
|
|
171
|
+
self
|
|
172
|
+
The fitted estimator.
|
|
173
|
+
"""
|
|
174
|
+
X_arr, y_arr = validate_data(self, X, y, dtype="numeric")
|
|
175
|
+
check_classification_targets(y_arr)
|
|
176
|
+
self._check_hyperparameters()
|
|
177
|
+
n_samples, n_features = X_arr.shape
|
|
178
|
+
if self.n_neighbors > n_samples:
|
|
179
|
+
raise ValueError(
|
|
180
|
+
"Expected n_neighbors <= n_samples, got "
|
|
181
|
+
f"n_neighbors={self.n_neighbors} and n_samples={n_samples}.",
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
self.classes_ = unique_labels(y_arr)
|
|
185
|
+
y_encoded = np.searchsorted(self.classes_, y_arr)
|
|
186
|
+
sizes = self._subspace_sizes(n_features)
|
|
187
|
+
cv = self._cross_validator(y_encoded)
|
|
188
|
+
|
|
189
|
+
features = np.arange(n_features)
|
|
190
|
+
self.feature_screening_scores_: NDArray[np.float64] | None = None
|
|
191
|
+
total = sum(comb(n_features, size) for size in sizes)
|
|
192
|
+
if self.max_candidates is not None and total > self.max_candidates:
|
|
193
|
+
singletons = [(index,) for index in range(n_features)]
|
|
194
|
+
screening = self._score_subspaces(singletons, X_arr, y_encoded, cv)
|
|
195
|
+
self.feature_screening_scores_ = screening
|
|
196
|
+
keep = self._n_features_to_keep(sizes, n_features, self.max_candidates)
|
|
197
|
+
features = np.sort(np.argsort(-screening, kind="stable")[:keep])
|
|
198
|
+
self.screened_features_ = features
|
|
199
|
+
|
|
200
|
+
feature_list = [int(index) for index in features]
|
|
201
|
+
candidates = [combo for size in sizes for combo in combinations(feature_list, size)]
|
|
202
|
+
scores = self._score_subspaces(candidates, X_arr, y_encoded, cv)
|
|
203
|
+
self.candidate_subspaces_ = candidates
|
|
204
|
+
self.candidate_scores_ = scores
|
|
205
|
+
|
|
206
|
+
order = np.argsort(-scores, kind="stable")
|
|
207
|
+
if self.n_subspaces is not None:
|
|
208
|
+
order = order[: self.n_subspaces]
|
|
209
|
+
self.subspaces_ = [candidates[index] for index in order]
|
|
210
|
+
self.subspace_scores_ = scores[order]
|
|
211
|
+
self.subspace_weights_ = self._weights(self.subspace_scores_)
|
|
212
|
+
self.estimators_ = [
|
|
213
|
+
self._make_knn().fit(X_arr[:, list(subspace)], y_encoded)
|
|
214
|
+
for subspace in self.subspaces_
|
|
215
|
+
]
|
|
216
|
+
self.feature_scores_ = self._feature_scores(n_features)
|
|
217
|
+
return self
|
|
218
|
+
|
|
219
|
+
def _check_hyperparameters(self) -> None:
|
|
220
|
+
if not _is_positive_integer(self.n_neighbors):
|
|
221
|
+
raise ValueError(f"n_neighbors must be a positive integer, got {self.n_neighbors!r}.")
|
|
222
|
+
sizes = self._requested_sizes()
|
|
223
|
+
if not sizes or any(size < 1 for size in sizes):
|
|
224
|
+
raise ValueError(
|
|
225
|
+
"subspace_size must be a positive integer or a non-empty sequence of "
|
|
226
|
+
f"positive integers, got {self.subspace_size!r}.",
|
|
227
|
+
)
|
|
228
|
+
if self.n_subspaces is not None and not _is_positive_integer(self.n_subspaces):
|
|
229
|
+
raise ValueError(
|
|
230
|
+
f"n_subspaces must be None or a positive integer, got {self.n_subspaces!r}.",
|
|
231
|
+
)
|
|
232
|
+
if self.max_candidates is not None and not _is_positive_integer(self.max_candidates):
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"max_candidates must be None or a positive integer, got {self.max_candidates!r}.",
|
|
235
|
+
)
|
|
236
|
+
if self.voting not in ("soft", "hard"):
|
|
237
|
+
raise ValueError(f"voting must be 'soft' or 'hard', got {self.voting!r}.")
|
|
238
|
+
if self.weighting not in ("score", "uniform"):
|
|
239
|
+
raise ValueError(f"weighting must be 'score' or 'uniform', got {self.weighting!r}.")
|
|
240
|
+
if _is_integer(self.cv) and self.cv < 2: # noqa: PLR2004
|
|
241
|
+
raise ValueError(f"cv must be at least 2 when given as an integer, got {self.cv!r}.")
|
|
242
|
+
|
|
243
|
+
def _requested_sizes(self) -> list[int]:
|
|
244
|
+
value: object = self.subspace_size
|
|
245
|
+
if _is_integer(value):
|
|
246
|
+
return [value]
|
|
247
|
+
if isinstance(value, str) or not isinstance(value, Iterable):
|
|
248
|
+
return []
|
|
249
|
+
sizes = list(value)
|
|
250
|
+
if not all(_is_integer(size) for size in sizes):
|
|
251
|
+
return []
|
|
252
|
+
return [int(size) for size in sizes]
|
|
253
|
+
|
|
254
|
+
def _subspace_sizes(self, n_features: int) -> list[int]:
|
|
255
|
+
requested = sorted(set(self._requested_sizes()))
|
|
256
|
+
sizes = [size for size in requested if size <= n_features]
|
|
257
|
+
if not sizes:
|
|
258
|
+
raise ValueError(
|
|
259
|
+
f"No feature subspace of size {requested} fits in n_features={n_features}; "
|
|
260
|
+
"reduce subspace_size or add features.",
|
|
261
|
+
)
|
|
262
|
+
return sizes
|
|
263
|
+
|
|
264
|
+
def _cross_validator(self, y_encoded: NDArray[np.intp]) -> BaseCrossValidator | None:
|
|
265
|
+
"""Return the splitter used to score subspaces, or None for resubstitution scoring."""
|
|
266
|
+
if not _is_integer(self.cv):
|
|
267
|
+
return self.cv
|
|
268
|
+
n_samples = len(y_encoded)
|
|
269
|
+
smallest_class = int(np.bincount(y_encoded).min())
|
|
270
|
+
n_splits = min(self.cv, smallest_class)
|
|
271
|
+
if n_splits < 2: # noqa: PLR2004
|
|
272
|
+
return None
|
|
273
|
+
smallest_training_fold = n_samples - ceil(n_samples / n_splits)
|
|
274
|
+
if smallest_training_fold < self.n_neighbors:
|
|
275
|
+
return None
|
|
276
|
+
return StratifiedKFold(n_splits=n_splits)
|
|
277
|
+
|
|
278
|
+
def _make_knn(self) -> KNeighborsClassifier:
|
|
279
|
+
return KNeighborsClassifier(
|
|
280
|
+
n_neighbors=self.n_neighbors,
|
|
281
|
+
weights=self.knn_weights,
|
|
282
|
+
metric=self.metric,
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
def _score_subspaces(
|
|
286
|
+
self,
|
|
287
|
+
subspaces: Sequence[tuple[int, ...]],
|
|
288
|
+
X: NDArray[np.float64],
|
|
289
|
+
y_encoded: NDArray[np.intp],
|
|
290
|
+
cv: BaseCrossValidator | None,
|
|
291
|
+
) -> NDArray[np.float64]:
|
|
292
|
+
if cv is None:
|
|
293
|
+
return self._resubstitution_scores(subspaces, X, y_encoded)
|
|
294
|
+
scores = np.empty(len(subspaces), dtype=np.float64)
|
|
295
|
+
for position, subspace in enumerate(subspaces):
|
|
296
|
+
fold_scores = cross_val_score(
|
|
297
|
+
self._make_knn(),
|
|
298
|
+
X[:, list(subspace)],
|
|
299
|
+
y_encoded,
|
|
300
|
+
cv=cv,
|
|
301
|
+
scoring=self.scoring,
|
|
302
|
+
n_jobs=self.n_jobs,
|
|
303
|
+
error_score="raise",
|
|
304
|
+
)
|
|
305
|
+
scores[position] = float(np.mean(fold_scores))
|
|
306
|
+
return scores
|
|
307
|
+
|
|
308
|
+
def _resubstitution_scores(
|
|
309
|
+
self,
|
|
310
|
+
subspaces: Sequence[tuple[int, ...]],
|
|
311
|
+
X: NDArray[np.float64],
|
|
312
|
+
y_encoded: NDArray[np.intp],
|
|
313
|
+
) -> NDArray[np.float64]:
|
|
314
|
+
"""Score each subspace on its own training data.
|
|
315
|
+
|
|
316
|
+
Used only when the training set is too small to cross-validate.
|
|
317
|
+
"""
|
|
318
|
+
scorer = get_scorer(self.scoring)
|
|
319
|
+
scores = np.empty(len(subspaces), dtype=np.float64)
|
|
320
|
+
for position, subspace in enumerate(subspaces):
|
|
321
|
+
X_sub = X[:, list(subspace)]
|
|
322
|
+
model = self._make_knn().fit(X_sub, y_encoded)
|
|
323
|
+
scores[position] = float(scorer(model, X_sub, y_encoded))
|
|
324
|
+
return scores
|
|
325
|
+
|
|
326
|
+
@staticmethod
|
|
327
|
+
def _n_features_to_keep(sizes: Sequence[int], n_features: int, max_candidates: int) -> int:
|
|
328
|
+
"""Largest feature count whose subspace count stays within ``max_candidates``."""
|
|
329
|
+
floor = max(sizes)
|
|
330
|
+
keep = floor
|
|
331
|
+
for count in range(floor, n_features + 1):
|
|
332
|
+
if sum(comb(count, size) for size in sizes) <= max_candidates:
|
|
333
|
+
keep = count
|
|
334
|
+
else:
|
|
335
|
+
break
|
|
336
|
+
return keep
|
|
337
|
+
|
|
338
|
+
def _weights(self, scores: NDArray[np.float64]) -> NDArray[np.float64]:
|
|
339
|
+
if self.weighting == "score":
|
|
340
|
+
clipped = np.clip(scores, 0.0, None)
|
|
341
|
+
if clipped.sum() > 0.0:
|
|
342
|
+
return np.asarray(clipped / clipped.sum(), dtype=np.float64)
|
|
343
|
+
return np.full(len(scores), 1.0 / len(scores), dtype=np.float64)
|
|
344
|
+
|
|
345
|
+
def _feature_scores(self, n_features: int) -> NDArray[np.float64]:
|
|
346
|
+
totals = np.zeros(n_features, dtype=np.float64)
|
|
347
|
+
counts = np.zeros(n_features, dtype=np.float64)
|
|
348
|
+
for subspace, score in zip(self.subspaces_, self.subspace_scores_, strict=True):
|
|
349
|
+
for feature in subspace:
|
|
350
|
+
totals[feature] += score
|
|
351
|
+
counts[feature] += 1.0
|
|
352
|
+
return np.asarray(np.divide(totals, counts, out=np.zeros_like(totals), where=counts > 0))
|
|
353
|
+
|
|
354
|
+
# --------------------------------------------------------------- predicting
|
|
355
|
+
|
|
356
|
+
def predict_proba(self, X: ArrayLike) -> NDArray[np.float64]:
|
|
357
|
+
"""Return the weighted average of the subspace models' class probabilities.
|
|
358
|
+
|
|
359
|
+
Parameters
|
|
360
|
+
----------
|
|
361
|
+
X : array-like of shape (n_samples, n_features)
|
|
362
|
+
Samples to classify.
|
|
363
|
+
|
|
364
|
+
Returns
|
|
365
|
+
-------
|
|
366
|
+
ndarray of shape (n_samples, n_classes)
|
|
367
|
+
Class probabilities aligned with ``classes_``; each row sums to one.
|
|
368
|
+
"""
|
|
369
|
+
check_is_fitted(self)
|
|
370
|
+
X_arr = validate_data(self, X, reset=False, dtype="numeric")
|
|
371
|
+
votes = self._subspace_votes(X_arr)
|
|
372
|
+
proba = np.tensordot(self.subspace_weights_, votes, axes=1)
|
|
373
|
+
return self._normalise(proba)
|
|
374
|
+
|
|
375
|
+
def predict(self, X: ArrayLike) -> NDArray[Any]:
|
|
376
|
+
"""Return the class with the highest ensemble probability for each sample.
|
|
377
|
+
|
|
378
|
+
Parameters
|
|
379
|
+
----------
|
|
380
|
+
X : array-like of shape (n_samples, n_features)
|
|
381
|
+
Samples to classify.
|
|
382
|
+
|
|
383
|
+
Returns
|
|
384
|
+
-------
|
|
385
|
+
ndarray of shape (n_samples,)
|
|
386
|
+
Predicted class labels.
|
|
387
|
+
"""
|
|
388
|
+
proba = self.predict_proba(X)
|
|
389
|
+
return np.asarray(self.classes_[np.argmax(proba, axis=1)])
|
|
390
|
+
|
|
391
|
+
def explain(self, X: ArrayLike) -> list[Explanation]:
|
|
392
|
+
"""Explain the prediction for every sample in ``X``.
|
|
393
|
+
|
|
394
|
+
Parameters
|
|
395
|
+
----------
|
|
396
|
+
X : array-like of shape (n_samples, n_features)
|
|
397
|
+
Samples to explain.
|
|
398
|
+
|
|
399
|
+
Returns
|
|
400
|
+
-------
|
|
401
|
+
list of Explanation
|
|
402
|
+
One explanation per sample, each listing the vote of every subspace
|
|
403
|
+
used by the ensemble, ordered from the highest to the lowest weight.
|
|
404
|
+
"""
|
|
405
|
+
check_is_fitted(self)
|
|
406
|
+
X_arr = validate_data(self, X, reset=False, dtype="numeric")
|
|
407
|
+
votes = self._subspace_votes(X_arr)
|
|
408
|
+
proba = self._normalise(np.tensordot(self.subspace_weights_, votes, axes=1))
|
|
409
|
+
names = self._feature_names()
|
|
410
|
+
explanations: list[Explanation] = []
|
|
411
|
+
for row in range(X_arr.shape[0]):
|
|
412
|
+
row_votes = tuple(
|
|
413
|
+
SubspaceVote(
|
|
414
|
+
features=subspace,
|
|
415
|
+
feature_names=tuple(names[index] for index in subspace),
|
|
416
|
+
score=float(score),
|
|
417
|
+
weight=float(weight),
|
|
418
|
+
probabilities=votes[position, row],
|
|
419
|
+
prediction=self.classes_[int(np.argmax(votes[position, row]))],
|
|
420
|
+
)
|
|
421
|
+
for position, (subspace, score, weight) in enumerate(
|
|
422
|
+
zip(
|
|
423
|
+
self.subspaces_, self.subspace_scores_, self.subspace_weights_, strict=True
|
|
424
|
+
),
|
|
425
|
+
)
|
|
426
|
+
)
|
|
427
|
+
explanations.append(
|
|
428
|
+
Explanation(
|
|
429
|
+
prediction=self.classes_[int(np.argmax(proba[row]))],
|
|
430
|
+
probabilities=proba[row],
|
|
431
|
+
classes=self.classes_,
|
|
432
|
+
votes=row_votes,
|
|
433
|
+
),
|
|
434
|
+
)
|
|
435
|
+
return explanations
|
|
436
|
+
|
|
437
|
+
def _subspace_votes(self, X: NDArray[np.float64]) -> NDArray[np.float64]:
|
|
438
|
+
"""Return an array of shape (n_selected, n_samples, n_classes) with each subspace's vote."""
|
|
439
|
+
n_classes = len(self.classes_)
|
|
440
|
+
votes = np.empty((len(self.estimators_), X.shape[0], n_classes), dtype=np.float64)
|
|
441
|
+
for position, (estimator, subspace) in enumerate(
|
|
442
|
+
zip(self.estimators_, self.subspaces_, strict=True),
|
|
443
|
+
):
|
|
444
|
+
X_sub = X[:, list(subspace)]
|
|
445
|
+
if self.voting == "hard":
|
|
446
|
+
votes[position] = np.eye(n_classes)[estimator.predict(X_sub)]
|
|
447
|
+
else:
|
|
448
|
+
votes[position] = estimator.predict_proba(X_sub)
|
|
449
|
+
return votes
|
|
450
|
+
|
|
451
|
+
def _normalise(self, proba: NDArray[np.float64]) -> NDArray[np.float64]:
|
|
452
|
+
row_sums = proba.sum(axis=1, keepdims=True)
|
|
453
|
+
uniform = np.full_like(proba, 1.0 / proba.shape[1])
|
|
454
|
+
return np.asarray(np.divide(proba, row_sums, out=uniform, where=row_sums > 0))
|
|
455
|
+
|
|
456
|
+
def _feature_names(self) -> list[str]:
|
|
457
|
+
names = getattr(self, "feature_names_in_", None)
|
|
458
|
+
if names is None:
|
|
459
|
+
return [f"x{index}" for index in range(self.n_features_in_)]
|
|
460
|
+
return [str(name) for name in names]
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _is_integer(value: object) -> TypeGuard[int]:
|
|
464
|
+
return isinstance(value, Integral) and not isinstance(value, bool)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _is_positive_integer(value: object) -> TypeGuard[int]:
|
|
468
|
+
return _is_integer(value) and value > 0
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Structured explanations of individual predictions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
import numpy as np
|
|
10
|
+
from numpy.typing import NDArray
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class SubspaceVote:
|
|
15
|
+
"""The contribution of one feature subspace to a single prediction.
|
|
16
|
+
|
|
17
|
+
Attributes
|
|
18
|
+
----------
|
|
19
|
+
features : tuple of int
|
|
20
|
+
Column indices of the features that span the subspace.
|
|
21
|
+
feature_names : tuple of str
|
|
22
|
+
Names of those features, taken from ``feature_names_in_`` when the
|
|
23
|
+
estimator was fitted on a data frame and ``x<i>`` otherwise.
|
|
24
|
+
score : float
|
|
25
|
+
Cross-validated score of the subspace on the training data.
|
|
26
|
+
weight : float
|
|
27
|
+
Normalised weight of the subspace in the ensemble vote. Weights sum to one
|
|
28
|
+
across the subspaces used for prediction.
|
|
29
|
+
probabilities : ndarray of shape (n_classes,)
|
|
30
|
+
Class probabilities produced by this subspace alone, aligned with
|
|
31
|
+
``Explanation.classes``. Under hard voting this is a one-hot vector.
|
|
32
|
+
prediction : Any
|
|
33
|
+
Class label predicted by this subspace alone.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
features: tuple[int, ...]
|
|
37
|
+
feature_names: tuple[str, ...]
|
|
38
|
+
score: float
|
|
39
|
+
weight: float
|
|
40
|
+
probabilities: NDArray[np.float64]
|
|
41
|
+
prediction: Any
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class Explanation:
|
|
46
|
+
"""How the ensemble arrived at the prediction for one sample.
|
|
47
|
+
|
|
48
|
+
Attributes
|
|
49
|
+
----------
|
|
50
|
+
prediction : Any
|
|
51
|
+
Class label predicted by the ensemble.
|
|
52
|
+
probabilities : ndarray of shape (n_classes,)
|
|
53
|
+
Ensemble class probabilities, the weighted average of the votes.
|
|
54
|
+
classes : ndarray of shape (n_classes,)
|
|
55
|
+
Class labels, in the order used by ``probabilities``.
|
|
56
|
+
votes : tuple of SubspaceVote
|
|
57
|
+
One entry per subspace used for prediction, ordered from the highest to
|
|
58
|
+
the lowest weight.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
prediction: Any
|
|
62
|
+
probabilities: NDArray[np.float64]
|
|
63
|
+
classes: NDArray[Any]
|
|
64
|
+
votes: tuple[SubspaceVote, ...]
|
|
65
|
+
|
|
66
|
+
def agreement(self) -> float:
|
|
67
|
+
"""Return the weighted fraction of subspaces that voted for the ensemble prediction."""
|
|
68
|
+
return float(
|
|
69
|
+
sum(vote.weight for vote in self.votes if vote.prediction == self.prediction),
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
def to_records(self) -> list[dict[str, Any]]:
|
|
73
|
+
"""Return the votes as plain dictionaries, one per subspace.
|
|
74
|
+
|
|
75
|
+
The result is suitable for ``pandas.DataFrame.from_records`` and contains,
|
|
76
|
+
for every vote, the feature names, the subspace score and weight, the
|
|
77
|
+
subspace prediction, whether it agrees with the ensemble, and one
|
|
78
|
+
``p(<class>)`` column per class.
|
|
79
|
+
"""
|
|
80
|
+
records: list[dict[str, Any]] = []
|
|
81
|
+
for vote in self.votes:
|
|
82
|
+
record: dict[str, Any] = {
|
|
83
|
+
"features": ", ".join(vote.feature_names),
|
|
84
|
+
"score": vote.score,
|
|
85
|
+
"weight": vote.weight,
|
|
86
|
+
"prediction": vote.prediction,
|
|
87
|
+
"agrees": bool(vote.prediction == self.prediction),
|
|
88
|
+
}
|
|
89
|
+
for label, probability in zip(self.classes, vote.probabilities, strict=True):
|
|
90
|
+
record[f"p({label})"] = float(probability)
|
|
91
|
+
records.append(record)
|
|
92
|
+
return records
|
subspaceknn/plotting.py
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""Visualise the subspaces of a fitted :class:`~subspaceknn.SubspaceKNNClassifier`.
|
|
2
|
+
|
|
3
|
+
This module needs matplotlib, which is an optional dependency::
|
|
4
|
+
|
|
5
|
+
pip install "subspaceknn[plot]"
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import TYPE_CHECKING, Any, cast
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
from sklearn.utils.validation import check_is_fitted, validate_data
|
|
14
|
+
|
|
15
|
+
if TYPE_CHECKING:
|
|
16
|
+
from collections.abc import Sequence
|
|
17
|
+
|
|
18
|
+
from matplotlib.axes import Axes
|
|
19
|
+
from matplotlib.figure import Figure
|
|
20
|
+
from mpl_toolkits.mplot3d import Axes3D
|
|
21
|
+
from numpy.typing import ArrayLike, NDArray
|
|
22
|
+
from sklearn.neighbors import KNeighborsClassifier
|
|
23
|
+
|
|
24
|
+
from subspaceknn._classifier import SubspaceKNNClassifier
|
|
25
|
+
|
|
26
|
+
_PALETTE = (
|
|
27
|
+
"tab:blue",
|
|
28
|
+
"tab:orange",
|
|
29
|
+
"tab:green",
|
|
30
|
+
"tab:red",
|
|
31
|
+
"tab:purple",
|
|
32
|
+
"tab:brown",
|
|
33
|
+
"tab:pink",
|
|
34
|
+
"tab:gray",
|
|
35
|
+
"tab:olive",
|
|
36
|
+
"tab:cyan",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def plot_subspaces(
|
|
41
|
+
estimator: SubspaceKNNClassifier,
|
|
42
|
+
X: ArrayLike,
|
|
43
|
+
y: ArrayLike,
|
|
44
|
+
*,
|
|
45
|
+
sample: ArrayLike | None = None,
|
|
46
|
+
n_subspaces: int | None = None,
|
|
47
|
+
grid_resolution: int = 100,
|
|
48
|
+
panel_size: float = 4.0,
|
|
49
|
+
) -> Figure:
|
|
50
|
+
"""Draw the training data in the best subspaces, with decision regions where possible.
|
|
51
|
+
|
|
52
|
+
Parameters
|
|
53
|
+
----------
|
|
54
|
+
estimator : SubspaceKNNClassifier
|
|
55
|
+
A fitted estimator.
|
|
56
|
+
X : array-like of shape (n_samples, n_features)
|
|
57
|
+
Data to draw, usually the training data.
|
|
58
|
+
y : array-like of shape (n_samples,)
|
|
59
|
+
Class labels of ``X``.
|
|
60
|
+
sample : array-like of shape (n_features,), optional
|
|
61
|
+
A single sample to highlight with a star in every panel, for example a
|
|
62
|
+
test point whose prediction is being explained.
|
|
63
|
+
n_subspaces : int, optional
|
|
64
|
+
Number of panels, from the best subspace onwards. Defaults to all
|
|
65
|
+
subspaces used for prediction.
|
|
66
|
+
grid_resolution : int, default=100
|
|
67
|
+
Number of grid points per axis for the decision regions of one- and
|
|
68
|
+
two-dimensional subspaces.
|
|
69
|
+
panel_size : float, default=4.0
|
|
70
|
+
Width and height of each panel in inches.
|
|
71
|
+
|
|
72
|
+
Returns
|
|
73
|
+
-------
|
|
74
|
+
matplotlib.figure.Figure
|
|
75
|
+
The figure; it is not shown or saved.
|
|
76
|
+
|
|
77
|
+
Raises
|
|
78
|
+
------
|
|
79
|
+
ImportError
|
|
80
|
+
If matplotlib is not installed.
|
|
81
|
+
"""
|
|
82
|
+
try:
|
|
83
|
+
import matplotlib.pyplot as plt # noqa: PLC0415
|
|
84
|
+
except ImportError as error: # pragma: no cover - exercised only without matplotlib
|
|
85
|
+
raise ImportError(
|
|
86
|
+
"plot_subspaces needs matplotlib; install it with 'pip install subspaceknn[plot]'.",
|
|
87
|
+
) from error
|
|
88
|
+
|
|
89
|
+
check_is_fitted(estimator)
|
|
90
|
+
X_arr = validate_data(estimator, X, reset=False, dtype="numeric")
|
|
91
|
+
y_arr = np.asarray(y)
|
|
92
|
+
sample_arr = None if sample is None else np.asarray(sample, dtype=np.float64).reshape(-1)
|
|
93
|
+
names = estimator._feature_names() # noqa: SLF001
|
|
94
|
+
subspaces = estimator.subspaces_ if n_subspaces is None else estimator.subspaces_[:n_subspaces]
|
|
95
|
+
n_panels = max(len(subspaces), 1)
|
|
96
|
+
|
|
97
|
+
fig = plt.figure(figsize=(panel_size * n_panels, panel_size))
|
|
98
|
+
for panel, subspace in enumerate(subspaces, start=1):
|
|
99
|
+
estimator_index = estimator.subspaces_.index(subspace)
|
|
100
|
+
model = estimator.estimators_[estimator_index]
|
|
101
|
+
score = float(estimator.subspace_scores_[estimator_index])
|
|
102
|
+
title = f"{', '.join(names[index] for index in subspace)}\nscore {score:.3f}"
|
|
103
|
+
if len(subspace) == 1:
|
|
104
|
+
ax = fig.add_subplot(1, n_panels, panel)
|
|
105
|
+
_draw_1d(
|
|
106
|
+
ax,
|
|
107
|
+
model=model,
|
|
108
|
+
classes=estimator.classes_,
|
|
109
|
+
X=X_arr,
|
|
110
|
+
y=y_arr,
|
|
111
|
+
subspace=subspace,
|
|
112
|
+
sample=sample_arr,
|
|
113
|
+
grid_resolution=grid_resolution,
|
|
114
|
+
)
|
|
115
|
+
elif len(subspace) == 2: # noqa: PLR2004
|
|
116
|
+
ax = fig.add_subplot(1, n_panels, panel)
|
|
117
|
+
_draw_2d(
|
|
118
|
+
ax,
|
|
119
|
+
model=model,
|
|
120
|
+
classes=estimator.classes_,
|
|
121
|
+
X=X_arr,
|
|
122
|
+
y=y_arr,
|
|
123
|
+
subspace=subspace,
|
|
124
|
+
sample=sample_arr,
|
|
125
|
+
grid_resolution=grid_resolution,
|
|
126
|
+
)
|
|
127
|
+
elif len(subspace) == 3: # noqa: PLR2004
|
|
128
|
+
axes3d = cast("Axes3D", fig.add_subplot(1, n_panels, panel, projection="3d"))
|
|
129
|
+
_draw_3d(
|
|
130
|
+
axes3d,
|
|
131
|
+
classes=estimator.classes_,
|
|
132
|
+
X=X_arr,
|
|
133
|
+
y=y_arr,
|
|
134
|
+
subspace=subspace,
|
|
135
|
+
sample=sample_arr,
|
|
136
|
+
)
|
|
137
|
+
ax = axes3d
|
|
138
|
+
else:
|
|
139
|
+
ax = fig.add_subplot(1, n_panels, panel)
|
|
140
|
+
_draw_2d(
|
|
141
|
+
ax,
|
|
142
|
+
model=None,
|
|
143
|
+
classes=estimator.classes_,
|
|
144
|
+
X=X_arr,
|
|
145
|
+
y=y_arr,
|
|
146
|
+
subspace=subspace[:2],
|
|
147
|
+
sample=sample_arr,
|
|
148
|
+
grid_resolution=grid_resolution,
|
|
149
|
+
)
|
|
150
|
+
title += f" (first two of {len(subspace)} features)"
|
|
151
|
+
ax.set_title(title)
|
|
152
|
+
ax.legend(loc="best", fontsize="small")
|
|
153
|
+
fig.tight_layout()
|
|
154
|
+
return fig
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _colour(index: int) -> str:
|
|
158
|
+
return _PALETTE[index % len(_PALETTE)]
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _draw_1d(
|
|
162
|
+
ax: Axes,
|
|
163
|
+
*,
|
|
164
|
+
model: KNeighborsClassifier,
|
|
165
|
+
classes: NDArray[Any],
|
|
166
|
+
X: NDArray[np.float64],
|
|
167
|
+
y: NDArray[Any],
|
|
168
|
+
subspace: Sequence[int],
|
|
169
|
+
sample: NDArray[np.float64] | None,
|
|
170
|
+
grid_resolution: int,
|
|
171
|
+
) -> None:
|
|
172
|
+
column = X[:, subspace[0]]
|
|
173
|
+
low, high = _padded_range(column)
|
|
174
|
+
grid = np.linspace(low, high, grid_resolution)
|
|
175
|
+
regions = model.predict(grid.reshape(-1, 1))
|
|
176
|
+
for index in range(len(grid) - 1):
|
|
177
|
+
ax.axvspan(
|
|
178
|
+
grid[index], grid[index + 1], color=_colour(int(regions[index])), alpha=0.12, lw=0
|
|
179
|
+
)
|
|
180
|
+
for class_index, label in enumerate(classes):
|
|
181
|
+
mask = y == label
|
|
182
|
+
ax.scatter(
|
|
183
|
+
column[mask],
|
|
184
|
+
np.full(mask.sum(), class_index),
|
|
185
|
+
s=18,
|
|
186
|
+
alpha=0.6,
|
|
187
|
+
c=_colour(class_index),
|
|
188
|
+
label=str(label),
|
|
189
|
+
)
|
|
190
|
+
if sample is not None:
|
|
191
|
+
ax.axvline(sample[subspace[0]], color="black", lw=1.2, ls="--", label="sample")
|
|
192
|
+
ax.set_yticks(range(len(classes)))
|
|
193
|
+
ax.set_yticklabels([str(label) for label in classes])
|
|
194
|
+
ax.set_xlim(low, high)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _draw_2d(
|
|
198
|
+
ax: Axes,
|
|
199
|
+
*,
|
|
200
|
+
model: KNeighborsClassifier | None,
|
|
201
|
+
classes: NDArray[Any],
|
|
202
|
+
X: NDArray[np.float64],
|
|
203
|
+
y: NDArray[Any],
|
|
204
|
+
subspace: Sequence[int],
|
|
205
|
+
sample: NDArray[np.float64] | None,
|
|
206
|
+
grid_resolution: int,
|
|
207
|
+
) -> None:
|
|
208
|
+
first, second = subspace[0], subspace[1]
|
|
209
|
+
x_low, x_high = _padded_range(X[:, first])
|
|
210
|
+
y_low, y_high = _padded_range(X[:, second])
|
|
211
|
+
if model is not None:
|
|
212
|
+
mesh_x, mesh_y = np.meshgrid(
|
|
213
|
+
np.linspace(x_low, x_high, grid_resolution),
|
|
214
|
+
np.linspace(y_low, y_high, grid_resolution),
|
|
215
|
+
)
|
|
216
|
+
grid = np.column_stack([mesh_x.ravel(), mesh_y.ravel()])
|
|
217
|
+
regions = model.predict(grid).reshape(mesh_x.shape)
|
|
218
|
+
levels = np.arange(len(classes) + 1) - 0.5
|
|
219
|
+
colours = [_colour(index) for index in range(len(classes))]
|
|
220
|
+
ax.contourf(mesh_x, mesh_y, regions, levels=levels, colors=colours, alpha=0.12)
|
|
221
|
+
for class_index, label in enumerate(classes):
|
|
222
|
+
mask = y == label
|
|
223
|
+
ax.scatter(
|
|
224
|
+
X[mask, first],
|
|
225
|
+
X[mask, second],
|
|
226
|
+
s=18,
|
|
227
|
+
alpha=0.6,
|
|
228
|
+
c=_colour(class_index),
|
|
229
|
+
label=str(label),
|
|
230
|
+
)
|
|
231
|
+
if sample is not None:
|
|
232
|
+
ax.plot(
|
|
233
|
+
sample[first], sample[second], marker="*", markersize=16, color="black", label="sample"
|
|
234
|
+
)
|
|
235
|
+
ax.set_xlim(x_low, x_high)
|
|
236
|
+
ax.set_ylim(y_low, y_high)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _draw_3d(
|
|
240
|
+
ax: Axes3D,
|
|
241
|
+
*,
|
|
242
|
+
classes: NDArray[Any],
|
|
243
|
+
X: NDArray[np.float64],
|
|
244
|
+
y: NDArray[Any],
|
|
245
|
+
subspace: Sequence[int],
|
|
246
|
+
sample: NDArray[np.float64] | None,
|
|
247
|
+
) -> None:
|
|
248
|
+
first, second, third = subspace[0], subspace[1], subspace[2]
|
|
249
|
+
for class_index, label in enumerate(classes):
|
|
250
|
+
mask = y == label
|
|
251
|
+
ax.scatter(
|
|
252
|
+
X[mask, first],
|
|
253
|
+
X[mask, second],
|
|
254
|
+
X[mask, third],
|
|
255
|
+
s=14,
|
|
256
|
+
alpha=0.6,
|
|
257
|
+
c=_colour(class_index),
|
|
258
|
+
label=str(label),
|
|
259
|
+
)
|
|
260
|
+
if sample is not None:
|
|
261
|
+
ax.scatter(
|
|
262
|
+
[sample[first]],
|
|
263
|
+
[sample[second]],
|
|
264
|
+
[sample[third]],
|
|
265
|
+
marker="*",
|
|
266
|
+
s=200,
|
|
267
|
+
c="black",
|
|
268
|
+
label="sample",
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _padded_range(values: NDArray[np.float64]) -> tuple[float, float]:
|
|
273
|
+
low, high = float(np.min(values)), float(np.max(values))
|
|
274
|
+
pad = 0.05 * (high - low) if high > low else 0.5
|
|
275
|
+
return low - pad, high + pad
|
subspaceknn/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: subspaceknn
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Interpretable k-nearest-neighbour classification by ensembling kNN models fitted on low-dimensional feature subspaces.
|
|
5
|
+
Project-URL: Homepage, https://github.com/DiogoRibeiro7/subspaceknn
|
|
6
|
+
Project-URL: Repository, https://github.com/DiogoRibeiro7/subspaceknn
|
|
7
|
+
Project-URL: Issues, https://github.com/DiogoRibeiro7/subspaceknn/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/DiogoRibeiro7/subspaceknn/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Diogo Ribeiro
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: ensemble,explainability,interpretable-machine-learning,knn,nearest-neighbours,scikit-learn
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Requires-Dist: numpy>=1.26
|
|
27
|
+
Requires-Dist: scikit-learn>=1.6
|
|
28
|
+
Provides-Extra: plot
|
|
29
|
+
Requires-Dist: matplotlib>=3.8; extra == 'plot'
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# subspaceknn
|
|
33
|
+
|
|
34
|
+
[](https://github.com/DiogoRibeiro7/subspaceknn/actions/workflows/ci.yml)
|
|
35
|
+
[](https://pypi.org/project/subspaceknn/)
|
|
36
|
+
[](https://github.com/DiogoRibeiro7/subspaceknn)
|
|
37
|
+
[](LICENSE)
|
|
38
|
+
|
|
39
|
+
Interpretable k-nearest-neighbour classification by ensembling kNN models fitted on low-dimensional feature subspaces.
|
|
40
|
+
|
|
41
|
+
`SubspaceKNNClassifier` fits one k-nearest-neighbour model per small subset of features, ranks those subspaces by cross-validated performance, and lets the best of them vote, each weighted by its score. Because every member of the ensemble lives in a space of one, two or three features, a prediction can be explained by showing the neighbourhoods that produced it, and each subspace can be drawn with its decision regions.
|
|
42
|
+
|
|
43
|
+
The method generalises the *interpretable kNN* (ikNN) idea of [Brett Kennedy](https://github.com/Brett-Kennedy/ikNN), described in his article [Interpretable kNN (ikNN)](https://towardsdatascience.com/interpretable-knn-iknn-33d38402b8fc), from pairs of features to subspaces of any small size. This package is an independent implementation written from the description of the method. It shares no code, text or results with the original.
|
|
44
|
+
|
|
45
|
+
## Installation
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
pip install subspaceknn # core: numpy and scikit-learn
|
|
49
|
+
pip install "subspaceknn[plot]" # adds matplotlib for plot_subspaces
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
The package supports Python 3.10 to 3.13 and scikit-learn 1.6 or newer.
|
|
53
|
+
|
|
54
|
+
## Quick start
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from sklearn.datasets import load_iris
|
|
58
|
+
from sklearn.model_selection import train_test_split
|
|
59
|
+
|
|
60
|
+
from subspaceknn import SubspaceKNNClassifier
|
|
61
|
+
|
|
62
|
+
iris = load_iris(as_frame=True)
|
|
63
|
+
X, y = iris.data, iris.target_names[iris.target]
|
|
64
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0, stratify=y)
|
|
65
|
+
|
|
66
|
+
clf = SubspaceKNNClassifier(subspace_size=(1, 2), n_subspaces=4).fit(X_train, y_train)
|
|
67
|
+
print(clf.score(X_test, y_test))
|
|
68
|
+
|
|
69
|
+
for subspace, score, weight in zip(clf.subspaces_, clf.subspace_scores_, clf.subspace_weights_):
|
|
70
|
+
print(list(X.columns[list(subspace)]), f"score={score:.3f}", f"weight={weight:.3f}")
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The estimator follows the scikit-learn contract, so it works inside `Pipeline`, `GridSearchCV` and `cross_val_score`, accepts data frames, and exposes `predict`, `predict_proba` and `score`. Feature scales matter for nearest neighbours, so put a `StandardScaler` in front of it unless the features are already comparable.
|
|
74
|
+
|
|
75
|
+
### Explaining a prediction
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
explanation = clf.explain(X_test.iloc[:1])[0]
|
|
79
|
+
print(explanation.prediction, explanation.agreement())
|
|
80
|
+
for vote in explanation.votes:
|
|
81
|
+
print(vote.feature_names, vote.prediction, f"weight={vote.weight:.3f}")
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
`explain` returns one `Explanation` per sample. It holds the ensemble prediction and probabilities and one `SubspaceVote` per subspace with the features involved, that subspace's own prediction and probabilities, its cross-validated score and its voting weight. `agreement()` is the total weight of the subspaces that voted for the final prediction, and `to_records()` produces rows ready for `pandas.DataFrame.from_records`.
|
|
85
|
+
|
|
86
|
+
### Drawing the subspaces
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from subspaceknn.plotting import plot_subspaces
|
|
90
|
+
|
|
91
|
+
fig = plot_subspaces(clf, X_train, y_train, sample=X_test.iloc[0].to_numpy())
|
|
92
|
+
fig.savefig("subspaces.png")
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
One panel per subspace: a strip plot with decision intervals for one feature, a scatter plot with decision regions for two, a 3-D scatter for three. The highlighted sample is the point being explained.
|
|
96
|
+
|
|
97
|
+
## How it works
|
|
98
|
+
|
|
99
|
+
For subspace sizes `d` in `subspace_size`, the estimator enumerates every `d`-subset of the features, fits a `KNeighborsClassifier` on each subset and scores it with stratified cross-validation on the training data (macro-F1 by default). The `n_subspaces` best subsets form the ensemble. For a new sample the class probabilities are the weighted average of the subspace models' probabilities,
|
|
100
|
+
|
|
101
|
+
```text
|
|
102
|
+
p(c | x) = sum_s w_s * p_s(c | x), w_s = score_s / sum_t score_t,
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
and the prediction is the class with the largest probability. `voting="hard"` replaces `p_s` by the one-hot prediction of each subspace and `weighting="uniform"` replaces `w_s` by equal weights.
|
|
106
|
+
|
|
107
|
+
The number of subsets grows combinatorially with the number of features, so `max_candidates` caps how many are cross-validated. Above the cap, features are screened by the cross-validated score of their one-dimensional model, and only the best-scoring features are combined, as many as keep the candidate count within the cap. Everything is deterministic: subsets are enumerated in lexicographic order and ties keep that order.
|
|
108
|
+
|
|
109
|
+
The full description, including how tiny training sets are handled, is in [docs/method.md](docs/method.md).
|
|
110
|
+
|
|
111
|
+
## Parameters
|
|
112
|
+
|
|
113
|
+
| Parameter | Default | Meaning |
|
|
114
|
+
| --- | --- | --- |
|
|
115
|
+
| `n_neighbors` | `5` | Neighbours used by every subspace model. |
|
|
116
|
+
| `subspace_size` | `2` | Size of each subspace, or a sequence of sizes to enumerate together. |
|
|
117
|
+
| `n_subspaces` | `5` | Number of best subspaces that vote; `None` uses all candidates. |
|
|
118
|
+
| `max_candidates` | `100` | Cap on cross-validated subspaces; triggers feature screening above it. |
|
|
119
|
+
| `voting` | `"soft"` | `"soft"` averages probabilities, `"hard"` averages one-hot votes. |
|
|
120
|
+
| `weighting` | `"score"` | Weight by cross-validated score, or `"uniform"`. |
|
|
121
|
+
| `cv` | `5` | Folds, or any scikit-learn splitter, used to score subspaces. |
|
|
122
|
+
| `scoring` | `"f1_macro"` | Any scikit-learn scorer; higher must be better. |
|
|
123
|
+
| `knn_weights`, `metric` | `"uniform"`, `"minkowski"` | Passed to the subspace models. |
|
|
124
|
+
|
|
125
|
+
Fitted attributes include `subspaces_`, `subspace_scores_`, `subspace_weights_`, `estimators_`, the full `candidate_subspaces_` with `candidate_scores_`, the `screened_features_`, and `feature_scores_`, a coarse feature-relevance measure. See the class docstring for the complete list.
|
|
126
|
+
|
|
127
|
+
## Does interpretability cost accuracy?
|
|
128
|
+
|
|
129
|
+
Five-fold stratified cross-validated macro-F1 on scikit-learn's toy datasets, features standardised, everything else at its defaults:
|
|
130
|
+
|
|
131
|
+
| Dataset | kNN | Subspaces of size 2 | Sizes 1, 2 and 3 (8 subspaces) |
|
|
132
|
+
| --- | ---: | ---: | ---: |
|
|
133
|
+
| iris (4 features) | 0.953 | 0.953 | 0.953 |
|
|
134
|
+
| wine (13 features) | 0.960 | 0.945 | 0.967 |
|
|
135
|
+
| breast cancer (30 features) | 0.962 | 0.946 | 0.943 |
|
|
136
|
+
|
|
137
|
+
The ensemble stays within a couple of points of plain kNN while every one of its votes is a picture. The test suite asserts this stays true. Details in [docs/benchmark.md](docs/benchmark.md).
|
|
138
|
+
|
|
139
|
+
## Limitations
|
|
140
|
+
|
|
141
|
+
- Features must be numeric and are used as given; encode categorical features and scale everything first.
|
|
142
|
+
- Candidate subspaces grow as the binomial coefficient of the feature count; rely on `max_candidates` or a sequence of small sizes for wide data.
|
|
143
|
+
- The voting weights are cross-validated scores, not calibrated probabilities. Treat `predict_proba` as a ranking rather than a probability estimate.
|
|
144
|
+
- Classification only.
|
|
145
|
+
|
|
146
|
+
## Development
|
|
147
|
+
|
|
148
|
+
```sh
|
|
149
|
+
git clone https://github.com/DiogoRibeiro7/subspaceknn.git
|
|
150
|
+
cd subspaceknn
|
|
151
|
+
uv sync --all-extras
|
|
152
|
+
uv run ruff check . && uv run ruff format --check .
|
|
153
|
+
uv run mypy
|
|
154
|
+
uv run pytest
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
The test suite runs scikit-learn's estimator contract (`check_estimator`) against two configurations, plus behavioural, explanation, plotting and benchmark tests. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
158
|
+
|
|
159
|
+
## License and attribution
|
|
160
|
+
|
|
161
|
+
MIT, see [LICENSE](LICENSE). The method is due to Brett Kennedy's ikNN; this implementation, its generalisation to arbitrary subspace sizes, and everything in this repository were written independently.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
subspaceknn/__init__.py,sha256=1x2Dw8xwtSuk6pw51UO3pg7WqfrJkau_Wcld1VRakds,740
|
|
2
|
+
subspaceknn/_classifier.py,sha256=GLHQ9O9EqdpM8yVJOp8RByW1lswKMvQsBpQv-tTizwA,20208
|
|
3
|
+
subspaceknn/_explanation.py,sha256=lvo-Lgn5Nvbj3jxw6JIPfknnwav6sTp3Sx1m6zuCCDg,3302
|
|
4
|
+
subspaceknn/plotting.py,sha256=WQbuVp6WE3BWQ8bt0iCh3wwVZBNC_roQ0I9NU4-Wa5w,8698
|
|
5
|
+
subspaceknn/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
subspaceknn-0.1.0.dist-info/METADATA,sha256=8yhcCUQV8q3InOUHdn83q1-PDGw_n0NGbJ2Jodp3QDI,9250
|
|
7
|
+
subspaceknn-0.1.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
|
|
8
|
+
subspaceknn-0.1.0.dist-info/licenses/LICENSE,sha256=jPYDkXzOXETff4tGRBkRvUJtBxPtpZKJ_ukTRSWkaVg,1070
|
|
9
|
+
subspaceknn-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Diogo Ribeiro
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|