deskit 1.2.5__tar.gz → 1.2.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deskit-1.2.5/src/deskit.egg-info → deskit-1.2.7}/PKG-INFO +12 -1
- {deskit-1.2.5 → deskit-1.2.7}/README.md +11 -0
- {deskit-1.2.5 → deskit-1.2.7}/pyproject.toml +1 -1
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/base/knnbase.py +5 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/base/predictbase.py +4 -1
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/dewsi.py +3 -7
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/dewsiv.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/dewst.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/dewsu.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/dewsv.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/knorae.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/knoraiu.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/knorau.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/lwsei.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/lwseu.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/ola.py +3 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/neighbors.py +0 -32
- {deskit-1.2.5 → deskit-1.2.7/src/deskit.egg-info}/PKG-INFO +12 -1
- {deskit-1.2.5 → deskit-1.2.7}/LICENSE +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/setup.cfg +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/__init__.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/_config.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/base/__init__.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/base/base.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/des/__init__.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/metrics.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/router.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit/utils.py +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit.egg-info/SOURCES.txt +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit.egg-info/dependency_links.txt +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit.egg-info/requires.txt +0 -0
- {deskit-1.2.5 → deskit-1.2.7}/src/deskit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deskit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.7
|
|
4
4
|
Summary: A Python library for Dynamic Ensemble Selection
|
|
5
5
|
Author: Tikhon Vodyanov
|
|
6
6
|
License-Expression: MIT
|
|
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
146
146
|
|
|
147
147
|
---
|
|
148
148
|
|
|
149
|
+
### Hyperparameter tuning
|
|
150
|
+
|
|
151
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
152
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
149
160
|
## Why deskit?
|
|
150
161
|
|
|
151
162
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
@@ -115,6 +115,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
115
115
|
|
|
116
116
|
---
|
|
117
117
|
|
|
118
|
+
### Hyperparameter tuning
|
|
119
|
+
|
|
120
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
121
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
118
129
|
## Why deskit?
|
|
119
130
|
|
|
120
131
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
@@ -57,6 +57,11 @@ class KNNBase(PredictBase, BaseRouter):
|
|
|
57
57
|
scores = self._compute_scores(y, preds_dict[name])
|
|
58
58
|
self.matrix[:, j] = scores if self.mode == 'max' else -scores
|
|
59
59
|
|
|
60
|
+
if self.task == 'classification':
|
|
61
|
+
self.classes_ = np.unique(y)
|
|
62
|
+
else:
|
|
63
|
+
self.classes_ = None
|
|
64
|
+
|
|
60
65
|
self.model.fit(features)
|
|
61
66
|
|
|
62
67
|
def _kneighbors(self, x, k=None, loo=False):
|
|
@@ -103,7 +103,10 @@ class PredictBase:
|
|
|
103
103
|
if self.task == 'classification':
|
|
104
104
|
# Probability arrays: blend per-class columns.
|
|
105
105
|
# preds_3d : (batch, n_models, n_classes)
|
|
106
|
-
|
|
106
|
+
|
|
107
|
+
n_classes = len(self.classes_)
|
|
108
|
+
preds_2d = np.stack(preds_list, axis=1) # (batch, n_models)
|
|
109
|
+
preds_3d = np.eye(n_classes)[preds_2d.astype(int)]
|
|
107
110
|
result = np.einsum("bm,bmc->bc", weights, preds_3d) # (batch, n_classes)
|
|
108
111
|
else:
|
|
109
112
|
# Scalar predictions: weighted average.
|
|
@@ -38,13 +38,9 @@ class DEWSI(KNNBase):
|
|
|
38
38
|
distance_metric : str
|
|
39
39
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
40
40
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
predict() and predict_weights() accept an optional ``loo=True`` keyword
|
|
45
|
-
for hyperparameter tuning directly on the DSEL this model was fit on:
|
|
46
|
-
it excludes each query point's own occurrence from its neighborhood so
|
|
47
|
-
it doesn't trivially neighbor itself at distance 0.
|
|
41
|
+
loo: bool
|
|
42
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
43
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
48
44
|
"""
|
|
49
45
|
|
|
50
46
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -40,6 +40,9 @@ class DEWSIV(KNNBase):
|
|
|
40
40
|
distance_metric : str
|
|
41
41
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
42
42
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
43
|
+
loo: bool
|
|
44
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
45
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
43
46
|
"""
|
|
44
47
|
|
|
45
48
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -43,6 +43,9 @@ class DEWST(KNNBase):
|
|
|
43
43
|
distance_metric : str
|
|
44
44
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
45
45
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
46
|
+
loo: bool
|
|
47
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
48
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
46
49
|
"""
|
|
47
50
|
|
|
48
51
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -34,6 +34,9 @@ class DEWSU(KNNBase):
|
|
|
34
34
|
distance_metric : str
|
|
35
35
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
36
36
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
37
|
+
loo: bool
|
|
38
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
39
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
37
40
|
"""
|
|
38
41
|
|
|
39
42
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -40,6 +40,9 @@ class DEWSV(KNNBase):
|
|
|
40
40
|
distance_metric : str
|
|
41
41
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
42
42
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
43
|
+
loo: bool
|
|
44
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
45
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
43
46
|
"""
|
|
44
47
|
|
|
45
48
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -29,6 +29,9 @@ class KNORAE(KNNBase):
|
|
|
29
29
|
distance_metric : str
|
|
30
30
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
31
31
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
32
|
+
loo: bool
|
|
33
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
34
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
32
35
|
"""
|
|
33
36
|
|
|
34
37
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -30,6 +30,9 @@ class KNORAIU(KNNBase):
|
|
|
30
30
|
distance_metric : str
|
|
31
31
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
32
32
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
33
|
+
loo: bool
|
|
34
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
35
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
33
36
|
"""
|
|
34
37
|
|
|
35
38
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -30,6 +30,9 @@ class KNORAU(KNNBase):
|
|
|
30
30
|
distance_metric : str
|
|
31
31
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
32
32
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
33
|
+
loo: bool
|
|
34
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
35
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
33
36
|
"""
|
|
34
37
|
|
|
35
38
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -22,6 +22,9 @@ class LWSEI(PredictBase):
|
|
|
22
22
|
distance_metric : str
|
|
23
23
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
24
24
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
25
|
+
loo: bool
|
|
26
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
27
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
25
28
|
"""
|
|
26
29
|
|
|
27
30
|
def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
|
|
@@ -22,6 +22,9 @@ class LWSEU(PredictBase):
|
|
|
22
22
|
distance_metric : str
|
|
23
23
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
24
24
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
25
|
+
loo: bool
|
|
26
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
27
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
25
28
|
"""
|
|
26
29
|
|
|
27
30
|
def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
|
|
@@ -25,6 +25,9 @@ class OLA(KNNBase):
|
|
|
25
25
|
distance_metric : str
|
|
26
26
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
27
27
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
28
|
+
loo: bool
|
|
29
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
30
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
28
31
|
"""
|
|
29
32
|
|
|
30
33
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -4,38 +4,6 @@ import warnings
|
|
|
4
4
|
# FAISS IVF k-means needs at least this many training samples per cell to converge.
|
|
5
5
|
_FAISS_MIN_SAMPLES_PER_CELL = 40
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
# ---------------------------------------------------------------------------
|
|
9
|
-
# Distance metric registry
|
|
10
|
-
# ---------------------------------------------------------------------------
|
|
11
|
-
|
|
12
|
-
# Metrics supported by each backend.
|
|
13
|
-
# 'euclidean' is the universal default and always available.
|
|
14
|
-
#
|
|
15
|
-
# Choosing a distance metric:
|
|
16
|
-
# euclidean – The standard L2 norm. Best default for most tabular data.
|
|
17
|
-
# manhattan – L1 norm (sum of absolute differences). More robust to
|
|
18
|
-
# outliers and tends to work better in moderately high-
|
|
19
|
-
# dimensional spaces because it doesn't square large diffs.
|
|
20
|
-
# chebyshev – L∞ norm (maximum absolute difference across features).
|
|
21
|
-
# Useful when a single feature dominating the distance is
|
|
22
|
-
# acceptable; common in game-grid / chess-style problems.
|
|
23
|
-
# minkowski – Generalisation of L1/L2 (controlled by p). p=1 →
|
|
24
|
-
# manhattan, p=2 → euclidean. Use when you want to tune
|
|
25
|
-
# between them.
|
|
26
|
-
# cosine – Angle between vectors, ignoring magnitude. Excellent for
|
|
27
|
-
# embeddings (text, image, audio) where direction matters
|
|
28
|
-
# more than raw scale.
|
|
29
|
-
# canberra – Weighted L1. Sensitive to small values near zero.
|
|
30
|
-
# braycurtis – Normalised L1 bounded to [0,1]. Common in ecology.
|
|
31
|
-
# jensenshannon – Symmetric KL divergence on probability distributions.
|
|
32
|
-
# Requires non-negative vectors. Supported by FAISS flat/
|
|
33
|
-
# HNSW/GPU indices natively.
|
|
34
|
-
# dot – Raw inner/dot product. Not a true metric; distances are
|
|
35
|
-
# not comparable across queries. Use for max inner-product
|
|
36
|
-
# search (recommendation systems). Prefer 'cosine' for
|
|
37
|
-
# normalised embeddings.
|
|
38
|
-
|
|
39
7
|
# Metrics that every backend supports natively.
|
|
40
8
|
_UNIVERSAL_METRICS = {'euclidean', 'manhattan', 'chebyshev', 'minkowski', 'cosine'}
|
|
41
9
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deskit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.7
|
|
4
4
|
Summary: A Python library for Dynamic Ensemble Selection
|
|
5
5
|
Author: Tikhon Vodyanov
|
|
6
6
|
License-Expression: MIT
|
|
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
146
146
|
|
|
147
147
|
---
|
|
148
148
|
|
|
149
|
+
### Hyperparameter tuning
|
|
150
|
+
|
|
151
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
152
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
149
160
|
## Why deskit?
|
|
150
161
|
|
|
151
162
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|