deskit 1.2.4__tar.gz → 1.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deskit-1.2.4/src/deskit.egg-info → deskit-1.2.6}/PKG-INFO +12 -1
- {deskit-1.2.4 → deskit-1.2.6}/README.md +11 -0
- {deskit-1.2.4 → deskit-1.2.6}/pyproject.toml +1 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsi.py +3 -7
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsiv.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewst.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsu.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsv.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knorae.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knoraiu.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knorau.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/lwsei.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/lwseu.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/ola.py +4 -1
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/neighbors.py +0 -32
- {deskit-1.2.4 → deskit-1.2.6/src/deskit.egg-info}/PKG-INFO +12 -1
- {deskit-1.2.4 → deskit-1.2.6}/LICENSE +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/setup.cfg +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/__init__.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/_config.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/__init__.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/base.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/knnbase.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/predictbase.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/__init__.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/metrics.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/router.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit/utils.py +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/SOURCES.txt +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/dependency_links.txt +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/requires.txt +0 -0
- {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deskit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.6
|
|
4
4
|
Summary: A Python library for Dynamic Ensemble Selection
|
|
5
5
|
Author: Tikhon Vodyanov
|
|
6
6
|
License-Expression: MIT
|
|
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
146
146
|
|
|
147
147
|
---
|
|
148
148
|
|
|
149
|
+
### Hyperparameter tuning
|
|
150
|
+
|
|
151
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
152
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
149
160
|
## Why deskit?
|
|
150
161
|
|
|
151
162
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
@@ -115,6 +115,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
115
115
|
|
|
116
116
|
---
|
|
117
117
|
|
|
118
|
+
### Hyperparameter tuning
|
|
119
|
+
|
|
120
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
121
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
118
129
|
## Why deskit?
|
|
119
130
|
|
|
120
131
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
@@ -38,13 +38,9 @@ class DEWSI(KNNBase):
|
|
|
38
38
|
distance_metric : str
|
|
39
39
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
40
40
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
predict() and predict_weights() accept an optional ``loo=True`` keyword
|
|
45
|
-
for hyperparameter tuning directly on the DSEL this model was fit on:
|
|
46
|
-
it excludes each query point's own occurrence from its neighborhood so
|
|
47
|
-
it doesn't trivially neighbor itself at distance 0.
|
|
41
|
+
loo: bool
|
|
42
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
43
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
48
44
|
"""
|
|
49
45
|
|
|
50
46
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -40,6 +40,9 @@ class DEWSIV(KNNBase):
|
|
|
40
40
|
distance_metric : str
|
|
41
41
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
42
42
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
43
|
+
loo: bool
|
|
44
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
45
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
43
46
|
"""
|
|
44
47
|
|
|
45
48
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -94,7 +97,7 @@ class DEWSIV(KNNBase):
|
|
|
94
97
|
(0.5 if self.mode == 'min' else 1.0))
|
|
95
98
|
th = threshold if threshold is not None else self.threshold
|
|
96
99
|
|
|
97
|
-
distances, indices = self.
|
|
100
|
+
distances, indices = self._kneighbors(x, k=k, loo=loo) # both (batch, k)
|
|
98
101
|
|
|
99
102
|
# Inverse-distance weights
|
|
100
103
|
inv_dist = 1.0 / np.maximum(distances, 1e-8) # (batch, k)
|
|
@@ -43,6 +43,9 @@ class DEWST(KNNBase):
|
|
|
43
43
|
distance_metric : str
|
|
44
44
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
45
45
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
46
|
+
loo: bool
|
|
47
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
48
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
46
49
|
"""
|
|
47
50
|
|
|
48
51
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -97,7 +100,7 @@ class DEWST(KNNBase):
|
|
|
97
100
|
th = threshold if threshold is not None else self.threshold
|
|
98
101
|
r2_th = r2_threshold if r2_threshold is not None else self.r2_threshold
|
|
99
102
|
|
|
100
|
-
distances, indices = self.
|
|
103
|
+
distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
|
|
101
104
|
k = distances.shape[1]
|
|
102
105
|
|
|
103
106
|
# Inverse-distance weights
|
|
@@ -34,6 +34,9 @@ class DEWSU(KNNBase):
|
|
|
34
34
|
distance_metric : str
|
|
35
35
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
36
36
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
37
|
+
loo: bool
|
|
38
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
39
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
37
40
|
"""
|
|
38
41
|
|
|
39
42
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -75,7 +78,7 @@ class DEWSU(KNNBase):
|
|
|
75
78
|
(0.5 if self.mode == 'min' else 1.0))
|
|
76
79
|
th = threshold if threshold is not None else self.threshold
|
|
77
80
|
|
|
78
|
-
_, indices = self.
|
|
81
|
+
_, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
|
|
79
82
|
|
|
80
83
|
# Average each model's scores over the K neighbors
|
|
81
84
|
avg_scores = self.matrix[indices].mean(axis=1) # (batch, n_models)
|
|
@@ -40,6 +40,9 @@ class DEWSV(KNNBase):
|
|
|
40
40
|
distance_metric : str
|
|
41
41
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
42
42
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
43
|
+
loo: bool
|
|
44
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
45
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
43
46
|
"""
|
|
44
47
|
|
|
45
48
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -94,7 +97,7 @@ class DEWSV(KNNBase):
|
|
|
94
97
|
(0.5 if self.mode == 'min' else 1.0))
|
|
95
98
|
th = threshold if threshold is not None else self.threshold
|
|
96
99
|
|
|
97
|
-
_, indices = self.
|
|
100
|
+
_, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
|
|
98
101
|
|
|
99
102
|
# Uniform average of each model's scores over K neighbors
|
|
100
103
|
neighbor_scores = self.matrix[indices] # (batch, k, n_models)
|
|
@@ -29,6 +29,9 @@ class KNORAE(KNNBase):
|
|
|
29
29
|
distance_metric : str
|
|
30
30
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
31
31
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
32
|
+
loo: bool
|
|
33
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
34
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
32
35
|
"""
|
|
33
36
|
|
|
34
37
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -64,7 +67,7 @@ class KNORAE(KNNBase):
|
|
|
64
67
|
th = threshold if threshold is not None else self.threshold
|
|
65
68
|
n_models = len(self.models)
|
|
66
69
|
|
|
67
|
-
_, indices = self.
|
|
70
|
+
_, indices = self._kneighbors(x, k=k, loo=loo)
|
|
68
71
|
k = indices.shape[1]
|
|
69
72
|
neighbor_scores = self.matrix[indices] # (batch, k, n_models)
|
|
70
73
|
|
|
@@ -30,6 +30,9 @@ class KNORAIU(KNNBase):
|
|
|
30
30
|
distance_metric : str
|
|
31
31
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
32
32
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
33
|
+
loo: bool
|
|
34
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
35
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
33
36
|
"""
|
|
34
37
|
|
|
35
38
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -64,7 +67,7 @@ class KNORAIU(KNNBase):
|
|
|
64
67
|
"""
|
|
65
68
|
th = threshold if threshold is not None else self.threshold
|
|
66
69
|
|
|
67
|
-
distances, indices = self.
|
|
70
|
+
distances, indices = self._kneighbors(x, k=k, loo=loo) # both (batch, k)
|
|
68
71
|
neighbor_scores = self.matrix[indices] # (batch, k, n_models)
|
|
69
72
|
|
|
70
73
|
# Normalize per neighbor: best model = 1.0, worst = 0.0
|
|
@@ -30,6 +30,9 @@ class KNORAU(KNNBase):
|
|
|
30
30
|
distance_metric : str
|
|
31
31
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
32
32
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
33
|
+
loo: bool
|
|
34
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
35
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
33
36
|
"""
|
|
34
37
|
|
|
35
38
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -64,7 +67,7 @@ class KNORAU(KNNBase):
|
|
|
64
67
|
"""
|
|
65
68
|
th = threshold if threshold is not None else self.threshold
|
|
66
69
|
|
|
67
|
-
_, indices = self.
|
|
70
|
+
_, indices = self._kneighbors(x, k=k, loo=loo)
|
|
68
71
|
neighbor_scores = self.matrix[indices] # (batch, k, n_models)
|
|
69
72
|
|
|
70
73
|
# Normalize per neighbor: best model = 1.0, worst = 0.0
|
|
@@ -22,6 +22,9 @@ class LWSEI(PredictBase):
|
|
|
22
22
|
distance_metric : str
|
|
23
23
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
24
24
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
25
|
+
loo: bool
|
|
26
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
27
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
25
28
|
"""
|
|
26
29
|
|
|
27
30
|
def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
|
|
@@ -87,7 +90,7 @@ class LWSEI(PredictBase):
|
|
|
87
90
|
n_models = len(self.models)
|
|
88
91
|
uniform = np.full(n_models, 1.0 / n_models)
|
|
89
92
|
|
|
90
|
-
distances, indices = self.
|
|
93
|
+
distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
|
|
91
94
|
weights_out = np.empty((batch_size, n_models))
|
|
92
95
|
|
|
93
96
|
for b in range(batch_size):
|
|
@@ -22,6 +22,9 @@ class LWSEU(PredictBase):
|
|
|
22
22
|
distance_metric : str
|
|
23
23
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
24
24
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
25
|
+
loo: bool
|
|
26
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
27
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
25
28
|
"""
|
|
26
29
|
|
|
27
30
|
def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
|
|
@@ -87,7 +90,7 @@ class LWSEU(PredictBase):
|
|
|
87
90
|
n_models = len(self.models)
|
|
88
91
|
uniform = np.full(n_models, 1.0 / n_models)
|
|
89
92
|
|
|
90
|
-
distances, indices = self.
|
|
93
|
+
distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
|
|
91
94
|
weights_out = np.empty((batch_size, n_models))
|
|
92
95
|
|
|
93
96
|
for b in range(batch_size):
|
|
@@ -25,6 +25,9 @@ class OLA(KNNBase):
|
|
|
25
25
|
distance_metric : str
|
|
26
26
|
Distance function to use for neighbor search. Default: 'euclidean'. See
|
|
27
27
|
neighbors.list_distance_metrics() for all options and per-backend availability.
|
|
28
|
+
loo: bool
|
|
29
|
+
Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
|
|
30
|
+
Ignores closest neighbor with a negligible distance to avoid overfitting.
|
|
28
31
|
"""
|
|
29
32
|
|
|
30
33
|
def __init__(self, task, metric='mae', mode='min', k=10,
|
|
@@ -63,7 +66,7 @@ class OLA(KNNBase):
|
|
|
63
66
|
"""
|
|
64
67
|
batch_size = x.shape[0]
|
|
65
68
|
|
|
66
|
-
_, indices = self.
|
|
69
|
+
_, indices = self._kneighbors(x, k=k, loo=loo)
|
|
67
70
|
avg_scores = self.matrix[indices].mean(axis=1) # (batch, n_models)
|
|
68
71
|
best_indices = np.argmax(avg_scores, axis=1)
|
|
69
72
|
|
|
@@ -4,38 +4,6 @@ import warnings
|
|
|
4
4
|
# FAISS IVF k-means needs at least this many training samples per cell to converge.
|
|
5
5
|
_FAISS_MIN_SAMPLES_PER_CELL = 40
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
# ---------------------------------------------------------------------------
|
|
9
|
-
# Distance metric registry
|
|
10
|
-
# ---------------------------------------------------------------------------
|
|
11
|
-
|
|
12
|
-
# Metrics supported by each backend.
|
|
13
|
-
# 'euclidean' is the universal default and always available.
|
|
14
|
-
#
|
|
15
|
-
# Choosing a distance metric:
|
|
16
|
-
# euclidean – The standard L2 norm. Best default for most tabular data.
|
|
17
|
-
# manhattan – L1 norm (sum of absolute differences). More robust to
|
|
18
|
-
# outliers and tends to work better in moderately high-
|
|
19
|
-
# dimensional spaces because it doesn't square large diffs.
|
|
20
|
-
# chebyshev – L∞ norm (maximum absolute difference across features).
|
|
21
|
-
# Useful when a single feature dominating the distance is
|
|
22
|
-
# acceptable; common in game-grid / chess-style problems.
|
|
23
|
-
# minkowski – Generalisation of L1/L2 (controlled by p). p=1 →
|
|
24
|
-
# manhattan, p=2 → euclidean. Use when you want to tune
|
|
25
|
-
# between them.
|
|
26
|
-
# cosine – Angle between vectors, ignoring magnitude. Excellent for
|
|
27
|
-
# embeddings (text, image, audio) where direction matters
|
|
28
|
-
# more than raw scale.
|
|
29
|
-
# canberra – Weighted L1. Sensitive to small values near zero.
|
|
30
|
-
# braycurtis – Normalised L1 bounded to [0,1]. Common in ecology.
|
|
31
|
-
# jensenshannon – Symmetric KL divergence on probability distributions.
|
|
32
|
-
# Requires non-negative vectors. Supported by FAISS flat/
|
|
33
|
-
# HNSW/GPU indices natively.
|
|
34
|
-
# dot – Raw inner/dot product. Not a true metric; distances are
|
|
35
|
-
# not comparable across queries. Use for max inner-product
|
|
36
|
-
# search (recommendation systems). Prefer 'cosine' for
|
|
37
|
-
# normalised embeddings.
|
|
38
|
-
|
|
39
7
|
# Metrics that every backend supports natively.
|
|
40
8
|
_UNIVERSAL_METRICS = {'euclidean', 'manhattan', 'chebyshev', 'minkowski', 'cosine'}
|
|
41
9
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deskit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.6
|
|
4
4
|
Summary: A Python library for Dynamic Ensemble Selection
|
|
5
5
|
Author: Tikhon Vodyanov
|
|
6
6
|
License-Expression: MIT
|
|
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
|
|
|
146
146
|
|
|
147
147
|
---
|
|
148
148
|
|
|
149
|
+
### Hyperparameter tuning
|
|
150
|
+
|
|
151
|
+
`deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
|
|
152
|
+
This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
results = router.predict(X_val, val_preds, loo=True)
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
149
160
|
## Why deskit?
|
|
150
161
|
|
|
151
162
|
Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|