deskit 1.2.4__tar.gz → 1.2.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {deskit-1.2.4/src/deskit.egg-info → deskit-1.2.6}/PKG-INFO +12 -1
  2. {deskit-1.2.4 → deskit-1.2.6}/README.md +11 -0
  3. {deskit-1.2.4 → deskit-1.2.6}/pyproject.toml +1 -1
  4. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsi.py +3 -7
  5. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsiv.py +4 -1
  6. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewst.py +4 -1
  7. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsu.py +4 -1
  8. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/dewsv.py +4 -1
  9. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knorae.py +4 -1
  10. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knoraiu.py +4 -1
  11. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/knorau.py +4 -1
  12. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/lwsei.py +4 -1
  13. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/lwseu.py +4 -1
  14. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/ola.py +4 -1
  15. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/neighbors.py +0 -32
  16. {deskit-1.2.4 → deskit-1.2.6/src/deskit.egg-info}/PKG-INFO +12 -1
  17. {deskit-1.2.4 → deskit-1.2.6}/LICENSE +0 -0
  18. {deskit-1.2.4 → deskit-1.2.6}/setup.cfg +0 -0
  19. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/__init__.py +0 -0
  20. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/_config.py +0 -0
  21. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/__init__.py +0 -0
  22. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/base.py +0 -0
  23. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/knnbase.py +0 -0
  24. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/base/predictbase.py +0 -0
  25. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/des/__init__.py +0 -0
  26. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/metrics.py +0 -0
  27. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/router.py +0 -0
  28. {deskit-1.2.4 → deskit-1.2.6}/src/deskit/utils.py +0 -0
  29. {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/SOURCES.txt +0 -0
  30. {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/dependency_links.txt +0 -0
  31. {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/requires.txt +0 -0
  32. {deskit-1.2.4 → deskit-1.2.6}/src/deskit.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deskit
3
- Version: 1.2.4
3
+ Version: 1.2.6
4
4
  Summary: A Python library for Dynamic Ensemble Selection
5
5
  Author: Tikhon Vodyanov
6
6
  License-Expression: MIT
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
146
146
 
147
147
  ---
148
148
 
149
+ ### Hyperparameter tuning
150
+
151
+ `deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
152
+ This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
153
+
154
+ ```python
155
+ results = router.predict(X_val, val_preds, loo=True)
156
+ ```
157
+
158
+ ---
159
+
149
160
  ## Why deskit?
150
161
 
151
162
  Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
@@ -115,6 +115,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
115
115
 
116
116
  ---
117
117
 
118
+ ### Hyperparameter tuning
119
+
120
+ `deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
121
+ This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
122
+
123
+ ```python
124
+ results = router.predict(X_val, val_preds, loo=True)
125
+ ```
126
+
127
+ ---
128
+
118
129
  ## Why deskit?
119
130
 
120
131
  Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "deskit"
7
- version = "1.2.4"
7
+ version = "1.2.6"
8
8
  description = "A Python library for Dynamic Ensemble Selection"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -38,13 +38,9 @@ class DEWSI(KNNBase):
38
38
  distance_metric : str
39
39
  Distance function to use for neighbor search. Default: 'euclidean'. See
40
40
  neighbors.list_distance_metrics() for all options and per-backend availability.
41
-
42
- Notes
43
- -----
44
- predict() and predict_weights() accept an optional ``loo=True`` keyword
45
- for hyperparameter tuning directly on the DSEL this model was fit on:
46
- it excludes each query point's own occurrence from its neighborhood so
47
- it doesn't trivially neighbor itself at distance 0.
41
+ loo: bool
42
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
43
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
48
44
  """
49
45
 
50
46
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -40,6 +40,9 @@ class DEWSIV(KNNBase):
40
40
  distance_metric : str
41
41
  Distance function to use for neighbor search. Default: 'euclidean'. See
42
42
  neighbors.list_distance_metrics() for all options and per-backend availability.
43
+ loo: bool
44
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
45
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
43
46
  """
44
47
 
45
48
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -94,7 +97,7 @@ class DEWSIV(KNNBase):
94
97
  (0.5 if self.mode == 'min' else 1.0))
95
98
  th = threshold if threshold is not None else self.threshold
96
99
 
97
- distances, indices = self.model._kneighbors(x, k=k, loo=loo) # both (batch, k)
100
+ distances, indices = self._kneighbors(x, k=k, loo=loo) # both (batch, k)
98
101
 
99
102
  # Inverse-distance weights
100
103
  inv_dist = 1.0 / np.maximum(distances, 1e-8) # (batch, k)
@@ -43,6 +43,9 @@ class DEWST(KNNBase):
43
43
  distance_metric : str
44
44
  Distance function to use for neighbor search. Default: 'euclidean'. See
45
45
  neighbors.list_distance_metrics() for all options and per-backend availability.
46
+ loo: bool
47
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
48
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
46
49
  """
47
50
 
48
51
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -97,7 +100,7 @@ class DEWST(KNNBase):
97
100
  th = threshold if threshold is not None else self.threshold
98
101
  r2_th = r2_threshold if r2_threshold is not None else self.r2_threshold
99
102
 
100
- distances, indices = self.model._kneighbors(x, k=k, loo=loo) # (batch, k)
103
+ distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
101
104
  k = distances.shape[1]
102
105
 
103
106
  # Inverse-distance weights
@@ -34,6 +34,9 @@ class DEWSU(KNNBase):
34
34
  distance_metric : str
35
35
  Distance function to use for neighbor search. Default: 'euclidean'. See
36
36
  neighbors.list_distance_metrics() for all options and per-backend availability.
37
+ loo: bool
38
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
39
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
37
40
  """
38
41
 
39
42
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -75,7 +78,7 @@ class DEWSU(KNNBase):
75
78
  (0.5 if self.mode == 'min' else 1.0))
76
79
  th = threshold if threshold is not None else self.threshold
77
80
 
78
- _, indices = self.model._kneighbors(x, k=k, loo=loo) # (batch, k)
81
+ _, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
79
82
 
80
83
  # Average each model's scores over the K neighbors
81
84
  avg_scores = self.matrix[indices].mean(axis=1) # (batch, n_models)
@@ -40,6 +40,9 @@ class DEWSV(KNNBase):
40
40
  distance_metric : str
41
41
  Distance function to use for neighbor search. Default: 'euclidean'. See
42
42
  neighbors.list_distance_metrics() for all options and per-backend availability.
43
+ loo: bool
44
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
45
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
43
46
  """
44
47
 
45
48
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -94,7 +97,7 @@ class DEWSV(KNNBase):
94
97
  (0.5 if self.mode == 'min' else 1.0))
95
98
  th = threshold if threshold is not None else self.threshold
96
99
 
97
- _, indices = self.model._kneighbors(x, k=k, loo=loo) # (batch, k)
100
+ _, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
98
101
 
99
102
  # Uniform average of each model's scores over K neighbors
100
103
  neighbor_scores = self.matrix[indices] # (batch, k, n_models)
@@ -29,6 +29,9 @@ class KNORAE(KNNBase):
29
29
  distance_metric : str
30
30
  Distance function to use for neighbor search. Default: 'euclidean'. See
31
31
  neighbors.list_distance_metrics() for all options and per-backend availability.
32
+ loo: bool
33
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
34
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
32
35
  """
33
36
 
34
37
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -64,7 +67,7 @@ class KNORAE(KNNBase):
64
67
  th = threshold if threshold is not None else self.threshold
65
68
  n_models = len(self.models)
66
69
 
67
- _, indices = self.model._kneighbors(x, k=k, loo=loo)
70
+ _, indices = self._kneighbors(x, k=k, loo=loo)
68
71
  k = indices.shape[1]
69
72
  neighbor_scores = self.matrix[indices] # (batch, k, n_models)
70
73
 
@@ -30,6 +30,9 @@ class KNORAIU(KNNBase):
30
30
  distance_metric : str
31
31
  Distance function to use for neighbor search. Default: 'euclidean'. See
32
32
  neighbors.list_distance_metrics() for all options and per-backend availability.
33
+ loo: bool
34
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
35
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
33
36
  """
34
37
 
35
38
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -64,7 +67,7 @@ class KNORAIU(KNNBase):
64
67
  """
65
68
  th = threshold if threshold is not None else self.threshold
66
69
 
67
- distances, indices = self.model._kneighbors(x, k=k, loo=loo) # both (batch, k)
70
+ distances, indices = self._kneighbors(x, k=k, loo=loo) # both (batch, k)
68
71
  neighbor_scores = self.matrix[indices] # (batch, k, n_models)
69
72
 
70
73
  # Normalize per neighbor: best model = 1.0, worst = 0.0
@@ -30,6 +30,9 @@ class KNORAU(KNNBase):
30
30
  distance_metric : str
31
31
  Distance function to use for neighbor search. Default: 'euclidean'. See
32
32
  neighbors.list_distance_metrics() for all options and per-backend availability.
33
+ loo: bool
34
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
35
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
33
36
  """
34
37
 
35
38
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -64,7 +67,7 @@ class KNORAU(KNNBase):
64
67
  """
65
68
  th = threshold if threshold is not None else self.threshold
66
69
 
67
- _, indices = self.model._kneighbors(x, k=k, loo=loo)
70
+ _, indices = self._kneighbors(x, k=k, loo=loo)
68
71
  neighbor_scores = self.matrix[indices] # (batch, k, n_models)
69
72
 
70
73
  # Normalize per neighbor: best model = 1.0, worst = 0.0
@@ -22,6 +22,9 @@ class LWSEI(PredictBase):
22
22
  distance_metric : str
23
23
  Distance function to use for neighbor search. Default: 'euclidean'. See
24
24
  neighbors.list_distance_metrics() for all options and per-backend availability.
25
+ loo: bool
26
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
27
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
25
28
  """
26
29
 
27
30
  def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
@@ -87,7 +90,7 @@ class LWSEI(PredictBase):
87
90
  n_models = len(self.models)
88
91
  uniform = np.full(n_models, 1.0 / n_models)
89
92
 
90
- distances, indices = self._finder._kneighbors(x, k=k, loo=loo) # (batch, k)
93
+ distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
91
94
  weights_out = np.empty((batch_size, n_models))
92
95
 
93
96
  for b in range(batch_size):
@@ -22,6 +22,9 @@ class LWSEU(PredictBase):
22
22
  distance_metric : str
23
23
  Distance function to use for neighbor search. Default: 'euclidean'. See
24
24
  neighbors.list_distance_metrics() for all options and per-backend availability.
25
+ loo: bool
26
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
27
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
25
28
  """
26
29
 
27
30
  def __init__(self, task, k=10, preset='balanced', distance_metric='euclidean', **kwargs):
@@ -87,7 +90,7 @@ class LWSEU(PredictBase):
87
90
  n_models = len(self.models)
88
91
  uniform = np.full(n_models, 1.0 / n_models)
89
92
 
90
- distances, indices = self._finder._kneighbors(x, k=k, loo=loo) # (batch, k)
93
+ distances, indices = self._kneighbors(x, k=k, loo=loo) # (batch, k)
91
94
  weights_out = np.empty((batch_size, n_models))
92
95
 
93
96
  for b in range(batch_size):
@@ -25,6 +25,9 @@ class OLA(KNNBase):
25
25
  distance_metric : str
26
26
  Distance function to use for neighbor search. Default: 'euclidean'. See
27
27
  neighbors.list_distance_metrics() for all options and per-backend availability.
28
+ loo: bool
29
+ Enables Leave One Out (LOO) for hyperparameter tuning on the DSEL set. Default: 'false'.
30
+ Ignores closest neighbor with a negligible distance to avoid overfitting.
28
31
  """
29
32
 
30
33
  def __init__(self, task, metric='mae', mode='min', k=10,
@@ -63,7 +66,7 @@ class OLA(KNNBase):
63
66
  """
64
67
  batch_size = x.shape[0]
65
68
 
66
- _, indices = self.model._kneighbors(x, k=k, loo=loo)
69
+ _, indices = self._kneighbors(x, k=k, loo=loo)
67
70
  avg_scores = self.matrix[indices].mean(axis=1) # (batch, n_models)
68
71
  best_indices = np.argmax(avg_scores, axis=1)
69
72
 
@@ -4,38 +4,6 @@ import warnings
4
4
  # FAISS IVF k-means needs at least this many training samples per cell to converge.
5
5
  _FAISS_MIN_SAMPLES_PER_CELL = 40
6
6
 
7
-
8
- # ---------------------------------------------------------------------------
9
- # Distance metric registry
10
- # ---------------------------------------------------------------------------
11
-
12
- # Metrics supported by each backend.
13
- # 'euclidean' is the universal default and always available.
14
- #
15
- # Choosing a distance metric:
16
- # euclidean – The standard L2 norm. Best default for most tabular data.
17
- # manhattan – L1 norm (sum of absolute differences). More robust to
18
- # outliers and tends to work better in moderately high-
19
- # dimensional spaces because it doesn't square large diffs.
20
- # chebyshev – L∞ norm (maximum absolute difference across features).
21
- # Useful when a single feature dominating the distance is
22
- # acceptable; common in game-grid / chess-style problems.
23
- # minkowski – Generalisation of L1/L2 (controlled by p). p=1 →
24
- # manhattan, p=2 → euclidean. Use when you want to tune
25
- # between them.
26
- # cosine – Angle between vectors, ignoring magnitude. Excellent for
27
- # embeddings (text, image, audio) where direction matters
28
- # more than raw scale.
29
- # canberra – Weighted L1. Sensitive to small values near zero.
30
- # braycurtis – Normalised L1 bounded to [0,1]. Common in ecology.
31
- # jensenshannon – Symmetric KL divergence on probability distributions.
32
- # Requires non-negative vectors. Supported by FAISS flat/
33
- # HNSW/GPU indices natively.
34
- # dot – Raw inner/dot product. Not a true metric; distances are
35
- # not comparable across queries. Use for max inner-product
36
- # search (recommendation systems). Prefer 'cosine' for
37
- # normalised embeddings.
38
-
39
7
  # Metrics that every backend supports natively.
40
8
  _UNIVERSAL_METRICS = {'euclidean', 'manhattan', 'chebyshev', 'minkowski', 'cosine'}
41
9
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deskit
3
- Version: 1.2.4
3
+ Version: 1.2.6
4
4
  Summary: A Python library for Dynamic Ensemble Selection
5
5
  Author: Tikhon Vodyanov
6
6
  License-Expression: MIT
@@ -146,6 +146,17 @@ weights = router.predict_weights(X_test, temperature=0.1)
146
146
 
147
147
  ---
148
148
 
149
+ ### Hyperparameter tuning
150
+
151
+ `deskit` supports hyperparameter tuning of KNN-based methods using Leave One Out (LOO) tuning.
152
+ This technique excludes the closest data point with a negligent distance from the one being used in predict(), allowing for `deskit` methods to be tuned from the same DSEL set it was fit on and removing the need for cross-validation as it sees the entire set throughout the tuning process.
153
+
154
+ ```python
155
+ results = router.predict(X_val, val_preds, loo=True)
156
+ ```
157
+
158
+ ---
159
+
149
160
  ## Why deskit?
150
161
 
151
162
  Most DES libraries are tied to scikit-learn. deskit only ever sees a numpy
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes