imblearn-resc 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,117 @@
1
+ Metadata-Version: 2.4
2
+ Name: imblearn_resc
3
+ Version: 0.1.0
4
+ Summary: Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning
5
+ Author-email: maksimkins <maks.aleshkov.04@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/maksimkins/resampling-methods
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
13
+ Requires-Python: >=3.11
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: numpy>=1.24.0
16
+ Requires-Dist: scipy>=1.10.0
17
+ Requires-Dist: scikit-learn>=1.4.0
18
+ Requires-Dist: imbalanced-learn>=0.12.0
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
21
+
22
+ # imblearn-resc
23
+
24
+ **Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
25
+
26
+ This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
27
+
28
+ ## 📦 Installation
29
+
30
+ You can install `imblearn-resc` directly from PyPI using pip:
31
+
32
+ ```bash
33
+ pip install imblearn-resc
34
+ ```
35
+
36
+ *Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
37
+
38
+ ---
39
+
40
+ ## 🚀 Quick Start & Usage
41
+
42
+ Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
43
+
44
+ * **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
45
+ * **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
46
+
47
+ ### Example: Complete Pipeline
48
+
49
+ Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
50
+
51
+ ```python
52
+ from sklearn.datasets import make_classification
53
+ from sklearn.ensemble import RandomForestClassifier
54
+ from sklearn.model_selection import train_test_split
55
+ from sklearn.metrics import classification_report
56
+
57
+ # 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
58
+ from imblearn.pipeline import Pipeline
59
+
60
+ # 2. Import the Re-SC Samplers and Transformer
61
+ from imblearn_resc.oversampling import ReSC, KMeansReSC
62
+ from imblearn_resc.preprocessing import ReSCTransformer
63
+
64
+ # Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
65
+ X, y = make_classification(
66
+ n_classes=2, class_sep=2, weights=[0.1, 0.9],
67
+ n_informative=3, n_redundant=1, flip_y=0,
68
+ n_features=5, n_clusters_per_class=1,
69
+ n_samples=1000, random_state=42
70
+ )
71
+
72
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
73
+
74
+ # ==========================================
75
+ # Option A: Standard ReSC Pipeline
76
+ # ==========================================
77
+ pipeline_resc = Pipeline([
78
+ ('sampler', ReSC(M=1.5, k=5, random_state=42)),
79
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
80
+ ('classifier', RandomForestClassifier(random_state=42))
81
+ ])
82
+
83
+ # Train and Predict
84
+ pipeline_resc.fit(X_train, y_train)
85
+ y_pred_resc = pipeline_resc.predict(X_test)
86
+
87
+ print("ReSC Classification Report:")
88
+ print(classification_report(y_test, y_pred_resc))
89
+
90
+
91
+ # ==========================================
92
+ # Option B: KMeansReSC Pipeline
93
+ # ==========================================
94
+ pipeline_kmeans = Pipeline([
95
+ ('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
96
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
97
+ ('classifier', RandomForestClassifier(random_state=42))
98
+ ])
99
+
100
+ # Train and Predict
101
+ pipeline_kmeans.fit(X_train, y_train)
102
+ y_pred_kmeans = pipeline_kmeans.predict(X_test)
103
+
104
+ print("KMeansReSC Classification Report:")
105
+ print(classification_report(y_test, y_pred_kmeans))
106
+ ```
107
+
108
+ ## 🧠 Key Parameters
109
+
110
+ ### `ReSC`
111
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
112
+ * **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
113
+ * **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
114
+
115
+ ### `KMeansReSC`
116
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
117
+ * **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
@@ -0,0 +1,96 @@
1
+ # imblearn-resc
2
+
3
+ **Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
4
+
5
+ This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
6
+
7
+ ## 📦 Installation
8
+
9
+ You can install `imblearn-resc` directly from PyPI using pip:
10
+
11
+ ```bash
12
+ pip install imblearn-resc
13
+ ```
14
+
15
+ *Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
16
+
17
+ ---
18
+
19
+ ## 🚀 Quick Start & Usage
20
+
21
+ Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
22
+
23
+ * **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
24
+ * **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
25
+
26
+ ### Example: Complete Pipeline
27
+
28
+ Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
29
+
30
+ ```python
31
+ from sklearn.datasets import make_classification
32
+ from sklearn.ensemble import RandomForestClassifier
33
+ from sklearn.model_selection import train_test_split
34
+ from sklearn.metrics import classification_report
35
+
36
+ # 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
37
+ from imblearn.pipeline import Pipeline
38
+
39
+ # 2. Import the Re-SC Samplers and Transformer
40
+ from imblearn_resc.oversampling import ReSC, KMeansReSC
41
+ from imblearn_resc.preprocessing import ReSCTransformer
42
+
43
+ # Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
44
+ X, y = make_classification(
45
+ n_classes=2, class_sep=2, weights=[0.1, 0.9],
46
+ n_informative=3, n_redundant=1, flip_y=0,
47
+ n_features=5, n_clusters_per_class=1,
48
+ n_samples=1000, random_state=42
49
+ )
50
+
51
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
52
+
53
+ # ==========================================
54
+ # Option A: Standard ReSC Pipeline
55
+ # ==========================================
56
+ pipeline_resc = Pipeline([
57
+ ('sampler', ReSC(M=1.5, k=5, random_state=42)),
58
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
59
+ ('classifier', RandomForestClassifier(random_state=42))
60
+ ])
61
+
62
+ # Train and Predict
63
+ pipeline_resc.fit(X_train, y_train)
64
+ y_pred_resc = pipeline_resc.predict(X_test)
65
+
66
+ print("ReSC Classification Report:")
67
+ print(classification_report(y_test, y_pred_resc))
68
+
69
+
70
+ # ==========================================
71
+ # Option B: KMeansReSC Pipeline
72
+ # ==========================================
73
+ pipeline_kmeans = Pipeline([
74
+ ('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
75
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
76
+ ('classifier', RandomForestClassifier(random_state=42))
77
+ ])
78
+
79
+ # Train and Predict
80
+ pipeline_kmeans.fit(X_train, y_train)
81
+ y_pred_kmeans = pipeline_kmeans.predict(X_test)
82
+
83
+ print("KMeansReSC Classification Report:")
84
+ print(classification_report(y_test, y_pred_kmeans))
85
+ ```
86
+
87
+ ## 🧠 Key Parameters
88
+
89
+ ### `ReSC`
90
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
91
+ * **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
92
+ * **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
93
+
94
+ ### `KMeansReSC`
95
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
96
+ * **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
File without changes
@@ -0,0 +1,3 @@
1
+ from ._resc_kmeans import KMeansReSC
2
+
3
+ __all__ = ["KMeansReSC"]
@@ -0,0 +1,117 @@
1
+ from typing import Optional, Union, List, Tuple
2
+ from numbers import Real, Integral
3
+
4
+ import numpy as np
5
+ from numpy.typing import NDArray
6
+
7
+ from sklearn.utils import check_random_state
8
+ from sklearn.utils._param_validation import Interval
9
+
10
+ from imblearn.base import BaseSampler
11
+
12
+ from .utils._resc_kmeans_utils import (
13
+ get_set_n_kmeans_re_sc,
14
+ kmeans_re_sc_concatenation
15
+ )
16
+
17
+ class KMeansReSC(BaseSampler):
18
+ """
19
+ Resampling based on Sample Concatenation (Re-SC) using K-Means clustering.
20
+
21
+ This algorithm addresses class imbalance by mapping the data into a higher-dimensional
22
+ (2d) concatenated feature space. It identifies "safe" majority samples using KNN,
23
+ determines the optimal number of clusters (k) via Silhouette Score, and uses the
24
+ resulting K-Means cluster centers as the majority subset (Set_N).
25
+
26
+ Attributes:
27
+ M (float): The maximum acceptable imbalance ratio threshold for the resulting dataset.
28
+ num_candidates_to_test (int): How many 'k' values to test during geometric tuning.
29
+ random_state (int, RandomState instance, default=None): Controls the randomization of the algorithm.
30
+
31
+ Methods:
32
+ _fit_resample(X, y): Core resampling logic that executes KMeansReSC and returns concatenated arrays.
33
+ get_feature_names_out(input_features): Generates output feature names for the 2d concatenated space.
34
+ """
35
+ _sampling_type = 'over-sampling'
36
+ _parameter_constraints = {
37
+ "M": [Interval(Real, 0, None, closed="left")],
38
+ "num_candidates_to_test": [Interval(Integral, 1, None, closed="left")],
39
+ "random_state": ["random_state"]
40
+ }
41
+ def __init__(self, M=1.5, num_candidates_to_test=5, random_state=None):
42
+ super().__init__()
43
+ self.M = M
44
+ self.num_candidates_to_test = num_candidates_to_test
45
+ self.random_state = random_state
46
+
47
+ def _fit_resample(
48
+ self,
49
+ X: NDArray[np.float64],
50
+ y: NDArray[np.int_]
51
+ ) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
52
+ """
53
+ Executes resampling logic for KMeansReSC.
54
+
55
+ Args:
56
+ X (numpy.typing.NDArray[np.float64]): 2D matrix containing the features of the original training dataset.
57
+ y (numpy.typing.NDArray[np.int_]): 1D array containing the target labels.
58
+
59
+ Returns:
60
+ Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
61
+ A tuple containing the resampled feature matrix (mapped to a 2d space)
62
+ and the corresponding label array.
63
+
64
+ Raises:
65
+ ValueError: If the dataset does not contain at least two distinct classes.
66
+ """
67
+ random_state_obj = check_random_state(self.random_state)
68
+ seed = random_state_obj.randint(0, 2**31 - 1)
69
+
70
+ labels, counts = np.unique(y, return_counts=True)
71
+ if len(labels) < 2:
72
+ raise ValueError("The target 'y' needs to have at least two classes.")
73
+
74
+ min_label = labels[np.argmin(counts)]
75
+ maj_label = labels[np.argmax(counts)]
76
+
77
+ X_set_n = get_set_n_kmeans_re_sc(
78
+ X=X,
79
+ y=y,
80
+ min_label=min_label,
81
+ maj_label=maj_label,
82
+ M=self.M,
83
+ num_candidates_to_test=self.num_candidates_to_test,
84
+ random_state=seed
85
+ )
86
+
87
+ X_resampled, y_resampled = kmeans_re_sc_concatenation(
88
+ X_min=X[y == min_label],
89
+ X_maj=X[y == maj_label],
90
+ X_set_n=X_set_n,
91
+ min_label=min_label,
92
+ maj_label=maj_label
93
+ )
94
+
95
+ return X_resampled, y_resampled
96
+
97
+ def get_feature_names_out(
98
+ self,
99
+ input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
100
+ ) -> NDArray[np.object_]:
101
+ """
102
+ Get output feature names for transformation.
103
+
104
+ Args:
105
+ input_features (Optional[Union[List[str], numpy.typing.NDArray[np.object_]]]):
106
+ Original input feature names. If None, generic names are generated.
107
+
108
+ Returns:
109
+ numpy.typing.NDArray[np.object_]: An array of strings containing the new feature
110
+ names for the 2d concatenated space.
111
+ """
112
+ if input_features is None:
113
+ input_features = [f"x{i}" for i in range(self.n_features_in_)]
114
+
115
+ out_features = [f"{name}_1" for name in input_features] + [f"{name}_2" for name in input_features]
116
+
117
+ return np.asarray(out_features, dtype=object)
@@ -0,0 +1,180 @@
1
+ from typing import Tuple, List
2
+
3
+ import numpy as np
4
+ from numpy.typing import NDArray
5
+
6
+ from sklearn.cluster import KMeans
7
+ from sklearn.neighbors import KNeighborsClassifier
8
+ from sklearn.metrics import silhouette_score
9
+
10
+ def find_best_k_geometric(
11
+ X_maj: NDArray[np.float64],
12
+ k_candidates: List[int],
13
+ random_state: int
14
+ ) -> int:
15
+ """
16
+ Finds the optimal number of clusters (k) using the Silhouette Score.
17
+
18
+ Evaluates a list of candidate values for k by applying K-Means clustering
19
+ and selecting the value that maximizes the Silhouette Score. If only one
20
+ valid candidate is provided or all candidates are invalid, it provides a
21
+ safe fallback.
22
+
23
+ Args:
24
+ X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the majority class.
25
+ k_candidates (List[int]): A list of integer candidate values for the number of clusters (k) to test.
26
+ random_state (int): Seed used by the random number generator for K-Means initialization to ensure reproducibility.
27
+
28
+ Returns:
29
+ int: The optimal number of clusters (k) selected from the candidates.
30
+ """
31
+ if len(k_candidates) == 1:
32
+ single_k = k_candidates[0] if k_candidates[0] != 0 else 1
33
+ return single_k
34
+
35
+ best_k = None
36
+ best_score = -2.0
37
+
38
+ for k in k_candidates:
39
+ if k < 2 or k >= len(X_maj):
40
+ continue
41
+
42
+ kmeans = KMeans(n_clusters=k, random_state=random_state, n_init='auto')
43
+ cluster_labels = kmeans.fit_predict(X_maj)
44
+
45
+ score = silhouette_score(X_maj, cluster_labels)
46
+
47
+ if score > best_score:
48
+ best_score = score
49
+ best_k = k
50
+
51
+ if best_k is None:
52
+ best_k = max(1, k_candidates[0])
53
+
54
+ return best_k
55
+
56
+
57
+ def get_set_n_kmeans_re_sc(
58
+ X: NDArray[np.float64],
59
+ y: NDArray[np.int_],
60
+ min_label: int,
61
+ maj_label: int,
62
+ M: float = 1.5,
63
+ num_candidates_to_test: int = 5,
64
+ random_state: int = 42
65
+ ) -> NDArray[np.float64]:
66
+ """
67
+ Generates the Set_N subset using K-Means clustering on 'safe' majority samples.
68
+
69
+ This function identifies 'safe' majority samples using a KNN classifier (samples
70
+ with >= 90% probability of belonging to the majority class). It then dynamically
71
+ generates a list of candidate values for the number of clusters based on the
72
+ imbalance ratio bounds, finds the optimal k using the Silhouette Score, and
73
+ returns the resulting K-Means cluster centers to be used as Set_N.
74
+
75
+ Args:
76
+ X (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the entire training dataset.
77
+ y (numpy.typing.NDArray[np.int_]): 1D NumPy array containing the target labels.
78
+ min_label (int): The target label assigned to the minority class.
79
+ maj_label (int): The target label assigned to the majority class.
80
+ M (float, optional): The maximum acceptable imbalance ratio threshold. Defaults to 1.5.
81
+ num_candidates_to_test (int, optional): The number of candidate values for k to evaluate during geometric tuning. Defaults to 5.
82
+ random_state (int, optional): Seed used by the random number generator for reproducibility. Defaults to 42.
83
+
84
+ Returns:
85
+ numpy.typing.NDArray[np.float64]: A 2D NumPy array containing the features of the generated majority subset (the K-Means cluster centers).
86
+
87
+ Raises:
88
+ ValueError: If either the minority or majority class contains zero samples.
89
+ """
90
+ X_min = X[y == min_label]
91
+ X_maj = X[y == maj_label]
92
+
93
+ n_maj = len(X_maj)
94
+ n_min = len(X_min)
95
+
96
+ if n_maj == 0 or n_min == 0:
97
+ raise ValueError("Both minority and majority classes must have at least one sample.")
98
+
99
+ n1 = int((n_min ** 2) / n_maj)
100
+ upper_bound = int(M * n1)
101
+
102
+ knn = KNeighborsClassifier(n_neighbors=5)
103
+ knn.fit(X, y)
104
+
105
+ maj_class_idx = np.where(knn.classes_ == maj_label)[0][0]
106
+
107
+ probs = knn.predict_proba(X_maj)
108
+ prob_majority = probs[:, maj_class_idx]
109
+
110
+ safe_mask = prob_majority >= 0.9
111
+ X_maj_safe = X_maj[safe_mask]
112
+
113
+ if len(X_maj_safe) == 0:
114
+ X_maj_safe = X_maj
115
+
116
+ step = max(1, (upper_bound - n1) // max(1, (num_candidates_to_test - 1)))
117
+ candidates = list(range(n1, upper_bound + 1, step))
118
+
119
+ best_k = find_best_k_geometric(X_maj_safe, candidates, random_state)
120
+ best_k = min(best_k, len(X_maj_safe))
121
+
122
+ kmeans = KMeans(n_clusters=best_k, random_state=random_state, n_init='auto')
123
+ kmeans.fit(X_maj_safe)
124
+
125
+ X_set_n = kmeans.cluster_centers_
126
+
127
+ return X_set_n
128
+
129
+ def kmeans_re_sc_concatenation(
130
+ X_min: NDArray[np.float64],
131
+ X_maj: NDArray[np.float64],
132
+ X_set_n: NDArray[np.float64],
133
+ min_label: int = 1,
134
+ maj_label: int = -1
135
+ ) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
136
+ """
137
+ Concatenates pairs of samples from the same class to map the dataset into a 2d dimensional space.
138
+
139
+ Args:
140
+ X_min (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the minority class.
141
+ X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the original majority class.
142
+ X_set_n (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the selected majority subset.
143
+ min_label (int, optional): The target label assigned to the minority class.
144
+ maj_label (int, optional): The target label assigned to the majority class.
145
+
146
+ Returns:
147
+ Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
148
+ A tuple containing:
149
+ - X_resampled: The concatenated 2D NumPy array with 2 * d features.
150
+ - y_resampled: The 1D NumPy array containing the target labels for the new samples.
151
+
152
+ Raises:
153
+ ValueError: If X_min is empty, as the minority class must contain at least one sample.
154
+ """
155
+ m = len(X_min)
156
+ if m == 0:
157
+ raise ValueError("X_min cannot be empty. Minority class must have at least one sample.")
158
+
159
+ P_repeat = np.repeat(X_min, m, axis=0)
160
+ P_tile = np.tile(X_min, (m, 1))
161
+ P_c = np.hstack([P_repeat, P_tile])
162
+
163
+ y_p_c = np.full(len(P_c), min_label, dtype=np.int_)
164
+
165
+ M = len(X_maj)
166
+ k = len(X_set_n)
167
+
168
+ if k > 0:
169
+ N_repeat = np.repeat(X_maj, k, axis=0)
170
+ Set_N_tile = np.tile(X_set_n, (M, 1))
171
+ N_c = np.hstack([N_repeat, Set_N_tile])
172
+ y_n_c = np.full(len(N_c), maj_label, dtype=np.int_)
173
+
174
+ X_c_array = np.vstack([P_c, N_c])
175
+ y_c_array = np.hstack([y_p_c, y_n_c])
176
+ else:
177
+ X_c_array = P_c
178
+ y_c_array = y_p_c
179
+
180
+ return X_c_array, y_c_array
@@ -0,0 +1,3 @@
1
+ from ._resc import ReSC
2
+
3
+ __all__ = ["ReSC"]
@@ -0,0 +1,131 @@
1
+ from typing import Optional, Union, List, Tuple
2
+ from numbers import Real, Integral
3
+
4
+ import numpy as np
5
+ from numpy.typing import NDArray
6
+
7
+ from sklearn.utils import check_random_state
8
+ from sklearn.utils._param_validation import Interval
9
+
10
+ from imblearn.base import BaseSampler
11
+
12
+ from .utils._resc_utils import (
13
+ calculate_set_n_size_re_sc,
14
+ get_set_n_random_weighted_re_sc,
15
+ re_sc_concatenation
16
+ )
17
+
18
+
19
+ class ReSC(BaseSampler):
20
+ """
21
+ Resampling based on Sample Concatenation (Re-SC) using density-weighted random sampling.
22
+
23
+ This algorithm addresses class imbalance by mapping the data into a higher-dimensional
24
+ (2d) concatenated feature space. It over-samples the minority class by concatenating
25
+ minority samples with themselves, and under-samples the majority class by pairing
26
+ original majority samples with a statistically determined subset (Set_N).
27
+
28
+ Attributes:
29
+ M (float): The maximum acceptable imbalance ratio threshold for the resulting dataset.
30
+ k (int): Number of nearest neighbors used to calculate majority sample weights.
31
+ alpha (float): Significance level for the Z-test used to compute the required statistical sample size.
32
+ epsilon (float): Acceptable tolerance error for representing the majority class distribution.
33
+ random_state (int, RandomState instance, default=None): Controls the randomization of the algorithm.
34
+
35
+ Methods:
36
+ _fit_resample(X, y): Core resampling logic that executes Re-SC1 and returns concatenated arrays.
37
+ get_feature_names_out(input_features): Generates output feature names for the 2d concatenated space.
38
+ """
39
+ _sampling_type = 'over-sampling'
40
+ _parameter_constraints = {
41
+ "M": [Interval(Real, 0, None, closed="left")],
42
+ "k": [Interval(Integral, 1, None, closed="left")],
43
+ "alpha": [Interval(Real, 0, 1, closed="both")],
44
+ "epsilon": [Interval(Real, 0, None, closed="neither")],
45
+ "random_state": ["random_state"]
46
+ }
47
+ def __init__(self, M=1.5, k=5, alpha=0.05, epsilon=0.05, random_state=None):
48
+ super().__init__()
49
+ self.M = M
50
+ self.k = k
51
+ self.alpha = alpha
52
+ self.epsilon = epsilon
53
+ self.random_state = random_state
54
+
55
+ def _fit_resample(
56
+ self,
57
+ X: NDArray[np.float64],
58
+ y: NDArray[np.int_]
59
+ ) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
60
+ """
61
+ Executes resampling logic for Re-SC.
62
+
63
+ Args:
64
+ X (numpy.typing.NDArray[np.float64]): 2D matrix containing the features of the original training dataset.
65
+ y (numpy.typing.NDArray[np.int_]): 1D array containing the target labels.
66
+
67
+ Returns:
68
+ Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
69
+ A tuple containing the resampled feature matrix (mapped to a 2d space)
70
+ and the corresponding label array.
71
+
72
+ Raises:
73
+ ValueError: If the dataset does not contain at least two distinct classes.
74
+ """
75
+ random_state = check_random_state(self.random_state)
76
+ np.random.seed(random_state.randint(0, 2**31 - 1))
77
+
78
+ labels, counts = np.unique(y, return_counts=True)
79
+ if len(labels) < 2:
80
+ raise ValueError("The target 'y' needs to have at least two classes.")
81
+
82
+ min_label = labels[np.argmin(counts)]
83
+ maj_label = labels[np.argmax(counts)]
84
+
85
+ X_min = X[y == min_label]
86
+ X_maj = X[y == maj_label]
87
+
88
+ target_size = calculate_set_n_size_re_sc(
89
+ X_maj=X_maj,
90
+ P=len(X_min),
91
+ alpha=self.alpha,
92
+ epsilon=self.epsilon,
93
+ M=self.M
94
+ )
95
+ X_set_n = get_set_n_random_weighted_re_sc(
96
+ X=X,
97
+ y=y,
98
+ n_size=target_size,
99
+ k=self.k
100
+ )
101
+ X_resampled, y_resampled = re_sc_concatenation(
102
+ X_min=X_min,
103
+ X_maj=X_maj,
104
+ X_set_n=X_set_n,
105
+ min_label=min_label,
106
+ maj_label=maj_label
107
+ )
108
+
109
+ return X_resampled, y_resampled
110
+
111
+ def get_feature_names_out(
112
+ self,
113
+ input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
114
+ ) -> NDArray[np.object_]:
115
+ """
116
+ Get output feature names for transformation.
117
+
118
+ Args:
119
+ input_features (Optional[Union[List[str], numpy.typing.NDArray[np.object_]]]):
120
+ Original input feature names. If None, generic names are generated.
121
+
122
+ Returns:
123
+ numpy.typing.NDArray[np.object_]: An array of strings containing the new feature
124
+ names for the 2d concatenated space.
125
+ """
126
+ if input_features is None:
127
+ input_features = [f"x{i}" for i in range(self.n_features_in_)]
128
+
129
+ out_features = [f"{name}_1" for name in input_features] + [f"{name}_2" for name in input_features]
130
+
131
+ return np.asarray(out_features, dtype=object)
@@ -0,0 +1,184 @@
1
+ from typing import Tuple
2
+
3
+ import numpy as np
4
+ from numpy.typing import NDArray
5
+
6
+ import scipy.stats as stats
7
+
8
+ from sklearn.neighbors import NearestNeighbors
9
+
10
+ def calculate_set_n_size_re_sc(
11
+ X_maj: NDArray[np.float64],
12
+ P: int,
13
+ alpha: float = 0.05,
14
+ epsilon: float = 0.05,
15
+ M: float = 1.5
16
+ ) -> int:
17
+ """
18
+ Calculates the target size for the majority subset (Set_N) in the Re-SC algorithm.
19
+
20
+ Args:
21
+ X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing features of the majority class.
22
+ P (int): number of samples in the minority class.
23
+ alpha (float): significance level for the Z-test used to compute statistical sample size.
24
+ epsilon (float): acceptable tolerance error for representing the majority class distribution
25
+ M (float): acceptable imbalance ratio threshold.
26
+
27
+ Returns:
28
+ int: The calculated number of majority samples for Set_N.
29
+
30
+ Raises:
31
+ ValueError: If X_maj is an empty array.
32
+ """
33
+ n_maj: int = len(X_maj)
34
+ if n_maj == 0:
35
+ raise ValueError("X_maj cannot be empty.")
36
+
37
+ n1 = (P ** 2) / n_maj
38
+ z_score = stats.norm.ppf(1 - alpha / 2)
39
+ sigma = np.std(X_maj, ddof=1)
40
+ z_sq = z_score ** 2
41
+ sigma_sq = sigma ** 2
42
+ epsilon_sq = epsilon ** 2
43
+
44
+ numerator = n_maj * z_sq * sigma_sq
45
+ denominator = (n_maj * epsilon_sq) + (z_sq * sigma_sq)
46
+ n2 = numerator / denominator
47
+
48
+ if n1 == 0:
49
+ return int(n2)
50
+
51
+ Pr = n2 / n1
52
+
53
+ if Pr < 1:
54
+ set_n_size = n1
55
+ elif 1 <= Pr <= M:
56
+ set_n_size = n2
57
+ else:
58
+ set_n_size = M * n1
59
+
60
+ return int(np.ceil(set_n_size))
61
+
62
+ def get_set_n_random_weighted_re_sc(
63
+ X: NDArray[np.float64],
64
+ y: NDArray[np.int_],
65
+ n_size: int,
66
+ k: int = 5
67
+ ) -> NDArray[np.float64]:
68
+ """
69
+ Selects a subset of majority class samples (Set_N) using a density-weighted random sampling strategy.
70
+
71
+ Args:
72
+ X (numpy.typing.NDArray[np.float64]): 2D NumPy array containing entire training dataset.
73
+ y (numpy.typing.NDArray[np.int_]): 1D NumPy array containing labels.
74
+ n_size (int): number of majority samples to select.
75
+ k (int, optional): number of nearest neighbors to evaluate for the weighting mechanism. Defaults to 5.
76
+
77
+ Returns:
78
+ numpy.typing.NDArray[np.float64]: A 2D NumPy array containing selected majority subset.
79
+
80
+ Raises:
81
+ ValueError: If there are no majority samples present in the dataset.
82
+ ValueError: If no valid majority samples are found.
83
+ """
84
+ labels, counts = np.unique(y, return_counts=True)
85
+ maj_label = labels[np.argmax(counts)]
86
+
87
+ maj_mask = (y == maj_label)
88
+ maj_indices = np.where(maj_mask)[0]
89
+
90
+ if len(maj_indices) == 0:
91
+ raise ValueError("No majority samples found in the dataset.")
92
+
93
+ X_mean = np.mean(X, axis=0)
94
+ X_std = np.std(X, axis=0)
95
+ X_std[X_std == 0] = 1.0
96
+ X_norm = (X - X_mean) / X_std
97
+
98
+ knn = NearestNeighbors(n_neighbors=k + 1).fit(X_norm)
99
+
100
+ X_maj_norm = X_norm[maj_indices]
101
+ distances, neighbor_idxs = knn.kneighbors(X_maj_norm)
102
+
103
+ weights = []
104
+ valid_indices = []
105
+
106
+ for i, n_idxs in enumerate(neighbor_idxs):
107
+ actual_neighbors = n_idxs[1:]
108
+ maj_neighbor_count = np.sum(y[actual_neighbors] == maj_label)
109
+ weight = maj_neighbor_count / k
110
+
111
+ if weight > 0:
112
+ weights.append(weight)
113
+ valid_indices.append(maj_indices[i])
114
+
115
+ weights_arr = np.array(weights, dtype=np.float64)
116
+ valid_indices_arr = np.array(valid_indices, dtype=np.int_)
117
+
118
+ if len(weights_arr) == 0:
119
+ raise ValueError("No valid majority samples found (all are surrounded by minority class).")
120
+
121
+ probs = weights_arr / np.sum(weights_arr)
122
+ actual_n_size = min(n_size, len(valid_indices_arr))
123
+
124
+ selected_indices = np.random.choice(
125
+ valid_indices_arr,
126
+ size=actual_n_size,
127
+ replace=False,
128
+ p=probs
129
+ )
130
+
131
+ return X[selected_indices]
132
+
133
+ def re_sc_concatenation(
134
+ X_min: NDArray[np.float64],
135
+ X_maj: NDArray[np.float64],
136
+ X_set_n: NDArray[np.float64],
137
+ min_label: int = 1,
138
+ maj_label: int = -1
139
+ ) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
140
+ """
141
+ Concatenates pairs of samples from the same class to map the dataset into a 2d dimensional space.
142
+
143
+ Args:
144
+ X_min (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the minority class.
145
+ X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the original majority class.
146
+ X_set_n (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the selected majority subset.
147
+ min_label (int, optional): The target label assigned to the minority class.
148
+ maj_label (int, optional): The target label assigned to the majority class.
149
+
150
+ Returns:
151
+ Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
152
+ A tuple containing:
153
+ - X_resampled: The concatenated 2D NumPy array with 2 * d features.
154
+ - y_resampled: The 1D NumPy array containing the target labels for the new samples.
155
+
156
+ Raises:
157
+ ValueError: If X_min is empty, as the minority class must contain at least one sample.
158
+ """
159
+ m = len(X_min)
160
+ if m == 0:
161
+ raise ValueError("X_min cannot be empty. Minority class must have at least one sample.")
162
+
163
+ P_repeat = np.repeat(X_min, m, axis=0)
164
+ P_tile = np.tile(X_min, (m, 1))
165
+ P_c = np.hstack([P_repeat, P_tile])
166
+
167
+ y_p_c = np.full(len(P_c), min_label, dtype=np.int_)
168
+
169
+ M = len(X_maj)
170
+ k = len(X_set_n)
171
+
172
+ if k > 0:
173
+ N_repeat = np.repeat(X_maj, k, axis=0)
174
+ Set_N_tile = np.tile(X_set_n, (M, 1))
175
+ N_c = np.hstack([N_repeat, Set_N_tile])
176
+ y_n_c = np.full(len(N_c), maj_label, dtype=np.int_)
177
+
178
+ X_c_array = np.vstack([P_c, N_c])
179
+ y_c_array = np.hstack([y_p_c, y_n_c])
180
+ else:
181
+ X_c_array = P_c
182
+ y_c_array = y_p_c
183
+
184
+ return X_c_array, y_c_array
@@ -0,0 +1,4 @@
1
+ from .ReSC import ReSC
2
+ from .KMeansReSC import KMeansReSC
3
+
4
+ __all__ = ["ReSC", "KMeansReSC"]
@@ -0,0 +1,3 @@
1
+ from ._resc_transformer import ReSCTransformer
2
+
3
+ __all__ = ["ReSCTransformer"]
@@ -0,0 +1,55 @@
1
+ from typing import Optional, Union, List
2
+
3
+ import numpy as np
4
+ from numpy.typing import NDArray
5
+
6
+ from sklearn.base import BaseEstimator, TransformerMixin
7
+ from sklearn.utils.validation import check_is_fitted, check_array, validate_data
8
+
9
+
10
+ class ReSCTransformer(TransformerMixin, BaseEstimator):
11
+ """
12
+ Transforms test data into a 2d concatenated space for prediction with Re-SC models.
13
+
14
+ This transformer must be placed immediately after a Re-SC sampler in a Pipeline.
15
+ - During training (fit_transform), it receives data that the Sampler has already
16
+ mapped to 2d space, so it simply passes it through.
17
+ - During testing (predict), the Sampler is bypassed, so this Transformer receives
18
+ data in the original d space. It duplicates the features (x -> [x, x]) to
19
+ match the 2d space the classifier expects.
20
+ """
21
+
22
+ def __init__(self):
23
+ pass
24
+
25
+ def fit(self, X: NDArray[np.float64], y: Optional[NDArray] = None) -> 'ReSCTransformer':
26
+ X = validate_data(self, X=X, reset=True, accept_sparse=False)
27
+ self.original_d_ = self.n_features_in_ // 2
28
+ self.is_fitted_ = True
29
+
30
+ return self
31
+
32
+ def transform(self, X: NDArray[np.float64]) -> NDArray[np.float64]:
33
+ check_is_fitted(self, 'is_fitted_')
34
+ X = check_array(X, accept_sparse=False)
35
+
36
+ if X.shape[1] == self.original_d_:
37
+ return np.hstack([X, X])
38
+
39
+ X = validate_data(self, X, reset=False, accept_sparse=False)
40
+
41
+ return X
42
+
43
+ def get_feature_names_out(
44
+ self,
45
+ input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
46
+ ) -> NDArray[np.object_]:
47
+ check_is_fitted(self, 'is_fitted_')
48
+
49
+ if input_features is None:
50
+ input_features = [f"x{i}" for i in range(self.original_d_)]
51
+
52
+ out_features = [f"{name}_1" for name in input_features] + \
53
+ [f"{name}_2" for name in input_features]
54
+
55
+ return np.asarray(out_features, dtype=object)
@@ -0,0 +1,117 @@
1
+ Metadata-Version: 2.4
2
+ Name: imblearn_resc
3
+ Version: 0.1.0
4
+ Summary: Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning
5
+ Author-email: maksimkins <maks.aleshkov.04@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/maksimkins/resampling-methods
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
13
+ Requires-Python: >=3.11
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: numpy>=1.24.0
16
+ Requires-Dist: scipy>=1.10.0
17
+ Requires-Dist: scikit-learn>=1.4.0
18
+ Requires-Dist: imbalanced-learn>=0.12.0
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
21
+
22
+ # imblearn-resc
23
+
24
+ **Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
25
+
26
+ This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
27
+
28
+ ## 📦 Installation
29
+
30
+ You can install `imblearn-resc` directly from PyPI using pip:
31
+
32
+ ```bash
33
+ pip install imblearn-resc
34
+ ```
35
+
36
+ *Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
37
+
38
+ ---
39
+
40
+ ## 🚀 Quick Start & Usage
41
+
42
+ Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
43
+
44
+ * **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
45
+ * **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
46
+
47
+ ### Example: Complete Pipeline
48
+
49
+ Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
50
+
51
+ ```python
52
+ from sklearn.datasets import make_classification
53
+ from sklearn.ensemble import RandomForestClassifier
54
+ from sklearn.model_selection import train_test_split
55
+ from sklearn.metrics import classification_report
56
+
57
+ # 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
58
+ from imblearn.pipeline import Pipeline
59
+
60
+ # 2. Import the Re-SC Samplers and Transformer
61
+ from imblearn_resc.oversampling import ReSC, KMeansReSC
62
+ from imblearn_resc.preprocessing import ReSCTransformer
63
+
64
+ # Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
65
+ X, y = make_classification(
66
+ n_classes=2, class_sep=2, weights=[0.1, 0.9],
67
+ n_informative=3, n_redundant=1, flip_y=0,
68
+ n_features=5, n_clusters_per_class=1,
69
+ n_samples=1000, random_state=42
70
+ )
71
+
72
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
73
+
74
+ # ==========================================
75
+ # Option A: Standard ReSC Pipeline
76
+ # ==========================================
77
+ pipeline_resc = Pipeline([
78
+ ('sampler', ReSC(M=1.5, k=5, random_state=42)),
79
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
80
+ ('classifier', RandomForestClassifier(random_state=42))
81
+ ])
82
+
83
+ # Train and Predict
84
+ pipeline_resc.fit(X_train, y_train)
85
+ y_pred_resc = pipeline_resc.predict(X_test)
86
+
87
+ print("ReSC Classification Report:")
88
+ print(classification_report(y_test, y_pred_resc))
89
+
90
+
91
+ # ==========================================
92
+ # Option B: KMeansReSC Pipeline
93
+ # ==========================================
94
+ pipeline_kmeans = Pipeline([
95
+ ('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
96
+ ('transformer', ReSCTransformer()), # <--- Mandatory!
97
+ ('classifier', RandomForestClassifier(random_state=42))
98
+ ])
99
+
100
+ # Train and Predict
101
+ pipeline_kmeans.fit(X_train, y_train)
102
+ y_pred_kmeans = pipeline_kmeans.predict(X_test)
103
+
104
+ print("KMeansReSC Classification Report:")
105
+ print(classification_report(y_test, y_pred_kmeans))
106
+ ```
107
+
108
+ ## 🧠 Key Parameters
109
+
110
+ ### `ReSC`
111
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
112
+ * **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
113
+ * **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
114
+
115
+ ### `KMeansReSC`
116
+ * **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
117
+ * **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
@@ -0,0 +1,19 @@
1
+ README.md
2
+ pyproject.toml
3
+ imblearn_resc/__init__.py
4
+ imblearn_resc.egg-info/PKG-INFO
5
+ imblearn_resc.egg-info/SOURCES.txt
6
+ imblearn_resc.egg-info/dependency_links.txt
7
+ imblearn_resc.egg-info/requires.txt
8
+ imblearn_resc.egg-info/top_level.txt
9
+ imblearn_resc/oversampling/__init__.py
10
+ imblearn_resc/oversampling/KMeansReSC/__init__.py
11
+ imblearn_resc/oversampling/KMeansReSC/_resc_kmeans.py
12
+ imblearn_resc/oversampling/KMeansReSC/utils/__init__.py
13
+ imblearn_resc/oversampling/KMeansReSC/utils/_resc_kmeans_utils.py
14
+ imblearn_resc/oversampling/ReSC/__init__.py
15
+ imblearn_resc/oversampling/ReSC/_resc.py
16
+ imblearn_resc/oversampling/ReSC/utils/__init__.py
17
+ imblearn_resc/oversampling/ReSC/utils/_resc_utils.py
18
+ imblearn_resc/preprocessing/__init__.py
19
+ imblearn_resc/preprocessing/_resc_transformer.py
@@ -0,0 +1,7 @@
1
+ numpy>=1.24.0
2
+ scipy>=1.10.0
3
+ scikit-learn>=1.4.0
4
+ imbalanced-learn>=0.12.0
5
+
6
+ [dev]
7
+ pytest>=7.0.0
@@ -0,0 +1 @@
1
+ imblearn_resc
@@ -0,0 +1,45 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "imblearn_resc"
7
+ version = "0.1.0"
8
+ description = "Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning"
9
+ authors = [
10
+ {name = "maksimkins", email = "maks.aleshkov.04@gmail.com"}
11
+ ]
12
+ readme = "README.md"
13
+ requires-python = ">=3.11"
14
+ license = {text = "MIT"}
15
+
16
+ classifiers = [
17
+ "Programming Language :: Python :: 3",
18
+ "License :: OSI Approved :: MIT License",
19
+ "Operating System :: OS Independent",
20
+ "Intended Audience :: Science/Research",
21
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
22
+ ]
23
+
24
+ dependencies = [
25
+ "numpy>=1.24.0",
26
+ "scipy>=1.10.0",
27
+ "scikit-learn>=1.4.0",
28
+ "imbalanced-learn>=0.12.0"
29
+ ]
30
+
31
+ [project.urls]
32
+ Homepage = "https://github.com/maksimkins/resampling-methods"
33
+
34
+
35
+ [project.optional-dependencies]
36
+ dev = [
37
+ "pytest>=7.0.0"
38
+ ]
39
+
40
+ [tool.setuptools.packages.find]
41
+ where = ["."]
42
+ include = ["imblearn_resc*"]
43
+
44
+ [tool.pytest.ini_options]
45
+ pythonpath = ["."]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+