imblearn-resc 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- imblearn_resc-0.1.0/PKG-INFO +117 -0
- imblearn_resc-0.1.0/README.md +96 -0
- imblearn_resc-0.1.0/imblearn_resc/__init__.py +0 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/KMeansReSC/__init__.py +3 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/KMeansReSC/_resc_kmeans.py +117 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/KMeansReSC/utils/__init__.py +0 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/KMeansReSC/utils/_resc_kmeans_utils.py +180 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/ReSC/__init__.py +3 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/ReSC/_resc.py +131 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/ReSC/utils/__init__.py +0 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/ReSC/utils/_resc_utils.py +184 -0
- imblearn_resc-0.1.0/imblearn_resc/oversampling/__init__.py +4 -0
- imblearn_resc-0.1.0/imblearn_resc/preprocessing/__init__.py +3 -0
- imblearn_resc-0.1.0/imblearn_resc/preprocessing/_resc_transformer.py +55 -0
- imblearn_resc-0.1.0/imblearn_resc.egg-info/PKG-INFO +117 -0
- imblearn_resc-0.1.0/imblearn_resc.egg-info/SOURCES.txt +19 -0
- imblearn_resc-0.1.0/imblearn_resc.egg-info/dependency_links.txt +1 -0
- imblearn_resc-0.1.0/imblearn_resc.egg-info/requires.txt +7 -0
- imblearn_resc-0.1.0/imblearn_resc.egg-info/top_level.txt +1 -0
- imblearn_resc-0.1.0/pyproject.toml +45 -0
- imblearn_resc-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: imblearn_resc
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning
|
|
5
|
+
Author-email: maksimkins <maks.aleshkov.04@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/maksimkins/resampling-methods
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
+
Requires-Python: >=3.11
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: numpy>=1.24.0
|
|
16
|
+
Requires-Dist: scipy>=1.10.0
|
|
17
|
+
Requires-Dist: scikit-learn>=1.4.0
|
|
18
|
+
Requires-Dist: imbalanced-learn>=0.12.0
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
21
|
+
|
|
22
|
+
# imblearn-resc
|
|
23
|
+
|
|
24
|
+
**Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
|
|
25
|
+
|
|
26
|
+
This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
|
|
27
|
+
|
|
28
|
+
## 📦 Installation
|
|
29
|
+
|
|
30
|
+
You can install `imblearn-resc` directly from PyPI using pip:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install imblearn-resc
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
*Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## 🚀 Quick Start & Usage
|
|
41
|
+
|
|
42
|
+
Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
|
|
43
|
+
|
|
44
|
+
* **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
|
|
45
|
+
* **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
|
|
46
|
+
|
|
47
|
+
### Example: Complete Pipeline
|
|
48
|
+
|
|
49
|
+
Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from sklearn.datasets import make_classification
|
|
53
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
54
|
+
from sklearn.model_selection import train_test_split
|
|
55
|
+
from sklearn.metrics import classification_report
|
|
56
|
+
|
|
57
|
+
# 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
|
|
58
|
+
from imblearn.pipeline import Pipeline
|
|
59
|
+
|
|
60
|
+
# 2. Import the Re-SC Samplers and Transformer
|
|
61
|
+
from imblearn_resc.oversampling import ReSC, KMeansReSC
|
|
62
|
+
from imblearn_resc.preprocessing import ReSCTransformer
|
|
63
|
+
|
|
64
|
+
# Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
|
|
65
|
+
X, y = make_classification(
|
|
66
|
+
n_classes=2, class_sep=2, weights=[0.1, 0.9],
|
|
67
|
+
n_informative=3, n_redundant=1, flip_y=0,
|
|
68
|
+
n_features=5, n_clusters_per_class=1,
|
|
69
|
+
n_samples=1000, random_state=42
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
|
|
73
|
+
|
|
74
|
+
# ==========================================
|
|
75
|
+
# Option A: Standard ReSC Pipeline
|
|
76
|
+
# ==========================================
|
|
77
|
+
pipeline_resc = Pipeline([
|
|
78
|
+
('sampler', ReSC(M=1.5, k=5, random_state=42)),
|
|
79
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
80
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
81
|
+
])
|
|
82
|
+
|
|
83
|
+
# Train and Predict
|
|
84
|
+
pipeline_resc.fit(X_train, y_train)
|
|
85
|
+
y_pred_resc = pipeline_resc.predict(X_test)
|
|
86
|
+
|
|
87
|
+
print("ReSC Classification Report:")
|
|
88
|
+
print(classification_report(y_test, y_pred_resc))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ==========================================
|
|
92
|
+
# Option B: KMeansReSC Pipeline
|
|
93
|
+
# ==========================================
|
|
94
|
+
pipeline_kmeans = Pipeline([
|
|
95
|
+
('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
|
|
96
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
97
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
98
|
+
])
|
|
99
|
+
|
|
100
|
+
# Train and Predict
|
|
101
|
+
pipeline_kmeans.fit(X_train, y_train)
|
|
102
|
+
y_pred_kmeans = pipeline_kmeans.predict(X_test)
|
|
103
|
+
|
|
104
|
+
print("KMeansReSC Classification Report:")
|
|
105
|
+
print(classification_report(y_test, y_pred_kmeans))
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## 🧠Key Parameters
|
|
109
|
+
|
|
110
|
+
### `ReSC`
|
|
111
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
112
|
+
* **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
|
|
113
|
+
* **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
|
|
114
|
+
|
|
115
|
+
### `KMeansReSC`
|
|
116
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
117
|
+
* **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# imblearn-resc
|
|
2
|
+
|
|
3
|
+
**Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
|
|
4
|
+
|
|
5
|
+
This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
|
|
6
|
+
|
|
7
|
+
## 📦 Installation
|
|
8
|
+
|
|
9
|
+
You can install `imblearn-resc` directly from PyPI using pip:
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install imblearn-resc
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
*Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
|
|
16
|
+
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
## 🚀 Quick Start & Usage
|
|
20
|
+
|
|
21
|
+
Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
|
|
22
|
+
|
|
23
|
+
* **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
|
|
24
|
+
* **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
|
|
25
|
+
|
|
26
|
+
### Example: Complete Pipeline
|
|
27
|
+
|
|
28
|
+
Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from sklearn.datasets import make_classification
|
|
32
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
33
|
+
from sklearn.model_selection import train_test_split
|
|
34
|
+
from sklearn.metrics import classification_report
|
|
35
|
+
|
|
36
|
+
# 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
|
|
37
|
+
from imblearn.pipeline import Pipeline
|
|
38
|
+
|
|
39
|
+
# 2. Import the Re-SC Samplers and Transformer
|
|
40
|
+
from imblearn_resc.oversampling import ReSC, KMeansReSC
|
|
41
|
+
from imblearn_resc.preprocessing import ReSCTransformer
|
|
42
|
+
|
|
43
|
+
# Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
|
|
44
|
+
X, y = make_classification(
|
|
45
|
+
n_classes=2, class_sep=2, weights=[0.1, 0.9],
|
|
46
|
+
n_informative=3, n_redundant=1, flip_y=0,
|
|
47
|
+
n_features=5, n_clusters_per_class=1,
|
|
48
|
+
n_samples=1000, random_state=42
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
|
|
52
|
+
|
|
53
|
+
# ==========================================
|
|
54
|
+
# Option A: Standard ReSC Pipeline
|
|
55
|
+
# ==========================================
|
|
56
|
+
pipeline_resc = Pipeline([
|
|
57
|
+
('sampler', ReSC(M=1.5, k=5, random_state=42)),
|
|
58
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
59
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
60
|
+
])
|
|
61
|
+
|
|
62
|
+
# Train and Predict
|
|
63
|
+
pipeline_resc.fit(X_train, y_train)
|
|
64
|
+
y_pred_resc = pipeline_resc.predict(X_test)
|
|
65
|
+
|
|
66
|
+
print("ReSC Classification Report:")
|
|
67
|
+
print(classification_report(y_test, y_pred_resc))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ==========================================
|
|
71
|
+
# Option B: KMeansReSC Pipeline
|
|
72
|
+
# ==========================================
|
|
73
|
+
pipeline_kmeans = Pipeline([
|
|
74
|
+
('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
|
|
75
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
76
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
77
|
+
])
|
|
78
|
+
|
|
79
|
+
# Train and Predict
|
|
80
|
+
pipeline_kmeans.fit(X_train, y_train)
|
|
81
|
+
y_pred_kmeans = pipeline_kmeans.predict(X_test)
|
|
82
|
+
|
|
83
|
+
print("KMeansReSC Classification Report:")
|
|
84
|
+
print(classification_report(y_test, y_pred_kmeans))
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## 🧠Key Parameters
|
|
88
|
+
|
|
89
|
+
### `ReSC`
|
|
90
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
91
|
+
* **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
|
|
92
|
+
* **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
|
|
93
|
+
|
|
94
|
+
### `KMeansReSC`
|
|
95
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
96
|
+
* **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
|
|
File without changes
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
from typing import Optional, Union, List, Tuple
|
|
2
|
+
from numbers import Real, Integral
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
from numpy.typing import NDArray
|
|
6
|
+
|
|
7
|
+
from sklearn.utils import check_random_state
|
|
8
|
+
from sklearn.utils._param_validation import Interval
|
|
9
|
+
|
|
10
|
+
from imblearn.base import BaseSampler
|
|
11
|
+
|
|
12
|
+
from .utils._resc_kmeans_utils import (
|
|
13
|
+
get_set_n_kmeans_re_sc,
|
|
14
|
+
kmeans_re_sc_concatenation
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
class KMeansReSC(BaseSampler):
|
|
18
|
+
"""
|
|
19
|
+
Resampling based on Sample Concatenation (Re-SC) using K-Means clustering.
|
|
20
|
+
|
|
21
|
+
This algorithm addresses class imbalance by mapping the data into a higher-dimensional
|
|
22
|
+
(2d) concatenated feature space. It identifies "safe" majority samples using KNN,
|
|
23
|
+
determines the optimal number of clusters (k) via Silhouette Score, and uses the
|
|
24
|
+
resulting K-Means cluster centers as the majority subset (Set_N).
|
|
25
|
+
|
|
26
|
+
Attributes:
|
|
27
|
+
M (float): The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
28
|
+
num_candidates_to_test (int): How many 'k' values to test during geometric tuning.
|
|
29
|
+
random_state (int, RandomState instance, default=None): Controls the randomization of the algorithm.
|
|
30
|
+
|
|
31
|
+
Methods:
|
|
32
|
+
_fit_resample(X, y): Core resampling logic that executes KMeansReSC and returns concatenated arrays.
|
|
33
|
+
get_feature_names_out(input_features): Generates output feature names for the 2d concatenated space.
|
|
34
|
+
"""
|
|
35
|
+
_sampling_type = 'over-sampling'
|
|
36
|
+
_parameter_constraints = {
|
|
37
|
+
"M": [Interval(Real, 0, None, closed="left")],
|
|
38
|
+
"num_candidates_to_test": [Interval(Integral, 1, None, closed="left")],
|
|
39
|
+
"random_state": ["random_state"]
|
|
40
|
+
}
|
|
41
|
+
def __init__(self, M=1.5, num_candidates_to_test=5, random_state=None):
|
|
42
|
+
super().__init__()
|
|
43
|
+
self.M = M
|
|
44
|
+
self.num_candidates_to_test = num_candidates_to_test
|
|
45
|
+
self.random_state = random_state
|
|
46
|
+
|
|
47
|
+
def _fit_resample(
|
|
48
|
+
self,
|
|
49
|
+
X: NDArray[np.float64],
|
|
50
|
+
y: NDArray[np.int_]
|
|
51
|
+
) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
|
|
52
|
+
"""
|
|
53
|
+
Executes resampling logic for KMeansReSC.
|
|
54
|
+
|
|
55
|
+
Args:
|
|
56
|
+
X (numpy.typing.NDArray[np.float64]): 2D matrix containing the features of the original training dataset.
|
|
57
|
+
y (numpy.typing.NDArray[np.int_]): 1D array containing the target labels.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
|
|
61
|
+
A tuple containing the resampled feature matrix (mapped to a 2d space)
|
|
62
|
+
and the corresponding label array.
|
|
63
|
+
|
|
64
|
+
Raises:
|
|
65
|
+
ValueError: If the dataset does not contain at least two distinct classes.
|
|
66
|
+
"""
|
|
67
|
+
random_state_obj = check_random_state(self.random_state)
|
|
68
|
+
seed = random_state_obj.randint(0, 2**31 - 1)
|
|
69
|
+
|
|
70
|
+
labels, counts = np.unique(y, return_counts=True)
|
|
71
|
+
if len(labels) < 2:
|
|
72
|
+
raise ValueError("The target 'y' needs to have at least two classes.")
|
|
73
|
+
|
|
74
|
+
min_label = labels[np.argmin(counts)]
|
|
75
|
+
maj_label = labels[np.argmax(counts)]
|
|
76
|
+
|
|
77
|
+
X_set_n = get_set_n_kmeans_re_sc(
|
|
78
|
+
X=X,
|
|
79
|
+
y=y,
|
|
80
|
+
min_label=min_label,
|
|
81
|
+
maj_label=maj_label,
|
|
82
|
+
M=self.M,
|
|
83
|
+
num_candidates_to_test=self.num_candidates_to_test,
|
|
84
|
+
random_state=seed
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
X_resampled, y_resampled = kmeans_re_sc_concatenation(
|
|
88
|
+
X_min=X[y == min_label],
|
|
89
|
+
X_maj=X[y == maj_label],
|
|
90
|
+
X_set_n=X_set_n,
|
|
91
|
+
min_label=min_label,
|
|
92
|
+
maj_label=maj_label
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
return X_resampled, y_resampled
|
|
96
|
+
|
|
97
|
+
def get_feature_names_out(
|
|
98
|
+
self,
|
|
99
|
+
input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
|
|
100
|
+
) -> NDArray[np.object_]:
|
|
101
|
+
"""
|
|
102
|
+
Get output feature names for transformation.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
input_features (Optional[Union[List[str], numpy.typing.NDArray[np.object_]]]):
|
|
106
|
+
Original input feature names. If None, generic names are generated.
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
numpy.typing.NDArray[np.object_]: An array of strings containing the new feature
|
|
110
|
+
names for the 2d concatenated space.
|
|
111
|
+
"""
|
|
112
|
+
if input_features is None:
|
|
113
|
+
input_features = [f"x{i}" for i in range(self.n_features_in_)]
|
|
114
|
+
|
|
115
|
+
out_features = [f"{name}_1" for name in input_features] + [f"{name}_2" for name in input_features]
|
|
116
|
+
|
|
117
|
+
return np.asarray(out_features, dtype=object)
|
|
File without changes
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
from typing import Tuple, List
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from numpy.typing import NDArray
|
|
5
|
+
|
|
6
|
+
from sklearn.cluster import KMeans
|
|
7
|
+
from sklearn.neighbors import KNeighborsClassifier
|
|
8
|
+
from sklearn.metrics import silhouette_score
|
|
9
|
+
|
|
10
|
+
def find_best_k_geometric(
|
|
11
|
+
X_maj: NDArray[np.float64],
|
|
12
|
+
k_candidates: List[int],
|
|
13
|
+
random_state: int
|
|
14
|
+
) -> int:
|
|
15
|
+
"""
|
|
16
|
+
Finds the optimal number of clusters (k) using the Silhouette Score.
|
|
17
|
+
|
|
18
|
+
Evaluates a list of candidate values for k by applying K-Means clustering
|
|
19
|
+
and selecting the value that maximizes the Silhouette Score. If only one
|
|
20
|
+
valid candidate is provided or all candidates are invalid, it provides a
|
|
21
|
+
safe fallback.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the majority class.
|
|
25
|
+
k_candidates (List[int]): A list of integer candidate values for the number of clusters (k) to test.
|
|
26
|
+
random_state (int): Seed used by the random number generator for K-Means initialization to ensure reproducibility.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
int: The optimal number of clusters (k) selected from the candidates.
|
|
30
|
+
"""
|
|
31
|
+
if len(k_candidates) == 1:
|
|
32
|
+
single_k = k_candidates[0] if k_candidates[0] != 0 else 1
|
|
33
|
+
return single_k
|
|
34
|
+
|
|
35
|
+
best_k = None
|
|
36
|
+
best_score = -2.0
|
|
37
|
+
|
|
38
|
+
for k in k_candidates:
|
|
39
|
+
if k < 2 or k >= len(X_maj):
|
|
40
|
+
continue
|
|
41
|
+
|
|
42
|
+
kmeans = KMeans(n_clusters=k, random_state=random_state, n_init='auto')
|
|
43
|
+
cluster_labels = kmeans.fit_predict(X_maj)
|
|
44
|
+
|
|
45
|
+
score = silhouette_score(X_maj, cluster_labels)
|
|
46
|
+
|
|
47
|
+
if score > best_score:
|
|
48
|
+
best_score = score
|
|
49
|
+
best_k = k
|
|
50
|
+
|
|
51
|
+
if best_k is None:
|
|
52
|
+
best_k = max(1, k_candidates[0])
|
|
53
|
+
|
|
54
|
+
return best_k
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def get_set_n_kmeans_re_sc(
|
|
58
|
+
X: NDArray[np.float64],
|
|
59
|
+
y: NDArray[np.int_],
|
|
60
|
+
min_label: int,
|
|
61
|
+
maj_label: int,
|
|
62
|
+
M: float = 1.5,
|
|
63
|
+
num_candidates_to_test: int = 5,
|
|
64
|
+
random_state: int = 42
|
|
65
|
+
) -> NDArray[np.float64]:
|
|
66
|
+
"""
|
|
67
|
+
Generates the Set_N subset using K-Means clustering on 'safe' majority samples.
|
|
68
|
+
|
|
69
|
+
This function identifies 'safe' majority samples using a KNN classifier (samples
|
|
70
|
+
with >= 90% probability of belonging to the majority class). It then dynamically
|
|
71
|
+
generates a list of candidate values for the number of clusters based on the
|
|
72
|
+
imbalance ratio bounds, finds the optimal k using the Silhouette Score, and
|
|
73
|
+
returns the resulting K-Means cluster centers to be used as Set_N.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
X (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the entire training dataset.
|
|
77
|
+
y (numpy.typing.NDArray[np.int_]): 1D NumPy array containing the target labels.
|
|
78
|
+
min_label (int): The target label assigned to the minority class.
|
|
79
|
+
maj_label (int): The target label assigned to the majority class.
|
|
80
|
+
M (float, optional): The maximum acceptable imbalance ratio threshold. Defaults to 1.5.
|
|
81
|
+
num_candidates_to_test (int, optional): The number of candidate values for k to evaluate during geometric tuning. Defaults to 5.
|
|
82
|
+
random_state (int, optional): Seed used by the random number generator for reproducibility. Defaults to 42.
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
numpy.typing.NDArray[np.float64]: A 2D NumPy array containing the features of the generated majority subset (the K-Means cluster centers).
|
|
86
|
+
|
|
87
|
+
Raises:
|
|
88
|
+
ValueError: If either the minority or majority class contains zero samples.
|
|
89
|
+
"""
|
|
90
|
+
X_min = X[y == min_label]
|
|
91
|
+
X_maj = X[y == maj_label]
|
|
92
|
+
|
|
93
|
+
n_maj = len(X_maj)
|
|
94
|
+
n_min = len(X_min)
|
|
95
|
+
|
|
96
|
+
if n_maj == 0 or n_min == 0:
|
|
97
|
+
raise ValueError("Both minority and majority classes must have at least one sample.")
|
|
98
|
+
|
|
99
|
+
n1 = int((n_min ** 2) / n_maj)
|
|
100
|
+
upper_bound = int(M * n1)
|
|
101
|
+
|
|
102
|
+
knn = KNeighborsClassifier(n_neighbors=5)
|
|
103
|
+
knn.fit(X, y)
|
|
104
|
+
|
|
105
|
+
maj_class_idx = np.where(knn.classes_ == maj_label)[0][0]
|
|
106
|
+
|
|
107
|
+
probs = knn.predict_proba(X_maj)
|
|
108
|
+
prob_majority = probs[:, maj_class_idx]
|
|
109
|
+
|
|
110
|
+
safe_mask = prob_majority >= 0.9
|
|
111
|
+
X_maj_safe = X_maj[safe_mask]
|
|
112
|
+
|
|
113
|
+
if len(X_maj_safe) == 0:
|
|
114
|
+
X_maj_safe = X_maj
|
|
115
|
+
|
|
116
|
+
step = max(1, (upper_bound - n1) // max(1, (num_candidates_to_test - 1)))
|
|
117
|
+
candidates = list(range(n1, upper_bound + 1, step))
|
|
118
|
+
|
|
119
|
+
best_k = find_best_k_geometric(X_maj_safe, candidates, random_state)
|
|
120
|
+
best_k = min(best_k, len(X_maj_safe))
|
|
121
|
+
|
|
122
|
+
kmeans = KMeans(n_clusters=best_k, random_state=random_state, n_init='auto')
|
|
123
|
+
kmeans.fit(X_maj_safe)
|
|
124
|
+
|
|
125
|
+
X_set_n = kmeans.cluster_centers_
|
|
126
|
+
|
|
127
|
+
return X_set_n
|
|
128
|
+
|
|
129
|
+
def kmeans_re_sc_concatenation(
|
|
130
|
+
X_min: NDArray[np.float64],
|
|
131
|
+
X_maj: NDArray[np.float64],
|
|
132
|
+
X_set_n: NDArray[np.float64],
|
|
133
|
+
min_label: int = 1,
|
|
134
|
+
maj_label: int = -1
|
|
135
|
+
) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
|
|
136
|
+
"""
|
|
137
|
+
Concatenates pairs of samples from the same class to map the dataset into a 2d dimensional space.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
X_min (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the minority class.
|
|
141
|
+
X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the original majority class.
|
|
142
|
+
X_set_n (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the selected majority subset.
|
|
143
|
+
min_label (int, optional): The target label assigned to the minority class.
|
|
144
|
+
maj_label (int, optional): The target label assigned to the majority class.
|
|
145
|
+
|
|
146
|
+
Returns:
|
|
147
|
+
Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
|
|
148
|
+
A tuple containing:
|
|
149
|
+
- X_resampled: The concatenated 2D NumPy array with 2 * d features.
|
|
150
|
+
- y_resampled: The 1D NumPy array containing the target labels for the new samples.
|
|
151
|
+
|
|
152
|
+
Raises:
|
|
153
|
+
ValueError: If X_min is empty, as the minority class must contain at least one sample.
|
|
154
|
+
"""
|
|
155
|
+
m = len(X_min)
|
|
156
|
+
if m == 0:
|
|
157
|
+
raise ValueError("X_min cannot be empty. Minority class must have at least one sample.")
|
|
158
|
+
|
|
159
|
+
P_repeat = np.repeat(X_min, m, axis=0)
|
|
160
|
+
P_tile = np.tile(X_min, (m, 1))
|
|
161
|
+
P_c = np.hstack([P_repeat, P_tile])
|
|
162
|
+
|
|
163
|
+
y_p_c = np.full(len(P_c), min_label, dtype=np.int_)
|
|
164
|
+
|
|
165
|
+
M = len(X_maj)
|
|
166
|
+
k = len(X_set_n)
|
|
167
|
+
|
|
168
|
+
if k > 0:
|
|
169
|
+
N_repeat = np.repeat(X_maj, k, axis=0)
|
|
170
|
+
Set_N_tile = np.tile(X_set_n, (M, 1))
|
|
171
|
+
N_c = np.hstack([N_repeat, Set_N_tile])
|
|
172
|
+
y_n_c = np.full(len(N_c), maj_label, dtype=np.int_)
|
|
173
|
+
|
|
174
|
+
X_c_array = np.vstack([P_c, N_c])
|
|
175
|
+
y_c_array = np.hstack([y_p_c, y_n_c])
|
|
176
|
+
else:
|
|
177
|
+
X_c_array = P_c
|
|
178
|
+
y_c_array = y_p_c
|
|
179
|
+
|
|
180
|
+
return X_c_array, y_c_array
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
from typing import Optional, Union, List, Tuple
|
|
2
|
+
from numbers import Real, Integral
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
from numpy.typing import NDArray
|
|
6
|
+
|
|
7
|
+
from sklearn.utils import check_random_state
|
|
8
|
+
from sklearn.utils._param_validation import Interval
|
|
9
|
+
|
|
10
|
+
from imblearn.base import BaseSampler
|
|
11
|
+
|
|
12
|
+
from .utils._resc_utils import (
|
|
13
|
+
calculate_set_n_size_re_sc,
|
|
14
|
+
get_set_n_random_weighted_re_sc,
|
|
15
|
+
re_sc_concatenation
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ReSC(BaseSampler):
|
|
20
|
+
"""
|
|
21
|
+
Resampling based on Sample Concatenation (Re-SC) using density-weighted random sampling.
|
|
22
|
+
|
|
23
|
+
This algorithm addresses class imbalance by mapping the data into a higher-dimensional
|
|
24
|
+
(2d) concatenated feature space. It over-samples the minority class by concatenating
|
|
25
|
+
minority samples with themselves, and under-samples the majority class by pairing
|
|
26
|
+
original majority samples with a statistically determined subset (Set_N).
|
|
27
|
+
|
|
28
|
+
Attributes:
|
|
29
|
+
M (float): The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
30
|
+
k (int): Number of nearest neighbors used to calculate majority sample weights.
|
|
31
|
+
alpha (float): Significance level for the Z-test used to compute the required statistical sample size.
|
|
32
|
+
epsilon (float): Acceptable tolerance error for representing the majority class distribution.
|
|
33
|
+
random_state (int, RandomState instance, default=None): Controls the randomization of the algorithm.
|
|
34
|
+
|
|
35
|
+
Methods:
|
|
36
|
+
_fit_resample(X, y): Core resampling logic that executes Re-SC1 and returns concatenated arrays.
|
|
37
|
+
get_feature_names_out(input_features): Generates output feature names for the 2d concatenated space.
|
|
38
|
+
"""
|
|
39
|
+
_sampling_type = 'over-sampling'
|
|
40
|
+
_parameter_constraints = {
|
|
41
|
+
"M": [Interval(Real, 0, None, closed="left")],
|
|
42
|
+
"k": [Interval(Integral, 1, None, closed="left")],
|
|
43
|
+
"alpha": [Interval(Real, 0, 1, closed="both")],
|
|
44
|
+
"epsilon": [Interval(Real, 0, None, closed="neither")],
|
|
45
|
+
"random_state": ["random_state"]
|
|
46
|
+
}
|
|
47
|
+
def __init__(self, M=1.5, k=5, alpha=0.05, epsilon=0.05, random_state=None):
|
|
48
|
+
super().__init__()
|
|
49
|
+
self.M = M
|
|
50
|
+
self.k = k
|
|
51
|
+
self.alpha = alpha
|
|
52
|
+
self.epsilon = epsilon
|
|
53
|
+
self.random_state = random_state
|
|
54
|
+
|
|
55
|
+
def _fit_resample(
|
|
56
|
+
self,
|
|
57
|
+
X: NDArray[np.float64],
|
|
58
|
+
y: NDArray[np.int_]
|
|
59
|
+
) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
|
|
60
|
+
"""
|
|
61
|
+
Executes resampling logic for Re-SC.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
X (numpy.typing.NDArray[np.float64]): 2D matrix containing the features of the original training dataset.
|
|
65
|
+
y (numpy.typing.NDArray[np.int_]): 1D array containing the target labels.
|
|
66
|
+
|
|
67
|
+
Returns:
|
|
68
|
+
Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
|
|
69
|
+
A tuple containing the resampled feature matrix (mapped to a 2d space)
|
|
70
|
+
and the corresponding label array.
|
|
71
|
+
|
|
72
|
+
Raises:
|
|
73
|
+
ValueError: If the dataset does not contain at least two distinct classes.
|
|
74
|
+
"""
|
|
75
|
+
random_state = check_random_state(self.random_state)
|
|
76
|
+
np.random.seed(random_state.randint(0, 2**31 - 1))
|
|
77
|
+
|
|
78
|
+
labels, counts = np.unique(y, return_counts=True)
|
|
79
|
+
if len(labels) < 2:
|
|
80
|
+
raise ValueError("The target 'y' needs to have at least two classes.")
|
|
81
|
+
|
|
82
|
+
min_label = labels[np.argmin(counts)]
|
|
83
|
+
maj_label = labels[np.argmax(counts)]
|
|
84
|
+
|
|
85
|
+
X_min = X[y == min_label]
|
|
86
|
+
X_maj = X[y == maj_label]
|
|
87
|
+
|
|
88
|
+
target_size = calculate_set_n_size_re_sc(
|
|
89
|
+
X_maj=X_maj,
|
|
90
|
+
P=len(X_min),
|
|
91
|
+
alpha=self.alpha,
|
|
92
|
+
epsilon=self.epsilon,
|
|
93
|
+
M=self.M
|
|
94
|
+
)
|
|
95
|
+
X_set_n = get_set_n_random_weighted_re_sc(
|
|
96
|
+
X=X,
|
|
97
|
+
y=y,
|
|
98
|
+
n_size=target_size,
|
|
99
|
+
k=self.k
|
|
100
|
+
)
|
|
101
|
+
X_resampled, y_resampled = re_sc_concatenation(
|
|
102
|
+
X_min=X_min,
|
|
103
|
+
X_maj=X_maj,
|
|
104
|
+
X_set_n=X_set_n,
|
|
105
|
+
min_label=min_label,
|
|
106
|
+
maj_label=maj_label
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
return X_resampled, y_resampled
|
|
110
|
+
|
|
111
|
+
def get_feature_names_out(
|
|
112
|
+
self,
|
|
113
|
+
input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
|
|
114
|
+
) -> NDArray[np.object_]:
|
|
115
|
+
"""
|
|
116
|
+
Get output feature names for transformation.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
input_features (Optional[Union[List[str], numpy.typing.NDArray[np.object_]]]):
|
|
120
|
+
Original input feature names. If None, generic names are generated.
|
|
121
|
+
|
|
122
|
+
Returns:
|
|
123
|
+
numpy.typing.NDArray[np.object_]: An array of strings containing the new feature
|
|
124
|
+
names for the 2d concatenated space.
|
|
125
|
+
"""
|
|
126
|
+
if input_features is None:
|
|
127
|
+
input_features = [f"x{i}" for i in range(self.n_features_in_)]
|
|
128
|
+
|
|
129
|
+
out_features = [f"{name}_1" for name in input_features] + [f"{name}_2" for name in input_features]
|
|
130
|
+
|
|
131
|
+
return np.asarray(out_features, dtype=object)
|
|
File without changes
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from numpy.typing import NDArray
|
|
5
|
+
|
|
6
|
+
import scipy.stats as stats
|
|
7
|
+
|
|
8
|
+
from sklearn.neighbors import NearestNeighbors
|
|
9
|
+
|
|
10
|
+
def calculate_set_n_size_re_sc(
|
|
11
|
+
X_maj: NDArray[np.float64],
|
|
12
|
+
P: int,
|
|
13
|
+
alpha: float = 0.05,
|
|
14
|
+
epsilon: float = 0.05,
|
|
15
|
+
M: float = 1.5
|
|
16
|
+
) -> int:
|
|
17
|
+
"""
|
|
18
|
+
Calculates the target size for the majority subset (Set_N) in the Re-SC algorithm.
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing features of the majority class.
|
|
22
|
+
P (int): number of samples in the minority class.
|
|
23
|
+
alpha (float): significance level for the Z-test used to compute statistical sample size.
|
|
24
|
+
epsilon (float): acceptable tolerance error for representing the majority class distribution
|
|
25
|
+
M (float): acceptable imbalance ratio threshold.
|
|
26
|
+
|
|
27
|
+
Returns:
|
|
28
|
+
int: The calculated number of majority samples for Set_N.
|
|
29
|
+
|
|
30
|
+
Raises:
|
|
31
|
+
ValueError: If X_maj is an empty array.
|
|
32
|
+
"""
|
|
33
|
+
n_maj: int = len(X_maj)
|
|
34
|
+
if n_maj == 0:
|
|
35
|
+
raise ValueError("X_maj cannot be empty.")
|
|
36
|
+
|
|
37
|
+
n1 = (P ** 2) / n_maj
|
|
38
|
+
z_score = stats.norm.ppf(1 - alpha / 2)
|
|
39
|
+
sigma = np.std(X_maj, ddof=1)
|
|
40
|
+
z_sq = z_score ** 2
|
|
41
|
+
sigma_sq = sigma ** 2
|
|
42
|
+
epsilon_sq = epsilon ** 2
|
|
43
|
+
|
|
44
|
+
numerator = n_maj * z_sq * sigma_sq
|
|
45
|
+
denominator = (n_maj * epsilon_sq) + (z_sq * sigma_sq)
|
|
46
|
+
n2 = numerator / denominator
|
|
47
|
+
|
|
48
|
+
if n1 == 0:
|
|
49
|
+
return int(n2)
|
|
50
|
+
|
|
51
|
+
Pr = n2 / n1
|
|
52
|
+
|
|
53
|
+
if Pr < 1:
|
|
54
|
+
set_n_size = n1
|
|
55
|
+
elif 1 <= Pr <= M:
|
|
56
|
+
set_n_size = n2
|
|
57
|
+
else:
|
|
58
|
+
set_n_size = M * n1
|
|
59
|
+
|
|
60
|
+
return int(np.ceil(set_n_size))
|
|
61
|
+
|
|
62
|
+
def get_set_n_random_weighted_re_sc(
|
|
63
|
+
X: NDArray[np.float64],
|
|
64
|
+
y: NDArray[np.int_],
|
|
65
|
+
n_size: int,
|
|
66
|
+
k: int = 5
|
|
67
|
+
) -> NDArray[np.float64]:
|
|
68
|
+
"""
|
|
69
|
+
Selects a subset of majority class samples (Set_N) using a density-weighted random sampling strategy.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
X (numpy.typing.NDArray[np.float64]): 2D NumPy array containing entire training dataset.
|
|
73
|
+
y (numpy.typing.NDArray[np.int_]): 1D NumPy array containing labels.
|
|
74
|
+
n_size (int): number of majority samples to select.
|
|
75
|
+
k (int, optional): number of nearest neighbors to evaluate for the weighting mechanism. Defaults to 5.
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
numpy.typing.NDArray[np.float64]: A 2D NumPy array containing selected majority subset.
|
|
79
|
+
|
|
80
|
+
Raises:
|
|
81
|
+
ValueError: If there are no majority samples present in the dataset.
|
|
82
|
+
ValueError: If no valid majority samples are found.
|
|
83
|
+
"""
|
|
84
|
+
labels, counts = np.unique(y, return_counts=True)
|
|
85
|
+
maj_label = labels[np.argmax(counts)]
|
|
86
|
+
|
|
87
|
+
maj_mask = (y == maj_label)
|
|
88
|
+
maj_indices = np.where(maj_mask)[0]
|
|
89
|
+
|
|
90
|
+
if len(maj_indices) == 0:
|
|
91
|
+
raise ValueError("No majority samples found in the dataset.")
|
|
92
|
+
|
|
93
|
+
X_mean = np.mean(X, axis=0)
|
|
94
|
+
X_std = np.std(X, axis=0)
|
|
95
|
+
X_std[X_std == 0] = 1.0
|
|
96
|
+
X_norm = (X - X_mean) / X_std
|
|
97
|
+
|
|
98
|
+
knn = NearestNeighbors(n_neighbors=k + 1).fit(X_norm)
|
|
99
|
+
|
|
100
|
+
X_maj_norm = X_norm[maj_indices]
|
|
101
|
+
distances, neighbor_idxs = knn.kneighbors(X_maj_norm)
|
|
102
|
+
|
|
103
|
+
weights = []
|
|
104
|
+
valid_indices = []
|
|
105
|
+
|
|
106
|
+
for i, n_idxs in enumerate(neighbor_idxs):
|
|
107
|
+
actual_neighbors = n_idxs[1:]
|
|
108
|
+
maj_neighbor_count = np.sum(y[actual_neighbors] == maj_label)
|
|
109
|
+
weight = maj_neighbor_count / k
|
|
110
|
+
|
|
111
|
+
if weight > 0:
|
|
112
|
+
weights.append(weight)
|
|
113
|
+
valid_indices.append(maj_indices[i])
|
|
114
|
+
|
|
115
|
+
weights_arr = np.array(weights, dtype=np.float64)
|
|
116
|
+
valid_indices_arr = np.array(valid_indices, dtype=np.int_)
|
|
117
|
+
|
|
118
|
+
if len(weights_arr) == 0:
|
|
119
|
+
raise ValueError("No valid majority samples found (all are surrounded by minority class).")
|
|
120
|
+
|
|
121
|
+
probs = weights_arr / np.sum(weights_arr)
|
|
122
|
+
actual_n_size = min(n_size, len(valid_indices_arr))
|
|
123
|
+
|
|
124
|
+
selected_indices = np.random.choice(
|
|
125
|
+
valid_indices_arr,
|
|
126
|
+
size=actual_n_size,
|
|
127
|
+
replace=False,
|
|
128
|
+
p=probs
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
return X[selected_indices]
|
|
132
|
+
|
|
133
|
+
def re_sc_concatenation(
|
|
134
|
+
X_min: NDArray[np.float64],
|
|
135
|
+
X_maj: NDArray[np.float64],
|
|
136
|
+
X_set_n: NDArray[np.float64],
|
|
137
|
+
min_label: int = 1,
|
|
138
|
+
maj_label: int = -1
|
|
139
|
+
) -> Tuple[NDArray[np.float64], NDArray[np.int_]]:
|
|
140
|
+
"""
|
|
141
|
+
Concatenates pairs of samples from the same class to map the dataset into a 2d dimensional space.
|
|
142
|
+
|
|
143
|
+
Args:
|
|
144
|
+
X_min (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the minority class.
|
|
145
|
+
X_maj (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the original majority class.
|
|
146
|
+
X_set_n (numpy.typing.NDArray[np.float64]): 2D NumPy array containing the features of the selected majority subset.
|
|
147
|
+
min_label (int, optional): The target label assigned to the minority class.
|
|
148
|
+
maj_label (int, optional): The target label assigned to the majority class.
|
|
149
|
+
|
|
150
|
+
Returns:
|
|
151
|
+
Tuple[numpy.typing.NDArray[np.float64], numpy.typing.NDArray[np.int_]]:
|
|
152
|
+
A tuple containing:
|
|
153
|
+
- X_resampled: The concatenated 2D NumPy array with 2 * d features.
|
|
154
|
+
- y_resampled: The 1D NumPy array containing the target labels for the new samples.
|
|
155
|
+
|
|
156
|
+
Raises:
|
|
157
|
+
ValueError: If X_min is empty, as the minority class must contain at least one sample.
|
|
158
|
+
"""
|
|
159
|
+
m = len(X_min)
|
|
160
|
+
if m == 0:
|
|
161
|
+
raise ValueError("X_min cannot be empty. Minority class must have at least one sample.")
|
|
162
|
+
|
|
163
|
+
P_repeat = np.repeat(X_min, m, axis=0)
|
|
164
|
+
P_tile = np.tile(X_min, (m, 1))
|
|
165
|
+
P_c = np.hstack([P_repeat, P_tile])
|
|
166
|
+
|
|
167
|
+
y_p_c = np.full(len(P_c), min_label, dtype=np.int_)
|
|
168
|
+
|
|
169
|
+
M = len(X_maj)
|
|
170
|
+
k = len(X_set_n)
|
|
171
|
+
|
|
172
|
+
if k > 0:
|
|
173
|
+
N_repeat = np.repeat(X_maj, k, axis=0)
|
|
174
|
+
Set_N_tile = np.tile(X_set_n, (M, 1))
|
|
175
|
+
N_c = np.hstack([N_repeat, Set_N_tile])
|
|
176
|
+
y_n_c = np.full(len(N_c), maj_label, dtype=np.int_)
|
|
177
|
+
|
|
178
|
+
X_c_array = np.vstack([P_c, N_c])
|
|
179
|
+
y_c_array = np.hstack([y_p_c, y_n_c])
|
|
180
|
+
else:
|
|
181
|
+
X_c_array = P_c
|
|
182
|
+
y_c_array = y_p_c
|
|
183
|
+
|
|
184
|
+
return X_c_array, y_c_array
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from typing import Optional, Union, List
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from numpy.typing import NDArray
|
|
5
|
+
|
|
6
|
+
from sklearn.base import BaseEstimator, TransformerMixin
|
|
7
|
+
from sklearn.utils.validation import check_is_fitted, check_array, validate_data
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ReSCTransformer(TransformerMixin, BaseEstimator):
|
|
11
|
+
"""
|
|
12
|
+
Transforms test data into a 2d concatenated space for prediction with Re-SC models.
|
|
13
|
+
|
|
14
|
+
This transformer must be placed immediately after a Re-SC sampler in a Pipeline.
|
|
15
|
+
- During training (fit_transform), it receives data that the Sampler has already
|
|
16
|
+
mapped to 2d space, so it simply passes it through.
|
|
17
|
+
- During testing (predict), the Sampler is bypassed, so this Transformer receives
|
|
18
|
+
data in the original d space. It duplicates the features (x -> [x, x]) to
|
|
19
|
+
match the 2d space the classifier expects.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(self):
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
def fit(self, X: NDArray[np.float64], y: Optional[NDArray] = None) -> 'ReSCTransformer':
|
|
26
|
+
X = validate_data(self, X=X, reset=True, accept_sparse=False)
|
|
27
|
+
self.original_d_ = self.n_features_in_ // 2
|
|
28
|
+
self.is_fitted_ = True
|
|
29
|
+
|
|
30
|
+
return self
|
|
31
|
+
|
|
32
|
+
def transform(self, X: NDArray[np.float64]) -> NDArray[np.float64]:
|
|
33
|
+
check_is_fitted(self, 'is_fitted_')
|
|
34
|
+
X = check_array(X, accept_sparse=False)
|
|
35
|
+
|
|
36
|
+
if X.shape[1] == self.original_d_:
|
|
37
|
+
return np.hstack([X, X])
|
|
38
|
+
|
|
39
|
+
X = validate_data(self, X, reset=False, accept_sparse=False)
|
|
40
|
+
|
|
41
|
+
return X
|
|
42
|
+
|
|
43
|
+
def get_feature_names_out(
|
|
44
|
+
self,
|
|
45
|
+
input_features: Optional[Union[List[str], NDArray[np.object_]]] = None
|
|
46
|
+
) -> NDArray[np.object_]:
|
|
47
|
+
check_is_fitted(self, 'is_fitted_')
|
|
48
|
+
|
|
49
|
+
if input_features is None:
|
|
50
|
+
input_features = [f"x{i}" for i in range(self.original_d_)]
|
|
51
|
+
|
|
52
|
+
out_features = [f"{name}_1" for name in input_features] + \
|
|
53
|
+
[f"{name}_2" for name in input_features]
|
|
54
|
+
|
|
55
|
+
return np.asarray(out_features, dtype=object)
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: imblearn_resc
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning
|
|
5
|
+
Author-email: maksimkins <maks.aleshkov.04@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/maksimkins/resampling-methods
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
+
Requires-Python: >=3.11
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: numpy>=1.24.0
|
|
16
|
+
Requires-Dist: scipy>=1.10.0
|
|
17
|
+
Requires-Dist: scikit-learn>=1.4.0
|
|
18
|
+
Requires-Dist: imbalanced-learn>=0.12.0
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
21
|
+
|
|
22
|
+
# imblearn-resc
|
|
23
|
+
|
|
24
|
+
**Re-SC (Resampling based on Sample Concatenation)** algorithms for imbalanced learning.
|
|
25
|
+
|
|
26
|
+
This package is fully compatible with the `scikit-learn` and `imbalanced-learn` ecosystems. It addresses class imbalance by mapping data into a higher-dimensional (2d) concatenated feature space, utilizing either density-weighted random sampling (`ReSC`) or K-Means clustering (`KMeansReSC`) to safely resample the majority and minority classes.
|
|
27
|
+
|
|
28
|
+
## 📦 Installation
|
|
29
|
+
|
|
30
|
+
You can install `imblearn-resc` directly from PyPI using pip:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install imblearn-resc
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
*Requires Python >=3.11, scikit-learn >=1.4.0, and imbalanced-learn >=0.12.0*
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## 🚀 Quick Start & Usage
|
|
41
|
+
|
|
42
|
+
Because Re-SC algorithms map your original features ($d$) into a concatenated feature space ($2d$), **you must always pair the Sampler with the `ReSCTransformer` inside an `imblearn` Pipeline.**
|
|
43
|
+
|
|
44
|
+
* **The Sampler** (`ReSC` or `KMeansReSC`) transforms the training data during `.fit_resample()`.
|
|
45
|
+
* **The Transformer** (`ReSCTransformer`) bypasses the training data, but safely duplicates the test data features ($x \rightarrow [x, x]$) during `.predict()` so your classifier receives the correct dimensions.
|
|
46
|
+
|
|
47
|
+
### Example: Complete Pipeline
|
|
48
|
+
|
|
49
|
+
Here is a full, runnable example of how to use `ReSC` and `KMeansReSC` with a standard machine learning classifier.
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from sklearn.datasets import make_classification
|
|
53
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
54
|
+
from sklearn.model_selection import train_test_split
|
|
55
|
+
from sklearn.metrics import classification_report
|
|
56
|
+
|
|
57
|
+
# 1. Import the pipeline from imbalanced-learn (NOT standard sklearn!)
|
|
58
|
+
from imblearn.pipeline import Pipeline
|
|
59
|
+
|
|
60
|
+
# 2. Import the Re-SC Samplers and Transformer
|
|
61
|
+
from imblearn_resc.oversampling import ReSC, KMeansReSC
|
|
62
|
+
from imblearn_resc.preprocessing import ReSCTransformer
|
|
63
|
+
|
|
64
|
+
# Generate a highly imbalanced dummy dataset (10% minority, 90% majority)
|
|
65
|
+
X, y = make_classification(
|
|
66
|
+
n_classes=2, class_sep=2, weights=[0.1, 0.9],
|
|
67
|
+
n_informative=3, n_redundant=1, flip_y=0,
|
|
68
|
+
n_features=5, n_clusters_per_class=1,
|
|
69
|
+
n_samples=1000, random_state=42
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
|
|
73
|
+
|
|
74
|
+
# ==========================================
|
|
75
|
+
# Option A: Standard ReSC Pipeline
|
|
76
|
+
# ==========================================
|
|
77
|
+
pipeline_resc = Pipeline([
|
|
78
|
+
('sampler', ReSC(M=1.5, k=5, random_state=42)),
|
|
79
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
80
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
81
|
+
])
|
|
82
|
+
|
|
83
|
+
# Train and Predict
|
|
84
|
+
pipeline_resc.fit(X_train, y_train)
|
|
85
|
+
y_pred_resc = pipeline_resc.predict(X_test)
|
|
86
|
+
|
|
87
|
+
print("ReSC Classification Report:")
|
|
88
|
+
print(classification_report(y_test, y_pred_resc))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ==========================================
|
|
92
|
+
# Option B: KMeansReSC Pipeline
|
|
93
|
+
# ==========================================
|
|
94
|
+
pipeline_kmeans = Pipeline([
|
|
95
|
+
('sampler', KMeansReSC(M=1.5, num_candidates_to_test=5, random_state=42)),
|
|
96
|
+
('transformer', ReSCTransformer()), # <--- Mandatory!
|
|
97
|
+
('classifier', RandomForestClassifier(random_state=42))
|
|
98
|
+
])
|
|
99
|
+
|
|
100
|
+
# Train and Predict
|
|
101
|
+
pipeline_kmeans.fit(X_train, y_train)
|
|
102
|
+
y_pred_kmeans = pipeline_kmeans.predict(X_test)
|
|
103
|
+
|
|
104
|
+
print("KMeansReSC Classification Report:")
|
|
105
|
+
print(classification_report(y_test, y_pred_kmeans))
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## 🧠Key Parameters
|
|
109
|
+
|
|
110
|
+
### `ReSC`
|
|
111
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
112
|
+
* **`k`** *(int, default=5)*: Number of nearest neighbors used to calculate majority sample weights.
|
|
113
|
+
* **`alpha`** *(float, default=0.05)*: Significance level for the Z-test used to compute the required statistical sample size.
|
|
114
|
+
|
|
115
|
+
### `KMeansReSC`
|
|
116
|
+
* **`M`** *(float, default=1.5)*: The maximum acceptable imbalance ratio threshold for the resulting dataset.
|
|
117
|
+
* **`num_candidates_to_test`** *(int, default=5)*: How many 'k' values (clusters) to test during geometric tuning using the Silhouette Score.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
imblearn_resc/__init__.py
|
|
4
|
+
imblearn_resc.egg-info/PKG-INFO
|
|
5
|
+
imblearn_resc.egg-info/SOURCES.txt
|
|
6
|
+
imblearn_resc.egg-info/dependency_links.txt
|
|
7
|
+
imblearn_resc.egg-info/requires.txt
|
|
8
|
+
imblearn_resc.egg-info/top_level.txt
|
|
9
|
+
imblearn_resc/oversampling/__init__.py
|
|
10
|
+
imblearn_resc/oversampling/KMeansReSC/__init__.py
|
|
11
|
+
imblearn_resc/oversampling/KMeansReSC/_resc_kmeans.py
|
|
12
|
+
imblearn_resc/oversampling/KMeansReSC/utils/__init__.py
|
|
13
|
+
imblearn_resc/oversampling/KMeansReSC/utils/_resc_kmeans_utils.py
|
|
14
|
+
imblearn_resc/oversampling/ReSC/__init__.py
|
|
15
|
+
imblearn_resc/oversampling/ReSC/_resc.py
|
|
16
|
+
imblearn_resc/oversampling/ReSC/utils/__init__.py
|
|
17
|
+
imblearn_resc/oversampling/ReSC/utils/_resc_utils.py
|
|
18
|
+
imblearn_resc/preprocessing/__init__.py
|
|
19
|
+
imblearn_resc/preprocessing/_resc_transformer.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
imblearn_resc
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "imblearn_resc"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Resampling based on Sample Concatenation (Re-SC) algorithms for imbalanced learning"
|
|
9
|
+
authors = [
|
|
10
|
+
{name = "maksimkins", email = "maks.aleshkov.04@gmail.com"}
|
|
11
|
+
]
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
requires-python = ">=3.11"
|
|
14
|
+
license = {text = "MIT"}
|
|
15
|
+
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Intended Audience :: Science/Research",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
dependencies = [
|
|
25
|
+
"numpy>=1.24.0",
|
|
26
|
+
"scipy>=1.10.0",
|
|
27
|
+
"scikit-learn>=1.4.0",
|
|
28
|
+
"imbalanced-learn>=0.12.0"
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://github.com/maksimkins/resampling-methods"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
dev = [
|
|
37
|
+
"pytest>=7.0.0"
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.packages.find]
|
|
41
|
+
where = ["."]
|
|
42
|
+
include = ["imblearn_resc*"]
|
|
43
|
+
|
|
44
|
+
[tool.pytest.ini_options]
|
|
45
|
+
pythonpath = ["."]
|