preorder4mlc 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- preorder4mlc/__init__.py +4 -0
- preorder4mlc/base_classifiers.py +242 -0
- preorder4mlc/config.py +115 -0
- preorder4mlc/constants.py +30 -0
- preorder4mlc/datasets4experiments.py +176 -0
- preorder4mlc/estimator.py +123 -0
- preorder4mlc/evaluation_metric.py +888 -0
- preorder4mlc/inference_models.py +502 -0
- preorder4mlc/searching_algorithms.py +517 -0
- preorder4mlc/solvers.py +178 -0
- preorder4mlc/training_orchestrator.py +474 -0
- preorder4mlc/utils/plot_figures.py +1014 -0
- preorder4mlc/utils/results_manager.py +167 -0
- preorder4mlc/utils/statistical_tests.py +411 -0
- preorder4mlc/utils/summarize_metrics.py +250 -0
- preorder4mlc/utils/suppress.py +55 -0
- preorder4mlc-1.0.0.dist-info/METADATA +213 -0
- preorder4mlc-1.0.0.dist-info/RECORD +21 -0
- preorder4mlc-1.0.0.dist-info/WHEEL +5 -0
- preorder4mlc-1.0.0.dist-info/licenses/LICENSE +21 -0
- preorder4mlc-1.0.0.dist-info/top_level.txt +1 -0
preorder4mlc/__init__.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Pairwise- and calibrated-classifier factories.
|
|
2
|
+
|
|
3
|
+
:class:`BaseClassifiers` builds the per-pair training datasets that
|
|
4
|
+
:class:`inference_models.PredictBOPOs` consumes: pairwise classifiers for
|
|
5
|
+
PRE_ORDER and PARTIAL_ORDER variants, and one calibrated classifier per
|
|
6
|
+
label for CLR. Estimator construction is delegated to
|
|
7
|
+
:class:`estimator.Estimator`, with fitting parallelised via joblib.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from logging import INFO, log
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
from joblib import Parallel, delayed
|
|
14
|
+
from numpy.typing import NDArray
|
|
15
|
+
from sklearn.base import BaseEstimator
|
|
16
|
+
|
|
17
|
+
from preorder4mlc.estimator import Estimator, train_classifier
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BaseClassifiers:
|
|
21
|
+
"""Base classifiers for multi-label classification with different order types.
|
|
22
|
+
|
|
23
|
+
Attributes:
|
|
24
|
+
base_learner (Estimator): The base learning algorithm
|
|
25
|
+
logger (logging.Logger): Logger instance
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self, name: str):
|
|
29
|
+
log(
|
|
30
|
+
INFO,
|
|
31
|
+
f"BaseClassifiers: Initializing base learner: {name}",
|
|
32
|
+
)
|
|
33
|
+
self.name = name
|
|
34
|
+
|
|
35
|
+
def get_classifier(self) -> BaseEstimator:
|
|
36
|
+
return Estimator(self.name).get_classifier() # type: ignore
|
|
37
|
+
|
|
38
|
+
def pairwise_calibrated_classifier(self, X: NDArray[np.float64], Y: NDArray[np.int32]):
|
|
39
|
+
"""Train pairwise calibrated classifiers.
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
X: Input features of shape (n_samples, n_features)
|
|
43
|
+
Y: Binary label matrix of shape (n_samples, n_labels)
|
|
44
|
+
|
|
45
|
+
X_train = [[1, 2, 3, 0], [3, 2, 4, 1], [4, 5, 3, 3], [7, 6, 3, 1]]
|
|
46
|
+
Y_train = [[1, 1, 0], [0, 1, 0], [1, 0, 0], [0, 0, 1]]
|
|
47
|
+
n_labels = 3
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
Tuple containing:
|
|
51
|
+
- Dictionary of pairwise classifiers
|
|
52
|
+
- List of calibrated classifiers
|
|
53
|
+
"""
|
|
54
|
+
n_instances, n_labels = Y.shape
|
|
55
|
+
|
|
56
|
+
# calibrated_classifiers is in fact is a (inverse) BR classifier
|
|
57
|
+
calibrated_classifiers = []
|
|
58
|
+
clr_dataset_classifier = {}
|
|
59
|
+
for k in range(n_labels):
|
|
60
|
+
# MCC = score for class 0; invert the label column so the trained
|
|
61
|
+
# classifier produces P(y_k = 0) directly (CLR convention).
|
|
62
|
+
# Example: Y[:, k] = [1, 0, 1, 0] -> MCC_y = [0, 1, 0, 1].
|
|
63
|
+
MCC_y = np.logical_not(Y[:, k]).astype(int)
|
|
64
|
+
clr_dataset_classifier[str(k)] = { # type: ignore
|
|
65
|
+
"X": X,
|
|
66
|
+
"Y": MCC_y,
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
log(
|
|
70
|
+
INFO,
|
|
71
|
+
f"\t - Training for {len(clr_dataset_classifier.keys())} calibrated_classifiers with {self.name}",
|
|
72
|
+
)
|
|
73
|
+
# run fit in parallel for each pair of labels
|
|
74
|
+
classifiers = Parallel(n_jobs=-1)(
|
|
75
|
+
delayed(train_classifier)(
|
|
76
|
+
clr_dataset_classifier[str(k)]["X"],
|
|
77
|
+
clr_dataset_classifier[str(k)]["Y"],
|
|
78
|
+
self.name,
|
|
79
|
+
) # type: ignore
|
|
80
|
+
for k in range(n_labels)
|
|
81
|
+
)
|
|
82
|
+
log(INFO, f"\t - Trained {n_labels} classifiers")
|
|
83
|
+
|
|
84
|
+
calibrated_classifiers = list(classifiers)
|
|
85
|
+
|
|
86
|
+
pairwise_classifiers = {}
|
|
87
|
+
single_label_pair = {}
|
|
88
|
+
dataset_classifier = {}
|
|
89
|
+
for i in range(n_labels - 1):
|
|
90
|
+
for j in range(i + 1, n_labels):
|
|
91
|
+
key = f"{i}_{j}"
|
|
92
|
+
MCC_X = []
|
|
93
|
+
MCC_y = []
|
|
94
|
+
|
|
95
|
+
for n in range(n_instances):
|
|
96
|
+
if Y[n, i] == 1 and Y[n, j] == 0:
|
|
97
|
+
MCC_X.append(X[n])
|
|
98
|
+
MCC_y.append(0)
|
|
99
|
+
elif Y[n, i] == 0 and Y[n, j] == 1:
|
|
100
|
+
MCC_X.append(X[n])
|
|
101
|
+
MCC_y.append(1)
|
|
102
|
+
|
|
103
|
+
dataset_classifier[key] = { # type: ignore
|
|
104
|
+
"X": MCC_X,
|
|
105
|
+
"Y": MCC_y,
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
log(
|
|
109
|
+
INFO,
|
|
110
|
+
f"\t - Training for {len(dataset_classifier.keys())} pairs with {self.name}",
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# Skip degenerate pairs (empty, single-sample, or single-class). LGBM
|
|
114
|
+
# rejects single-sample fits and any single-class fit; RF tolerates
|
|
115
|
+
# single-sample but not single-class. downstream predict_CLR guards
|
|
116
|
+
# against missing keys.
|
|
117
|
+
trainable_keys = [
|
|
118
|
+
k for k in dataset_classifier.keys()
|
|
119
|
+
if len(dataset_classifier[k]["Y"]) >= 2
|
|
120
|
+
and len(set(dataset_classifier[k]["Y"])) >= 2
|
|
121
|
+
]
|
|
122
|
+
skipped = len(dataset_classifier) - len(trainable_keys)
|
|
123
|
+
if skipped:
|
|
124
|
+
log(INFO, f"\t - Skipping {skipped} empty pairs (degenerate fold)")
|
|
125
|
+
|
|
126
|
+
classifiers = Parallel(n_jobs=-1)(
|
|
127
|
+
delayed(train_classifier)(
|
|
128
|
+
dataset_classifier[key]["X"],
|
|
129
|
+
dataset_classifier[key]["Y"],
|
|
130
|
+
self.name,
|
|
131
|
+
) # type: ignore
|
|
132
|
+
for key in trainable_keys
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
log(INFO, f"\t - Trained {len(trainable_keys)} classifiers")
|
|
136
|
+
|
|
137
|
+
pairwise_classifiers = dict(zip(trainable_keys, classifiers)) # type: ignore
|
|
138
|
+
|
|
139
|
+
for i in range(n_labels - 1):
|
|
140
|
+
for j in range(i + 1, n_labels):
|
|
141
|
+
key = f"{i}_{j}"
|
|
142
|
+
MCC_y = dataset_classifier[key]["Y"]
|
|
143
|
+
single_label_pair[key] = (
|
|
144
|
+
1
|
|
145
|
+
if len(np.unique(MCC_y)) == 1 and np.unique(MCC_y)[0] == 1
|
|
146
|
+
else (0 if len(np.unique(MCC_y)) == 1 and np.unique(MCC_y)[0] == 0 else None)
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
return pairwise_classifiers, calibrated_classifiers, single_label_pair
|
|
150
|
+
|
|
151
|
+
def pairwise_partial_order_classifier_fit(self, X, Y):
|
|
152
|
+
# This BaseClassifier provides pairwise_probability_information for learning partial orders
|
|
153
|
+
n_labels = len(Y[0])
|
|
154
|
+
n_instances, _ = Y.shape
|
|
155
|
+
dataset_classifier = {}
|
|
156
|
+
pairwise_classifiers = {}
|
|
157
|
+
|
|
158
|
+
for i in range(n_labels - 1):
|
|
159
|
+
for j in range(i + 1, n_labels):
|
|
160
|
+
key = f"{i}_{j}"
|
|
161
|
+
MCC_y = []
|
|
162
|
+
for n in range(n_instances):
|
|
163
|
+
if Y[n, i] == Y[n, j]:
|
|
164
|
+
MCC_y.append(2)
|
|
165
|
+
elif Y[n, i] == 1 and Y[n, j] == 0:
|
|
166
|
+
MCC_y.append(0)
|
|
167
|
+
elif Y[n, i] == 0 and Y[n, j] == 1:
|
|
168
|
+
MCC_y.append(1)
|
|
169
|
+
# X is read-only input for sklearn/LightGBM fit; share the
|
|
170
|
+
# reference across pairs instead of copying. For K=101 (mediamill)
|
|
171
|
+
# that's 5050 copies of a 42 MB array (~212 GB) avoided.
|
|
172
|
+
dataset_classifier[key] = { # type: ignore
|
|
173
|
+
"X": X,
|
|
174
|
+
"Y": MCC_y,
|
|
175
|
+
}
|
|
176
|
+
log(
|
|
177
|
+
INFO,
|
|
178
|
+
f"\t - Training for {len(dataset_classifier.keys())} pairs with {self.name}",
|
|
179
|
+
)
|
|
180
|
+
# run fit in parallel for each pair of labels
|
|
181
|
+
classifiers = Parallel(n_jobs=-1)(
|
|
182
|
+
delayed(train_classifier)(
|
|
183
|
+
dataset_classifier[key]["X"],
|
|
184
|
+
dataset_classifier[key]["Y"],
|
|
185
|
+
self.name,
|
|
186
|
+
) # type: ignore
|
|
187
|
+
for key in dataset_classifier.keys()
|
|
188
|
+
)
|
|
189
|
+
log(INFO, f"\t - Trained {len(dataset_classifier.keys())} classifiers")
|
|
190
|
+
|
|
191
|
+
pairwise_classifiers = dict(zip(dataset_classifier.keys(), classifiers)) # type: ignore
|
|
192
|
+
|
|
193
|
+
return pairwise_classifiers # type: ignore
|
|
194
|
+
|
|
195
|
+
def pairwise_pre_order_classifier_fit(self, X, Y) -> dict[str, Estimator]:
|
|
196
|
+
"""
|
|
197
|
+
This BaseClassifier provides pairwise_probability_information for learning preorders
|
|
198
|
+
For each pair of labels, we will train a classifier to predict the probability of the label
|
|
199
|
+
"""
|
|
200
|
+
n_instances, n_labels = Y.shape
|
|
201
|
+
log(INFO, f"\t - {n_labels} labels with {self.name}")
|
|
202
|
+
dataset_classifier = {}
|
|
203
|
+
for i in range(n_labels - 1):
|
|
204
|
+
for j in range(i + 1, n_labels):
|
|
205
|
+
key = f"{i}_{j}"
|
|
206
|
+
MCC_y = []
|
|
207
|
+
for n in range(n_instances):
|
|
208
|
+
if Y[n, i] == 0 and Y[n, j] == 0:
|
|
209
|
+
MCC_y.append(2)
|
|
210
|
+
elif Y[n, i] == 1 and Y[n, j] == 1:
|
|
211
|
+
MCC_y.append(3)
|
|
212
|
+
elif Y[n, i] == 1 and Y[n, j] == 0:
|
|
213
|
+
MCC_y.append(0)
|
|
214
|
+
elif Y[n, i] == 0 and Y[n, j] == 1:
|
|
215
|
+
MCC_y.append(1)
|
|
216
|
+
|
|
217
|
+
# X is read-only input for sklearn/LightGBM fit; share the
|
|
218
|
+
# reference across pairs instead of copying.
|
|
219
|
+
dataset_classifier[key] = { # type: ignore
|
|
220
|
+
"X": X,
|
|
221
|
+
"Y": MCC_y,
|
|
222
|
+
}
|
|
223
|
+
log(
|
|
224
|
+
INFO,
|
|
225
|
+
f"\t - Training {len(dataset_classifier.keys())} pairs with {self.name}",
|
|
226
|
+
)
|
|
227
|
+
# run fit in parallel for each pair of labels
|
|
228
|
+
classifiers = Parallel(n_jobs=-1)(
|
|
229
|
+
delayed(train_classifier)(
|
|
230
|
+
dataset_classifier[key]["X"],
|
|
231
|
+
dataset_classifier[key]["Y"],
|
|
232
|
+
self.name,
|
|
233
|
+
) # type: ignore
|
|
234
|
+
for key in dataset_classifier.keys()
|
|
235
|
+
)
|
|
236
|
+
log(INFO, f"\t - Trained {len(dataset_classifier.keys())} classifiers")
|
|
237
|
+
|
|
238
|
+
pairwise_classifiers = dict(zip(dataset_classifier.keys(), classifiers)) # type: ignore
|
|
239
|
+
|
|
240
|
+
# This is a dictionary of pairwise classifiers. [key] is a string of the form "i_j"
|
|
241
|
+
# where i and j are the indices of the labels in the label matrix Y
|
|
242
|
+
return pairwise_classifiers # type: ignore
|
preorder4mlc/config.py
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Run configuration for the BOPOs training pipeline.
|
|
2
|
+
|
|
3
|
+
Defines the typed config dataclasses (:class:`DatasetConfig`,
|
|
4
|
+
:class:`TrainingConfig`) and the per-run :class:`ConfigManager` factory
|
|
5
|
+
that maps a dataset key from the command line to its ARFF location, the
|
|
6
|
+
target label count, and the fixed list of noisy rates / base learners /
|
|
7
|
+
fold and repeat counts used throughout the paper.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from preorder4mlc.constants import BaseLearnerName
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class AlgorithmType(Enum):
|
|
18
|
+
BOPOS = "bopos"
|
|
19
|
+
CLR = "clr"
|
|
20
|
+
BR = "br"
|
|
21
|
+
CC = "cc"
|
|
22
|
+
ECC = "ecc"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class DatasetConfig:
|
|
27
|
+
name: str
|
|
28
|
+
file: str
|
|
29
|
+
n_labels: int
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class TrainingConfig:
|
|
34
|
+
data_path: str
|
|
35
|
+
results_dir: str
|
|
36
|
+
noisy_rates: list[float]
|
|
37
|
+
base_learners: list[BaseLearnerName]
|
|
38
|
+
total_repeat_times: int
|
|
39
|
+
number_folds: int
|
|
40
|
+
algorithms: list[AlgorithmType] # Which algorithms to run
|
|
41
|
+
# When set, orchestrator skips repeats/folds whose index does not match.
|
|
42
|
+
# Result pickle filename is suffixed with _r<repeat>_f<fold> to keep
|
|
43
|
+
# split partials distinct; merge_split_results.py recombines them.
|
|
44
|
+
repeat_idx_filter: int | None = None
|
|
45
|
+
fold_idx_filter: int | None = None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ConfigManager:
|
|
49
|
+
DATASET_CONFIGS = {
|
|
50
|
+
"chd_49": DatasetConfig("CHD_49", "CHD_49.arff", 6),
|
|
51
|
+
"emotions": DatasetConfig("emotions", "emotions.arff", 6),
|
|
52
|
+
"scene": DatasetConfig("scene", "scene.arff", 6),
|
|
53
|
+
"yeast": DatasetConfig("Yeast", "Yeast.arff", 14),
|
|
54
|
+
"water_quality": DatasetConfig("Water-quality", "Water-quality.arff", 14),
|
|
55
|
+
"humanpseaac": DatasetConfig("HumanPseAAC", "HumanPseAAC.arff", 14),
|
|
56
|
+
"gpositivepseaac": DatasetConfig("GpositivePseAAC", "GpositivePseAAC.arff", 4),
|
|
57
|
+
"plantpseaac": DatasetConfig("PlantPseAAC", "PlantPseAAC.arff", 12),
|
|
58
|
+
"viruspseaac": DatasetConfig("VirusPseAAC", "VirusPseAAC.arff", 6),
|
|
59
|
+
# enron (K=53). ARFF is NOT bundled — download from COMETA / MULAN
|
|
60
|
+
# and place at ./data/enron.arff. See REPRODUCE.md.
|
|
61
|
+
"enron": DatasetConfig("enron", "enron.arff", 53),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
@staticmethod
|
|
65
|
+
def get_dataset_config(dataset_name: str) -> DatasetConfig:
|
|
66
|
+
dataset_name = dataset_name.lower()
|
|
67
|
+
if dataset_name not in ConfigManager.DATASET_CONFIGS:
|
|
68
|
+
raise ValueError(f"Dataset {dataset_name} not found")
|
|
69
|
+
return ConfigManager.DATASET_CONFIGS[dataset_name]
|
|
70
|
+
|
|
71
|
+
@staticmethod
|
|
72
|
+
def get_training_config(args) -> TrainingConfig:
|
|
73
|
+
results_dir = args.results_dir if args.results_dir else "./results"
|
|
74
|
+
Path(results_dir).mkdir(parents=True, exist_ok=True)
|
|
75
|
+
|
|
76
|
+
# NOISY_RATES = [0.0, 0.1, 0.2, 0.3]
|
|
77
|
+
# If --noise_rate is passed on the CLI, restrict to that single level
|
|
78
|
+
# so the run can be split across slurm jobs by noise.
|
|
79
|
+
single_noise = getattr(args, "noise_rate", None)
|
|
80
|
+
if single_noise is not None:
|
|
81
|
+
NOISY_RATES = [float(single_noise)]
|
|
82
|
+
else:
|
|
83
|
+
NOISY_RATES = [
|
|
84
|
+
0.0,
|
|
85
|
+
0.1,
|
|
86
|
+
0.2,
|
|
87
|
+
0.3,
|
|
88
|
+
]
|
|
89
|
+
# Default is RF (paper-equivalent). Pass --base_learner LightGBM on
|
|
90
|
+
# the CLI to switch to LightGBM (paper also reports LGBM results).
|
|
91
|
+
BASE_LEARNERS = [BaseLearnerName.RF]
|
|
92
|
+
bl_override = getattr(args, "base_learner", None)
|
|
93
|
+
if bl_override:
|
|
94
|
+
BASE_LEARNERS = [BaseLearnerName(bl_override)]
|
|
95
|
+
ALGORITHMS = [
|
|
96
|
+
AlgorithmType.BOPOS,
|
|
97
|
+
AlgorithmType.CLR,
|
|
98
|
+
AlgorithmType.BR,
|
|
99
|
+
AlgorithmType.CC,
|
|
100
|
+
]
|
|
101
|
+
algo_override = getattr(args, "algorithm", None)
|
|
102
|
+
if algo_override:
|
|
103
|
+
ALGORITHMS = [AlgorithmType(algo_override)]
|
|
104
|
+
|
|
105
|
+
return TrainingConfig(
|
|
106
|
+
data_path="./data/",
|
|
107
|
+
results_dir=results_dir,
|
|
108
|
+
noisy_rates=NOISY_RATES,
|
|
109
|
+
base_learners=BASE_LEARNERS,
|
|
110
|
+
total_repeat_times=5,
|
|
111
|
+
number_folds=5,
|
|
112
|
+
algorithms=ALGORITHMS,
|
|
113
|
+
repeat_idx_filter=getattr(args, "repeat_idx", None),
|
|
114
|
+
fold_idx_filter=getattr(args, "fold_idx", None),
|
|
115
|
+
)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Shared constants and enum types.
|
|
2
|
+
|
|
3
|
+
Centralises the base learner names and target metric enums used across
|
|
4
|
+
:mod:`training_orchestrator`, :mod:`inference_models`, and
|
|
5
|
+
:mod:`searching_algorithms`, along with the canonical random seed that
|
|
6
|
+
controls every reproducible split and noise draw in the paper.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from enum import Enum
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class BaseLearnerName(Enum):
|
|
13
|
+
"""Identifier for the supported base learners."""
|
|
14
|
+
|
|
15
|
+
RF = "RF"
|
|
16
|
+
ETC = "ETC"
|
|
17
|
+
XGBoost = "XGBoost"
|
|
18
|
+
LightGBM = "LightGBM"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# Used by every split, fold, and Bernoulli noise draw. Do not change
|
|
22
|
+
# casually -- result directories in `results/` are tied to this seed.
|
|
23
|
+
RANDOM_STATE = 6
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class TargetMetric(Enum):
|
|
27
|
+
"""Loss the ILP search optimises for in :mod:`searching_algorithms`."""
|
|
28
|
+
|
|
29
|
+
Hamming = "Hamming"
|
|
30
|
+
Subset = "Subset"
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Dataset loading, splitting, and noise injection.
|
|
2
|
+
|
|
3
|
+
:class:`Datasets4Experiments` reads multi-label ARFF files, separates X
|
|
4
|
+
from Y based on whether labels are at the beginning or end of the file
|
|
5
|
+
(controlled by :data:`TARGET_IN_END_FILE_DATASETS`), and yields k-fold
|
|
6
|
+
splits with optional symmetric label-flip noise (Bernoulli with rate
|
|
7
|
+
``noisy_rate``) for the training fold of each split.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from logging import INFO, log
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
import pandas as pd
|
|
14
|
+
from scipy.io import arff
|
|
15
|
+
from scipy.stats import bernoulli
|
|
16
|
+
from sklearn.model_selection import KFold
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
import arff as liac_arff # liac-arff package — handles sparse ARFF
|
|
20
|
+
except ImportError: # pragma: no cover
|
|
21
|
+
liac_arff = None
|
|
22
|
+
|
|
23
|
+
TARGET_IN_END_FILE_DATASETS = [
|
|
24
|
+
"emotions.arff",
|
|
25
|
+
"scene.arff",
|
|
26
|
+
"flags.arff",
|
|
27
|
+
"VirusGO.arff",
|
|
28
|
+
"VirusPseAAC.arff",
|
|
29
|
+
"Yelp.arff",
|
|
30
|
+
"birds.arff",
|
|
31
|
+
"HumanPseAAC.arff",
|
|
32
|
+
"PlantGO.arff",
|
|
33
|
+
"GpositivePseAAC.arff",
|
|
34
|
+
"PlantPseAAC.arff",
|
|
35
|
+
# Medium-K additions (COMETA convention: labels at end of attribute list).
|
|
36
|
+
"enron.arff",
|
|
37
|
+
"medical.arff",
|
|
38
|
+
# Large-K additions (COMETA convention: labels at end of attribute list).
|
|
39
|
+
"CAL500.arff",
|
|
40
|
+
"mediamill.arff",
|
|
41
|
+
"bibtex.arff",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _load_sparse_arff_to_dataframe(path: str) -> pd.DataFrame:
|
|
46
|
+
"""Parse sparse-format ARFF via liac-arff and densify to a DataFrame.
|
|
47
|
+
|
|
48
|
+
Sparse rows look like ``{2 1, 5 1, ...}`` — every non-listed index is 0.
|
|
49
|
+
liac-arff returns a (n_samples, n_attributes) list-of-lists with zeros
|
|
50
|
+
filled in, which we wrap in a DataFrame keyed by the attribute names.
|
|
51
|
+
"""
|
|
52
|
+
with open(path) as f:
|
|
53
|
+
obj = liac_arff.load(f, return_type=liac_arff.DENSE_GEN)
|
|
54
|
+
attr_names = [a[0] for a in obj["attributes"]]
|
|
55
|
+
rows = list(obj["data"])
|
|
56
|
+
return pd.DataFrame(rows, columns=attr_names)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Datasets4Experiments:
|
|
60
|
+
|
|
61
|
+
def __init__(self, data_path: str, data_files: list[dict]):
|
|
62
|
+
self.data_path = data_path
|
|
63
|
+
|
|
64
|
+
self.data_files = []
|
|
65
|
+
self.n_labels_set = []
|
|
66
|
+
for item in data_files:
|
|
67
|
+
self.data_files.append(item["dataset_name"])
|
|
68
|
+
self.n_labels_set.append(item["n_labels_set"])
|
|
69
|
+
|
|
70
|
+
self.datasets: list[tuple[np.ndarray, np.ndarray, str]] = []
|
|
71
|
+
|
|
72
|
+
def load_datasets(self):
|
|
73
|
+
for file_name, n_labels in zip(self.data_files, self.n_labels_set):
|
|
74
|
+
full_path = f"{self.data_path}{file_name}"
|
|
75
|
+
log(INFO, f"Loading dataset from {full_path}")
|
|
76
|
+
# NPY path: dataset_name ends with "_features.npy"; sibling labels
|
|
77
|
+
# file lives at the same path with "_features" replaced by "_labels".
|
|
78
|
+
if file_name.endswith("_features.npy"):
|
|
79
|
+
X = np.load(full_path).astype(np.float32)
|
|
80
|
+
labels_path = full_path.replace("_features.npy", "_labels.npy")
|
|
81
|
+
Y = np.load(labels_path).astype(int)
|
|
82
|
+
if Y.shape[1] != n_labels:
|
|
83
|
+
raise ValueError(
|
|
84
|
+
f"{file_name}: labels file has {Y.shape[1]} columns, "
|
|
85
|
+
f"expected n_labels_set={n_labels}"
|
|
86
|
+
)
|
|
87
|
+
Y = np.where(Y < 0, 0, Y)
|
|
88
|
+
df_name = file_name.removesuffix("_features.npy")
|
|
89
|
+
self.datasets.append((X, Y, df_name))
|
|
90
|
+
continue
|
|
91
|
+
try:
|
|
92
|
+
data, _meta = arff.loadarff(full_path)
|
|
93
|
+
df = pd.DataFrame(data)
|
|
94
|
+
except (ValueError, NotImplementedError) as e:
|
|
95
|
+
# scipy.io.arff cannot parse sparse ARFF format
|
|
96
|
+
# ({idx val, idx val, ...}). Fall back to liac-arff which
|
|
97
|
+
# supports it, then densify.
|
|
98
|
+
if liac_arff is None:
|
|
99
|
+
raise RuntimeError(
|
|
100
|
+
f"scipy.io.arff failed on {file_name} ({e}). "
|
|
101
|
+
"Install liac-arff (`pip install liac-arff`) to handle "
|
|
102
|
+
"sparse ARFF datasets like bibtex."
|
|
103
|
+
) from e
|
|
104
|
+
log(INFO, f" scipy failed ({e!s}); retrying with liac-arff…")
|
|
105
|
+
df = _load_sparse_arff_to_dataframe(full_path)
|
|
106
|
+
|
|
107
|
+
is_target_in_end = any(
|
|
108
|
+
f.lower() == file_name.lower() for f in TARGET_IN_END_FILE_DATASETS
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
X, Y = self.preprocess_data(df, n_labels, is_target_in_end)
|
|
112
|
+
df_name = file_name.split(".")[0]
|
|
113
|
+
self.datasets.append((X, Y, df_name))
|
|
114
|
+
|
|
115
|
+
def preprocess_data(self, df, n_labels, is_target_in_end=False):
|
|
116
|
+
if is_target_in_end:
|
|
117
|
+
X = df.iloc[:, :-n_labels].to_numpy()
|
|
118
|
+
Y = df.iloc[:, -n_labels:].to_numpy().astype(int)
|
|
119
|
+
else:
|
|
120
|
+
X = df.iloc[:, n_labels:].to_numpy()
|
|
121
|
+
Y = df.iloc[:, :n_labels].to_numpy().astype(int)
|
|
122
|
+
|
|
123
|
+
# liac-arff returns object dtype when attributes are stored as strings
|
|
124
|
+
# in a sparse ARFF. Coerce to float for sklearn/LightGBM compatibility.
|
|
125
|
+
if X.dtype == object:
|
|
126
|
+
X = X.astype(float)
|
|
127
|
+
|
|
128
|
+
# Map sklearn-style -1 (negative) labels to 0 so downstream pairwise
|
|
129
|
+
# encoders see a clean {0,1} matrix.
|
|
130
|
+
Y = np.where(Y < 0, 0, Y)
|
|
131
|
+
|
|
132
|
+
return X, Y
|
|
133
|
+
|
|
134
|
+
def add_noise_to_labels(self, Y, noisy_rate):
|
|
135
|
+
"""
|
|
136
|
+
Adds noise to the dataset labels based on the specified noisy rate.
|
|
137
|
+
|
|
138
|
+
:param Y: The label matrix for a dataset.
|
|
139
|
+
:param noisy_rate: The rate at which noise should be added to the labels.
|
|
140
|
+
:return: The label matrix with added noise.
|
|
141
|
+
"""
|
|
142
|
+
n_instances, n_labels = Y.shape
|
|
143
|
+
for i in range(n_instances):
|
|
144
|
+
for j in range(n_labels):
|
|
145
|
+
if bernoulli.rvs(p=noisy_rate):
|
|
146
|
+
Y[i, j] = 1 - Y[i, j] # Flip the label to add noise
|
|
147
|
+
return Y
|
|
148
|
+
|
|
149
|
+
def kfold_split_with_noise(
|
|
150
|
+
self, dataset_index, n_splits=5, noisy_rate=0.0, random_state=None, shuffle=True
|
|
151
|
+
):
|
|
152
|
+
"""
|
|
153
|
+
Generates K-fold splits for a specific dataset and adds noise to the training labels.
|
|
154
|
+
|
|
155
|
+
:param dataset_index: Index of the dataset to split.
|
|
156
|
+
:param n_splits: Number of folds.
|
|
157
|
+
:param noisy_rate: Noise rate to be applied to the training set labels.
|
|
158
|
+
:param random_state: Random state for reproducibility.
|
|
159
|
+
:param shuffle: Whether to shuffle the data before splitting.
|
|
160
|
+
:return: Generator of K-fold splits (train_index, test_index) with noisy training labels.
|
|
161
|
+
"""
|
|
162
|
+
X, Y, _ = self.datasets[dataset_index]
|
|
163
|
+
kf = KFold(n_splits=n_splits, random_state=random_state, shuffle=shuffle)
|
|
164
|
+
for train_index, test_index in kf.split(X):
|
|
165
|
+
Y_train_noisy = self.add_noise_to_labels(Y[train_index].copy(), noisy_rate)
|
|
166
|
+
# I want to return x_train, y_train_noisy, x_test, y_test
|
|
167
|
+
yield X[train_index], Y_train_noisy, X[test_index], Y[test_index]
|
|
168
|
+
|
|
169
|
+
def get_datasets(self) -> list:
|
|
170
|
+
return self.datasets
|
|
171
|
+
|
|
172
|
+
def get_length(self) -> int:
|
|
173
|
+
return len(self.datasets)
|
|
174
|
+
|
|
175
|
+
def get_dataset_name(self, dataset_index) -> str:
|
|
176
|
+
return self.datasets[dataset_index][2]
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Uniform estimator interface over scikit-learn and LightGBM backends.
|
|
2
|
+
|
|
3
|
+
:class:`Estimator` is the single adapter every other module talks to;
|
|
4
|
+
its constructor accepts a :class:`constants.BaseLearnerName` and hides
|
|
5
|
+
the differences between :class:`sklearn.ensemble.RandomForestClassifier`,
|
|
6
|
+
:class:`sklearn.ensemble.ExtraTreesClassifier`,
|
|
7
|
+
:class:`sklearn.ensemble.GradientBoostingClassifier` (XGBoost in our
|
|
8
|
+
nomenclature), and :class:`lightgbm.LGBMClassifier`.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
from logging import INFO, basicConfig
|
|
13
|
+
|
|
14
|
+
from lightgbm import LGBMClassifier
|
|
15
|
+
from numpy.typing import NDArray
|
|
16
|
+
from sklearn.base import BaseEstimator
|
|
17
|
+
from sklearn.calibration import CalibratedClassifierCV
|
|
18
|
+
from sklearn.ensemble import (
|
|
19
|
+
ExtraTreesClassifier,
|
|
20
|
+
GradientBoostingClassifier,
|
|
21
|
+
RandomForestClassifier,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
from preorder4mlc.constants import RANDOM_STATE, BaseLearnerName
|
|
25
|
+
|
|
26
|
+
basicConfig(level=INFO) # type: ignore
|
|
27
|
+
|
|
28
|
+
number_of_cores: int = os.cpu_count() if os.cpu_count() is not None else 1 # type: ignore
|
|
29
|
+
# log(INFO, f"Number of cores: {number_of_cores}")
|
|
30
|
+
|
|
31
|
+
# Whether to wrap the base learner in CalibratedClassifierCV (isotonic).
|
|
32
|
+
# Opt-in: default OFF so existing pipelines / paper-equivalent runs are
|
|
33
|
+
# unchanged. Enable per-run via env (PREORDER_CALIBRATE=1) or constructor
|
|
34
|
+
# (Estimator(name, calibrate=True)). The A/B/C/D ablation in
|
|
35
|
+
# scripts/ablations/ablation_base_learner.py decides whether to flip the default.
|
|
36
|
+
CALIBRATE_PROBAS = os.environ.get("PREORDER_CALIBRATE", "0") not in ("0", "false", "False")
|
|
37
|
+
CALIBRATION_METHOD = os.environ.get("PREORDER_CALIBRATION_METHOD", "isotonic")
|
|
38
|
+
CALIBRATION_CV = int(os.environ.get("PREORDER_CALIBRATION_CV", "3"))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class Estimator:
|
|
42
|
+
def __init__(self, name: str, calibrate: bool | None = None):
|
|
43
|
+
self.name = name
|
|
44
|
+
self.calibrate = CALIBRATE_PROBAS if calibrate is None else calibrate
|
|
45
|
+
base = self.get_classifier()
|
|
46
|
+
if self.calibrate:
|
|
47
|
+
self.clf = CalibratedClassifierCV(
|
|
48
|
+
base, method=CALIBRATION_METHOD, cv=CALIBRATION_CV
|
|
49
|
+
)
|
|
50
|
+
else:
|
|
51
|
+
self.clf = base
|
|
52
|
+
|
|
53
|
+
def get_classifier(self) -> BaseEstimator | LGBMClassifier:
|
|
54
|
+
"""Get the classifier based on name with proper error handling."""
|
|
55
|
+
if self.name == BaseLearnerName.RF.value:
|
|
56
|
+
return RandomForestClassifier(random_state=RANDOM_STATE)
|
|
57
|
+
elif self.name == BaseLearnerName.ETC.value:
|
|
58
|
+
return ExtraTreesClassifier(random_state=RANDOM_STATE)
|
|
59
|
+
elif self.name == BaseLearnerName.XGBoost.value:
|
|
60
|
+
return GradientBoostingClassifier(random_state=RANDOM_STATE)
|
|
61
|
+
elif self.name == BaseLearnerName.LightGBM.value:
|
|
62
|
+
return LGBMClassifier(
|
|
63
|
+
random_state=RANDOM_STATE,
|
|
64
|
+
# n_jobs=1 because LGBM is always called inside a
|
|
65
|
+
# joblib.Parallel(n_jobs=-1) loop in base_classifiers.py and the
|
|
66
|
+
# ECC/LP path. Using n_jobs=cores caused thread oversubscription
|
|
67
|
+
# (~5x slower on yeast); see scripts/ablations/NOTES.md.
|
|
68
|
+
n_jobs=1,
|
|
69
|
+
verbose=-1,
|
|
70
|
+
num_leaves=20, # Moderate complexity
|
|
71
|
+
max_depth=6,
|
|
72
|
+
bagging_fraction=0.9,
|
|
73
|
+
feature_fraction=0.8,
|
|
74
|
+
learning_rate=0.1,
|
|
75
|
+
n_estimators=100,
|
|
76
|
+
min_child_samples=5, # Relaxed for small datasets
|
|
77
|
+
min_child_weight=0.0001, # Allow splits with low Hessian
|
|
78
|
+
min_split_gain=0.01, # Allow minimal gain splits
|
|
79
|
+
is_unbalance=True, # Handle label imbalance
|
|
80
|
+
device="cpu",
|
|
81
|
+
)
|
|
82
|
+
else:
|
|
83
|
+
raise ValueError(f"Unknown base learner: {self.name}")
|
|
84
|
+
|
|
85
|
+
def fit(self, X: NDArray, Y: NDArray):
|
|
86
|
+
"""Fit the classifier with proper error handling.
|
|
87
|
+
|
|
88
|
+
When calibration is enabled but the data is degenerate (single class
|
|
89
|
+
or too few samples per class to support cv folds), fall back to the
|
|
90
|
+
uncalibrated base estimator so the pairwise pipeline still runs.
|
|
91
|
+
"""
|
|
92
|
+
try:
|
|
93
|
+
self.clf.fit(X, Y) # type: ignore
|
|
94
|
+
except Exception as e:
|
|
95
|
+
if self.calibrate:
|
|
96
|
+
fallback = self.get_classifier()
|
|
97
|
+
try:
|
|
98
|
+
fallback.fit(X, Y) # type: ignore
|
|
99
|
+
self.clf = fallback
|
|
100
|
+
return
|
|
101
|
+
except Exception as e2:
|
|
102
|
+
raise ValueError(
|
|
103
|
+
f"Error training {self.name} (calibrated and uncalibrated both failed): {e2}"
|
|
104
|
+
) from e2
|
|
105
|
+
raise ValueError(f"Error training {self.name}: {e}") from e
|
|
106
|
+
|
|
107
|
+
def predict_proba(self, X: NDArray) -> NDArray:
|
|
108
|
+
"""Predict the probability of each class for each instance."""
|
|
109
|
+
prob = self.clf.predict_proba(X) # type: ignore
|
|
110
|
+
|
|
111
|
+
return prob # type: ignore
|
|
112
|
+
|
|
113
|
+
def classes_(self) -> list[int]:
|
|
114
|
+
return self.clf.classes_ # type: ignore
|
|
115
|
+
|
|
116
|
+
def predict(self, X: NDArray) -> NDArray:
|
|
117
|
+
return self.clf.predict(X) # type: ignore
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def train_classifier(X, Y, estimator_name):
|
|
121
|
+
classifier = Estimator(estimator_name) # Add n_jobs or other params here
|
|
122
|
+
classifier.fit(X, Y)
|
|
123
|
+
return classifier
|