preorder4mlc 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,4 @@
1
+ """Pre-order based multi-label classification (preorder4MLC).
2
+
3
+ Library package. CLI entry points live under ``scripts/``.
4
+ """
@@ -0,0 +1,242 @@
1
+ """Pairwise- and calibrated-classifier factories.
2
+
3
+ :class:`BaseClassifiers` builds the per-pair training datasets that
4
+ :class:`inference_models.PredictBOPOs` consumes: pairwise classifiers for
5
+ PRE_ORDER and PARTIAL_ORDER variants, and one calibrated classifier per
6
+ label for CLR. Estimator construction is delegated to
7
+ :class:`estimator.Estimator`, with fitting parallelised via joblib.
8
+ """
9
+
10
+ from logging import INFO, log
11
+
12
+ import numpy as np
13
+ from joblib import Parallel, delayed
14
+ from numpy.typing import NDArray
15
+ from sklearn.base import BaseEstimator
16
+
17
+ from preorder4mlc.estimator import Estimator, train_classifier
18
+
19
+
20
+ class BaseClassifiers:
21
+ """Base classifiers for multi-label classification with different order types.
22
+
23
+ Attributes:
24
+ base_learner (Estimator): The base learning algorithm
25
+ logger (logging.Logger): Logger instance
26
+ """
27
+
28
+ def __init__(self, name: str):
29
+ log(
30
+ INFO,
31
+ f"BaseClassifiers: Initializing base learner: {name}",
32
+ )
33
+ self.name = name
34
+
35
+ def get_classifier(self) -> BaseEstimator:
36
+ return Estimator(self.name).get_classifier() # type: ignore
37
+
38
+ def pairwise_calibrated_classifier(self, X: NDArray[np.float64], Y: NDArray[np.int32]):
39
+ """Train pairwise calibrated classifiers.
40
+
41
+ Args:
42
+ X: Input features of shape (n_samples, n_features)
43
+ Y: Binary label matrix of shape (n_samples, n_labels)
44
+
45
+ X_train = [[1, 2, 3, 0], [3, 2, 4, 1], [4, 5, 3, 3], [7, 6, 3, 1]]
46
+ Y_train = [[1, 1, 0], [0, 1, 0], [1, 0, 0], [0, 0, 1]]
47
+ n_labels = 3
48
+
49
+ Returns:
50
+ Tuple containing:
51
+ - Dictionary of pairwise classifiers
52
+ - List of calibrated classifiers
53
+ """
54
+ n_instances, n_labels = Y.shape
55
+
56
+ # calibrated_classifiers is in fact is a (inverse) BR classifier
57
+ calibrated_classifiers = []
58
+ clr_dataset_classifier = {}
59
+ for k in range(n_labels):
60
+ # MCC = score for class 0; invert the label column so the trained
61
+ # classifier produces P(y_k = 0) directly (CLR convention).
62
+ # Example: Y[:, k] = [1, 0, 1, 0] -> MCC_y = [0, 1, 0, 1].
63
+ MCC_y = np.logical_not(Y[:, k]).astype(int)
64
+ clr_dataset_classifier[str(k)] = { # type: ignore
65
+ "X": X,
66
+ "Y": MCC_y,
67
+ }
68
+
69
+ log(
70
+ INFO,
71
+ f"\t - Training for {len(clr_dataset_classifier.keys())} calibrated_classifiers with {self.name}",
72
+ )
73
+ # run fit in parallel for each pair of labels
74
+ classifiers = Parallel(n_jobs=-1)(
75
+ delayed(train_classifier)(
76
+ clr_dataset_classifier[str(k)]["X"],
77
+ clr_dataset_classifier[str(k)]["Y"],
78
+ self.name,
79
+ ) # type: ignore
80
+ for k in range(n_labels)
81
+ )
82
+ log(INFO, f"\t - Trained {n_labels} classifiers")
83
+
84
+ calibrated_classifiers = list(classifiers)
85
+
86
+ pairwise_classifiers = {}
87
+ single_label_pair = {}
88
+ dataset_classifier = {}
89
+ for i in range(n_labels - 1):
90
+ for j in range(i + 1, n_labels):
91
+ key = f"{i}_{j}"
92
+ MCC_X = []
93
+ MCC_y = []
94
+
95
+ for n in range(n_instances):
96
+ if Y[n, i] == 1 and Y[n, j] == 0:
97
+ MCC_X.append(X[n])
98
+ MCC_y.append(0)
99
+ elif Y[n, i] == 0 and Y[n, j] == 1:
100
+ MCC_X.append(X[n])
101
+ MCC_y.append(1)
102
+
103
+ dataset_classifier[key] = { # type: ignore
104
+ "X": MCC_X,
105
+ "Y": MCC_y,
106
+ }
107
+
108
+ log(
109
+ INFO,
110
+ f"\t - Training for {len(dataset_classifier.keys())} pairs with {self.name}",
111
+ )
112
+
113
+ # Skip degenerate pairs (empty, single-sample, or single-class). LGBM
114
+ # rejects single-sample fits and any single-class fit; RF tolerates
115
+ # single-sample but not single-class. downstream predict_CLR guards
116
+ # against missing keys.
117
+ trainable_keys = [
118
+ k for k in dataset_classifier.keys()
119
+ if len(dataset_classifier[k]["Y"]) >= 2
120
+ and len(set(dataset_classifier[k]["Y"])) >= 2
121
+ ]
122
+ skipped = len(dataset_classifier) - len(trainable_keys)
123
+ if skipped:
124
+ log(INFO, f"\t - Skipping {skipped} empty pairs (degenerate fold)")
125
+
126
+ classifiers = Parallel(n_jobs=-1)(
127
+ delayed(train_classifier)(
128
+ dataset_classifier[key]["X"],
129
+ dataset_classifier[key]["Y"],
130
+ self.name,
131
+ ) # type: ignore
132
+ for key in trainable_keys
133
+ )
134
+
135
+ log(INFO, f"\t - Trained {len(trainable_keys)} classifiers")
136
+
137
+ pairwise_classifiers = dict(zip(trainable_keys, classifiers)) # type: ignore
138
+
139
+ for i in range(n_labels - 1):
140
+ for j in range(i + 1, n_labels):
141
+ key = f"{i}_{j}"
142
+ MCC_y = dataset_classifier[key]["Y"]
143
+ single_label_pair[key] = (
144
+ 1
145
+ if len(np.unique(MCC_y)) == 1 and np.unique(MCC_y)[0] == 1
146
+ else (0 if len(np.unique(MCC_y)) == 1 and np.unique(MCC_y)[0] == 0 else None)
147
+ )
148
+
149
+ return pairwise_classifiers, calibrated_classifiers, single_label_pair
150
+
151
+ def pairwise_partial_order_classifier_fit(self, X, Y):
152
+ # This BaseClassifier provides pairwise_probability_information for learning partial orders
153
+ n_labels = len(Y[0])
154
+ n_instances, _ = Y.shape
155
+ dataset_classifier = {}
156
+ pairwise_classifiers = {}
157
+
158
+ for i in range(n_labels - 1):
159
+ for j in range(i + 1, n_labels):
160
+ key = f"{i}_{j}"
161
+ MCC_y = []
162
+ for n in range(n_instances):
163
+ if Y[n, i] == Y[n, j]:
164
+ MCC_y.append(2)
165
+ elif Y[n, i] == 1 and Y[n, j] == 0:
166
+ MCC_y.append(0)
167
+ elif Y[n, i] == 0 and Y[n, j] == 1:
168
+ MCC_y.append(1)
169
+ # X is read-only input for sklearn/LightGBM fit; share the
170
+ # reference across pairs instead of copying. For K=101 (mediamill)
171
+ # that's 5050 copies of a 42 MB array (~212 GB) avoided.
172
+ dataset_classifier[key] = { # type: ignore
173
+ "X": X,
174
+ "Y": MCC_y,
175
+ }
176
+ log(
177
+ INFO,
178
+ f"\t - Training for {len(dataset_classifier.keys())} pairs with {self.name}",
179
+ )
180
+ # run fit in parallel for each pair of labels
181
+ classifiers = Parallel(n_jobs=-1)(
182
+ delayed(train_classifier)(
183
+ dataset_classifier[key]["X"],
184
+ dataset_classifier[key]["Y"],
185
+ self.name,
186
+ ) # type: ignore
187
+ for key in dataset_classifier.keys()
188
+ )
189
+ log(INFO, f"\t - Trained {len(dataset_classifier.keys())} classifiers")
190
+
191
+ pairwise_classifiers = dict(zip(dataset_classifier.keys(), classifiers)) # type: ignore
192
+
193
+ return pairwise_classifiers # type: ignore
194
+
195
+ def pairwise_pre_order_classifier_fit(self, X, Y) -> dict[str, Estimator]:
196
+ """
197
+ This BaseClassifier provides pairwise_probability_information for learning preorders
198
+ For each pair of labels, we will train a classifier to predict the probability of the label
199
+ """
200
+ n_instances, n_labels = Y.shape
201
+ log(INFO, f"\t - {n_labels} labels with {self.name}")
202
+ dataset_classifier = {}
203
+ for i in range(n_labels - 1):
204
+ for j in range(i + 1, n_labels):
205
+ key = f"{i}_{j}"
206
+ MCC_y = []
207
+ for n in range(n_instances):
208
+ if Y[n, i] == 0 and Y[n, j] == 0:
209
+ MCC_y.append(2)
210
+ elif Y[n, i] == 1 and Y[n, j] == 1:
211
+ MCC_y.append(3)
212
+ elif Y[n, i] == 1 and Y[n, j] == 0:
213
+ MCC_y.append(0)
214
+ elif Y[n, i] == 0 and Y[n, j] == 1:
215
+ MCC_y.append(1)
216
+
217
+ # X is read-only input for sklearn/LightGBM fit; share the
218
+ # reference across pairs instead of copying.
219
+ dataset_classifier[key] = { # type: ignore
220
+ "X": X,
221
+ "Y": MCC_y,
222
+ }
223
+ log(
224
+ INFO,
225
+ f"\t - Training {len(dataset_classifier.keys())} pairs with {self.name}",
226
+ )
227
+ # run fit in parallel for each pair of labels
228
+ classifiers = Parallel(n_jobs=-1)(
229
+ delayed(train_classifier)(
230
+ dataset_classifier[key]["X"],
231
+ dataset_classifier[key]["Y"],
232
+ self.name,
233
+ ) # type: ignore
234
+ for key in dataset_classifier.keys()
235
+ )
236
+ log(INFO, f"\t - Trained {len(dataset_classifier.keys())} classifiers")
237
+
238
+ pairwise_classifiers = dict(zip(dataset_classifier.keys(), classifiers)) # type: ignore
239
+
240
+ # This is a dictionary of pairwise classifiers. [key] is a string of the form "i_j"
241
+ # where i and j are the indices of the labels in the label matrix Y
242
+ return pairwise_classifiers # type: ignore
preorder4mlc/config.py ADDED
@@ -0,0 +1,115 @@
1
+ """Run configuration for the BOPOs training pipeline.
2
+
3
+ Defines the typed config dataclasses (:class:`DatasetConfig`,
4
+ :class:`TrainingConfig`) and the per-run :class:`ConfigManager` factory
5
+ that maps a dataset key from the command line to its ARFF location, the
6
+ target label count, and the fixed list of noisy rates / base learners /
7
+ fold and repeat counts used throughout the paper.
8
+ """
9
+
10
+ from dataclasses import dataclass
11
+ from enum import Enum
12
+ from pathlib import Path
13
+
14
+ from preorder4mlc.constants import BaseLearnerName
15
+
16
+
17
+ class AlgorithmType(Enum):
18
+ BOPOS = "bopos"
19
+ CLR = "clr"
20
+ BR = "br"
21
+ CC = "cc"
22
+ ECC = "ecc"
23
+
24
+
25
+ @dataclass
26
+ class DatasetConfig:
27
+ name: str
28
+ file: str
29
+ n_labels: int
30
+
31
+
32
+ @dataclass
33
+ class TrainingConfig:
34
+ data_path: str
35
+ results_dir: str
36
+ noisy_rates: list[float]
37
+ base_learners: list[BaseLearnerName]
38
+ total_repeat_times: int
39
+ number_folds: int
40
+ algorithms: list[AlgorithmType] # Which algorithms to run
41
+ # When set, orchestrator skips repeats/folds whose index does not match.
42
+ # Result pickle filename is suffixed with _r<repeat>_f<fold> to keep
43
+ # split partials distinct; merge_split_results.py recombines them.
44
+ repeat_idx_filter: int | None = None
45
+ fold_idx_filter: int | None = None
46
+
47
+
48
+ class ConfigManager:
49
+ DATASET_CONFIGS = {
50
+ "chd_49": DatasetConfig("CHD_49", "CHD_49.arff", 6),
51
+ "emotions": DatasetConfig("emotions", "emotions.arff", 6),
52
+ "scene": DatasetConfig("scene", "scene.arff", 6),
53
+ "yeast": DatasetConfig("Yeast", "Yeast.arff", 14),
54
+ "water_quality": DatasetConfig("Water-quality", "Water-quality.arff", 14),
55
+ "humanpseaac": DatasetConfig("HumanPseAAC", "HumanPseAAC.arff", 14),
56
+ "gpositivepseaac": DatasetConfig("GpositivePseAAC", "GpositivePseAAC.arff", 4),
57
+ "plantpseaac": DatasetConfig("PlantPseAAC", "PlantPseAAC.arff", 12),
58
+ "viruspseaac": DatasetConfig("VirusPseAAC", "VirusPseAAC.arff", 6),
59
+ # enron (K=53). ARFF is NOT bundled — download from COMETA / MULAN
60
+ # and place at ./data/enron.arff. See REPRODUCE.md.
61
+ "enron": DatasetConfig("enron", "enron.arff", 53),
62
+ }
63
+
64
+ @staticmethod
65
+ def get_dataset_config(dataset_name: str) -> DatasetConfig:
66
+ dataset_name = dataset_name.lower()
67
+ if dataset_name not in ConfigManager.DATASET_CONFIGS:
68
+ raise ValueError(f"Dataset {dataset_name} not found")
69
+ return ConfigManager.DATASET_CONFIGS[dataset_name]
70
+
71
+ @staticmethod
72
+ def get_training_config(args) -> TrainingConfig:
73
+ results_dir = args.results_dir if args.results_dir else "./results"
74
+ Path(results_dir).mkdir(parents=True, exist_ok=True)
75
+
76
+ # NOISY_RATES = [0.0, 0.1, 0.2, 0.3]
77
+ # If --noise_rate is passed on the CLI, restrict to that single level
78
+ # so the run can be split across slurm jobs by noise.
79
+ single_noise = getattr(args, "noise_rate", None)
80
+ if single_noise is not None:
81
+ NOISY_RATES = [float(single_noise)]
82
+ else:
83
+ NOISY_RATES = [
84
+ 0.0,
85
+ 0.1,
86
+ 0.2,
87
+ 0.3,
88
+ ]
89
+ # Default is RF (paper-equivalent). Pass --base_learner LightGBM on
90
+ # the CLI to switch to LightGBM (paper also reports LGBM results).
91
+ BASE_LEARNERS = [BaseLearnerName.RF]
92
+ bl_override = getattr(args, "base_learner", None)
93
+ if bl_override:
94
+ BASE_LEARNERS = [BaseLearnerName(bl_override)]
95
+ ALGORITHMS = [
96
+ AlgorithmType.BOPOS,
97
+ AlgorithmType.CLR,
98
+ AlgorithmType.BR,
99
+ AlgorithmType.CC,
100
+ ]
101
+ algo_override = getattr(args, "algorithm", None)
102
+ if algo_override:
103
+ ALGORITHMS = [AlgorithmType(algo_override)]
104
+
105
+ return TrainingConfig(
106
+ data_path="./data/",
107
+ results_dir=results_dir,
108
+ noisy_rates=NOISY_RATES,
109
+ base_learners=BASE_LEARNERS,
110
+ total_repeat_times=5,
111
+ number_folds=5,
112
+ algorithms=ALGORITHMS,
113
+ repeat_idx_filter=getattr(args, "repeat_idx", None),
114
+ fold_idx_filter=getattr(args, "fold_idx", None),
115
+ )
@@ -0,0 +1,30 @@
1
+ """Shared constants and enum types.
2
+
3
+ Centralises the base learner names and target metric enums used across
4
+ :mod:`training_orchestrator`, :mod:`inference_models`, and
5
+ :mod:`searching_algorithms`, along with the canonical random seed that
6
+ controls every reproducible split and noise draw in the paper.
7
+ """
8
+
9
+ from enum import Enum
10
+
11
+
12
+ class BaseLearnerName(Enum):
13
+ """Identifier for the supported base learners."""
14
+
15
+ RF = "RF"
16
+ ETC = "ETC"
17
+ XGBoost = "XGBoost"
18
+ LightGBM = "LightGBM"
19
+
20
+
21
+ # Used by every split, fold, and Bernoulli noise draw. Do not change
22
+ # casually -- result directories in `results/` are tied to this seed.
23
+ RANDOM_STATE = 6
24
+
25
+
26
+ class TargetMetric(Enum):
27
+ """Loss the ILP search optimises for in :mod:`searching_algorithms`."""
28
+
29
+ Hamming = "Hamming"
30
+ Subset = "Subset"
@@ -0,0 +1,176 @@
1
+ """Dataset loading, splitting, and noise injection.
2
+
3
+ :class:`Datasets4Experiments` reads multi-label ARFF files, separates X
4
+ from Y based on whether labels are at the beginning or end of the file
5
+ (controlled by :data:`TARGET_IN_END_FILE_DATASETS`), and yields k-fold
6
+ splits with optional symmetric label-flip noise (Bernoulli with rate
7
+ ``noisy_rate``) for the training fold of each split.
8
+ """
9
+
10
+ from logging import INFO, log
11
+
12
+ import numpy as np
13
+ import pandas as pd
14
+ from scipy.io import arff
15
+ from scipy.stats import bernoulli
16
+ from sklearn.model_selection import KFold
17
+
18
+ try:
19
+ import arff as liac_arff # liac-arff package — handles sparse ARFF
20
+ except ImportError: # pragma: no cover
21
+ liac_arff = None
22
+
23
+ TARGET_IN_END_FILE_DATASETS = [
24
+ "emotions.arff",
25
+ "scene.arff",
26
+ "flags.arff",
27
+ "VirusGO.arff",
28
+ "VirusPseAAC.arff",
29
+ "Yelp.arff",
30
+ "birds.arff",
31
+ "HumanPseAAC.arff",
32
+ "PlantGO.arff",
33
+ "GpositivePseAAC.arff",
34
+ "PlantPseAAC.arff",
35
+ # Medium-K additions (COMETA convention: labels at end of attribute list).
36
+ "enron.arff",
37
+ "medical.arff",
38
+ # Large-K additions (COMETA convention: labels at end of attribute list).
39
+ "CAL500.arff",
40
+ "mediamill.arff",
41
+ "bibtex.arff",
42
+ ]
43
+
44
+
45
+ def _load_sparse_arff_to_dataframe(path: str) -> pd.DataFrame:
46
+ """Parse sparse-format ARFF via liac-arff and densify to a DataFrame.
47
+
48
+ Sparse rows look like ``{2 1, 5 1, ...}`` — every non-listed index is 0.
49
+ liac-arff returns a (n_samples, n_attributes) list-of-lists with zeros
50
+ filled in, which we wrap in a DataFrame keyed by the attribute names.
51
+ """
52
+ with open(path) as f:
53
+ obj = liac_arff.load(f, return_type=liac_arff.DENSE_GEN)
54
+ attr_names = [a[0] for a in obj["attributes"]]
55
+ rows = list(obj["data"])
56
+ return pd.DataFrame(rows, columns=attr_names)
57
+
58
+
59
+ class Datasets4Experiments:
60
+
61
+ def __init__(self, data_path: str, data_files: list[dict]):
62
+ self.data_path = data_path
63
+
64
+ self.data_files = []
65
+ self.n_labels_set = []
66
+ for item in data_files:
67
+ self.data_files.append(item["dataset_name"])
68
+ self.n_labels_set.append(item["n_labels_set"])
69
+
70
+ self.datasets: list[tuple[np.ndarray, np.ndarray, str]] = []
71
+
72
+ def load_datasets(self):
73
+ for file_name, n_labels in zip(self.data_files, self.n_labels_set):
74
+ full_path = f"{self.data_path}{file_name}"
75
+ log(INFO, f"Loading dataset from {full_path}")
76
+ # NPY path: dataset_name ends with "_features.npy"; sibling labels
77
+ # file lives at the same path with "_features" replaced by "_labels".
78
+ if file_name.endswith("_features.npy"):
79
+ X = np.load(full_path).astype(np.float32)
80
+ labels_path = full_path.replace("_features.npy", "_labels.npy")
81
+ Y = np.load(labels_path).astype(int)
82
+ if Y.shape[1] != n_labels:
83
+ raise ValueError(
84
+ f"{file_name}: labels file has {Y.shape[1]} columns, "
85
+ f"expected n_labels_set={n_labels}"
86
+ )
87
+ Y = np.where(Y < 0, 0, Y)
88
+ df_name = file_name.removesuffix("_features.npy")
89
+ self.datasets.append((X, Y, df_name))
90
+ continue
91
+ try:
92
+ data, _meta = arff.loadarff(full_path)
93
+ df = pd.DataFrame(data)
94
+ except (ValueError, NotImplementedError) as e:
95
+ # scipy.io.arff cannot parse sparse ARFF format
96
+ # ({idx val, idx val, ...}). Fall back to liac-arff which
97
+ # supports it, then densify.
98
+ if liac_arff is None:
99
+ raise RuntimeError(
100
+ f"scipy.io.arff failed on {file_name} ({e}). "
101
+ "Install liac-arff (`pip install liac-arff`) to handle "
102
+ "sparse ARFF datasets like bibtex."
103
+ ) from e
104
+ log(INFO, f" scipy failed ({e!s}); retrying with liac-arff…")
105
+ df = _load_sparse_arff_to_dataframe(full_path)
106
+
107
+ is_target_in_end = any(
108
+ f.lower() == file_name.lower() for f in TARGET_IN_END_FILE_DATASETS
109
+ )
110
+
111
+ X, Y = self.preprocess_data(df, n_labels, is_target_in_end)
112
+ df_name = file_name.split(".")[0]
113
+ self.datasets.append((X, Y, df_name))
114
+
115
+ def preprocess_data(self, df, n_labels, is_target_in_end=False):
116
+ if is_target_in_end:
117
+ X = df.iloc[:, :-n_labels].to_numpy()
118
+ Y = df.iloc[:, -n_labels:].to_numpy().astype(int)
119
+ else:
120
+ X = df.iloc[:, n_labels:].to_numpy()
121
+ Y = df.iloc[:, :n_labels].to_numpy().astype(int)
122
+
123
+ # liac-arff returns object dtype when attributes are stored as strings
124
+ # in a sparse ARFF. Coerce to float for sklearn/LightGBM compatibility.
125
+ if X.dtype == object:
126
+ X = X.astype(float)
127
+
128
+ # Map sklearn-style -1 (negative) labels to 0 so downstream pairwise
129
+ # encoders see a clean {0,1} matrix.
130
+ Y = np.where(Y < 0, 0, Y)
131
+
132
+ return X, Y
133
+
134
+ def add_noise_to_labels(self, Y, noisy_rate):
135
+ """
136
+ Adds noise to the dataset labels based on the specified noisy rate.
137
+
138
+ :param Y: The label matrix for a dataset.
139
+ :param noisy_rate: The rate at which noise should be added to the labels.
140
+ :return: The label matrix with added noise.
141
+ """
142
+ n_instances, n_labels = Y.shape
143
+ for i in range(n_instances):
144
+ for j in range(n_labels):
145
+ if bernoulli.rvs(p=noisy_rate):
146
+ Y[i, j] = 1 - Y[i, j] # Flip the label to add noise
147
+ return Y
148
+
149
+ def kfold_split_with_noise(
150
+ self, dataset_index, n_splits=5, noisy_rate=0.0, random_state=None, shuffle=True
151
+ ):
152
+ """
153
+ Generates K-fold splits for a specific dataset and adds noise to the training labels.
154
+
155
+ :param dataset_index: Index of the dataset to split.
156
+ :param n_splits: Number of folds.
157
+ :param noisy_rate: Noise rate to be applied to the training set labels.
158
+ :param random_state: Random state for reproducibility.
159
+ :param shuffle: Whether to shuffle the data before splitting.
160
+ :return: Generator of K-fold splits (train_index, test_index) with noisy training labels.
161
+ """
162
+ X, Y, _ = self.datasets[dataset_index]
163
+ kf = KFold(n_splits=n_splits, random_state=random_state, shuffle=shuffle)
164
+ for train_index, test_index in kf.split(X):
165
+ Y_train_noisy = self.add_noise_to_labels(Y[train_index].copy(), noisy_rate)
166
+ # I want to return x_train, y_train_noisy, x_test, y_test
167
+ yield X[train_index], Y_train_noisy, X[test_index], Y[test_index]
168
+
169
+ def get_datasets(self) -> list:
170
+ return self.datasets
171
+
172
+ def get_length(self) -> int:
173
+ return len(self.datasets)
174
+
175
+ def get_dataset_name(self, dataset_index) -> str:
176
+ return self.datasets[dataset_index][2]
@@ -0,0 +1,123 @@
1
+ """Uniform estimator interface over scikit-learn and LightGBM backends.
2
+
3
+ :class:`Estimator` is the single adapter every other module talks to;
4
+ its constructor accepts a :class:`constants.BaseLearnerName` and hides
5
+ the differences between :class:`sklearn.ensemble.RandomForestClassifier`,
6
+ :class:`sklearn.ensemble.ExtraTreesClassifier`,
7
+ :class:`sklearn.ensemble.GradientBoostingClassifier` (XGBoost in our
8
+ nomenclature), and :class:`lightgbm.LGBMClassifier`.
9
+ """
10
+
11
+ import os
12
+ from logging import INFO, basicConfig
13
+
14
+ from lightgbm import LGBMClassifier
15
+ from numpy.typing import NDArray
16
+ from sklearn.base import BaseEstimator
17
+ from sklearn.calibration import CalibratedClassifierCV
18
+ from sklearn.ensemble import (
19
+ ExtraTreesClassifier,
20
+ GradientBoostingClassifier,
21
+ RandomForestClassifier,
22
+ )
23
+
24
+ from preorder4mlc.constants import RANDOM_STATE, BaseLearnerName
25
+
26
+ basicConfig(level=INFO) # type: ignore
27
+
28
+ number_of_cores: int = os.cpu_count() if os.cpu_count() is not None else 1 # type: ignore
29
+ # log(INFO, f"Number of cores: {number_of_cores}")
30
+
31
+ # Whether to wrap the base learner in CalibratedClassifierCV (isotonic).
32
+ # Opt-in: default OFF so existing pipelines / paper-equivalent runs are
33
+ # unchanged. Enable per-run via env (PREORDER_CALIBRATE=1) or constructor
34
+ # (Estimator(name, calibrate=True)). The A/B/C/D ablation in
35
+ # scripts/ablations/ablation_base_learner.py decides whether to flip the default.
36
+ CALIBRATE_PROBAS = os.environ.get("PREORDER_CALIBRATE", "0") not in ("0", "false", "False")
37
+ CALIBRATION_METHOD = os.environ.get("PREORDER_CALIBRATION_METHOD", "isotonic")
38
+ CALIBRATION_CV = int(os.environ.get("PREORDER_CALIBRATION_CV", "3"))
39
+
40
+
41
+ class Estimator:
42
+ def __init__(self, name: str, calibrate: bool | None = None):
43
+ self.name = name
44
+ self.calibrate = CALIBRATE_PROBAS if calibrate is None else calibrate
45
+ base = self.get_classifier()
46
+ if self.calibrate:
47
+ self.clf = CalibratedClassifierCV(
48
+ base, method=CALIBRATION_METHOD, cv=CALIBRATION_CV
49
+ )
50
+ else:
51
+ self.clf = base
52
+
53
+ def get_classifier(self) -> BaseEstimator | LGBMClassifier:
54
+ """Get the classifier based on name with proper error handling."""
55
+ if self.name == BaseLearnerName.RF.value:
56
+ return RandomForestClassifier(random_state=RANDOM_STATE)
57
+ elif self.name == BaseLearnerName.ETC.value:
58
+ return ExtraTreesClassifier(random_state=RANDOM_STATE)
59
+ elif self.name == BaseLearnerName.XGBoost.value:
60
+ return GradientBoostingClassifier(random_state=RANDOM_STATE)
61
+ elif self.name == BaseLearnerName.LightGBM.value:
62
+ return LGBMClassifier(
63
+ random_state=RANDOM_STATE,
64
+ # n_jobs=1 because LGBM is always called inside a
65
+ # joblib.Parallel(n_jobs=-1) loop in base_classifiers.py and the
66
+ # ECC/LP path. Using n_jobs=cores caused thread oversubscription
67
+ # (~5x slower on yeast); see scripts/ablations/NOTES.md.
68
+ n_jobs=1,
69
+ verbose=-1,
70
+ num_leaves=20, # Moderate complexity
71
+ max_depth=6,
72
+ bagging_fraction=0.9,
73
+ feature_fraction=0.8,
74
+ learning_rate=0.1,
75
+ n_estimators=100,
76
+ min_child_samples=5, # Relaxed for small datasets
77
+ min_child_weight=0.0001, # Allow splits with low Hessian
78
+ min_split_gain=0.01, # Allow minimal gain splits
79
+ is_unbalance=True, # Handle label imbalance
80
+ device="cpu",
81
+ )
82
+ else:
83
+ raise ValueError(f"Unknown base learner: {self.name}")
84
+
85
+ def fit(self, X: NDArray, Y: NDArray):
86
+ """Fit the classifier with proper error handling.
87
+
88
+ When calibration is enabled but the data is degenerate (single class
89
+ or too few samples per class to support cv folds), fall back to the
90
+ uncalibrated base estimator so the pairwise pipeline still runs.
91
+ """
92
+ try:
93
+ self.clf.fit(X, Y) # type: ignore
94
+ except Exception as e:
95
+ if self.calibrate:
96
+ fallback = self.get_classifier()
97
+ try:
98
+ fallback.fit(X, Y) # type: ignore
99
+ self.clf = fallback
100
+ return
101
+ except Exception as e2:
102
+ raise ValueError(
103
+ f"Error training {self.name} (calibrated and uncalibrated both failed): {e2}"
104
+ ) from e2
105
+ raise ValueError(f"Error training {self.name}: {e}") from e
106
+
107
+ def predict_proba(self, X: NDArray) -> NDArray:
108
+ """Predict the probability of each class for each instance."""
109
+ prob = self.clf.predict_proba(X) # type: ignore
110
+
111
+ return prob # type: ignore
112
+
113
+ def classes_(self) -> list[int]:
114
+ return self.clf.classes_ # type: ignore
115
+
116
+ def predict(self, X: NDArray) -> NDArray:
117
+ return self.clf.predict(X) # type: ignore
118
+
119
+
120
+ def train_classifier(X, Y, estimator_name):
121
+ classifier = Estimator(estimator_name) # Add n_jobs or other params here
122
+ classifier.fit(X, Y)
123
+ return classifier