binboost 0.1.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {binboost-0.1.0 → binboost-0.2.2}/LICENSE +1 -1
- {binboost-0.1.0 → binboost-0.2.2}/PKG-INFO +2 -5
- binboost-0.2.2/binboost/__init__.py +5 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost/binarizer.py +1 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost/binboost.py +134 -51
- {binboost-0.1.0 → binboost-0.2.2}/binboost/loss.py +23 -7
- binboost-0.2.2/binboost/multiclass.py +78 -0
- binboost-0.2.2/binboost/rule.py +217 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost.egg-info/PKG-INFO +2 -5
- {binboost-0.1.0 → binboost-0.2.2}/binboost.egg-info/SOURCES.txt +1 -0
- {binboost-0.1.0 → binboost-0.2.2}/pyproject.toml +1 -4
- {binboost-0.1.0 → binboost-0.2.2}/setup.py +2 -3
- binboost-0.1.0/binboost/__init__.py +0 -5
- binboost-0.1.0/binboost/rule.py +0 -168
- {binboost-0.1.0 → binboost-0.2.2}/README.md +0 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost.egg-info/dependency_links.txt +0 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost.egg-info/requires.txt +0 -0
- {binboost-0.1.0 → binboost-0.2.2}/binboost.egg-info/top_level.txt +0 -0
- {binboost-0.1.0 → binboost-0.2.2}/setup.cfg +0 -0
|
@@ -1,11 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: binboost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
|
|
5
|
-
|
|
6
|
-
Author: BinBoost Authors
|
|
5
|
+
Author: Rangga Wahyu Pratama
|
|
7
6
|
License: MIT
|
|
8
|
-
Project-URL: Homepage, https://github.com/username/binboost
|
|
9
7
|
Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
|
|
10
8
|
Classifier: Programming Language :: Python :: 3
|
|
11
9
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -18,7 +16,6 @@ License-File: LICENSE
|
|
|
18
16
|
Requires-Dist: numpy>=1.21.0
|
|
19
17
|
Requires-Dist: pandas>=1.3.0
|
|
20
18
|
Dynamic: author
|
|
21
|
-
Dynamic: home-page
|
|
22
19
|
Dynamic: license-file
|
|
23
20
|
Dynamic: requires-python
|
|
24
21
|
|
|
@@ -9,7 +9,7 @@ class BinBoost:
|
|
|
9
9
|
"""
|
|
10
10
|
BinBoost: Gradient Boosting Berbasis Aturan Logika Adaptif.
|
|
11
11
|
|
|
12
|
-
Algoritma klasifikasi biner yang membangun ensemble aturan logika
|
|
12
|
+
Algoritma klasifikasi biner yang membangun ensemble aturan logika
|
|
13
13
|
murni (AND, OR, XOR) dengan binarisasi fitur numerik adaptif
|
|
14
14
|
berbasis gradien pada setiap iterasi boosting.
|
|
15
15
|
|
|
@@ -18,7 +18,7 @@ class BinBoost:
|
|
|
18
18
|
n_estimators : int, default=100
|
|
19
19
|
Jumlah iterasi boosting.
|
|
20
20
|
|
|
21
|
-
learning_rate : float, default=0.
|
|
21
|
+
learning_rate : float, default=0.2
|
|
22
22
|
Faktor penyusutan kontribusi setiap aturan.
|
|
23
23
|
|
|
24
24
|
loss : str, default='logistic'
|
|
@@ -33,7 +33,7 @@ class BinBoost:
|
|
|
33
33
|
focal_alpha : float, default=0.25
|
|
34
34
|
Parameter alpha pada Focal Loss. Hanya berlaku saat loss='focal'.
|
|
35
35
|
|
|
36
|
-
max_rule_length : int, default=
|
|
36
|
+
max_rule_length : int, default=4
|
|
37
37
|
Jumlah maksimum fitur dalam satu aturan.
|
|
38
38
|
|
|
39
39
|
operators : list, default=['AND', 'OR']
|
|
@@ -45,8 +45,10 @@ class BinBoost:
|
|
|
45
45
|
rule_complexity_penalty : float, default=0.0
|
|
46
46
|
Penalti bobot proporsional dengan panjang aturan.
|
|
47
47
|
|
|
48
|
-
feature_selection_threshold : float, default=0.
|
|
49
|
-
Ambang batas
|
|
48
|
+
feature_selection_threshold : float, default=0.0
|
|
49
|
+
Ambang batas persentil gain fitur untuk seleksi kandidat per iterasi.
|
|
50
|
+
Nilai 0.0 berarti semua fitur diikutsertakan. Nilai 0.5 berarti hanya
|
|
51
|
+
fitur dengan gain di atas median yang masuk kandidat.
|
|
50
52
|
|
|
51
53
|
beam_width : int, default=5
|
|
52
54
|
Jumlah kandidat aturan terbaik yang dipertahankan per langkah beam search.
|
|
@@ -54,9 +56,12 @@ class BinBoost:
|
|
|
54
56
|
subsample : float, default=0.8
|
|
55
57
|
Fraksi data yang digunakan per iterasi boosting.
|
|
56
58
|
|
|
57
|
-
min_samples_rule : int or float, default=
|
|
59
|
+
min_samples_rule : int or float, default=5
|
|
58
60
|
Jumlah minimum sampel yang harus memenuhi sebuah aturan.
|
|
59
61
|
|
|
62
|
+
lambda0 : float, default=3.0
|
|
63
|
+
Konstanta regularisasi dasar untuk bobot Newton adaptif (λ_R = lambda0 · L / sqrt(n_R+1)).
|
|
64
|
+
|
|
60
65
|
binarize_strategy : str, default='gradient'
|
|
61
66
|
Strategi binarisasi fitur numerik. Pilihan: 'gradient', 'quantile',
|
|
62
67
|
'uniform', 'kmeans'.
|
|
@@ -76,13 +81,18 @@ class BinBoost:
|
|
|
76
81
|
warm_start : bool, default=False
|
|
77
82
|
Jika True maka pelatihan dilanjutkan dari kondisi model sebelumnya.
|
|
78
83
|
|
|
84
|
+
threshold : float or 'auto', default='auto'
|
|
85
|
+
Threshold prediksi kelas. Nilai 'auto' menggunakan indeks Youden
|
|
86
|
+
untuk mencari threshold optimal pada data training.
|
|
87
|
+
|
|
79
88
|
Atribut yang tersedia Setelah fit
|
|
80
89
|
--------------------------------
|
|
81
90
|
estimators_ : list of Rule
|
|
82
91
|
Daftar aturan yang dipelajari.
|
|
83
92
|
|
|
84
|
-
rule_weights_ : ndarray of float
|
|
85
|
-
Bobot
|
|
93
|
+
rule_weights_ : ndarray of float shape (n_estimators_, 2)
|
|
94
|
+
Bobot dua sisi setiap aturan. Kolom 0 adalah w0 (tidak memenuhi aturan)
|
|
95
|
+
dan kolom 1 adalah w1 (memenuhi aturan).
|
|
86
96
|
|
|
87
97
|
rules_ : list of str
|
|
88
98
|
Representasi teks setiap aturan.
|
|
@@ -112,25 +122,27 @@ class BinBoost:
|
|
|
112
122
|
def __init__(
|
|
113
123
|
self,
|
|
114
124
|
n_estimators=100,
|
|
115
|
-
learning_rate=0.
|
|
125
|
+
learning_rate=0.2,
|
|
116
126
|
loss='logistic',
|
|
117
127
|
poly_epsilon=1.0,
|
|
118
128
|
focal_gamma=2.0,
|
|
119
129
|
focal_alpha=0.25,
|
|
120
|
-
max_rule_length=
|
|
130
|
+
max_rule_length=4,
|
|
121
131
|
operators=None,
|
|
122
132
|
use_xor=False,
|
|
123
133
|
rule_complexity_penalty=0.0,
|
|
124
|
-
feature_selection_threshold=0.
|
|
134
|
+
feature_selection_threshold=0.0,
|
|
125
135
|
beam_width=5,
|
|
126
136
|
subsample=0.8,
|
|
127
|
-
min_samples_rule=
|
|
137
|
+
min_samples_rule=5,
|
|
138
|
+
lambda0=3.0,
|
|
128
139
|
binarize_strategy='gradient',
|
|
129
140
|
n_thresholds='auto',
|
|
130
141
|
n_iter_no_change=None,
|
|
131
142
|
tol=1e-4,
|
|
132
143
|
random_state=None,
|
|
133
144
|
warm_start=False,
|
|
145
|
+
threshold='auto',
|
|
134
146
|
):
|
|
135
147
|
self.n_estimators = n_estimators
|
|
136
148
|
self.learning_rate = learning_rate
|
|
@@ -146,12 +158,15 @@ class BinBoost:
|
|
|
146
158
|
self.beam_width = beam_width
|
|
147
159
|
self.subsample = subsample
|
|
148
160
|
self.min_samples_rule = min_samples_rule
|
|
161
|
+
self.lambda0 = lambda0
|
|
149
162
|
self.binarize_strategy = binarize_strategy
|
|
150
163
|
self.n_thresholds = n_thresholds
|
|
151
164
|
self.n_iter_no_change = n_iter_no_change
|
|
152
165
|
self.tol = tol
|
|
153
166
|
self.random_state = random_state
|
|
154
167
|
self.warm_start = warm_start
|
|
168
|
+
self.threshold = threshold
|
|
169
|
+
self._threshold_fitted = 0.5
|
|
155
170
|
|
|
156
171
|
def _validate_input(self, X, y=None):
|
|
157
172
|
# Ubah input ke numpy array dan pastikan dimensinya benar
|
|
@@ -195,10 +210,11 @@ class BinBoost:
|
|
|
195
210
|
continue
|
|
196
211
|
# a. Dua fitur bersifat mutually exclusive jika tidak pernah bernilai 1 bersamaan
|
|
197
212
|
both_one = np.sum((X[:, i] == 1) & (X[:, j] == 1))
|
|
213
|
+
# b. Batasi ukuran grup maksimum 20 agar tidak memblokir terlalu banyak fitur
|
|
198
214
|
if both_one == 0:
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
215
|
+
if len(group) < 20 and \
|
|
216
|
+
np.all(np.isin(np.unique(X[:, i]), [0.0, 1.0])) and \
|
|
217
|
+
np.all(np.isin(np.unique(X[:, j]), [0.0, 1.0])):
|
|
202
218
|
group.append(j)
|
|
203
219
|
used.add(j)
|
|
204
220
|
if len(group) > 1:
|
|
@@ -221,6 +237,40 @@ class BinBoost:
|
|
|
221
237
|
ops.append('XOR')
|
|
222
238
|
return ops
|
|
223
239
|
|
|
240
|
+
def _seleksi_fitur(self, X_bin, gradients, hessians):
|
|
241
|
+
dot_gf = X_bin.T @ gradients
|
|
242
|
+
H_f = X_bin.T @ hessians
|
|
243
|
+
n_f = X_bin.sum(axis=0)
|
|
244
|
+
lambda_f = self.lambda0 / np.sqrt(n_f + 1.0)
|
|
245
|
+
gains = (dot_gf ** 2) / (H_f + lambda_f + 1e-10)
|
|
246
|
+
|
|
247
|
+
if self.feature_selection_threshold <= 0.0:
|
|
248
|
+
return np.arange(X_bin.shape[1])
|
|
249
|
+
|
|
250
|
+
nilai_persentil = np.percentile(gains, self.feature_selection_threshold * 100)
|
|
251
|
+
selected = np.where(gains >= nilai_persentil)[0]
|
|
252
|
+
if len(selected) == 0:
|
|
253
|
+
selected = np.array([np.argmax(gains)])
|
|
254
|
+
return selected
|
|
255
|
+
|
|
256
|
+
def _hitung_bobot_dua_sisi(self, h, gradients, hessians, rule_length):
|
|
257
|
+
mask_1 = h.astype(bool)
|
|
258
|
+
mask_0 = ~mask_1
|
|
259
|
+
|
|
260
|
+
sum_g1 = gradients[mask_1].sum() if mask_1.any() else 0.0
|
|
261
|
+
sum_H1 = hessians[mask_1].sum() if mask_1.any() else 0.0
|
|
262
|
+
n1 = float(mask_1.sum())
|
|
263
|
+
lambda_1 = self.lambda0 * rule_length / np.sqrt(n1 + 1.0)
|
|
264
|
+
w1 = sum_g1 / (sum_H1 + lambda_1 + 1e-10) if n1 > 0 else 0.0
|
|
265
|
+
|
|
266
|
+
sum_g0 = gradients[mask_0].sum() if mask_0.any() else 0.0
|
|
267
|
+
sum_H0 = hessians[mask_0].sum() if mask_0.any() else 0.0
|
|
268
|
+
n0 = float(mask_0.sum())
|
|
269
|
+
lambda_0 = self.lambda0 * rule_length / np.sqrt(n0 + 1.0)
|
|
270
|
+
w0 = sum_g0 / (sum_H0 + lambda_0 + 1e-10) if n0 > 0 else 0.0
|
|
271
|
+
|
|
272
|
+
return w1, w0
|
|
273
|
+
|
|
224
274
|
def fit(self, X, y, sample_weight=None):
|
|
225
275
|
"""
|
|
226
276
|
Latih BinBoost pada data X dan label y.
|
|
@@ -258,16 +308,17 @@ class BinBoost:
|
|
|
258
308
|
numeric_cols = self._detect_numeric_cols(X)
|
|
259
309
|
self.ohe_groups_ = self._detect_ohe_groups(X)
|
|
260
310
|
|
|
261
|
-
|
|
311
|
+
self._binarizer = GradientBinarizer(
|
|
262
312
|
strategy=self.binarize_strategy,
|
|
263
313
|
n_thresholds=self.n_thresholds
|
|
264
314
|
)
|
|
315
|
+
binarizer = self._binarizer
|
|
265
316
|
|
|
266
317
|
rule_finder = BeamSearchRuleFinder(
|
|
267
318
|
operators=ops,
|
|
268
319
|
max_rule_length=self.max_rule_length,
|
|
269
320
|
beam_width=self.beam_width,
|
|
270
|
-
feature_selection_threshold=
|
|
321
|
+
feature_selection_threshold=0.0,
|
|
271
322
|
min_samples_rule=self.min_samples_rule,
|
|
272
323
|
rule_complexity_penalty=self.rule_complexity_penalty,
|
|
273
324
|
ohe_groups=self.ohe_groups_,
|
|
@@ -286,16 +337,22 @@ class BinBoost:
|
|
|
286
337
|
for m in range(self.n_estimators):
|
|
287
338
|
# 1. Hitung gradien dari fungsi loss yang dipilih
|
|
288
339
|
gradients = loss_fn.gradient(y, self._F)
|
|
340
|
+
hessians = loss_fn.hessian(y, self._F)
|
|
289
341
|
|
|
290
342
|
# 2. Ambil subsample data untuk iterasi ini
|
|
291
343
|
idx = self._subsample_indices(n_samples, rng)
|
|
292
344
|
X_sub = X[idx]
|
|
293
345
|
g_sub = gradients[idx]
|
|
346
|
+
h_sub = hessians[idx]
|
|
294
347
|
|
|
295
348
|
# 3. Binarisasi fitur numerik menggunakan strategi yang dipilih
|
|
296
349
|
X_bin, thresholds_iter = binarizer.transform(X_sub, g_sub, numeric_cols)
|
|
297
350
|
|
|
298
|
-
# 4.
|
|
351
|
+
# 4. Seleksi fitur kandidat menggunakan kriteria gain yang konsisten
|
|
352
|
+
candidate_features = self._seleksi_fitur(X_bin, g_sub, h_sub)
|
|
353
|
+
rule_finder.forced_candidate_features = candidate_features
|
|
354
|
+
|
|
355
|
+
# 5. Cari aturan terbaik menggunakan beam search
|
|
299
356
|
rule = rule_finder.find_best_rule(X_bin, g_sub)
|
|
300
357
|
if rule is None:
|
|
301
358
|
break
|
|
@@ -304,21 +361,24 @@ class BinBoost:
|
|
|
304
361
|
list(self.feature_names_in_) if self.feature_names_in_ is not None else None
|
|
305
362
|
)
|
|
306
363
|
|
|
307
|
-
#
|
|
364
|
+
# 6. Perbarui prediksi model untuk seluruh data menggunakan update dua sisi
|
|
308
365
|
X_bin_full, thresholds_full = binarizer.transform(X, gradients, numeric_cols)
|
|
309
366
|
h_full = rule.evaluate(X_bin_full)
|
|
310
|
-
|
|
367
|
+
w1, w0 = self._hitung_bobot_dua_sisi(h_full, gradients, hessians, len(rule.features))
|
|
368
|
+
|
|
369
|
+
# 7. Update F menggunakan kontribusi dua sisi: w1 untuk yang memenuhi, w0 untuk yang tidak
|
|
370
|
+
self._F = self._F + self.learning_rate * (w1 * h_full + w0 * (1.0 - h_full))
|
|
311
371
|
|
|
312
|
-
#
|
|
372
|
+
# 8. Simpan aturan dengan bobot dua sisi dan statistik iterasi ini
|
|
313
373
|
self.estimators_.append(rule)
|
|
314
|
-
self.rule_weights_.append(
|
|
374
|
+
self.rule_weights_.append(np.array([w0, w1]))
|
|
315
375
|
for col, t in thresholds_full.items():
|
|
316
376
|
self.thresholds_[col].append(t)
|
|
317
377
|
|
|
318
378
|
current_score = loss_fn.loss(y, self._F)
|
|
319
379
|
self.train_score_.append(current_score)
|
|
320
380
|
|
|
321
|
-
#
|
|
381
|
+
# 9. Periksa kondisi early stopping jika diaktifkan
|
|
322
382
|
if self.n_iter_no_change is not None:
|
|
323
383
|
if current_score < best_score - self.tol:
|
|
324
384
|
best_score = current_score
|
|
@@ -334,13 +394,19 @@ class BinBoost:
|
|
|
334
394
|
self.rules_ = [r.to_string(self.feature_names_in_) for r in self.estimators_]
|
|
335
395
|
self._compute_feature_importances(n_features)
|
|
336
396
|
self.is_fitted_ = True
|
|
397
|
+
# Tentukan threshold prediksi optimal setelah seluruh iterasi selesai
|
|
398
|
+
if self.threshold == 'auto':
|
|
399
|
+
self._threshold_fitted = self._cari_threshold_optimal(X, y)
|
|
400
|
+
else:
|
|
401
|
+
self._threshold_fitted = float(self.threshold)
|
|
337
402
|
return self
|
|
338
403
|
|
|
339
404
|
def _compute_feature_importances(self, n_features):
|
|
340
405
|
# Hitung skor kepentingan fitur dari frekuensi kemunculan dikali rata-rata bobot absolut
|
|
341
406
|
importances = np.zeros(n_features)
|
|
342
|
-
for rule in self.estimators_:
|
|
343
|
-
gain
|
|
407
|
+
for rule, weights in zip(self.estimators_, self.rule_weights_):
|
|
408
|
+
# Gunakan selisih absolut w1 dan w0 sebagai ukuran gain aturan
|
|
409
|
+
gain = np.abs(weights[1] - weights[0])
|
|
344
410
|
for f in rule.features:
|
|
345
411
|
importances[f] += gain
|
|
346
412
|
total = importances.sum()
|
|
@@ -350,21 +416,41 @@ class BinBoost:
|
|
|
350
416
|
if not getattr(self, 'is_fitted_', False):
|
|
351
417
|
raise RuntimeError("Model belum dilatih. Panggil fit terlebih dahulu.")
|
|
352
418
|
|
|
419
|
+
def _cari_threshold_optimal(self, X, y):
|
|
420
|
+
# Cari threshold prediksi optimal menggunakan indeks Youden pada data training
|
|
421
|
+
proba = self.predict_proba(X)[:, 1]
|
|
422
|
+
ambang_kandidat = np.sort(np.unique(proba))
|
|
423
|
+
skor_terbaik = -np.inf
|
|
424
|
+
threshold_terbaik = 0.5
|
|
425
|
+
pos = np.sum(y == 1)
|
|
426
|
+
neg = np.sum(y == 0)
|
|
427
|
+
if pos == 0 or neg == 0:
|
|
428
|
+
return 0.5
|
|
429
|
+
for t in ambang_kandidat:
|
|
430
|
+
pred = (proba >= t).astype(int)
|
|
431
|
+
tp = np.sum((pred == 1) & (y == 1))
|
|
432
|
+
tn = np.sum((pred == 0) & (y == 0))
|
|
433
|
+
sensitivity = tp / pos
|
|
434
|
+
specificity = tn / neg
|
|
435
|
+
# Indeks Youden memaksimalkan sensitivity tambah specificity dikurangi 1
|
|
436
|
+
youden = sensitivity + specificity - 1
|
|
437
|
+
if youden > skor_terbaik:
|
|
438
|
+
skor_terbaik = youden
|
|
439
|
+
threshold_terbaik = t
|
|
440
|
+
return float(threshold_terbaik)
|
|
441
|
+
|
|
353
442
|
def _decision_function(self, X):
|
|
354
|
-
# Hitung nilai F akhir untuk seluruh sampel
|
|
443
|
+
# Hitung nilai F akhir untuk seluruh sampel menggunakan update dua sisi
|
|
355
444
|
self._check_is_fitted()
|
|
356
445
|
X, _ = self._validate_input(X)
|
|
357
446
|
F = np.zeros(X.shape[0])
|
|
358
|
-
binarizer = GradientBinarizer(
|
|
359
|
-
strategy=self.binarize_strategy,
|
|
360
|
-
n_thresholds=self.n_thresholds
|
|
361
|
-
)
|
|
362
447
|
numeric_cols = self._detect_numeric_cols(X)
|
|
363
448
|
dummy_grad = np.ones(X.shape[0])
|
|
364
|
-
X_bin, _ =
|
|
365
|
-
for rule,
|
|
449
|
+
X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
|
|
450
|
+
for rule, weights in zip(self.estimators_, self.rule_weights_):
|
|
366
451
|
h = rule.evaluate(X_bin)
|
|
367
|
-
|
|
452
|
+
w0, w1 = weights[0], weights[1]
|
|
453
|
+
F += self.learning_rate * (w1 * h + w0 * (1.0 - h))
|
|
368
454
|
return F
|
|
369
455
|
|
|
370
456
|
def predict_proba(self, X):
|
|
@@ -398,7 +484,7 @@ class BinBoost:
|
|
|
398
484
|
ndarray of shape (n_samples,)
|
|
399
485
|
"""
|
|
400
486
|
proba = self.predict_proba(X)
|
|
401
|
-
return (proba[:, 1] >=
|
|
487
|
+
return (proba[:, 1] >= self._threshold_fitted).astype(int)
|
|
402
488
|
|
|
403
489
|
def score(self, X, y):
|
|
404
490
|
"""
|
|
@@ -422,17 +508,14 @@ class BinBoost:
|
|
|
422
508
|
"""
|
|
423
509
|
self._check_is_fitted()
|
|
424
510
|
X, _ = self._validate_input(X)
|
|
425
|
-
binarizer = GradientBinarizer(
|
|
426
|
-
strategy=self.binarize_strategy,
|
|
427
|
-
n_thresholds=self.n_thresholds
|
|
428
|
-
)
|
|
429
511
|
numeric_cols = self._detect_numeric_cols(X)
|
|
430
512
|
dummy_grad = np.ones(X.shape[0])
|
|
431
|
-
X_bin, _ =
|
|
513
|
+
X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
|
|
432
514
|
F = np.zeros(X.shape[0])
|
|
433
|
-
for rule,
|
|
515
|
+
for rule, weights in zip(self.estimators_, self.rule_weights_):
|
|
434
516
|
h = rule.evaluate(X_bin)
|
|
435
|
-
|
|
517
|
+
w0, w1 = weights[0], weights[1]
|
|
518
|
+
F += self.learning_rate * (w1 * h + w0 * (1.0 - h))
|
|
436
519
|
p = sigmoid(F)
|
|
437
520
|
yield np.column_stack([1 - p, p])
|
|
438
521
|
|
|
@@ -445,7 +528,7 @@ class BinBoost:
|
|
|
445
528
|
ndarray of shape (n_samples,) per iterasi
|
|
446
529
|
"""
|
|
447
530
|
for proba in self.staged_predict_proba(X):
|
|
448
|
-
yield (proba[:, 1] >=
|
|
531
|
+
yield (proba[:, 1] >= self._threshold_fitted).astype(int)
|
|
449
532
|
|
|
450
533
|
def apply(self, X):
|
|
451
534
|
"""
|
|
@@ -457,13 +540,9 @@ class BinBoost:
|
|
|
457
540
|
"""
|
|
458
541
|
self._check_is_fitted()
|
|
459
542
|
X, _ = self._validate_input(X)
|
|
460
|
-
binarizer = GradientBinarizer(
|
|
461
|
-
strategy=self.binarize_strategy,
|
|
462
|
-
n_thresholds=self.n_thresholds
|
|
463
|
-
)
|
|
464
543
|
numeric_cols = self._detect_numeric_cols(X)
|
|
465
544
|
dummy_grad = np.ones(X.shape[0])
|
|
466
|
-
X_bin, _ =
|
|
545
|
+
X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
|
|
467
546
|
results = []
|
|
468
547
|
for rule in self.estimators_:
|
|
469
548
|
results.append(rule.evaluate(X_bin))
|
|
@@ -492,12 +571,14 @@ class BinBoost:
|
|
|
492
571
|
'beam_width': self.beam_width,
|
|
493
572
|
'subsample': self.subsample,
|
|
494
573
|
'min_samples_rule': self.min_samples_rule,
|
|
574
|
+
'lambda0': self.lambda0,
|
|
495
575
|
'binarize_strategy': self.binarize_strategy,
|
|
496
576
|
'n_thresholds': self.n_thresholds,
|
|
497
577
|
'n_iter_no_change': self.n_iter_no_change,
|
|
498
578
|
'tol': self.tol,
|
|
499
579
|
'random_state': self.random_state,
|
|
500
580
|
'warm_start': self.warm_start,
|
|
581
|
+
'threshold': self.threshold,
|
|
501
582
|
}
|
|
502
583
|
|
|
503
584
|
def set_params(self, **params):
|
|
@@ -517,7 +598,7 @@ class BinBoost:
|
|
|
517
598
|
@property
|
|
518
599
|
def rule_summary_(self):
|
|
519
600
|
"""
|
|
520
|
-
Tabel ringkasan semua aturan beserta bobot
|
|
601
|
+
Tabel ringkasan semua aturan beserta bobot dua sisi setiap aturan.
|
|
521
602
|
|
|
522
603
|
Kembalian
|
|
523
604
|
----------
|
|
@@ -525,12 +606,14 @@ class BinBoost:
|
|
|
525
606
|
"""
|
|
526
607
|
self._check_is_fitted()
|
|
527
608
|
summary = []
|
|
528
|
-
for i, rule in enumerate(self.estimators_):
|
|
609
|
+
for i, (rule, weights) in enumerate(zip(self.estimators_, self.rule_weights_)):
|
|
529
610
|
summary.append({
|
|
530
611
|
'iterasi': i + 1,
|
|
531
612
|
'aturan': self.rules_[i],
|
|
532
|
-
'
|
|
613
|
+
'bobot_memenuhi': round(float(weights[1]), 6),
|
|
614
|
+
'bobot_tidak_memenuhi': round(float(weights[0]), 6),
|
|
533
615
|
'panjang_aturan': len(rule.features),
|
|
616
|
+
'threshold_prediksi': round(self._threshold_fitted, 6),
|
|
534
617
|
})
|
|
535
618
|
return summary
|
|
536
619
|
|
|
@@ -565,4 +648,4 @@ class BinBoost:
|
|
|
565
648
|
else f"X{f}"
|
|
566
649
|
)
|
|
567
650
|
usage[name] = usage.get(name, 0) + 1
|
|
568
|
-
return usage
|
|
651
|
+
return usage
|
|
@@ -1,14 +1,12 @@
|
|
|
1
|
+
# loss.py
|
|
1
2
|
import numpy as np
|
|
2
3
|
|
|
3
|
-
|
|
4
4
|
def sigmoid(x):
|
|
5
5
|
# Fungsi sigmoid dengan penjagaan numerik agar tidak overflow
|
|
6
6
|
return np.where(x >= 0, 1 / (1 + np.exp(-x)), np.exp(x) / (1 + np.exp(x)))
|
|
7
7
|
|
|
8
|
-
|
|
9
8
|
class LogisticLoss:
|
|
10
9
|
# Loss biner standar berbasis log-likelihood
|
|
11
|
-
|
|
12
10
|
def loss(self, y, F):
|
|
13
11
|
p = sigmoid(F)
|
|
14
12
|
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
@@ -18,16 +16,21 @@ class LogisticLoss:
|
|
|
18
16
|
# 1. Gradien negatif sebagai arah penurunan loss
|
|
19
17
|
p = sigmoid(F)
|
|
20
18
|
return y - p
|
|
19
|
+
|
|
20
|
+
def hessian(self, y, F):
|
|
21
|
+
# Turunan kedua logistic loss terhadap F
|
|
22
|
+
p = sigmoid(F)
|
|
23
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
24
|
+
h = p * (1 - p)
|
|
25
|
+
return np.maximum(h, 1e-10)
|
|
21
26
|
|
|
22
27
|
def init_F(self, y):
|
|
23
28
|
# Inisialisasi F0 dari proporsi kelas positif
|
|
24
29
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
25
30
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
26
31
|
|
|
27
|
-
|
|
28
32
|
class FocalLoss:
|
|
29
33
|
# Focal Loss untuk data tidak seimbang
|
|
30
|
-
|
|
31
34
|
def __init__(self, gamma=2.0, alpha=0.25):
|
|
32
35
|
self.gamma = gamma
|
|
33
36
|
self.alpha = alpha
|
|
@@ -50,12 +53,20 @@ class FocalLoss:
|
|
|
50
53
|
weight = alpha_t * (1 - pt) ** self.gamma
|
|
51
54
|
grad = weight * (y - p) + self.gamma * weight * pt * np.log(pt) * np.where(y == 1, -(1 - p), p)
|
|
52
55
|
return grad
|
|
56
|
+
|
|
57
|
+
def hessian(self, y, F):
|
|
58
|
+
p = sigmoid(F)
|
|
59
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
60
|
+
pt = np.where(y == 1, p, 1 - p)
|
|
61
|
+
alpha_t = np.where(y == 1, self.alpha, 1 - self.alpha)
|
|
62
|
+
beta = alpha_t * (1 - pt) ** self.gamma
|
|
63
|
+
h = beta * p * (1 - p)
|
|
64
|
+
return np.maximum(h, 1e-10)
|
|
53
65
|
|
|
54
66
|
def init_F(self, y):
|
|
55
67
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
56
68
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
57
69
|
|
|
58
|
-
|
|
59
70
|
class PolyLoss:
|
|
60
71
|
# PolyLoss yang berbasis ekspansi polinomial
|
|
61
72
|
def __init__(self, epsilon=1.0):
|
|
@@ -78,12 +89,17 @@ class PolyLoss:
|
|
|
78
89
|
# b. Suku koreksi dari ekspansi polinomial pertama
|
|
79
90
|
grad_poly = self.epsilon * np.where(y == 1, p * (1 - p), -p * (1 - p))
|
|
80
91
|
return grad_ce + grad_poly
|
|
92
|
+
|
|
93
|
+
def hessian(self, y, F):
|
|
94
|
+
p = sigmoid(F)
|
|
95
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
96
|
+
h = p * (1 - p) * (1 + self.epsilon * (2 * y - 1) * (2 * p - 1))
|
|
97
|
+
return np.maximum(h, 1e-10)
|
|
81
98
|
|
|
82
99
|
def init_F(self, y):
|
|
83
100
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
84
101
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
85
102
|
|
|
86
|
-
|
|
87
103
|
def get_loss(name, **kwargs):
|
|
88
104
|
# Kembalikan objek loss sesuai nama yang dipilih pengguna
|
|
89
105
|
if name == 'logistic':
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# multiclass.py
|
|
2
|
+
import numpy as np
|
|
3
|
+
from binboost import BinBoost
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class BinBoostOvR:
|
|
7
|
+
def __init__(self, mode='multiclass', **binboost_params):
|
|
8
|
+
self.mode = mode
|
|
9
|
+
self.binboost_params = binboost_params
|
|
10
|
+
|
|
11
|
+
def fit(self, X, y):
|
|
12
|
+
X = np.asarray(X, dtype=np.float64)
|
|
13
|
+
y = np.asarray(y)
|
|
14
|
+
self.classes_ = np.unique(y)
|
|
15
|
+
self.n_classes_ = len(self.classes_)
|
|
16
|
+
if self.n_classes_ < 2:
|
|
17
|
+
raise ValueError("Minimal harus ada 2 kelas berbeda pada label y.")
|
|
18
|
+
|
|
19
|
+
self.estimators_ovr_ = []
|
|
20
|
+
for kelas in self.classes_:
|
|
21
|
+
y_bin = (y == kelas).astype(np.float64)
|
|
22
|
+
model_k = BinBoost(**self.binboost_params)
|
|
23
|
+
model_k.fit(X, y_bin)
|
|
24
|
+
self.estimators_ovr_.append(model_k)
|
|
25
|
+
return self
|
|
26
|
+
|
|
27
|
+
def _raw_scores(self, X):
|
|
28
|
+
return np.column_stack([m.predict_proba(X)[:, 1] for m in self.estimators_ovr_])
|
|
29
|
+
|
|
30
|
+
def predict_proba(self, X):
|
|
31
|
+
raw = self._raw_scores(X)
|
|
32
|
+
if self.mode == 'multilabel':
|
|
33
|
+
# Tidak dinormalisasi — tiap kelas independen
|
|
34
|
+
return raw
|
|
35
|
+
# mode='multiclass' — normalisasi supaya tiap baris berjumlah 1
|
|
36
|
+
total = raw.sum(axis=1, keepdims=True)
|
|
37
|
+
semua_nol = (total.flatten() <= 1e-10)
|
|
38
|
+
total_aman = np.where(total <= 1e-10, 1.0, total)
|
|
39
|
+
probs = raw / total_aman
|
|
40
|
+
if semua_nol.any():
|
|
41
|
+
probs[semua_nol] = 1.0 / self.n_classes_
|
|
42
|
+
return probs
|
|
43
|
+
|
|
44
|
+
def predict(self, X):
|
|
45
|
+
if self.mode == 'multilabel':
|
|
46
|
+
# Tiap kelas pakai threshold Youden J miliknya sendiri
|
|
47
|
+
preds = np.column_stack([m.predict(X) for m in self.estimators_ovr_])
|
|
48
|
+
return preds
|
|
49
|
+
probs = self.predict_proba(X)
|
|
50
|
+
idx = np.argmax(probs, axis=1)
|
|
51
|
+
return self.classes_[idx]
|
|
52
|
+
|
|
53
|
+
def score(self, X, y):
|
|
54
|
+
y = np.asarray(y)
|
|
55
|
+
preds = self.predict(X)
|
|
56
|
+
return np.mean(preds == y)
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def rule_summary_(self):
|
|
60
|
+
rows = []
|
|
61
|
+
for kelas, model_k in zip(self.classes_, self.estimators_ovr_):
|
|
62
|
+
for r in model_k.rule_summary_:
|
|
63
|
+
r = dict(r)
|
|
64
|
+
r['kelas'] = kelas
|
|
65
|
+
rows.append(r)
|
|
66
|
+
return rows
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def feature_usage_(self):
|
|
70
|
+
usage_total = {}
|
|
71
|
+
for model_k in self.estimators_ovr_:
|
|
72
|
+
for k, v in model_k.feature_usage_.items():
|
|
73
|
+
usage_total[k] = usage_total.get(k, 0) + v
|
|
74
|
+
return usage_total
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def n_rules_(self):
|
|
78
|
+
return sum(m.n_rules_ for m in self.estimators_ovr_)
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
# rule.py
|
|
2
|
+
import numpy as np
|
|
3
|
+
|
|
4
|
+
class Rule:
|
|
5
|
+
# Satu aturan logika dalam ensemble BinBoost, dengan dukungan negasi (NOT) per fitur
|
|
6
|
+
def __init__(self, features, operators, weight, negations=None, feature_names=None):
|
|
7
|
+
self.features = features
|
|
8
|
+
self.operators = operators
|
|
9
|
+
self.weight = weight
|
|
10
|
+
# Daftar boolean sepanjang features: True berarti fitur itu dipakai dalam bentuk negasi (NOT)
|
|
11
|
+
self.negations = negations if negations is not None else [False] * len(features)
|
|
12
|
+
self.feature_names = feature_names
|
|
13
|
+
|
|
14
|
+
def _kolom(self, X_bin, idx_dalam_features):
|
|
15
|
+
fitur_idx = self.features[idx_dalam_features]
|
|
16
|
+
kolom = X_bin[:, fitur_idx].astype(bool)
|
|
17
|
+
if self.negations[idx_dalam_features]:
|
|
18
|
+
kolom = ~kolom
|
|
19
|
+
return kolom
|
|
20
|
+
|
|
21
|
+
def evaluate(self, X_bin):
|
|
22
|
+
result = self._kolom(X_bin, 0)
|
|
23
|
+
for i, op in enumerate(self.operators):
|
|
24
|
+
next_col = self._kolom(X_bin, i + 1)
|
|
25
|
+
if op == 'AND':
|
|
26
|
+
result = result & next_col
|
|
27
|
+
elif op == 'OR':
|
|
28
|
+
result = result | next_col
|
|
29
|
+
elif op == 'XOR':
|
|
30
|
+
result = result ^ next_col
|
|
31
|
+
return result.astype(np.float64)
|
|
32
|
+
|
|
33
|
+
def to_string(self, feature_names=None):
|
|
34
|
+
names = feature_names if feature_names is not None else self.feature_names
|
|
35
|
+
if names is None:
|
|
36
|
+
base_parts = [f"X{f}" for f in self.features]
|
|
37
|
+
else:
|
|
38
|
+
base_parts = [str(names[f]) for f in self.features]
|
|
39
|
+
parts = [f"NOT {p}" if neg else p for p, neg in zip(base_parts, self.negations)]
|
|
40
|
+
if len(parts) == 1:
|
|
41
|
+
return parts[0]
|
|
42
|
+
expr = parts[0]
|
|
43
|
+
for i, op in enumerate(self.operators):
|
|
44
|
+
expr = f"({expr} {op} {parts[i + 1]})"
|
|
45
|
+
return expr
|
|
46
|
+
|
|
47
|
+
def __repr__(self):
|
|
48
|
+
return f"Rule({self.to_string()}, weight={self.weight:.4f})"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class BeamSearchRuleFinder:
|
|
52
|
+
def __init__(self, operators, max_rule_length, beam_width,
|
|
53
|
+
feature_selection_threshold, min_samples_rule,
|
|
54
|
+
rule_complexity_penalty, ohe_groups=None, allow_negation=True):
|
|
55
|
+
self.operators = operators
|
|
56
|
+
self.max_rule_length = max_rule_length
|
|
57
|
+
self.beam_width = beam_width
|
|
58
|
+
self.feature_selection_threshold = feature_selection_threshold
|
|
59
|
+
self.min_samples_rule = min_samples_rule
|
|
60
|
+
self.rule_complexity_penalty = rule_complexity_penalty
|
|
61
|
+
self.ohe_groups = ohe_groups
|
|
62
|
+
self.allow_negation = allow_negation # False -> perilaku identik versi sebelum ini
|
|
63
|
+
|
|
64
|
+
self._feature_to_group = {}
|
|
65
|
+
if ohe_groups is not None:
|
|
66
|
+
for gid, members in ohe_groups.items():
|
|
67
|
+
for f in members:
|
|
68
|
+
self._feature_to_group[f] = gid
|
|
69
|
+
|
|
70
|
+
def _select_features(self, X_bin, gradients):
|
|
71
|
+
dot_products = np.abs(X_bin.T @ gradients)
|
|
72
|
+
col_sums = X_bin.sum(axis=0) + 1e-10
|
|
73
|
+
scores = dot_products / col_sums
|
|
74
|
+
selected = np.where(scores >= self.feature_selection_threshold)[0]
|
|
75
|
+
if len(selected) == 0:
|
|
76
|
+
selected = np.array([np.argmax(scores)])
|
|
77
|
+
return selected
|
|
78
|
+
|
|
79
|
+
def _compute_scores_batch(self, H_batch, gradients):
|
|
80
|
+
dot_gh = H_batch @ gradients
|
|
81
|
+
dot_hh = (H_batch * H_batch).sum(axis=1)
|
|
82
|
+
valid = dot_hh > 1e-10
|
|
83
|
+
w = np.where(valid, dot_gh / np.where(dot_hh > 1e-10, dot_hh, 1.0), 0.0)
|
|
84
|
+
residual = gradients[np.newaxis, :] - w[:, np.newaxis] * H_batch
|
|
85
|
+
scores = (residual * residual).sum(axis=1)
|
|
86
|
+
scores = np.where(valid, scores, np.inf)
|
|
87
|
+
return w, scores, valid
|
|
88
|
+
|
|
89
|
+
def _min_samples_int(self, n_samples):
|
|
90
|
+
if isinstance(self.min_samples_rule, float) and self.min_samples_rule < 1.0:
|
|
91
|
+
return max(1, int(self.min_samples_rule * n_samples))
|
|
92
|
+
return int(self.min_samples_rule)
|
|
93
|
+
|
|
94
|
+
def find_best_rule(self, X_bin, gradients):
|
|
95
|
+
n_samples, n_features = X_bin.shape
|
|
96
|
+
min_s = self._min_samples_int(n_samples)
|
|
97
|
+
if hasattr(self, 'forced_candidate_features') and self.forced_candidate_features is not None:
|
|
98
|
+
candidate_features = self.forced_candidate_features
|
|
99
|
+
else:
|
|
100
|
+
candidate_features = self._select_features(X_bin, gradients)
|
|
101
|
+
|
|
102
|
+
if len(candidate_features) == 0:
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
X_bin_bool = X_bin.astype(bool)
|
|
106
|
+
polaritas_list = [False, True] if self.allow_negation else [False]
|
|
107
|
+
|
|
108
|
+
# 1. Inisialisasi beam panjang satu — coba dua polaritas (asli & negasi) per fitur
|
|
109
|
+
H_list, meta_list = [], []
|
|
110
|
+
for cf in candidate_features:
|
|
111
|
+
kolom_pos = X_bin_bool[:, int(cf)]
|
|
112
|
+
for neg in polaritas_list:
|
|
113
|
+
kolom = ~kolom_pos if neg else kolom_pos
|
|
114
|
+
H_list.append(kolom.astype(np.float64))
|
|
115
|
+
meta_list.append((int(cf), neg))
|
|
116
|
+
|
|
117
|
+
H_init = np.array(H_list)
|
|
118
|
+
support_init = H_init.sum(axis=1)
|
|
119
|
+
valid_init = support_init >= min_s
|
|
120
|
+
if not valid_init.any():
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
w_init, scores_init, w_valid = self._compute_scores_batch(H_init, gradients)
|
|
124
|
+
valid_mask = valid_init & w_valid
|
|
125
|
+
if not valid_mask.any():
|
|
126
|
+
return None
|
|
127
|
+
|
|
128
|
+
candidates = []
|
|
129
|
+
for i, (cf, neg) in enumerate(meta_list):
|
|
130
|
+
if valid_mask[i]:
|
|
131
|
+
pen = scores_init[i] * (1 + self.rule_complexity_penalty * 1)
|
|
132
|
+
candidates.append((pen, [cf], [], [neg], H_init[i].astype(bool)))
|
|
133
|
+
|
|
134
|
+
if not candidates:
|
|
135
|
+
return None
|
|
136
|
+
|
|
137
|
+
candidates.sort(key=lambda x: x[0])
|
|
138
|
+
beam = candidates[:self.beam_width]
|
|
139
|
+
best_score, best_feats, best_ops, best_negs, best_h = beam[0]
|
|
140
|
+
|
|
141
|
+
# 2. Perluas aturan hingga panjang maksimum
|
|
142
|
+
for length in range(2, self.max_rule_length + 1):
|
|
143
|
+
new_candidates = []
|
|
144
|
+
|
|
145
|
+
for beam_score, beam_feats, beam_ops, beam_negs, beam_h in beam:
|
|
146
|
+
beam_feat_set = set(beam_feats)
|
|
147
|
+
|
|
148
|
+
expandable = []
|
|
149
|
+
for cf in candidate_features:
|
|
150
|
+
cf_int = int(cf)
|
|
151
|
+
if cf_int in beam_feat_set:
|
|
152
|
+
continue
|
|
153
|
+
g_new = self._feature_to_group.get(cf_int)
|
|
154
|
+
if g_new is not None:
|
|
155
|
+
if any(self._feature_to_group.get(ef) == g_new for ef in beam_feats):
|
|
156
|
+
continue
|
|
157
|
+
expandable.append(cf_int)
|
|
158
|
+
|
|
159
|
+
if not expandable:
|
|
160
|
+
continue
|
|
161
|
+
|
|
162
|
+
cols_pos = X_bin_bool[:, expandable].T
|
|
163
|
+
kandidat_kolom = [(cols_pos, False)]
|
|
164
|
+
if self.allow_negation:
|
|
165
|
+
kandidat_kolom.append((~cols_pos, True))
|
|
166
|
+
|
|
167
|
+
for cols, neg_flag in kandidat_kolom:
|
|
168
|
+
for op in self.operators:
|
|
169
|
+
beam_h_2d = np.broadcast_to(beam_h, cols.shape)
|
|
170
|
+
if op == 'AND':
|
|
171
|
+
H_new = (beam_h_2d & cols).astype(np.float64)
|
|
172
|
+
elif op == 'OR':
|
|
173
|
+
H_new = (beam_h_2d | cols).astype(np.float64)
|
|
174
|
+
elif op == 'XOR':
|
|
175
|
+
H_new = (beam_h_2d ^ cols).astype(np.float64)
|
|
176
|
+
else:
|
|
177
|
+
continue
|
|
178
|
+
|
|
179
|
+
support_new = H_new.sum(axis=1)
|
|
180
|
+
valid_sup = support_new >= min_s
|
|
181
|
+
if not valid_sup.any():
|
|
182
|
+
continue
|
|
183
|
+
|
|
184
|
+
w_new, scores_new, w_valid_new = self._compute_scores_batch(H_new, gradients)
|
|
185
|
+
valid_combined = valid_sup & w_valid_new
|
|
186
|
+
|
|
187
|
+
for j, cf_int in enumerate(expandable):
|
|
188
|
+
if not valid_combined[j]:
|
|
189
|
+
continue
|
|
190
|
+
pen = scores_new[j] * (1 + self.rule_complexity_penalty * length)
|
|
191
|
+
new_candidates.append((
|
|
192
|
+
pen,
|
|
193
|
+
beam_feats + [cf_int],
|
|
194
|
+
beam_ops + [op],
|
|
195
|
+
beam_negs + [neg_flag],
|
|
196
|
+
H_new[j].astype(bool)
|
|
197
|
+
))
|
|
198
|
+
|
|
199
|
+
if not new_candidates:
|
|
200
|
+
break
|
|
201
|
+
|
|
202
|
+
new_candidates.sort(key=lambda x: x[0])
|
|
203
|
+
new_candidates = new_candidates[:self.beam_width]
|
|
204
|
+
|
|
205
|
+
if new_candidates[0][0] < best_score:
|
|
206
|
+
best_score, best_feats, best_ops, best_negs, best_h = new_candidates[0]
|
|
207
|
+
beam = new_candidates
|
|
208
|
+
else:
|
|
209
|
+
break
|
|
210
|
+
|
|
211
|
+
h_final = best_h.astype(np.float64)
|
|
212
|
+
denom = np.dot(h_final, h_final)
|
|
213
|
+
if denom < 1e-10:
|
|
214
|
+
return None
|
|
215
|
+
w_final = np.dot(gradients, h_final) / denom
|
|
216
|
+
|
|
217
|
+
return Rule(best_feats, best_ops, w_final, negations=best_negs)
|
|
@@ -1,11 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: binboost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
|
|
5
|
-
|
|
6
|
-
Author: BinBoost Authors
|
|
5
|
+
Author: Rangga Wahyu Pratama
|
|
7
6
|
License: MIT
|
|
8
|
-
Project-URL: Homepage, https://github.com/username/binboost
|
|
9
7
|
Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
|
|
10
8
|
Classifier: Programming Language :: Python :: 3
|
|
11
9
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -18,7 +16,6 @@ License-File: LICENSE
|
|
|
18
16
|
Requires-Dist: numpy>=1.21.0
|
|
19
17
|
Requires-Dist: pandas>=1.3.0
|
|
20
18
|
Dynamic: author
|
|
21
|
-
Dynamic: home-page
|
|
22
19
|
Dynamic: license-file
|
|
23
20
|
Dynamic: requires-python
|
|
24
21
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "binboost"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.2"
|
|
8
8
|
description = "Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.8"
|
|
@@ -21,6 +21,3 @@ classifiers = [
|
|
|
21
21
|
"Intended Audience :: Science/Research",
|
|
22
22
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
23
23
|
]
|
|
24
|
-
|
|
25
|
-
[project.urls]
|
|
26
|
-
Homepage = "https://github.com/username/binboost"
|
|
@@ -2,12 +2,11 @@ from setuptools import setup, find_packages
|
|
|
2
2
|
|
|
3
3
|
setup(
|
|
4
4
|
name="binboost",
|
|
5
|
-
version="0.
|
|
6
|
-
author="
|
|
5
|
+
version="0.2.2",
|
|
6
|
+
author="Rangga Wahyu Pratama",
|
|
7
7
|
description="Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner",
|
|
8
8
|
long_description=open("README.md", encoding="utf-8").read(),
|
|
9
9
|
long_description_content_type="text/markdown",
|
|
10
|
-
url="https://github.com/username/binboost",
|
|
11
10
|
packages=find_packages(),
|
|
12
11
|
python_requires=">=3.8",
|
|
13
12
|
install_requires=[
|
binboost-0.1.0/binboost/rule.py
DELETED
|
@@ -1,168 +0,0 @@
|
|
|
1
|
-
import numpy as np
|
|
2
|
-
from itertools import combinations
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
class Rule:
|
|
6
|
-
# Satu aturan logika dalam ensemble BinBoost
|
|
7
|
-
|
|
8
|
-
def __init__(self, features, operators, weight, feature_names=None):
|
|
9
|
-
# 1. Simpan komponen aturan
|
|
10
|
-
# a. Indeks fitur yang terlibat
|
|
11
|
-
self.features = features
|
|
12
|
-
# b. Daftar operator antar fitur
|
|
13
|
-
self.operators = operators
|
|
14
|
-
# c. Bobot optimal hasil perhitungan w_m*
|
|
15
|
-
self.weight = weight
|
|
16
|
-
self.feature_names = feature_names
|
|
17
|
-
|
|
18
|
-
def evaluate(self, X_bin):
|
|
19
|
-
# Hitung keluaran aturan untuk setiap sampel dan kembalikan larik biner
|
|
20
|
-
result = X_bin[:, self.features[0]].astype(bool)
|
|
21
|
-
for i, op in enumerate(self.operators):
|
|
22
|
-
next_col = X_bin[:, self.features[i + 1]].astype(bool)
|
|
23
|
-
if op == 'AND':
|
|
24
|
-
result = result & next_col
|
|
25
|
-
elif op == 'OR':
|
|
26
|
-
result = result | next_col
|
|
27
|
-
elif op == 'XOR':
|
|
28
|
-
result = result ^ next_col
|
|
29
|
-
return result.astype(np.float64)
|
|
30
|
-
|
|
31
|
-
def to_string(self, feature_names=None):
|
|
32
|
-
# Kembalikan representasi teks aturan yang dapat dibaca manusia
|
|
33
|
-
names = feature_names if feature_names is not None else self.feature_names
|
|
34
|
-
if names is None:
|
|
35
|
-
parts = [f"X{f}" for f in self.features]
|
|
36
|
-
else:
|
|
37
|
-
parts = [names[f] for f in self.features]
|
|
38
|
-
if len(parts) == 1:
|
|
39
|
-
return parts[0]
|
|
40
|
-
expr = parts[0]
|
|
41
|
-
for i, op in enumerate(self.operators):
|
|
42
|
-
expr = f"({expr} {op} {parts[i + 1]})"
|
|
43
|
-
return expr
|
|
44
|
-
|
|
45
|
-
def __repr__(self):
|
|
46
|
-
return f"Rule({self.to_string()}, weight={self.weight:.4f})"
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
class BeamSearchRuleFinder:
|
|
50
|
-
# Pencari aturan terbaik menggunakan beam search berbasis gradien
|
|
51
|
-
|
|
52
|
-
def __init__(self, operators, max_rule_length, beam_width,
|
|
53
|
-
feature_selection_threshold, min_samples_rule,
|
|
54
|
-
rule_complexity_penalty, ohe_groups=None):
|
|
55
|
-
self.operators = operators
|
|
56
|
-
self.max_rule_length = max_rule_length
|
|
57
|
-
self.beam_width = beam_width
|
|
58
|
-
self.feature_selection_threshold = feature_selection_threshold
|
|
59
|
-
self.min_samples_rule = min_samples_rule
|
|
60
|
-
self.rule_complexity_penalty = rule_complexity_penalty
|
|
61
|
-
self.ohe_groups = ohe_groups
|
|
62
|
-
|
|
63
|
-
def _select_features(self, X_bin, gradients):
|
|
64
|
-
# 1. Pilih fitur kandidat berdasarkan skor korelasi gradien
|
|
65
|
-
n_samples, n_features = X_bin.shape
|
|
66
|
-
scores = []
|
|
67
|
-
for f in range(n_features):
|
|
68
|
-
col = X_bin[:, f]
|
|
69
|
-
score = np.abs(np.dot(gradients, col)) / (np.sum(col) + 1e-10)
|
|
70
|
-
scores.append(score)
|
|
71
|
-
scores = np.array(scores)
|
|
72
|
-
selected = np.where(scores >= self.feature_selection_threshold)[0]
|
|
73
|
-
if len(selected) == 0:
|
|
74
|
-
selected = np.array([np.argmax(scores)])
|
|
75
|
-
return selected
|
|
76
|
-
|
|
77
|
-
def _same_ohe_group(self, f1, f2):
|
|
78
|
-
# Periksa apakah dua fitur berasal dari grup OneHotEncoding yang sama
|
|
79
|
-
if self.ohe_groups is None:
|
|
80
|
-
return False
|
|
81
|
-
for group in self.ohe_groups.values():
|
|
82
|
-
if f1 in group and f2 in group:
|
|
83
|
-
return True
|
|
84
|
-
return False
|
|
85
|
-
|
|
86
|
-
def _compute_score(self, h, gradients):
|
|
87
|
-
# 1. Hitung bobot optimal w_m* dan skor galat kuadrat terkecil
|
|
88
|
-
denom = np.dot(h, h)
|
|
89
|
-
if denom < 1e-10:
|
|
90
|
-
return None, np.inf
|
|
91
|
-
w = np.dot(gradients, h) / denom
|
|
92
|
-
penalty = self.rule_complexity_penalty
|
|
93
|
-
score = np.sum((gradients - w * h) ** 2)
|
|
94
|
-
return w, score
|
|
95
|
-
|
|
96
|
-
def _is_valid(self, h, n_samples):
|
|
97
|
-
# Periksa apakah jumlah sampel yang memenuhi aturan mencukupi min_samples_rule
|
|
98
|
-
support = int(np.sum(h))
|
|
99
|
-
if isinstance(self.min_samples_rule, float) and self.min_samples_rule < 1.0:
|
|
100
|
-
min_s = int(self.min_samples_rule * n_samples)
|
|
101
|
-
else:
|
|
102
|
-
min_s = int(self.min_samples_rule)
|
|
103
|
-
return support >= min_s
|
|
104
|
-
|
|
105
|
-
def find_best_rule(self, X_bin, gradients):
|
|
106
|
-
# 1. Jalankan beam search untuk menemukan aturan terbaik
|
|
107
|
-
n_samples = X_bin.shape[0]
|
|
108
|
-
candidate_features = self._select_features(X_bin, gradients)
|
|
109
|
-
|
|
110
|
-
# a. Inisialisasi beam dengan aturan panjang satu
|
|
111
|
-
beam = []
|
|
112
|
-
for f in candidate_features:
|
|
113
|
-
h = X_bin[:, f].astype(np.float64)
|
|
114
|
-
if not self._is_valid(h, n_samples):
|
|
115
|
-
continue
|
|
116
|
-
w, score = self._compute_score(h, gradients)
|
|
117
|
-
if w is not None:
|
|
118
|
-
beam.append(([f], [], w, score))
|
|
119
|
-
|
|
120
|
-
if not beam:
|
|
121
|
-
return None
|
|
122
|
-
|
|
123
|
-
beam.sort(key=lambda x: x[3])
|
|
124
|
-
beam = beam[:self.beam_width]
|
|
125
|
-
best_candidate = beam[0]
|
|
126
|
-
|
|
127
|
-
# b. Perluas aturan hingga panjang maksimum
|
|
128
|
-
for length in range(2, self.max_rule_length + 1):
|
|
129
|
-
new_beam = []
|
|
130
|
-
for (feats, ops, w_prev, score_prev) in beam:
|
|
131
|
-
for f in candidate_features:
|
|
132
|
-
if f in feats:
|
|
133
|
-
continue
|
|
134
|
-
# c. Terapkan constraint mutual exclusivity OHE
|
|
135
|
-
skip = False
|
|
136
|
-
for existing_f in feats:
|
|
137
|
-
if self._same_ohe_group(existing_f, f):
|
|
138
|
-
skip = True
|
|
139
|
-
break
|
|
140
|
-
if skip:
|
|
141
|
-
continue
|
|
142
|
-
for op in self.operators:
|
|
143
|
-
new_feats = feats + [f]
|
|
144
|
-
new_ops = ops + [op]
|
|
145
|
-
tmp_rule = Rule(new_feats, new_ops, 0.0)
|
|
146
|
-
h = tmp_rule.evaluate(X_bin)
|
|
147
|
-
if not self._is_valid(h, n_samples):
|
|
148
|
-
continue
|
|
149
|
-
w, score = self._compute_score(h, gradients)
|
|
150
|
-
# d. Terapkan penalti kompleksitas berdasarkan panjang aturan
|
|
151
|
-
penalized_score = score * (1 + self.rule_complexity_penalty * length)
|
|
152
|
-
if w is not None:
|
|
153
|
-
new_beam.append((new_feats, new_ops, w, penalized_score))
|
|
154
|
-
|
|
155
|
-
if not new_beam:
|
|
156
|
-
break
|
|
157
|
-
|
|
158
|
-
new_beam.sort(key=lambda x: x[3])
|
|
159
|
-
new_beam = new_beam[:self.beam_width]
|
|
160
|
-
|
|
161
|
-
if new_beam[0][3] < best_candidate[3]:
|
|
162
|
-
best_candidate = new_beam[0]
|
|
163
|
-
beam = new_beam
|
|
164
|
-
else:
|
|
165
|
-
break
|
|
166
|
-
|
|
167
|
-
feats, ops, w, score = best_candidate
|
|
168
|
-
return Rule(feats, ops, w)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|