binboost 0.1.2__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {binboost-0.1.2 → binboost-0.2.1}/LICENSE +1 -1
- {binboost-0.1.2 → binboost-0.2.1}/PKG-INFO +2 -2
- binboost-0.2.1/binboost/__init__.py +5 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost/binboost.py +31 -21
- {binboost-0.1.2 → binboost-0.2.1}/binboost/loss.py +22 -7
- {binboost-0.1.2 → binboost-0.2.1}/binboost.egg-info/PKG-INFO +2 -2
- {binboost-0.1.2 → binboost-0.2.1}/pyproject.toml +1 -1
- {binboost-0.1.2 → binboost-0.2.1}/setup.py +2 -2
- binboost-0.1.2/binboost/__init__.py +0 -5
- {binboost-0.1.2 → binboost-0.2.1}/README.md +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost/binarizer.py +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost/rule.py +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost.egg-info/SOURCES.txt +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost.egg-info/dependency_links.txt +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost.egg-info/requires.txt +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/binboost.egg-info/top_level.txt +0 -0
- {binboost-0.1.2 → binboost-0.2.1}/setup.cfg +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: binboost
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
|
|
5
|
-
Author:
|
|
5
|
+
Author: Rangga Wahyu Pratama
|
|
6
6
|
License: MIT
|
|
7
7
|
Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
|
|
8
8
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -59,6 +59,9 @@ class BinBoost:
|
|
|
59
59
|
min_samples_rule : int or float, default=5
|
|
60
60
|
Jumlah minimum sampel yang harus memenuhi sebuah aturan.
|
|
61
61
|
|
|
62
|
+
lambda0 : float, default=1.0
|
|
63
|
+
Konstanta regularisasi dasar untuk bobot Newton adaptif (λ_R = lambda0 · L / sqrt(n_R+1)).
|
|
64
|
+
|
|
62
65
|
binarize_strategy : str, default='gradient'
|
|
63
66
|
Strategi binarisasi fitur numerik. Pilihan: 'gradient', 'quantile',
|
|
64
67
|
'uniform', 'kmeans'.
|
|
@@ -132,6 +135,7 @@ class BinBoost:
|
|
|
132
135
|
beam_width=5,
|
|
133
136
|
subsample=0.8,
|
|
134
137
|
min_samples_rule=5,
|
|
138
|
+
lambda0=1.0,
|
|
135
139
|
binarize_strategy='gradient',
|
|
136
140
|
n_thresholds='auto',
|
|
137
141
|
n_iter_no_change=None,
|
|
@@ -154,6 +158,7 @@ class BinBoost:
|
|
|
154
158
|
self.beam_width = beam_width
|
|
155
159
|
self.subsample = subsample
|
|
156
160
|
self.min_samples_rule = min_samples_rule
|
|
161
|
+
self.lambda0 = lambda0
|
|
157
162
|
self.binarize_strategy = binarize_strategy
|
|
158
163
|
self.n_thresholds = n_thresholds
|
|
159
164
|
self.n_iter_no_change = n_iter_no_change
|
|
@@ -232,37 +237,38 @@ class BinBoost:
|
|
|
232
237
|
ops.append('XOR')
|
|
233
238
|
return ops
|
|
234
239
|
|
|
235
|
-
def _seleksi_fitur(self, X_bin, gradients):
|
|
236
|
-
# 1. Pilih fitur kandidat menggunakan kriteria gain yang sama dengan seleksi aturan
|
|
237
|
-
# a. Hitung gain setiap fitur: (Σgᵢfᵢ)² / Σfᵢ² konsisten dengan w_m* aturan
|
|
240
|
+
def _seleksi_fitur(self, X_bin, gradients, hessians):
|
|
238
241
|
dot_gf = X_bin.T @ gradients
|
|
239
|
-
|
|
240
|
-
|
|
242
|
+
H_f = X_bin.T @ hessians
|
|
243
|
+
n_f = X_bin.sum(axis=0)
|
|
244
|
+
lambda_f = self.lambda0 / np.sqrt(n_f + 1.0)
|
|
245
|
+
gains = (dot_gf ** 2) / (H_f + lambda_f + 1e-10)
|
|
241
246
|
|
|
242
247
|
if self.feature_selection_threshold <= 0.0:
|
|
243
|
-
# b. Ambil semua fitur jika threshold 0.0
|
|
244
248
|
return np.arange(X_bin.shape[1])
|
|
245
249
|
|
|
246
|
-
# c. Gunakan threshold sebagai persentil relatif dari distribusi gain iterasi ini
|
|
247
250
|
nilai_persentil = np.percentile(gains, self.feature_selection_threshold * 100)
|
|
248
251
|
selected = np.where(gains >= nilai_persentil)[0]
|
|
249
252
|
if len(selected) == 0:
|
|
250
253
|
selected = np.array([np.argmax(gains)])
|
|
251
254
|
return selected
|
|
252
255
|
|
|
253
|
-
def _hitung_bobot_dua_sisi(self, h, gradients):
|
|
254
|
-
# 1. Hitung bobot optimal untuk dua region secara terpisah
|
|
255
|
-
# a. Region 1: sampel yang memenuhi aturan (h=1)
|
|
256
|
+
def _hitung_bobot_dua_sisi(self, h, gradients, hessians, rule_length):
|
|
256
257
|
mask_1 = h.astype(bool)
|
|
257
258
|
mask_0 = ~mask_1
|
|
258
|
-
|
|
259
|
+
|
|
259
260
|
sum_g1 = gradients[mask_1].sum() if mask_1.any() else 0.0
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
261
|
+
sum_H1 = hessians[mask_1].sum() if mask_1.any() else 0.0
|
|
262
|
+
n1 = float(mask_1.sum())
|
|
263
|
+
lambda_1 = self.lambda0 * rule_length / np.sqrt(n1 + 1.0)
|
|
264
|
+
w1 = sum_g1 / (sum_H1 + lambda_1 + 1e-10) if n1 > 0 else 0.0
|
|
265
|
+
|
|
263
266
|
sum_g0 = gradients[mask_0].sum() if mask_0.any() else 0.0
|
|
264
|
-
|
|
265
|
-
|
|
267
|
+
sum_H0 = hessians[mask_0].sum() if mask_0.any() else 0.0
|
|
268
|
+
n0 = float(mask_0.sum())
|
|
269
|
+
lambda_0 = self.lambda0 * rule_length / np.sqrt(n0 + 1.0)
|
|
270
|
+
w0 = sum_g0 / (sum_H0 + lambda_0 + 1e-10) if n0 > 0 else 0.0
|
|
271
|
+
|
|
266
272
|
return w1, w0
|
|
267
273
|
|
|
268
274
|
def fit(self, X, y, sample_weight=None):
|
|
@@ -331,17 +337,19 @@ class BinBoost:
|
|
|
331
337
|
for m in range(self.n_estimators):
|
|
332
338
|
# 1. Hitung gradien dari fungsi loss yang dipilih
|
|
333
339
|
gradients = loss_fn.gradient(y, self._F)
|
|
340
|
+
hessians = loss_fn.hessian(y, self._F)
|
|
334
341
|
|
|
335
342
|
# 2. Ambil subsample data untuk iterasi ini
|
|
336
343
|
idx = self._subsample_indices(n_samples, rng)
|
|
337
344
|
X_sub = X[idx]
|
|
338
345
|
g_sub = gradients[idx]
|
|
346
|
+
h_sub = hessians[idx]
|
|
339
347
|
|
|
340
348
|
# 3. Binarisasi fitur numerik menggunakan strategi yang dipilih
|
|
341
349
|
X_bin, thresholds_iter = binarizer.transform(X_sub, g_sub, numeric_cols)
|
|
342
350
|
|
|
343
351
|
# 4. Seleksi fitur kandidat menggunakan kriteria gain yang konsisten
|
|
344
|
-
candidate_features = self._seleksi_fitur(X_bin, g_sub)
|
|
352
|
+
candidate_features = self._seleksi_fitur(X_bin, g_sub, h_sub)
|
|
345
353
|
rule_finder.forced_candidate_features = candidate_features
|
|
346
354
|
|
|
347
355
|
# 5. Cari aturan terbaik menggunakan beam search
|
|
@@ -356,11 +364,12 @@ class BinBoost:
|
|
|
356
364
|
# 6. Perbarui prediksi model untuk seluruh data menggunakan update dua sisi
|
|
357
365
|
X_bin_full, thresholds_full = binarizer.transform(X, gradients, numeric_cols)
|
|
358
366
|
h_full = rule.evaluate(X_bin_full)
|
|
359
|
-
w1, w0 = self._hitung_bobot_dua_sisi(h_full, gradients)
|
|
360
|
-
|
|
367
|
+
w1, w0 = self._hitung_bobot_dua_sisi(h_full, gradients, hessians, len(rule.features))
|
|
368
|
+
|
|
369
|
+
# 7. Update F menggunakan kontribusi dua sisi: w1 untuk yang memenuhi, w0 untuk yang tidak
|
|
361
370
|
self._F = self._F + self.learning_rate * (w1 * h_full + w0 * (1.0 - h_full))
|
|
362
371
|
|
|
363
|
-
#
|
|
372
|
+
# 8. Simpan aturan dengan bobot dua sisi dan statistik iterasi ini
|
|
364
373
|
self.estimators_.append(rule)
|
|
365
374
|
self.rule_weights_.append(np.array([w0, w1]))
|
|
366
375
|
for col, t in thresholds_full.items():
|
|
@@ -369,7 +378,7 @@ class BinBoost:
|
|
|
369
378
|
current_score = loss_fn.loss(y, self._F)
|
|
370
379
|
self.train_score_.append(current_score)
|
|
371
380
|
|
|
372
|
-
#
|
|
381
|
+
# 9. Periksa kondisi early stopping jika diaktifkan
|
|
373
382
|
if self.n_iter_no_change is not None:
|
|
374
383
|
if current_score < best_score - self.tol:
|
|
375
384
|
best_score = current_score
|
|
@@ -562,6 +571,7 @@ class BinBoost:
|
|
|
562
571
|
'beam_width': self.beam_width,
|
|
563
572
|
'subsample': self.subsample,
|
|
564
573
|
'min_samples_rule': self.min_samples_rule,
|
|
574
|
+
'lambda0': self.lambda0,
|
|
565
575
|
'binarize_strategy': self.binarize_strategy,
|
|
566
576
|
'n_thresholds': self.n_thresholds,
|
|
567
577
|
'n_iter_no_change': self.n_iter_no_change,
|
|
@@ -1,15 +1,12 @@
|
|
|
1
1
|
# loss.py
|
|
2
2
|
import numpy as np
|
|
3
3
|
|
|
4
|
-
|
|
5
4
|
def sigmoid(x):
|
|
6
5
|
# Fungsi sigmoid dengan penjagaan numerik agar tidak overflow
|
|
7
6
|
return np.where(x >= 0, 1 / (1 + np.exp(-x)), np.exp(x) / (1 + np.exp(x)))
|
|
8
7
|
|
|
9
|
-
|
|
10
8
|
class LogisticLoss:
|
|
11
9
|
# Loss biner standar berbasis log-likelihood
|
|
12
|
-
|
|
13
10
|
def loss(self, y, F):
|
|
14
11
|
p = sigmoid(F)
|
|
15
12
|
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
@@ -19,16 +16,21 @@ class LogisticLoss:
|
|
|
19
16
|
# 1. Gradien negatif sebagai arah penurunan loss
|
|
20
17
|
p = sigmoid(F)
|
|
21
18
|
return y - p
|
|
19
|
+
|
|
20
|
+
def hessian(self, y, F):
|
|
21
|
+
# Turunan kedua logistic loss terhadap F
|
|
22
|
+
p = sigmoid(F)
|
|
23
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
24
|
+
h = p * (1 - p)
|
|
25
|
+
return np.maximum(h, 1e-10)
|
|
22
26
|
|
|
23
27
|
def init_F(self, y):
|
|
24
28
|
# Inisialisasi F0 dari proporsi kelas positif
|
|
25
29
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
26
30
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
27
31
|
|
|
28
|
-
|
|
29
32
|
class FocalLoss:
|
|
30
33
|
# Focal Loss untuk data tidak seimbang
|
|
31
|
-
|
|
32
34
|
def __init__(self, gamma=2.0, alpha=0.25):
|
|
33
35
|
self.gamma = gamma
|
|
34
36
|
self.alpha = alpha
|
|
@@ -51,12 +53,20 @@ class FocalLoss:
|
|
|
51
53
|
weight = alpha_t * (1 - pt) ** self.gamma
|
|
52
54
|
grad = weight * (y - p) + self.gamma * weight * pt * np.log(pt) * np.where(y == 1, -(1 - p), p)
|
|
53
55
|
return grad
|
|
56
|
+
|
|
57
|
+
def hessian(self, y, F):
|
|
58
|
+
p = sigmoid(F)
|
|
59
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
60
|
+
pt = np.where(y == 1, p, 1 - p)
|
|
61
|
+
alpha_t = np.where(y == 1, self.alpha, 1 - self.alpha)
|
|
62
|
+
beta = alpha_t * (1 - pt) ** self.gamma
|
|
63
|
+
h = beta * p * (1 - p)
|
|
64
|
+
return np.maximum(h, 1e-10)
|
|
54
65
|
|
|
55
66
|
def init_F(self, y):
|
|
56
67
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
57
68
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
58
69
|
|
|
59
|
-
|
|
60
70
|
class PolyLoss:
|
|
61
71
|
# PolyLoss yang berbasis ekspansi polinomial
|
|
62
72
|
def __init__(self, epsilon=1.0):
|
|
@@ -79,12 +89,17 @@ class PolyLoss:
|
|
|
79
89
|
# b. Suku koreksi dari ekspansi polinomial pertama
|
|
80
90
|
grad_poly = self.epsilon * np.where(y == 1, p * (1 - p), -p * (1 - p))
|
|
81
91
|
return grad_ce + grad_poly
|
|
92
|
+
|
|
93
|
+
def hessian(self, y, F):
|
|
94
|
+
p = sigmoid(F)
|
|
95
|
+
p = np.clip(p, 1e-15, 1 - 1e-15)
|
|
96
|
+
h = p * (1 - p) * (1 + self.epsilon * (2 * y - 1) * (2 * p - 1))
|
|
97
|
+
return np.maximum(h, 1e-10)
|
|
82
98
|
|
|
83
99
|
def init_F(self, y):
|
|
84
100
|
p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
|
|
85
101
|
return np.full(len(y), np.log(p / (1 - p)))
|
|
86
102
|
|
|
87
|
-
|
|
88
103
|
def get_loss(name, **kwargs):
|
|
89
104
|
# Kembalikan objek loss sesuai nama yang dipilih pengguna
|
|
90
105
|
if name == 'logistic':
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: binboost
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
|
|
5
|
-
Author:
|
|
5
|
+
Author: Rangga Wahyu Pratama
|
|
6
6
|
License: MIT
|
|
7
7
|
Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
|
|
8
8
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -2,8 +2,8 @@ from setuptools import setup, find_packages
|
|
|
2
2
|
|
|
3
3
|
setup(
|
|
4
4
|
name="binboost",
|
|
5
|
-
version="0.1
|
|
6
|
-
author="
|
|
5
|
+
version="0.2.1",
|
|
6
|
+
author="Rangga Wahyu Pratama",
|
|
7
7
|
description="Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner",
|
|
8
8
|
long_description=open("README.md", encoding="utf-8").read(),
|
|
9
9
|
long_description_content_type="text/markdown",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|