binboost 0.1.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  MIT License
2
2
 
3
- Copyright (c) 2024 BinBoost Authors
3
+ Copyright (c) 2026 BinBoost Authors
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
@@ -1,11 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: binboost
3
- Version: 0.1.0
3
+ Version: 0.2.2
4
4
  Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
5
- Home-page: https://github.com/username/binboost
6
- Author: BinBoost Authors
5
+ Author: Rangga Wahyu Pratama
7
6
  License: MIT
8
- Project-URL: Homepage, https://github.com/username/binboost
9
7
  Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
10
8
  Classifier: Programming Language :: Python :: 3
11
9
  Classifier: License :: OSI Approved :: MIT License
@@ -18,7 +16,6 @@ License-File: LICENSE
18
16
  Requires-Dist: numpy>=1.21.0
19
17
  Requires-Dist: pandas>=1.3.0
20
18
  Dynamic: author
21
- Dynamic: home-page
22
19
  Dynamic: license-file
23
20
  Dynamic: requires-python
24
21
 
@@ -0,0 +1,5 @@
1
+ from .binboost import BinBoost
2
+
3
+ __version__ = "0.2.2"
4
+ __author__ = "Rangga Wahyu Pratama"
5
+ __all__ = ["BinBoost"]
@@ -1,3 +1,4 @@
1
+ # Binarizer.py
1
2
  import numpy as np
2
3
 
3
4
 
@@ -9,7 +9,7 @@ class BinBoost:
9
9
  """
10
10
  BinBoost: Gradient Boosting Berbasis Aturan Logika Adaptif.
11
11
 
12
- Algoritma klasifikasi biner yang membangun ensemble aturan logika
12
+ Algoritma klasifikasi biner yang membangun ensemble aturan logika
13
13
  murni (AND, OR, XOR) dengan binarisasi fitur numerik adaptif
14
14
  berbasis gradien pada setiap iterasi boosting.
15
15
 
@@ -18,7 +18,7 @@ class BinBoost:
18
18
  n_estimators : int, default=100
19
19
  Jumlah iterasi boosting.
20
20
 
21
- learning_rate : float, default=0.1
21
+ learning_rate : float, default=0.2
22
22
  Faktor penyusutan kontribusi setiap aturan.
23
23
 
24
24
  loss : str, default='logistic'
@@ -33,7 +33,7 @@ class BinBoost:
33
33
  focal_alpha : float, default=0.25
34
34
  Parameter alpha pada Focal Loss. Hanya berlaku saat loss='focal'.
35
35
 
36
- max_rule_length : int, default=2
36
+ max_rule_length : int, default=4
37
37
  Jumlah maksimum fitur dalam satu aturan.
38
38
 
39
39
  operators : list, default=['AND', 'OR']
@@ -45,8 +45,10 @@ class BinBoost:
45
45
  rule_complexity_penalty : float, default=0.0
46
46
  Penalti bobot proporsional dengan panjang aturan.
47
47
 
48
- feature_selection_threshold : float, default=0.01
49
- Ambang batas skor korelasi gradien minimum agar fitur masuk kandidat.
48
+ feature_selection_threshold : float, default=0.0
49
+ Ambang batas persentil gain fitur untuk seleksi kandidat per iterasi.
50
+ Nilai 0.0 berarti semua fitur diikutsertakan. Nilai 0.5 berarti hanya
51
+ fitur dengan gain di atas median yang masuk kandidat.
50
52
 
51
53
  beam_width : int, default=5
52
54
  Jumlah kandidat aturan terbaik yang dipertahankan per langkah beam search.
@@ -54,9 +56,12 @@ class BinBoost:
54
56
  subsample : float, default=0.8
55
57
  Fraksi data yang digunakan per iterasi boosting.
56
58
 
57
- min_samples_rule : int or float, default=10
59
+ min_samples_rule : int or float, default=5
58
60
  Jumlah minimum sampel yang harus memenuhi sebuah aturan.
59
61
 
62
+ lambda0 : float, default=3.0
63
+ Konstanta regularisasi dasar untuk bobot Newton adaptif (λ_R = lambda0 · L / sqrt(n_R+1)).
64
+
60
65
  binarize_strategy : str, default='gradient'
61
66
  Strategi binarisasi fitur numerik. Pilihan: 'gradient', 'quantile',
62
67
  'uniform', 'kmeans'.
@@ -76,13 +81,18 @@ class BinBoost:
76
81
  warm_start : bool, default=False
77
82
  Jika True maka pelatihan dilanjutkan dari kondisi model sebelumnya.
78
83
 
84
+ threshold : float or 'auto', default='auto'
85
+ Threshold prediksi kelas. Nilai 'auto' menggunakan indeks Youden
86
+ untuk mencari threshold optimal pada data training.
87
+
79
88
  Atribut yang tersedia Setelah fit
80
89
  --------------------------------
81
90
  estimators_ : list of Rule
82
91
  Daftar aturan yang dipelajari.
83
92
 
84
- rule_weights_ : ndarray of float
85
- Bobot optimal setiap aturan.
93
+ rule_weights_ : ndarray of float shape (n_estimators_, 2)
94
+ Bobot dua sisi setiap aturan. Kolom 0 adalah w0 (tidak memenuhi aturan)
95
+ dan kolom 1 adalah w1 (memenuhi aturan).
86
96
 
87
97
  rules_ : list of str
88
98
  Representasi teks setiap aturan.
@@ -112,25 +122,27 @@ class BinBoost:
112
122
  def __init__(
113
123
  self,
114
124
  n_estimators=100,
115
- learning_rate=0.1,
125
+ learning_rate=0.2,
116
126
  loss='logistic',
117
127
  poly_epsilon=1.0,
118
128
  focal_gamma=2.0,
119
129
  focal_alpha=0.25,
120
- max_rule_length=2,
130
+ max_rule_length=4,
121
131
  operators=None,
122
132
  use_xor=False,
123
133
  rule_complexity_penalty=0.0,
124
- feature_selection_threshold=0.01,
134
+ feature_selection_threshold=0.0,
125
135
  beam_width=5,
126
136
  subsample=0.8,
127
- min_samples_rule=10,
137
+ min_samples_rule=5,
138
+ lambda0=3.0,
128
139
  binarize_strategy='gradient',
129
140
  n_thresholds='auto',
130
141
  n_iter_no_change=None,
131
142
  tol=1e-4,
132
143
  random_state=None,
133
144
  warm_start=False,
145
+ threshold='auto',
134
146
  ):
135
147
  self.n_estimators = n_estimators
136
148
  self.learning_rate = learning_rate
@@ -146,12 +158,15 @@ class BinBoost:
146
158
  self.beam_width = beam_width
147
159
  self.subsample = subsample
148
160
  self.min_samples_rule = min_samples_rule
161
+ self.lambda0 = lambda0
149
162
  self.binarize_strategy = binarize_strategy
150
163
  self.n_thresholds = n_thresholds
151
164
  self.n_iter_no_change = n_iter_no_change
152
165
  self.tol = tol
153
166
  self.random_state = random_state
154
167
  self.warm_start = warm_start
168
+ self.threshold = threshold
169
+ self._threshold_fitted = 0.5
155
170
 
156
171
  def _validate_input(self, X, y=None):
157
172
  # Ubah input ke numpy array dan pastikan dimensinya benar
@@ -195,10 +210,11 @@ class BinBoost:
195
210
  continue
196
211
  # a. Dua fitur bersifat mutually exclusive jika tidak pernah bernilai 1 bersamaan
197
212
  both_one = np.sum((X[:, i] == 1) & (X[:, j] == 1))
213
+ # b. Batasi ukuran grup maksimum 20 agar tidak memblokir terlalu banyak fitur
198
214
  if both_one == 0:
199
- # b. Pastikan keduanya memang fitur biner
200
- if np.all(np.isin(np.unique(X[:, i]), [0.0, 1.0])) and \
201
- np.all(np.isin(np.unique(X[:, j]), [0.0, 1.0])):
215
+ if len(group) < 20 and \
216
+ np.all(np.isin(np.unique(X[:, i]), [0.0, 1.0])) and \
217
+ np.all(np.isin(np.unique(X[:, j]), [0.0, 1.0])):
202
218
  group.append(j)
203
219
  used.add(j)
204
220
  if len(group) > 1:
@@ -221,6 +237,40 @@ class BinBoost:
221
237
  ops.append('XOR')
222
238
  return ops
223
239
 
240
+ def _seleksi_fitur(self, X_bin, gradients, hessians):
241
+ dot_gf = X_bin.T @ gradients
242
+ H_f = X_bin.T @ hessians
243
+ n_f = X_bin.sum(axis=0)
244
+ lambda_f = self.lambda0 / np.sqrt(n_f + 1.0)
245
+ gains = (dot_gf ** 2) / (H_f + lambda_f + 1e-10)
246
+
247
+ if self.feature_selection_threshold <= 0.0:
248
+ return np.arange(X_bin.shape[1])
249
+
250
+ nilai_persentil = np.percentile(gains, self.feature_selection_threshold * 100)
251
+ selected = np.where(gains >= nilai_persentil)[0]
252
+ if len(selected) == 0:
253
+ selected = np.array([np.argmax(gains)])
254
+ return selected
255
+
256
+ def _hitung_bobot_dua_sisi(self, h, gradients, hessians, rule_length):
257
+ mask_1 = h.astype(bool)
258
+ mask_0 = ~mask_1
259
+
260
+ sum_g1 = gradients[mask_1].sum() if mask_1.any() else 0.0
261
+ sum_H1 = hessians[mask_1].sum() if mask_1.any() else 0.0
262
+ n1 = float(mask_1.sum())
263
+ lambda_1 = self.lambda0 * rule_length / np.sqrt(n1 + 1.0)
264
+ w1 = sum_g1 / (sum_H1 + lambda_1 + 1e-10) if n1 > 0 else 0.0
265
+
266
+ sum_g0 = gradients[mask_0].sum() if mask_0.any() else 0.0
267
+ sum_H0 = hessians[mask_0].sum() if mask_0.any() else 0.0
268
+ n0 = float(mask_0.sum())
269
+ lambda_0 = self.lambda0 * rule_length / np.sqrt(n0 + 1.0)
270
+ w0 = sum_g0 / (sum_H0 + lambda_0 + 1e-10) if n0 > 0 else 0.0
271
+
272
+ return w1, w0
273
+
224
274
  def fit(self, X, y, sample_weight=None):
225
275
  """
226
276
  Latih BinBoost pada data X dan label y.
@@ -258,16 +308,17 @@ class BinBoost:
258
308
  numeric_cols = self._detect_numeric_cols(X)
259
309
  self.ohe_groups_ = self._detect_ohe_groups(X)
260
310
 
261
- binarizer = GradientBinarizer(
311
+ self._binarizer = GradientBinarizer(
262
312
  strategy=self.binarize_strategy,
263
313
  n_thresholds=self.n_thresholds
264
314
  )
315
+ binarizer = self._binarizer
265
316
 
266
317
  rule_finder = BeamSearchRuleFinder(
267
318
  operators=ops,
268
319
  max_rule_length=self.max_rule_length,
269
320
  beam_width=self.beam_width,
270
- feature_selection_threshold=self.feature_selection_threshold,
321
+ feature_selection_threshold=0.0,
271
322
  min_samples_rule=self.min_samples_rule,
272
323
  rule_complexity_penalty=self.rule_complexity_penalty,
273
324
  ohe_groups=self.ohe_groups_,
@@ -286,16 +337,22 @@ class BinBoost:
286
337
  for m in range(self.n_estimators):
287
338
  # 1. Hitung gradien dari fungsi loss yang dipilih
288
339
  gradients = loss_fn.gradient(y, self._F)
340
+ hessians = loss_fn.hessian(y, self._F)
289
341
 
290
342
  # 2. Ambil subsample data untuk iterasi ini
291
343
  idx = self._subsample_indices(n_samples, rng)
292
344
  X_sub = X[idx]
293
345
  g_sub = gradients[idx]
346
+ h_sub = hessians[idx]
294
347
 
295
348
  # 3. Binarisasi fitur numerik menggunakan strategi yang dipilih
296
349
  X_bin, thresholds_iter = binarizer.transform(X_sub, g_sub, numeric_cols)
297
350
 
298
- # 4. Cari aturan terbaik menggunakan beam search
351
+ # 4. Seleksi fitur kandidat menggunakan kriteria gain yang konsisten
352
+ candidate_features = self._seleksi_fitur(X_bin, g_sub, h_sub)
353
+ rule_finder.forced_candidate_features = candidate_features
354
+
355
+ # 5. Cari aturan terbaik menggunakan beam search
299
356
  rule = rule_finder.find_best_rule(X_bin, g_sub)
300
357
  if rule is None:
301
358
  break
@@ -304,21 +361,24 @@ class BinBoost:
304
361
  list(self.feature_names_in_) if self.feature_names_in_ is not None else None
305
362
  )
306
363
 
307
- # 5. Perbarui prediksi model untuk seluruh data
364
+ # 6. Perbarui prediksi model untuk seluruh data menggunakan update dua sisi
308
365
  X_bin_full, thresholds_full = binarizer.transform(X, gradients, numeric_cols)
309
366
  h_full = rule.evaluate(X_bin_full)
310
- self._F = self._F + self.learning_rate * rule.weight * h_full
367
+ w1, w0 = self._hitung_bobot_dua_sisi(h_full, gradients, hessians, len(rule.features))
368
+
369
+ # 7. Update F menggunakan kontribusi dua sisi: w1 untuk yang memenuhi, w0 untuk yang tidak
370
+ self._F = self._F + self.learning_rate * (w1 * h_full + w0 * (1.0 - h_full))
311
371
 
312
- # 6. Simpan aturan dan statistik iterasi ini
372
+ # 8. Simpan aturan dengan bobot dua sisi dan statistik iterasi ini
313
373
  self.estimators_.append(rule)
314
- self.rule_weights_.append(rule.weight)
374
+ self.rule_weights_.append(np.array([w0, w1]))
315
375
  for col, t in thresholds_full.items():
316
376
  self.thresholds_[col].append(t)
317
377
 
318
378
  current_score = loss_fn.loss(y, self._F)
319
379
  self.train_score_.append(current_score)
320
380
 
321
- # 7. Periksa kondisi early stopping jika diaktifkan
381
+ # 9. Periksa kondisi early stopping jika diaktifkan
322
382
  if self.n_iter_no_change is not None:
323
383
  if current_score < best_score - self.tol:
324
384
  best_score = current_score
@@ -334,13 +394,19 @@ class BinBoost:
334
394
  self.rules_ = [r.to_string(self.feature_names_in_) for r in self.estimators_]
335
395
  self._compute_feature_importances(n_features)
336
396
  self.is_fitted_ = True
397
+ # Tentukan threshold prediksi optimal setelah seluruh iterasi selesai
398
+ if self.threshold == 'auto':
399
+ self._threshold_fitted = self._cari_threshold_optimal(X, y)
400
+ else:
401
+ self._threshold_fitted = float(self.threshold)
337
402
  return self
338
403
 
339
404
  def _compute_feature_importances(self, n_features):
340
405
  # Hitung skor kepentingan fitur dari frekuensi kemunculan dikali rata-rata bobot absolut
341
406
  importances = np.zeros(n_features)
342
- for rule in self.estimators_:
343
- gain = np.abs(rule.weight)
407
+ for rule, weights in zip(self.estimators_, self.rule_weights_):
408
+ # Gunakan selisih absolut w1 dan w0 sebagai ukuran gain aturan
409
+ gain = np.abs(weights[1] - weights[0])
344
410
  for f in rule.features:
345
411
  importances[f] += gain
346
412
  total = importances.sum()
@@ -350,21 +416,41 @@ class BinBoost:
350
416
  if not getattr(self, 'is_fitted_', False):
351
417
  raise RuntimeError("Model belum dilatih. Panggil fit terlebih dahulu.")
352
418
 
419
+ def _cari_threshold_optimal(self, X, y):
420
+ # Cari threshold prediksi optimal menggunakan indeks Youden pada data training
421
+ proba = self.predict_proba(X)[:, 1]
422
+ ambang_kandidat = np.sort(np.unique(proba))
423
+ skor_terbaik = -np.inf
424
+ threshold_terbaik = 0.5
425
+ pos = np.sum(y == 1)
426
+ neg = np.sum(y == 0)
427
+ if pos == 0 or neg == 0:
428
+ return 0.5
429
+ for t in ambang_kandidat:
430
+ pred = (proba >= t).astype(int)
431
+ tp = np.sum((pred == 1) & (y == 1))
432
+ tn = np.sum((pred == 0) & (y == 0))
433
+ sensitivity = tp / pos
434
+ specificity = tn / neg
435
+ # Indeks Youden memaksimalkan sensitivity tambah specificity dikurangi 1
436
+ youden = sensitivity + specificity - 1
437
+ if youden > skor_terbaik:
438
+ skor_terbaik = youden
439
+ threshold_terbaik = t
440
+ return float(threshold_terbaik)
441
+
353
442
  def _decision_function(self, X):
354
- # Hitung nilai F akhir untuk seluruh sampel
443
+ # Hitung nilai F akhir untuk seluruh sampel menggunakan update dua sisi
355
444
  self._check_is_fitted()
356
445
  X, _ = self._validate_input(X)
357
446
  F = np.zeros(X.shape[0])
358
- binarizer = GradientBinarizer(
359
- strategy=self.binarize_strategy,
360
- n_thresholds=self.n_thresholds
361
- )
362
447
  numeric_cols = self._detect_numeric_cols(X)
363
448
  dummy_grad = np.ones(X.shape[0])
364
- X_bin, _ = binarizer.transform(X, dummy_grad, numeric_cols)
365
- for rule, w in zip(self.estimators_, self.rule_weights_):
449
+ X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
450
+ for rule, weights in zip(self.estimators_, self.rule_weights_):
366
451
  h = rule.evaluate(X_bin)
367
- F += self.learning_rate * w * h
452
+ w0, w1 = weights[0], weights[1]
453
+ F += self.learning_rate * (w1 * h + w0 * (1.0 - h))
368
454
  return F
369
455
 
370
456
  def predict_proba(self, X):
@@ -398,7 +484,7 @@ class BinBoost:
398
484
  ndarray of shape (n_samples,)
399
485
  """
400
486
  proba = self.predict_proba(X)
401
- return (proba[:, 1] >= 0.5).astype(int)
487
+ return (proba[:, 1] >= self._threshold_fitted).astype(int)
402
488
 
403
489
  def score(self, X, y):
404
490
  """
@@ -422,17 +508,14 @@ class BinBoost:
422
508
  """
423
509
  self._check_is_fitted()
424
510
  X, _ = self._validate_input(X)
425
- binarizer = GradientBinarizer(
426
- strategy=self.binarize_strategy,
427
- n_thresholds=self.n_thresholds
428
- )
429
511
  numeric_cols = self._detect_numeric_cols(X)
430
512
  dummy_grad = np.ones(X.shape[0])
431
- X_bin, _ = binarizer.transform(X, dummy_grad, numeric_cols)
513
+ X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
432
514
  F = np.zeros(X.shape[0])
433
- for rule, w in zip(self.estimators_, self.rule_weights_):
515
+ for rule, weights in zip(self.estimators_, self.rule_weights_):
434
516
  h = rule.evaluate(X_bin)
435
- F += self.learning_rate * w * h
517
+ w0, w1 = weights[0], weights[1]
518
+ F += self.learning_rate * (w1 * h + w0 * (1.0 - h))
436
519
  p = sigmoid(F)
437
520
  yield np.column_stack([1 - p, p])
438
521
 
@@ -445,7 +528,7 @@ class BinBoost:
445
528
  ndarray of shape (n_samples,) per iterasi
446
529
  """
447
530
  for proba in self.staged_predict_proba(X):
448
- yield (proba[:, 1] >= 0.5).astype(int)
531
+ yield (proba[:, 1] >= self._threshold_fitted).astype(int)
449
532
 
450
533
  def apply(self, X):
451
534
  """
@@ -457,13 +540,9 @@ class BinBoost:
457
540
  """
458
541
  self._check_is_fitted()
459
542
  X, _ = self._validate_input(X)
460
- binarizer = GradientBinarizer(
461
- strategy=self.binarize_strategy,
462
- n_thresholds=self.n_thresholds
463
- )
464
543
  numeric_cols = self._detect_numeric_cols(X)
465
544
  dummy_grad = np.ones(X.shape[0])
466
- X_bin, _ = binarizer.transform(X, dummy_grad, numeric_cols)
545
+ X_bin, _ = self._binarizer.transform(X, dummy_grad, numeric_cols)
467
546
  results = []
468
547
  for rule in self.estimators_:
469
548
  results.append(rule.evaluate(X_bin))
@@ -492,12 +571,14 @@ class BinBoost:
492
571
  'beam_width': self.beam_width,
493
572
  'subsample': self.subsample,
494
573
  'min_samples_rule': self.min_samples_rule,
574
+ 'lambda0': self.lambda0,
495
575
  'binarize_strategy': self.binarize_strategy,
496
576
  'n_thresholds': self.n_thresholds,
497
577
  'n_iter_no_change': self.n_iter_no_change,
498
578
  'tol': self.tol,
499
579
  'random_state': self.random_state,
500
580
  'warm_start': self.warm_start,
581
+ 'threshold': self.threshold,
501
582
  }
502
583
 
503
584
  def set_params(self, **params):
@@ -517,7 +598,7 @@ class BinBoost:
517
598
  @property
518
599
  def rule_summary_(self):
519
600
  """
520
- Tabel ringkasan semua aturan beserta bobot, cakupan, dan dukungan.
601
+ Tabel ringkasan semua aturan beserta bobot dua sisi setiap aturan.
521
602
 
522
603
  Kembalian
523
604
  ----------
@@ -525,12 +606,14 @@ class BinBoost:
525
606
  """
526
607
  self._check_is_fitted()
527
608
  summary = []
528
- for i, rule in enumerate(self.estimators_):
609
+ for i, (rule, weights) in enumerate(zip(self.estimators_, self.rule_weights_)):
529
610
  summary.append({
530
611
  'iterasi': i + 1,
531
612
  'aturan': self.rules_[i],
532
- 'bobot': round(float(self.rule_weights_[i]), 6),
613
+ 'bobot_memenuhi': round(float(weights[1]), 6),
614
+ 'bobot_tidak_memenuhi': round(float(weights[0]), 6),
533
615
  'panjang_aturan': len(rule.features),
616
+ 'threshold_prediksi': round(self._threshold_fitted, 6),
534
617
  })
535
618
  return summary
536
619
 
@@ -565,4 +648,4 @@ class BinBoost:
565
648
  else f"X{f}"
566
649
  )
567
650
  usage[name] = usage.get(name, 0) + 1
568
- return usage
651
+ return usage
@@ -1,14 +1,12 @@
1
+ # loss.py
1
2
  import numpy as np
2
3
 
3
-
4
4
  def sigmoid(x):
5
5
  # Fungsi sigmoid dengan penjagaan numerik agar tidak overflow
6
6
  return np.where(x >= 0, 1 / (1 + np.exp(-x)), np.exp(x) / (1 + np.exp(x)))
7
7
 
8
-
9
8
  class LogisticLoss:
10
9
  # Loss biner standar berbasis log-likelihood
11
-
12
10
  def loss(self, y, F):
13
11
  p = sigmoid(F)
14
12
  p = np.clip(p, 1e-15, 1 - 1e-15)
@@ -18,16 +16,21 @@ class LogisticLoss:
18
16
  # 1. Gradien negatif sebagai arah penurunan loss
19
17
  p = sigmoid(F)
20
18
  return y - p
19
+
20
+ def hessian(self, y, F):
21
+ # Turunan kedua logistic loss terhadap F
22
+ p = sigmoid(F)
23
+ p = np.clip(p, 1e-15, 1 - 1e-15)
24
+ h = p * (1 - p)
25
+ return np.maximum(h, 1e-10)
21
26
 
22
27
  def init_F(self, y):
23
28
  # Inisialisasi F0 dari proporsi kelas positif
24
29
  p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
25
30
  return np.full(len(y), np.log(p / (1 - p)))
26
31
 
27
-
28
32
  class FocalLoss:
29
33
  # Focal Loss untuk data tidak seimbang
30
-
31
34
  def __init__(self, gamma=2.0, alpha=0.25):
32
35
  self.gamma = gamma
33
36
  self.alpha = alpha
@@ -50,12 +53,20 @@ class FocalLoss:
50
53
  weight = alpha_t * (1 - pt) ** self.gamma
51
54
  grad = weight * (y - p) + self.gamma * weight * pt * np.log(pt) * np.where(y == 1, -(1 - p), p)
52
55
  return grad
56
+
57
+ def hessian(self, y, F):
58
+ p = sigmoid(F)
59
+ p = np.clip(p, 1e-15, 1 - 1e-15)
60
+ pt = np.where(y == 1, p, 1 - p)
61
+ alpha_t = np.where(y == 1, self.alpha, 1 - self.alpha)
62
+ beta = alpha_t * (1 - pt) ** self.gamma
63
+ h = beta * p * (1 - p)
64
+ return np.maximum(h, 1e-10)
53
65
 
54
66
  def init_F(self, y):
55
67
  p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
56
68
  return np.full(len(y), np.log(p / (1 - p)))
57
69
 
58
-
59
70
  class PolyLoss:
60
71
  # PolyLoss yang berbasis ekspansi polinomial
61
72
  def __init__(self, epsilon=1.0):
@@ -78,12 +89,17 @@ class PolyLoss:
78
89
  # b. Suku koreksi dari ekspansi polinomial pertama
79
90
  grad_poly = self.epsilon * np.where(y == 1, p * (1 - p), -p * (1 - p))
80
91
  return grad_ce + grad_poly
92
+
93
+ def hessian(self, y, F):
94
+ p = sigmoid(F)
95
+ p = np.clip(p, 1e-15, 1 - 1e-15)
96
+ h = p * (1 - p) * (1 + self.epsilon * (2 * y - 1) * (2 * p - 1))
97
+ return np.maximum(h, 1e-10)
81
98
 
82
99
  def init_F(self, y):
83
100
  p = np.clip(np.mean(y), 1e-15, 1 - 1e-15)
84
101
  return np.full(len(y), np.log(p / (1 - p)))
85
102
 
86
-
87
103
  def get_loss(name, **kwargs):
88
104
  # Kembalikan objek loss sesuai nama yang dipilih pengguna
89
105
  if name == 'logistic':
@@ -0,0 +1,78 @@
1
+ # multiclass.py
2
+ import numpy as np
3
+ from binboost import BinBoost
4
+
5
+
6
+ class BinBoostOvR:
7
+ def __init__(self, mode='multiclass', **binboost_params):
8
+ self.mode = mode
9
+ self.binboost_params = binboost_params
10
+
11
+ def fit(self, X, y):
12
+ X = np.asarray(X, dtype=np.float64)
13
+ y = np.asarray(y)
14
+ self.classes_ = np.unique(y)
15
+ self.n_classes_ = len(self.classes_)
16
+ if self.n_classes_ < 2:
17
+ raise ValueError("Minimal harus ada 2 kelas berbeda pada label y.")
18
+
19
+ self.estimators_ovr_ = []
20
+ for kelas in self.classes_:
21
+ y_bin = (y == kelas).astype(np.float64)
22
+ model_k = BinBoost(**self.binboost_params)
23
+ model_k.fit(X, y_bin)
24
+ self.estimators_ovr_.append(model_k)
25
+ return self
26
+
27
+ def _raw_scores(self, X):
28
+ return np.column_stack([m.predict_proba(X)[:, 1] for m in self.estimators_ovr_])
29
+
30
+ def predict_proba(self, X):
31
+ raw = self._raw_scores(X)
32
+ if self.mode == 'multilabel':
33
+ # Tidak dinormalisasi — tiap kelas independen
34
+ return raw
35
+ # mode='multiclass' — normalisasi supaya tiap baris berjumlah 1
36
+ total = raw.sum(axis=1, keepdims=True)
37
+ semua_nol = (total.flatten() <= 1e-10)
38
+ total_aman = np.where(total <= 1e-10, 1.0, total)
39
+ probs = raw / total_aman
40
+ if semua_nol.any():
41
+ probs[semua_nol] = 1.0 / self.n_classes_
42
+ return probs
43
+
44
+ def predict(self, X):
45
+ if self.mode == 'multilabel':
46
+ # Tiap kelas pakai threshold Youden J miliknya sendiri
47
+ preds = np.column_stack([m.predict(X) for m in self.estimators_ovr_])
48
+ return preds
49
+ probs = self.predict_proba(X)
50
+ idx = np.argmax(probs, axis=1)
51
+ return self.classes_[idx]
52
+
53
+ def score(self, X, y):
54
+ y = np.asarray(y)
55
+ preds = self.predict(X)
56
+ return np.mean(preds == y)
57
+
58
+ @property
59
+ def rule_summary_(self):
60
+ rows = []
61
+ for kelas, model_k in zip(self.classes_, self.estimators_ovr_):
62
+ for r in model_k.rule_summary_:
63
+ r = dict(r)
64
+ r['kelas'] = kelas
65
+ rows.append(r)
66
+ return rows
67
+
68
+ @property
69
+ def feature_usage_(self):
70
+ usage_total = {}
71
+ for model_k in self.estimators_ovr_:
72
+ for k, v in model_k.feature_usage_.items():
73
+ usage_total[k] = usage_total.get(k, 0) + v
74
+ return usage_total
75
+
76
+ @property
77
+ def n_rules_(self):
78
+ return sum(m.n_rules_ for m in self.estimators_ovr_)
@@ -0,0 +1,217 @@
1
+ # rule.py
2
+ import numpy as np
3
+
4
+ class Rule:
5
+ # Satu aturan logika dalam ensemble BinBoost, dengan dukungan negasi (NOT) per fitur
6
+ def __init__(self, features, operators, weight, negations=None, feature_names=None):
7
+ self.features = features
8
+ self.operators = operators
9
+ self.weight = weight
10
+ # Daftar boolean sepanjang features: True berarti fitur itu dipakai dalam bentuk negasi (NOT)
11
+ self.negations = negations if negations is not None else [False] * len(features)
12
+ self.feature_names = feature_names
13
+
14
+ def _kolom(self, X_bin, idx_dalam_features):
15
+ fitur_idx = self.features[idx_dalam_features]
16
+ kolom = X_bin[:, fitur_idx].astype(bool)
17
+ if self.negations[idx_dalam_features]:
18
+ kolom = ~kolom
19
+ return kolom
20
+
21
+ def evaluate(self, X_bin):
22
+ result = self._kolom(X_bin, 0)
23
+ for i, op in enumerate(self.operators):
24
+ next_col = self._kolom(X_bin, i + 1)
25
+ if op == 'AND':
26
+ result = result & next_col
27
+ elif op == 'OR':
28
+ result = result | next_col
29
+ elif op == 'XOR':
30
+ result = result ^ next_col
31
+ return result.astype(np.float64)
32
+
33
+ def to_string(self, feature_names=None):
34
+ names = feature_names if feature_names is not None else self.feature_names
35
+ if names is None:
36
+ base_parts = [f"X{f}" for f in self.features]
37
+ else:
38
+ base_parts = [str(names[f]) for f in self.features]
39
+ parts = [f"NOT {p}" if neg else p for p, neg in zip(base_parts, self.negations)]
40
+ if len(parts) == 1:
41
+ return parts[0]
42
+ expr = parts[0]
43
+ for i, op in enumerate(self.operators):
44
+ expr = f"({expr} {op} {parts[i + 1]})"
45
+ return expr
46
+
47
+ def __repr__(self):
48
+ return f"Rule({self.to_string()}, weight={self.weight:.4f})"
49
+
50
+
51
+ class BeamSearchRuleFinder:
52
+ def __init__(self, operators, max_rule_length, beam_width,
53
+ feature_selection_threshold, min_samples_rule,
54
+ rule_complexity_penalty, ohe_groups=None, allow_negation=True):
55
+ self.operators = operators
56
+ self.max_rule_length = max_rule_length
57
+ self.beam_width = beam_width
58
+ self.feature_selection_threshold = feature_selection_threshold
59
+ self.min_samples_rule = min_samples_rule
60
+ self.rule_complexity_penalty = rule_complexity_penalty
61
+ self.ohe_groups = ohe_groups
62
+ self.allow_negation = allow_negation # False -> perilaku identik versi sebelum ini
63
+
64
+ self._feature_to_group = {}
65
+ if ohe_groups is not None:
66
+ for gid, members in ohe_groups.items():
67
+ for f in members:
68
+ self._feature_to_group[f] = gid
69
+
70
+ def _select_features(self, X_bin, gradients):
71
+ dot_products = np.abs(X_bin.T @ gradients)
72
+ col_sums = X_bin.sum(axis=0) + 1e-10
73
+ scores = dot_products / col_sums
74
+ selected = np.where(scores >= self.feature_selection_threshold)[0]
75
+ if len(selected) == 0:
76
+ selected = np.array([np.argmax(scores)])
77
+ return selected
78
+
79
+ def _compute_scores_batch(self, H_batch, gradients):
80
+ dot_gh = H_batch @ gradients
81
+ dot_hh = (H_batch * H_batch).sum(axis=1)
82
+ valid = dot_hh > 1e-10
83
+ w = np.where(valid, dot_gh / np.where(dot_hh > 1e-10, dot_hh, 1.0), 0.0)
84
+ residual = gradients[np.newaxis, :] - w[:, np.newaxis] * H_batch
85
+ scores = (residual * residual).sum(axis=1)
86
+ scores = np.where(valid, scores, np.inf)
87
+ return w, scores, valid
88
+
89
+ def _min_samples_int(self, n_samples):
90
+ if isinstance(self.min_samples_rule, float) and self.min_samples_rule < 1.0:
91
+ return max(1, int(self.min_samples_rule * n_samples))
92
+ return int(self.min_samples_rule)
93
+
94
+ def find_best_rule(self, X_bin, gradients):
95
+ n_samples, n_features = X_bin.shape
96
+ min_s = self._min_samples_int(n_samples)
97
+ if hasattr(self, 'forced_candidate_features') and self.forced_candidate_features is not None:
98
+ candidate_features = self.forced_candidate_features
99
+ else:
100
+ candidate_features = self._select_features(X_bin, gradients)
101
+
102
+ if len(candidate_features) == 0:
103
+ return None
104
+
105
+ X_bin_bool = X_bin.astype(bool)
106
+ polaritas_list = [False, True] if self.allow_negation else [False]
107
+
108
+ # 1. Inisialisasi beam panjang satu — coba dua polaritas (asli & negasi) per fitur
109
+ H_list, meta_list = [], []
110
+ for cf in candidate_features:
111
+ kolom_pos = X_bin_bool[:, int(cf)]
112
+ for neg in polaritas_list:
113
+ kolom = ~kolom_pos if neg else kolom_pos
114
+ H_list.append(kolom.astype(np.float64))
115
+ meta_list.append((int(cf), neg))
116
+
117
+ H_init = np.array(H_list)
118
+ support_init = H_init.sum(axis=1)
119
+ valid_init = support_init >= min_s
120
+ if not valid_init.any():
121
+ return None
122
+
123
+ w_init, scores_init, w_valid = self._compute_scores_batch(H_init, gradients)
124
+ valid_mask = valid_init & w_valid
125
+ if not valid_mask.any():
126
+ return None
127
+
128
+ candidates = []
129
+ for i, (cf, neg) in enumerate(meta_list):
130
+ if valid_mask[i]:
131
+ pen = scores_init[i] * (1 + self.rule_complexity_penalty * 1)
132
+ candidates.append((pen, [cf], [], [neg], H_init[i].astype(bool)))
133
+
134
+ if not candidates:
135
+ return None
136
+
137
+ candidates.sort(key=lambda x: x[0])
138
+ beam = candidates[:self.beam_width]
139
+ best_score, best_feats, best_ops, best_negs, best_h = beam[0]
140
+
141
+ # 2. Perluas aturan hingga panjang maksimum
142
+ for length in range(2, self.max_rule_length + 1):
143
+ new_candidates = []
144
+
145
+ for beam_score, beam_feats, beam_ops, beam_negs, beam_h in beam:
146
+ beam_feat_set = set(beam_feats)
147
+
148
+ expandable = []
149
+ for cf in candidate_features:
150
+ cf_int = int(cf)
151
+ if cf_int in beam_feat_set:
152
+ continue
153
+ g_new = self._feature_to_group.get(cf_int)
154
+ if g_new is not None:
155
+ if any(self._feature_to_group.get(ef) == g_new for ef in beam_feats):
156
+ continue
157
+ expandable.append(cf_int)
158
+
159
+ if not expandable:
160
+ continue
161
+
162
+ cols_pos = X_bin_bool[:, expandable].T
163
+ kandidat_kolom = [(cols_pos, False)]
164
+ if self.allow_negation:
165
+ kandidat_kolom.append((~cols_pos, True))
166
+
167
+ for cols, neg_flag in kandidat_kolom:
168
+ for op in self.operators:
169
+ beam_h_2d = np.broadcast_to(beam_h, cols.shape)
170
+ if op == 'AND':
171
+ H_new = (beam_h_2d & cols).astype(np.float64)
172
+ elif op == 'OR':
173
+ H_new = (beam_h_2d | cols).astype(np.float64)
174
+ elif op == 'XOR':
175
+ H_new = (beam_h_2d ^ cols).astype(np.float64)
176
+ else:
177
+ continue
178
+
179
+ support_new = H_new.sum(axis=1)
180
+ valid_sup = support_new >= min_s
181
+ if not valid_sup.any():
182
+ continue
183
+
184
+ w_new, scores_new, w_valid_new = self._compute_scores_batch(H_new, gradients)
185
+ valid_combined = valid_sup & w_valid_new
186
+
187
+ for j, cf_int in enumerate(expandable):
188
+ if not valid_combined[j]:
189
+ continue
190
+ pen = scores_new[j] * (1 + self.rule_complexity_penalty * length)
191
+ new_candidates.append((
192
+ pen,
193
+ beam_feats + [cf_int],
194
+ beam_ops + [op],
195
+ beam_negs + [neg_flag],
196
+ H_new[j].astype(bool)
197
+ ))
198
+
199
+ if not new_candidates:
200
+ break
201
+
202
+ new_candidates.sort(key=lambda x: x[0])
203
+ new_candidates = new_candidates[:self.beam_width]
204
+
205
+ if new_candidates[0][0] < best_score:
206
+ best_score, best_feats, best_ops, best_negs, best_h = new_candidates[0]
207
+ beam = new_candidates
208
+ else:
209
+ break
210
+
211
+ h_final = best_h.astype(np.float64)
212
+ denom = np.dot(h_final, h_final)
213
+ if denom < 1e-10:
214
+ return None
215
+ w_final = np.dot(gradients, h_final) / denom
216
+
217
+ return Rule(best_feats, best_ops, w_final, negations=best_negs)
@@ -1,11 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: binboost
3
- Version: 0.1.0
3
+ Version: 0.2.2
4
4
  Summary: Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner
5
- Home-page: https://github.com/username/binboost
6
- Author: BinBoost Authors
5
+ Author: Rangga Wahyu Pratama
7
6
  License: MIT
8
- Project-URL: Homepage, https://github.com/username/binboost
9
7
  Keywords: gradient boosting,logical rules,binary features,interpretable machine learning
10
8
  Classifier: Programming Language :: Python :: 3
11
9
  Classifier: License :: OSI Approved :: MIT License
@@ -18,7 +16,6 @@ License-File: LICENSE
18
16
  Requires-Dist: numpy>=1.21.0
19
17
  Requires-Dist: pandas>=1.3.0
20
18
  Dynamic: author
21
- Dynamic: home-page
22
19
  Dynamic: license-file
23
20
  Dynamic: requires-python
24
21
 
@@ -6,6 +6,7 @@ binboost/__init__.py
6
6
  binboost/binarizer.py
7
7
  binboost/binboost.py
8
8
  binboost/loss.py
9
+ binboost/multiclass.py
9
10
  binboost/rule.py
10
11
  binboost.egg-info/PKG-INFO
11
12
  binboost.egg-info/SOURCES.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "binboost"
7
- version = "0.1.0"
7
+ version = "0.2.2"
8
8
  description = "Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
@@ -21,6 +21,3 @@ classifiers = [
21
21
  "Intended Audience :: Science/Research",
22
22
  "Topic :: Scientific/Engineering :: Artificial Intelligence",
23
23
  ]
24
-
25
- [project.urls]
26
- Homepage = "https://github.com/username/binboost"
@@ -2,12 +2,11 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="binboost",
5
- version="0.1.0",
6
- author="BinBoost Authors",
5
+ version="0.2.2",
6
+ author="Rangga Wahyu Pratama",
7
7
  description="Gradient Boosting Berbasis Aturan Logika Adaptif untuk Fitur Biner",
8
8
  long_description=open("README.md", encoding="utf-8").read(),
9
9
  long_description_content_type="text/markdown",
10
- url="https://github.com/username/binboost",
11
10
  packages=find_packages(),
12
11
  python_requires=">=3.8",
13
12
  install_requires=[
@@ -1,5 +0,0 @@
1
- from .binboost import BinBoost
2
-
3
- __version__ = "0.1.0"
4
- __author__ = "BinBoost Authors"
5
- __all__ = ["BinBoost"]
@@ -1,168 +0,0 @@
1
- import numpy as np
2
- from itertools import combinations
3
-
4
-
5
- class Rule:
6
- # Satu aturan logika dalam ensemble BinBoost
7
-
8
- def __init__(self, features, operators, weight, feature_names=None):
9
- # 1. Simpan komponen aturan
10
- # a. Indeks fitur yang terlibat
11
- self.features = features
12
- # b. Daftar operator antar fitur
13
- self.operators = operators
14
- # c. Bobot optimal hasil perhitungan w_m*
15
- self.weight = weight
16
- self.feature_names = feature_names
17
-
18
- def evaluate(self, X_bin):
19
- # Hitung keluaran aturan untuk setiap sampel dan kembalikan larik biner
20
- result = X_bin[:, self.features[0]].astype(bool)
21
- for i, op in enumerate(self.operators):
22
- next_col = X_bin[:, self.features[i + 1]].astype(bool)
23
- if op == 'AND':
24
- result = result & next_col
25
- elif op == 'OR':
26
- result = result | next_col
27
- elif op == 'XOR':
28
- result = result ^ next_col
29
- return result.astype(np.float64)
30
-
31
- def to_string(self, feature_names=None):
32
- # Kembalikan representasi teks aturan yang dapat dibaca manusia
33
- names = feature_names if feature_names is not None else self.feature_names
34
- if names is None:
35
- parts = [f"X{f}" for f in self.features]
36
- else:
37
- parts = [names[f] for f in self.features]
38
- if len(parts) == 1:
39
- return parts[0]
40
- expr = parts[0]
41
- for i, op in enumerate(self.operators):
42
- expr = f"({expr} {op} {parts[i + 1]})"
43
- return expr
44
-
45
- def __repr__(self):
46
- return f"Rule({self.to_string()}, weight={self.weight:.4f})"
47
-
48
-
49
- class BeamSearchRuleFinder:
50
- # Pencari aturan terbaik menggunakan beam search berbasis gradien
51
-
52
- def __init__(self, operators, max_rule_length, beam_width,
53
- feature_selection_threshold, min_samples_rule,
54
- rule_complexity_penalty, ohe_groups=None):
55
- self.operators = operators
56
- self.max_rule_length = max_rule_length
57
- self.beam_width = beam_width
58
- self.feature_selection_threshold = feature_selection_threshold
59
- self.min_samples_rule = min_samples_rule
60
- self.rule_complexity_penalty = rule_complexity_penalty
61
- self.ohe_groups = ohe_groups
62
-
63
- def _select_features(self, X_bin, gradients):
64
- # 1. Pilih fitur kandidat berdasarkan skor korelasi gradien
65
- n_samples, n_features = X_bin.shape
66
- scores = []
67
- for f in range(n_features):
68
- col = X_bin[:, f]
69
- score = np.abs(np.dot(gradients, col)) / (np.sum(col) + 1e-10)
70
- scores.append(score)
71
- scores = np.array(scores)
72
- selected = np.where(scores >= self.feature_selection_threshold)[0]
73
- if len(selected) == 0:
74
- selected = np.array([np.argmax(scores)])
75
- return selected
76
-
77
- def _same_ohe_group(self, f1, f2):
78
- # Periksa apakah dua fitur berasal dari grup OneHotEncoding yang sama
79
- if self.ohe_groups is None:
80
- return False
81
- for group in self.ohe_groups.values():
82
- if f1 in group and f2 in group:
83
- return True
84
- return False
85
-
86
- def _compute_score(self, h, gradients):
87
- # 1. Hitung bobot optimal w_m* dan skor galat kuadrat terkecil
88
- denom = np.dot(h, h)
89
- if denom < 1e-10:
90
- return None, np.inf
91
- w = np.dot(gradients, h) / denom
92
- penalty = self.rule_complexity_penalty
93
- score = np.sum((gradients - w * h) ** 2)
94
- return w, score
95
-
96
- def _is_valid(self, h, n_samples):
97
- # Periksa apakah jumlah sampel yang memenuhi aturan mencukupi min_samples_rule
98
- support = int(np.sum(h))
99
- if isinstance(self.min_samples_rule, float) and self.min_samples_rule < 1.0:
100
- min_s = int(self.min_samples_rule * n_samples)
101
- else:
102
- min_s = int(self.min_samples_rule)
103
- return support >= min_s
104
-
105
- def find_best_rule(self, X_bin, gradients):
106
- # 1. Jalankan beam search untuk menemukan aturan terbaik
107
- n_samples = X_bin.shape[0]
108
- candidate_features = self._select_features(X_bin, gradients)
109
-
110
- # a. Inisialisasi beam dengan aturan panjang satu
111
- beam = []
112
- for f in candidate_features:
113
- h = X_bin[:, f].astype(np.float64)
114
- if not self._is_valid(h, n_samples):
115
- continue
116
- w, score = self._compute_score(h, gradients)
117
- if w is not None:
118
- beam.append(([f], [], w, score))
119
-
120
- if not beam:
121
- return None
122
-
123
- beam.sort(key=lambda x: x[3])
124
- beam = beam[:self.beam_width]
125
- best_candidate = beam[0]
126
-
127
- # b. Perluas aturan hingga panjang maksimum
128
- for length in range(2, self.max_rule_length + 1):
129
- new_beam = []
130
- for (feats, ops, w_prev, score_prev) in beam:
131
- for f in candidate_features:
132
- if f in feats:
133
- continue
134
- # c. Terapkan constraint mutual exclusivity OHE
135
- skip = False
136
- for existing_f in feats:
137
- if self._same_ohe_group(existing_f, f):
138
- skip = True
139
- break
140
- if skip:
141
- continue
142
- for op in self.operators:
143
- new_feats = feats + [f]
144
- new_ops = ops + [op]
145
- tmp_rule = Rule(new_feats, new_ops, 0.0)
146
- h = tmp_rule.evaluate(X_bin)
147
- if not self._is_valid(h, n_samples):
148
- continue
149
- w, score = self._compute_score(h, gradients)
150
- # d. Terapkan penalti kompleksitas berdasarkan panjang aturan
151
- penalized_score = score * (1 + self.rule_complexity_penalty * length)
152
- if w is not None:
153
- new_beam.append((new_feats, new_ops, w, penalized_score))
154
-
155
- if not new_beam:
156
- break
157
-
158
- new_beam.sort(key=lambda x: x[3])
159
- new_beam = new_beam[:self.beam_width]
160
-
161
- if new_beam[0][3] < best_candidate[3]:
162
- best_candidate = new_beam[0]
163
- beam = new_beam
164
- else:
165
- break
166
-
167
- feats, ops, w, score = best_candidate
168
- return Rule(feats, ops, w)
File without changes
File without changes