hgp-lib 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hgp_lib/__init__.py +3 -0
- hgp_lib/algorithms/__init__.py +3 -0
- hgp_lib/algorithms/boolean_gp.py +629 -0
- hgp_lib/benchmarkers/__init__.py +3 -0
- hgp_lib/benchmarkers/gp_benchmarker.py +198 -0
- hgp_lib/benchmarkers/progress.py +203 -0
- hgp_lib/benchmarkers/runner.py +182 -0
- hgp_lib/configs/__init__.py +12 -0
- hgp_lib/configs/benchmarker_config.py +150 -0
- hgp_lib/configs/boolean_gp_config.py +215 -0
- hgp_lib/configs/trainer_config.py +115 -0
- hgp_lib/crossover/__init__.py +4 -0
- hgp_lib/crossover/crossover_executor.py +191 -0
- hgp_lib/crossover/crossover_factory.py +124 -0
- hgp_lib/metrics/__init__.py +19 -0
- hgp_lib/metrics/core.py +161 -0
- hgp_lib/metrics/history.py +120 -0
- hgp_lib/metrics/results.py +447 -0
- hgp_lib/mutations/__init__.py +32 -0
- hgp_lib/mutations/base_mutation.py +72 -0
- hgp_lib/mutations/literal_mutations.py +318 -0
- hgp_lib/mutations/mutation_executor.py +176 -0
- hgp_lib/mutations/mutation_factory.py +199 -0
- hgp_lib/mutations/operator_mutations.py +218 -0
- hgp_lib/mutations/utils.py +3 -0
- hgp_lib/populations/__init__.py +24 -0
- hgp_lib/populations/base_strategy.py +26 -0
- hgp_lib/populations/generator.py +97 -0
- hgp_lib/populations/populations_factory.py +104 -0
- hgp_lib/populations/sampling.py +376 -0
- hgp_lib/populations/strategies.py +221 -0
- hgp_lib/preprocessing/__init__.py +4 -0
- hgp_lib/preprocessing/binarizer.py +418 -0
- hgp_lib/preprocessing/utils.py +68 -0
- hgp_lib/rules/__init__.py +13 -0
- hgp_lib/rules/literals.py +114 -0
- hgp_lib/rules/low_memory_operators.py +138 -0
- hgp_lib/rules/operators.py +170 -0
- hgp_lib/rules/rules.py +274 -0
- hgp_lib/rules/utils.py +229 -0
- hgp_lib/selections/__init__.py +5 -0
- hgp_lib/selections/base_selection.py +87 -0
- hgp_lib/selections/roulette_selection.py +106 -0
- hgp_lib/selections/tournament_selection.py +148 -0
- hgp_lib/trainers/__init__.py +3 -0
- hgp_lib/trainers/gp_trainer.py +151 -0
- hgp_lib/utils/__init__.py +5 -0
- hgp_lib/utils/metrics.py +244 -0
- hgp_lib/utils/validation.py +206 -0
- hgp_lib-0.0.1.dist-info/METADATA +180 -0
- hgp_lib-0.0.1.dist-info/RECORD +54 -0
- hgp_lib-0.0.1.dist-info/WHEEL +5 -0
- hgp_lib-0.0.1.dist-info/licenses/LICENSE +201 -0
- hgp_lib-0.0.1.dist-info/top_level.txt +1 -0
hgp_lib/__init__.py
ADDED
|
@@ -0,0 +1,629 @@
|
|
|
1
|
+
from dataclasses import replace
|
|
2
|
+
from typing import Callable, List
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
from hgp_lib.utils.metrics import confusion_matrix
|
|
6
|
+
from numpy import ndarray
|
|
7
|
+
|
|
8
|
+
from ..configs import BooleanGPConfig, validate_gp_config
|
|
9
|
+
from ..metrics import GenerationMetrics
|
|
10
|
+
from ..rules import Rule
|
|
11
|
+
from ..selections import TournamentSelection
|
|
12
|
+
from ..utils.metrics import optimize_scorers_for_data
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class BooleanGP:
|
|
16
|
+
"""
|
|
17
|
+
Boolean Genetic Programming algorithm for evolving rule-based classifiers.
|
|
18
|
+
|
|
19
|
+
This class implements a genetic programming algorithm that evolves a population of
|
|
20
|
+
boolean rules to optimize a fitness function. Each generation applies crossover and
|
|
21
|
+
mutation operations, evaluates the population, and selects the best individuals.
|
|
22
|
+
|
|
23
|
+
The algorithm tracks the current best rule (best in the current run). When enabled,
|
|
24
|
+
regeneration resets the population if no improvement is observed for a specified
|
|
25
|
+
number of epochs.
|
|
26
|
+
|
|
27
|
+
Training data and labels are provided via `BooleanGPConfig`. The number of features
|
|
28
|
+
(`num_features`) is derived from the data shape and passed to the configured
|
|
29
|
+
`population_factory` and `mutation_factory` for runtime construction of the
|
|
30
|
+
`PopulationGenerator` and `MutationExecutor`.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
config (BooleanGPConfig): Configuration containing `train_data`,
|
|
34
|
+
`train_labels`, `score_fn`, `population_factory`,
|
|
35
|
+
`mutation_factory`, and optional components.
|
|
36
|
+
|
|
37
|
+
Examples:
|
|
38
|
+
>>> import numpy as np
|
|
39
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
40
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
41
|
+
>>>
|
|
42
|
+
>>> def accuracy(predictions, labels):
|
|
43
|
+
... return np.mean(predictions == labels)
|
|
44
|
+
>>>
|
|
45
|
+
>>> train_data = np.array([[True, False, True, False], [False, True, False, True]])
|
|
46
|
+
>>> train_labels = np.array([1, 0])
|
|
47
|
+
>>> config = BooleanGPConfig(
|
|
48
|
+
... score_fn=accuracy,
|
|
49
|
+
... train_data=train_data,
|
|
50
|
+
... train_labels=train_labels,
|
|
51
|
+
... optimize_scorer=False,
|
|
52
|
+
... )
|
|
53
|
+
>>> gp = BooleanGP(config)
|
|
54
|
+
>>> gen_metrics = gp.step()
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def __init__(self, config: BooleanGPConfig, current_depth: int = 0):
|
|
58
|
+
validate_gp_config(config)
|
|
59
|
+
|
|
60
|
+
train_data = config.train_data
|
|
61
|
+
train_labels = config.train_labels
|
|
62
|
+
# TODO: We should add in documentation that our score_fn follows the sklearn
|
|
63
|
+
# standard of (predictions, labels) and sample_weight support is recommended for optimization.
|
|
64
|
+
# Careful! the sklearn pattern is labels, predictions!
|
|
65
|
+
|
|
66
|
+
score_fn = config.score_fn
|
|
67
|
+
self._original_score_fn = score_fn
|
|
68
|
+
|
|
69
|
+
if config.optimize_scorer:
|
|
70
|
+
score_fn, train_cm, train_data, train_labels = optimize_scorers_for_data(
|
|
71
|
+
config.score_fn,
|
|
72
|
+
confusion_matrix,
|
|
73
|
+
data=config.train_data,
|
|
74
|
+
labels=config.train_labels,
|
|
75
|
+
)
|
|
76
|
+
else:
|
|
77
|
+
train_cm = confusion_matrix
|
|
78
|
+
|
|
79
|
+
self.score_fn = score_fn
|
|
80
|
+
self.train_cm = train_cm
|
|
81
|
+
self.complexity_penalty = config.complexity_penalty
|
|
82
|
+
self.train_data = train_data
|
|
83
|
+
self.train_labels = train_labels
|
|
84
|
+
|
|
85
|
+
self.current_depth = current_depth
|
|
86
|
+
num_features = train_data.shape[1]
|
|
87
|
+
|
|
88
|
+
self.population_generator = config.population_factory.create(
|
|
89
|
+
num_features, score_fn, train_data, train_labels
|
|
90
|
+
)
|
|
91
|
+
self.mutation_executor = config.mutation_factory.create(
|
|
92
|
+
num_features, config.check_valid
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
self.crossover_executor = config.crossover_factory.create(config.check_valid)
|
|
96
|
+
|
|
97
|
+
selection = config.selection
|
|
98
|
+
if selection is None:
|
|
99
|
+
selection = TournamentSelection()
|
|
100
|
+
|
|
101
|
+
self.selection = selection
|
|
102
|
+
self.regeneration = config.regeneration
|
|
103
|
+
self.regeneration_patience = config.regeneration_patience
|
|
104
|
+
|
|
105
|
+
self.population = self.population_generator.generate()
|
|
106
|
+
self.population_size = len(self.population)
|
|
107
|
+
self._top_k = config.top_k_transfer
|
|
108
|
+
|
|
109
|
+
self.best_score = -float("inf")
|
|
110
|
+
self.global_best_score = -float("inf")
|
|
111
|
+
self.best_rule: Rule | None = None
|
|
112
|
+
self.global_best_rule: Rule | None = None
|
|
113
|
+
self.best_not_improved_epochs = 0
|
|
114
|
+
self._epoch = -1
|
|
115
|
+
|
|
116
|
+
self.config = config
|
|
117
|
+
self.child_populations: List["BooleanGP"] = []
|
|
118
|
+
self.feature_mapping: dict[int, int] | None = None
|
|
119
|
+
self._transfer_size: int = 0
|
|
120
|
+
self.parent_rule_indices: List[int] = []
|
|
121
|
+
|
|
122
|
+
if config.max_depth > current_depth:
|
|
123
|
+
self._create_child_populations()
|
|
124
|
+
|
|
125
|
+
def _create_child_populations(self) -> None:
|
|
126
|
+
"""Create child populations using the sampling strategy.
|
|
127
|
+
|
|
128
|
+
Calls sample() once for all children to ensure correct overlap/partitioning
|
|
129
|
+
behavior controlled by the replace parameter. Each SamplingResult in the
|
|
130
|
+
returned list is used to configure one child population.
|
|
131
|
+
"""
|
|
132
|
+
results = self.config.sampling_strategy.sample(
|
|
133
|
+
self.train_data,
|
|
134
|
+
self.train_labels,
|
|
135
|
+
self.config.num_child_populations,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
for result in results:
|
|
139
|
+
child_config = replace(
|
|
140
|
+
self.config,
|
|
141
|
+
train_data=result.data,
|
|
142
|
+
train_labels=result.labels,
|
|
143
|
+
)
|
|
144
|
+
child = BooleanGP(child_config, current_depth=self.current_depth + 1)
|
|
145
|
+
child.feature_mapping = result.feature_mapping
|
|
146
|
+
|
|
147
|
+
self.child_populations.append(child)
|
|
148
|
+
|
|
149
|
+
def step(self) -> GenerationMetrics:
|
|
150
|
+
"""
|
|
151
|
+
Perform one training step (generation) of the genetic programming algorithm.
|
|
152
|
+
|
|
153
|
+
Each step applies crossover to create offspring, mutates the population, evaluates
|
|
154
|
+
all rules, updates the best rule, and selects individuals for the next generation.
|
|
155
|
+
If regeneration is enabled and no improvement has been observed for
|
|
156
|
+
``regeneration_patience`` epochs, the population is regenerated.
|
|
157
|
+
|
|
158
|
+
Returns:
|
|
159
|
+
GenerationMetrics: Metrics for this generation including rules, scores, etc.
|
|
160
|
+
|
|
161
|
+
Examples:
|
|
162
|
+
>>> import numpy as np
|
|
163
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
164
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
165
|
+
>>> def accuracy(predictions, labels):
|
|
166
|
+
... return np.mean(predictions == labels)
|
|
167
|
+
>>> data = np.array([[True, False], [False, True], [True, True], [False, False]])
|
|
168
|
+
>>> labels = np.array([1, 0, 1, 0])
|
|
169
|
+
>>> config = BooleanGPConfig(
|
|
170
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
171
|
+
... optimize_scorer=False,
|
|
172
|
+
... )
|
|
173
|
+
>>> gp = BooleanGP(config)
|
|
174
|
+
>>> metrics = gp.step()
|
|
175
|
+
>>> isinstance(metrics.best_train_score, float)
|
|
176
|
+
True
|
|
177
|
+
>>> len(metrics.train_scores) > 0
|
|
178
|
+
True
|
|
179
|
+
"""
|
|
180
|
+
self._forward()
|
|
181
|
+
return self._backward()
|
|
182
|
+
|
|
183
|
+
def _forward(self) -> None:
|
|
184
|
+
"""
|
|
185
|
+
Perform the forward pass: collect child rules, apply crossover, then mutate.
|
|
186
|
+
|
|
187
|
+
Recursively calls ``_forward()`` on all child populations, collects their top-K
|
|
188
|
+
rules into a crossover pool alongside the current population, produces offspring
|
|
189
|
+
via crossover, and applies mutations to the expanded population.
|
|
190
|
+
|
|
191
|
+
Examples:
|
|
192
|
+
>>> import numpy as np
|
|
193
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
194
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
195
|
+
>>> def accuracy(p, l): return np.mean(p == l)
|
|
196
|
+
>>> data = np.array([[True, False], [False, True], [True, True], [False, False]])
|
|
197
|
+
>>> labels = np.array([1, 0, 1, 0])
|
|
198
|
+
>>> config = BooleanGPConfig(
|
|
199
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
200
|
+
... optimize_scorer=False,
|
|
201
|
+
... )
|
|
202
|
+
>>> gp = BooleanGP(config)
|
|
203
|
+
>>> pop_before = len(gp.population)
|
|
204
|
+
>>> gp._forward()
|
|
205
|
+
>>> len(gp.population) >= pop_before
|
|
206
|
+
True
|
|
207
|
+
"""
|
|
208
|
+
for child in self.child_populations:
|
|
209
|
+
child._forward()
|
|
210
|
+
|
|
211
|
+
crossover_pool = []
|
|
212
|
+
feature_mappings = []
|
|
213
|
+
self._transfer_size = 0
|
|
214
|
+
|
|
215
|
+
for child in self.child_populations:
|
|
216
|
+
crossover_pool.extend(child._get_top_k_rules())
|
|
217
|
+
self._transfer_size += child._top_k
|
|
218
|
+
feature_mappings.extend([child.feature_mapping] * child._top_k)
|
|
219
|
+
|
|
220
|
+
crossover_pool.extend(self.population)
|
|
221
|
+
feature_mappings.extend([None] * len(self.population))
|
|
222
|
+
offspring, self.parent_rule_indices = self.crossover_executor.apply(
|
|
223
|
+
crossover_pool, feature_mappings
|
|
224
|
+
)
|
|
225
|
+
self.population += offspring
|
|
226
|
+
self.mutation_executor.apply(self.population)
|
|
227
|
+
|
|
228
|
+
def _backward(self, parent_scores: ndarray | None = None) -> GenerationMetrics:
|
|
229
|
+
"""
|
|
230
|
+
Perform the backward pass: evaluate, propagate feedback, and select the next generation.
|
|
231
|
+
|
|
232
|
+
Evaluates all rules against training data, optionally incorporates feedback from a
|
|
233
|
+
parent population, propagates signals to child populations, and creates the next
|
|
234
|
+
generation via selection.
|
|
235
|
+
|
|
236
|
+
Args:
|
|
237
|
+
parent_scores (ndarray | None):
|
|
238
|
+
Optional scores propagated from a parent population. Used in hierarchical GP
|
|
239
|
+
to reward child rules that contributed to successful offspring. Default: `None`.
|
|
240
|
+
|
|
241
|
+
Returns:
|
|
242
|
+
GenerationMetrics: Metrics for this generation.
|
|
243
|
+
|
|
244
|
+
Examples:
|
|
245
|
+
>>> import numpy as np
|
|
246
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
247
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
248
|
+
>>> def accuracy(p, l): return np.mean(p == l)
|
|
249
|
+
>>> data = np.array([[True, False], [False, True], [True, True], [False, False]])
|
|
250
|
+
>>> labels = np.array([1, 0, 1, 0])
|
|
251
|
+
>>> config = BooleanGPConfig(
|
|
252
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
253
|
+
... optimize_scorer=False,
|
|
254
|
+
... )
|
|
255
|
+
>>> gp = BooleanGP(config)
|
|
256
|
+
>>> gp._forward()
|
|
257
|
+
>>> metrics = gp._backward()
|
|
258
|
+
>>> isinstance(metrics.best_train_score, float)
|
|
259
|
+
True
|
|
260
|
+
"""
|
|
261
|
+
scores = self.evaluate_population(
|
|
262
|
+
self.train_data, self.train_labels, self.score_fn
|
|
263
|
+
)
|
|
264
|
+
if parent_scores is not None:
|
|
265
|
+
self._apply_feedback(scores, parent_scores)
|
|
266
|
+
|
|
267
|
+
if self.child_populations:
|
|
268
|
+
child_feedbacks = self._generate_child_feedback(scores)
|
|
269
|
+
children_metrics = [
|
|
270
|
+
child._backward(feedback)
|
|
271
|
+
for child, feedback in zip(self.child_populations, child_feedbacks)
|
|
272
|
+
]
|
|
273
|
+
else:
|
|
274
|
+
children_metrics = []
|
|
275
|
+
|
|
276
|
+
return self._new_generation(scores, children_metrics)
|
|
277
|
+
|
|
278
|
+
def _get_top_k_rules(self) -> List[Rule]:
|
|
279
|
+
"""Retrieve the top-K rules for transfer to the parent population.
|
|
280
|
+
|
|
281
|
+
Returns the first `_top_k` rules from the population, which are expected
|
|
282
|
+
to be the highest-scoring rules. After each generation (except the first),
|
|
283
|
+
non-root populations sort their rules by score in descending order during
|
|
284
|
+
`_new_generation()`, ensuring that indices 0 through `_top_k - 1` contain
|
|
285
|
+
the best-performing rules.
|
|
286
|
+
|
|
287
|
+
Note:
|
|
288
|
+
During the first epoch (before any `backward()` call completes), the
|
|
289
|
+
population has not yet been sorted by score. In this case, the returned
|
|
290
|
+
rules are simply the first `_top_k` rules from the initial population,
|
|
291
|
+
which are effectively random. This is acceptable since all populations
|
|
292
|
+
start with randomly generated rules.
|
|
293
|
+
|
|
294
|
+
Returns:
|
|
295
|
+
List[Rule]: The top-K rules from this population, to be used in the
|
|
296
|
+
parent's crossover pool during hierarchical GP.
|
|
297
|
+
|
|
298
|
+
Examples:
|
|
299
|
+
>>> import numpy as np
|
|
300
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
301
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
302
|
+
>>> from hgp_lib.populations import FeatureSamplingStrategy
|
|
303
|
+
>>> from sklearn.metrics import accuracy_score
|
|
304
|
+
>>> data = np.random.rand(50, 10) > 0.5
|
|
305
|
+
>>> labels = np.random.randint(0, 2, 50)
|
|
306
|
+
>>> config = BooleanGPConfig(
|
|
307
|
+
... score_fn=accuracy_score,
|
|
308
|
+
... train_data=data,
|
|
309
|
+
... train_labels=labels,
|
|
310
|
+
... max_depth=1,
|
|
311
|
+
... num_child_populations=2,
|
|
312
|
+
... sampling_strategy=FeatureSamplingStrategy(),
|
|
313
|
+
... top_k_transfer=5,
|
|
314
|
+
... )
|
|
315
|
+
>>> gp = BooleanGP(config)
|
|
316
|
+
>>> child = gp.child_populations[0]
|
|
317
|
+
>>> top_rules = child._get_top_k_rules()
|
|
318
|
+
>>> len(top_rules)
|
|
319
|
+
5
|
|
320
|
+
"""
|
|
321
|
+
return self.population[: self._top_k]
|
|
322
|
+
|
|
323
|
+
def _apply_feedback(self, scores: ndarray, parent_scores: ndarray) -> ndarray:
|
|
324
|
+
"""Apply incoming feedback from parent to the first self._top_k scores."""
|
|
325
|
+
parent_scores *= self.config.feedback_strength
|
|
326
|
+
if self.config.feedback_type == "multiplicative":
|
|
327
|
+
scores[: self._top_k] *= 1 + parent_scores
|
|
328
|
+
else:
|
|
329
|
+
scores[: self._top_k] += parent_scores
|
|
330
|
+
return scores
|
|
331
|
+
|
|
332
|
+
def _generate_child_feedback(self, scores: ndarray) -> ndarray:
|
|
333
|
+
"""Generate feedback signals for each child from offspring performance.
|
|
334
|
+
|
|
335
|
+
Each parent that came from a child population receives a signal based on
|
|
336
|
+
how well the offspring it helped create performed. The signal is normalized
|
|
337
|
+
relative to the population's score distribution.
|
|
338
|
+
|
|
339
|
+
Args:
|
|
340
|
+
scores (ndarray): Fitness scores for all rules in the current population,
|
|
341
|
+
including both the original population and offspring from crossover.
|
|
342
|
+
|
|
343
|
+
Returns:
|
|
344
|
+
ndarray: A 2D array of shape (num_children, top_k) containing feedback
|
|
345
|
+
signals for each child population's transferred rules.
|
|
346
|
+
"""
|
|
347
|
+
num_children = len(self.child_populations)
|
|
348
|
+
|
|
349
|
+
mean_s = float(np.mean(scores))
|
|
350
|
+
min_s = float(np.min(scores))
|
|
351
|
+
max_s = float(np.max(scores))
|
|
352
|
+
|
|
353
|
+
if max_s == min_s:
|
|
354
|
+
return np.zeros((num_children, self._top_k))
|
|
355
|
+
|
|
356
|
+
# Offspring are situated after the original parent population in the scores array
|
|
357
|
+
offspring_scores = scores[self.population_size :]
|
|
358
|
+
above = offspring_scores >= mean_s
|
|
359
|
+
signals = np.where(
|
|
360
|
+
above,
|
|
361
|
+
(offspring_scores - mean_s) / (max_s - mean_s),
|
|
362
|
+
(offspring_scores - mean_s) / (mean_s - min_s),
|
|
363
|
+
)
|
|
364
|
+
child_feedbacks = np.zeros((num_children, self._top_k))
|
|
365
|
+
child_counts = np.zeros((num_children, self._top_k))
|
|
366
|
+
|
|
367
|
+
# parent_rule_indices stores TWO indices per offspring (one for each parent).
|
|
368
|
+
# For offspring i, parent_rule_indices[2*i] and parent_rule_indices[2*i + 1]
|
|
369
|
+
# are the indices of its two parents in the crossover pool.
|
|
370
|
+
# Therefore, j // 2 maps from the parent_rule_indices position to the
|
|
371
|
+
# corresponding offspring index in offspring_scores.
|
|
372
|
+
for j, parent_idx in enumerate(self.parent_rule_indices):
|
|
373
|
+
if parent_idx < self._transfer_size:
|
|
374
|
+
# parent_idx < _transfer_size means this parent came from a child population
|
|
375
|
+
# Determine which child and which local rule index within that child
|
|
376
|
+
child_idx, local_idx = divmod(parent_idx, self._top_k)
|
|
377
|
+
offspring_idx = j // 2 # Two parent indices per offspring
|
|
378
|
+
child_feedbacks[child_idx, local_idx] += signals[offspring_idx]
|
|
379
|
+
child_counts[child_idx, local_idx] += 1
|
|
380
|
+
|
|
381
|
+
mask = child_counts > 0
|
|
382
|
+
child_feedbacks[mask] /= child_counts[mask]
|
|
383
|
+
|
|
384
|
+
return child_feedbacks
|
|
385
|
+
|
|
386
|
+
def _compute_regularized_scores(
|
|
387
|
+
self, scores: ndarray, complexities: List[int]
|
|
388
|
+
) -> ndarray:
|
|
389
|
+
"""
|
|
390
|
+
Compute regularized scores: ``score - complexity_penalty * ln(complexity)``.
|
|
391
|
+
|
|
392
|
+
When ``complexity_penalty`` is zero, returns ``scores`` unchanged.
|
|
393
|
+
|
|
394
|
+
Args:
|
|
395
|
+
scores (ndarray):
|
|
396
|
+
Raw fitness scores for the population.
|
|
397
|
+
complexities (List[int]):
|
|
398
|
+
Number of nodes in each rule.
|
|
399
|
+
|
|
400
|
+
Returns:
|
|
401
|
+
ndarray: Regularized scores for selection.
|
|
402
|
+
|
|
403
|
+
Examples:
|
|
404
|
+
>>> import numpy as np
|
|
405
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
406
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
407
|
+
>>> def accuracy(p, l): return np.mean(p == l)
|
|
408
|
+
>>> data = np.array([[True, False], [False, True]])
|
|
409
|
+
>>> labels = np.array([1, 0])
|
|
410
|
+
>>> config = BooleanGPConfig(
|
|
411
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
412
|
+
... optimize_scorer=False,
|
|
413
|
+
... )
|
|
414
|
+
>>> gp = BooleanGP(config)
|
|
415
|
+
>>> gp.complexity_penalty = 0.0
|
|
416
|
+
>>> scores = np.array([0.9, 0.8])
|
|
417
|
+
>>> np.allclose(gp._compute_regularized_scores(scores, [3, 5]), scores)
|
|
418
|
+
True
|
|
419
|
+
>>> gp.complexity_penalty = 0.1
|
|
420
|
+
>>> reg = gp._compute_regularized_scores(np.array([1.0, 1.0]), [3, 5])
|
|
421
|
+
>>> bool(reg[0] > reg[1])
|
|
422
|
+
True
|
|
423
|
+
"""
|
|
424
|
+
if self.complexity_penalty == 0:
|
|
425
|
+
return scores
|
|
426
|
+
return scores - self.complexity_penalty * np.log(complexities)
|
|
427
|
+
|
|
428
|
+
def _new_generation(
|
|
429
|
+
self, scores: ndarray, children_metrics: List[GenerationMetrics]
|
|
430
|
+
) -> GenerationMetrics:
|
|
431
|
+
"""
|
|
432
|
+
Creates a new generation by selecting individuals and optionally regenerating the population.
|
|
433
|
+
|
|
434
|
+
Updates the best rule tracking, checks if regeneration is needed, and selects the next
|
|
435
|
+
generation using the configured selection strategy. If regeneration is triggered,
|
|
436
|
+
the population is completely regenerated and best tracking is reset.
|
|
437
|
+
|
|
438
|
+
Args:
|
|
439
|
+
scores (ndarray):
|
|
440
|
+
Fitness scores for all rules in the current population. Must have the same
|
|
441
|
+
length as `self.population`.
|
|
442
|
+
|
|
443
|
+
Returns:
|
|
444
|
+
GenerationMetrics: Metrics about the generation step.
|
|
445
|
+
"""
|
|
446
|
+
best_idx = int(np.argmax(scores))
|
|
447
|
+
current_best = float(scores[best_idx])
|
|
448
|
+
current_best_rule = self.population[best_idx].copy()
|
|
449
|
+
|
|
450
|
+
self._update_best(current_best, current_best_rule)
|
|
451
|
+
|
|
452
|
+
regenerated = False
|
|
453
|
+
if (
|
|
454
|
+
self.regeneration
|
|
455
|
+
and self.best_not_improved_epochs >= self.regeneration_patience
|
|
456
|
+
):
|
|
457
|
+
regenerated = True
|
|
458
|
+
|
|
459
|
+
self._epoch += 1
|
|
460
|
+
|
|
461
|
+
# Create GenerationMetrics
|
|
462
|
+
complexities = [len(rule) for rule in self.population]
|
|
463
|
+
|
|
464
|
+
metrics = GenerationMetrics.from_population(
|
|
465
|
+
best_idx=best_idx,
|
|
466
|
+
best_rule=current_best_rule,
|
|
467
|
+
train_scores=scores.tolist(),
|
|
468
|
+
complexities=complexities,
|
|
469
|
+
child_population_generation_metrics=children_metrics,
|
|
470
|
+
)
|
|
471
|
+
|
|
472
|
+
if regenerated:
|
|
473
|
+
self.population = self.population_generator.generate()
|
|
474
|
+
self.best_score = -float("inf")
|
|
475
|
+
self.best_not_improved_epochs = 0
|
|
476
|
+
else:
|
|
477
|
+
regularized_scores = self._compute_regularized_scores(scores, complexities)
|
|
478
|
+
self.population, selected_scores = self.selection.select(
|
|
479
|
+
self.population, regularized_scores, self.population_size
|
|
480
|
+
)
|
|
481
|
+
# Non-root populations need reordering so top-K rules are at the front
|
|
482
|
+
# for transfer to parent population during the next forward pass.
|
|
483
|
+
if self.current_depth > 0 and self._top_k < self.population_size:
|
|
484
|
+
# top_k must be positive if current_depth > 0
|
|
485
|
+
sorted_indices = np.argpartition(-selected_scores, self._top_k)
|
|
486
|
+
self.population = [self.population[i] for i in sorted_indices]
|
|
487
|
+
|
|
488
|
+
return metrics
|
|
489
|
+
|
|
490
|
+
def evaluate_population(
|
|
491
|
+
self,
|
|
492
|
+
data: ndarray,
|
|
493
|
+
labels: ndarray,
|
|
494
|
+
score_fn: Callable[[ndarray, ndarray], float],
|
|
495
|
+
) -> ndarray:
|
|
496
|
+
"""
|
|
497
|
+
Evaluate all rules in the population against the given data.
|
|
498
|
+
|
|
499
|
+
Args:
|
|
500
|
+
data (ndarray):
|
|
501
|
+
Data to evaluate rules on (2D boolean array).
|
|
502
|
+
labels (ndarray):
|
|
503
|
+
True labels (1D integer array).
|
|
504
|
+
score_fn (Callable[[ndarray, ndarray], float]):
|
|
505
|
+
Function to compute fitness scores.
|
|
506
|
+
|
|
507
|
+
Returns:
|
|
508
|
+
ndarray: Array of fitness scores, one for each rule in the population.
|
|
509
|
+
|
|
510
|
+
Examples:
|
|
511
|
+
>>> import numpy as np
|
|
512
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
513
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
514
|
+
>>> def accuracy(predictions, labels):
|
|
515
|
+
... return np.mean(predictions == labels)
|
|
516
|
+
>>> data = np.array([[True, False], [False, True], [True, True], [False, False]])
|
|
517
|
+
>>> labels = np.array([1, 0, 1, 0])
|
|
518
|
+
>>> config = BooleanGPConfig(
|
|
519
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
520
|
+
... optimize_scorer=False,
|
|
521
|
+
... )
|
|
522
|
+
>>> gp = BooleanGP(config)
|
|
523
|
+
>>> scores = gp.evaluate_population(data, labels, accuracy)
|
|
524
|
+
>>> len(scores) == len(gp.population)
|
|
525
|
+
True
|
|
526
|
+
>>> all(0.0 <= s <= 1.0 for s in scores)
|
|
527
|
+
True
|
|
528
|
+
"""
|
|
529
|
+
return np.array(
|
|
530
|
+
[score_fn(rule.evaluate(data), labels) for rule in self.population]
|
|
531
|
+
)
|
|
532
|
+
|
|
533
|
+
def _update_best(self, current_best: float, current_best_rule: Rule):
|
|
534
|
+
"""
|
|
535
|
+
Update the best rule tracking based on the current generation's best.
|
|
536
|
+
|
|
537
|
+
If ``current_best`` is greater than or equal to the stored best score, resets the
|
|
538
|
+
stagnation counter and stores the new best. Otherwise, increments the counter.
|
|
539
|
+
|
|
540
|
+
Args:
|
|
541
|
+
current_best (float):
|
|
542
|
+
The best fitness score from the current generation.
|
|
543
|
+
current_best_rule (Rule):
|
|
544
|
+
The rule that achieved the best score.
|
|
545
|
+
|
|
546
|
+
Examples:
|
|
547
|
+
>>> import numpy as np
|
|
548
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
549
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
550
|
+
>>> from hgp_lib.rules import Literal
|
|
551
|
+
>>> def accuracy(p, l): return np.mean(p == l)
|
|
552
|
+
>>> data = np.array([[True, False], [False, True]])
|
|
553
|
+
>>> labels = np.array([1, 0])
|
|
554
|
+
>>> config = BooleanGPConfig(
|
|
555
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
556
|
+
... optimize_scorer=False,
|
|
557
|
+
... )
|
|
558
|
+
>>> gp = BooleanGP(config)
|
|
559
|
+
>>> gp._update_best(0.8, Literal(value=0))
|
|
560
|
+
>>> gp.best_score
|
|
561
|
+
0.8
|
|
562
|
+
>>> gp._update_best(0.5, Literal(value=1))
|
|
563
|
+
>>> gp.best_score
|
|
564
|
+
0.8
|
|
565
|
+
>>> gp.best_not_improved_epochs
|
|
566
|
+
1
|
|
567
|
+
"""
|
|
568
|
+
if current_best >= self.best_score:
|
|
569
|
+
self.best_not_improved_epochs = 0
|
|
570
|
+
self.best_score = current_best
|
|
571
|
+
self.best_rule = current_best_rule
|
|
572
|
+
if current_best >= self.global_best_score:
|
|
573
|
+
self.global_best_score = current_best
|
|
574
|
+
self.global_best_rule = current_best_rule
|
|
575
|
+
else:
|
|
576
|
+
self.best_not_improved_epochs += 1
|
|
577
|
+
|
|
578
|
+
def evaluate_best(
|
|
579
|
+
self,
|
|
580
|
+
data: ndarray,
|
|
581
|
+
labels: ndarray,
|
|
582
|
+
score_fn: Callable[[ndarray, ndarray], float] | None = None,
|
|
583
|
+
) -> float:
|
|
584
|
+
"""
|
|
585
|
+
Evaluate the global best rule on validation or test data.
|
|
586
|
+
|
|
587
|
+
Args:
|
|
588
|
+
data (ndarray):
|
|
589
|
+
Validation/test data (2D boolean array).
|
|
590
|
+
labels (ndarray):
|
|
591
|
+
Validation/test labels (1D integer array).
|
|
592
|
+
score_fn (Callable[[ndarray, ndarray], float] | None):
|
|
593
|
+
Optional scoring function. Uses the original (non-optimized) scorer when
|
|
594
|
+
``None``, since the optimized scorer has ``sample_weight`` bound to training
|
|
595
|
+
data. Default: `None`.
|
|
596
|
+
|
|
597
|
+
Returns:
|
|
598
|
+
float: Score of the global best rule on the provided data.
|
|
599
|
+
|
|
600
|
+
Raises:
|
|
601
|
+
RuntimeError: If no best rule is available (run at least one step first).
|
|
602
|
+
|
|
603
|
+
Examples:
|
|
604
|
+
>>> import numpy as np
|
|
605
|
+
>>> from hgp_lib.configs import BooleanGPConfig
|
|
606
|
+
>>> from hgp_lib.algorithms import BooleanGP
|
|
607
|
+
>>> def accuracy(predictions, labels):
|
|
608
|
+
... return np.mean(predictions == labels)
|
|
609
|
+
>>> data = np.array([[True, False], [False, True], [True, True], [False, False]])
|
|
610
|
+
>>> labels = np.array([1, 0, 1, 0])
|
|
611
|
+
>>> config = BooleanGPConfig(
|
|
612
|
+
... score_fn=accuracy, train_data=data, train_labels=labels,
|
|
613
|
+
... optimize_scorer=False,
|
|
614
|
+
... )
|
|
615
|
+
>>> gp = BooleanGP(config)
|
|
616
|
+
>>> _ = gp.step()
|
|
617
|
+
>>> score = gp.evaluate_best(data, labels)
|
|
618
|
+
>>> isinstance(score, float)
|
|
619
|
+
True
|
|
620
|
+
"""
|
|
621
|
+
if self.global_best_rule is None:
|
|
622
|
+
raise RuntimeError("No best rule available. Run at least one step first.")
|
|
623
|
+
|
|
624
|
+
fn = self._original_score_fn if score_fn is None else score_fn
|
|
625
|
+
return float(fn(self.global_best_rule.evaluate(data), labels))
|
|
626
|
+
|
|
627
|
+
@property
|
|
628
|
+
def original_score_fn(self):
|
|
629
|
+
return self._original_score_fn
|