hdlib 2.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hdlib/__init__.py +16 -0
- hdlib/arithmetic/__init__.py +265 -0
- hdlib/arithmetic/quantum.py +1257 -0
- hdlib/model/__init__.py +12 -0
- hdlib/model/classification.py +1719 -0
- hdlib/model/clustering.py +172 -0
- hdlib/model/graph.py +764 -0
- hdlib/model/regression.py +306 -0
- hdlib/space.py +864 -0
- hdlib/vector.py +608 -0
- hdlib-2.1.0.data/scripts/chopin2.py +473 -0
- hdlib-2.1.0.dist-info/METADATA +137 -0
- hdlib-2.1.0.dist-info/RECORD +16 -0
- hdlib-2.1.0.dist-info/WHEEL +5 -0
- hdlib-2.1.0.dist-info/licenses/LICENSE +21 -0
- hdlib-2.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1719 @@
|
|
|
1
|
+
"""Classification with hdlib.
|
|
2
|
+
|
|
3
|
+
It implements the __hdlib.model.classification.ClassificationModel__ class object which allows to generate, fit, and test a classification model
|
|
4
|
+
built according to the Hyperdimensional Computing (HDC) paradigm as described in _Cumbo et al. 2020_ https://doi.org/10.3390/a13090233.
|
|
5
|
+
|
|
6
|
+
It also implements a stepwise regression model as backward and forward variable elimination techniques for selecting
|
|
7
|
+
relevant features in a dataset according to the same HDC paradigm.
|
|
8
|
+
|
|
9
|
+
The quantum version of this classification model is also provided here in __hdlib.model.classification.QuantumClassificationModel__
|
|
10
|
+
as described in _Cumbo et al. 2025_ https://doi.org/10.48550/arXiv.2511.12664."""
|
|
11
|
+
|
|
12
|
+
import copy
|
|
13
|
+
import itertools
|
|
14
|
+
import multiprocessing as mp
|
|
15
|
+
import os
|
|
16
|
+
import statistics
|
|
17
|
+
from math import log2
|
|
18
|
+
from functools import partial
|
|
19
|
+
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
20
|
+
from contextlib import nullcontext
|
|
21
|
+
|
|
22
|
+
import numpy as np
|
|
23
|
+
from sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score
|
|
24
|
+
from sklearn.model_selection import StratifiedKFold
|
|
25
|
+
from qiskit import QuantumCircuit, QuantumRegister, transpile
|
|
26
|
+
from qiskit.quantum_info import Statevector
|
|
27
|
+
from qiskit_aer import AerSimulator
|
|
28
|
+
from qiskit_aer.noise import NoiseModel
|
|
29
|
+
from qiskit_ibm_runtime import (
|
|
30
|
+
QiskitRuntimeService,
|
|
31
|
+
Sampler,
|
|
32
|
+
SamplerOptions,
|
|
33
|
+
Session
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
from hdlib import __version__
|
|
37
|
+
from hdlib.space import Space
|
|
38
|
+
from hdlib.vector import Vector
|
|
39
|
+
from hdlib.arithmetic import bundle, permute
|
|
40
|
+
|
|
41
|
+
# Quantum functions
|
|
42
|
+
from hdlib.arithmetic.quantum import (
|
|
43
|
+
encode as quantum_encode,
|
|
44
|
+
bundle as quantum_bundle,
|
|
45
|
+
permute as quantum_permute,
|
|
46
|
+
compress_circuit,
|
|
47
|
+
run_compute_uncompute_test,
|
|
48
|
+
get_circuit_metrics,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class ClassificationModel(object):
|
|
53
|
+
"""Supervised Classification Model."""
|
|
54
|
+
|
|
55
|
+
def __init__(
|
|
56
|
+
self,
|
|
57
|
+
size: int=10000,
|
|
58
|
+
levels: int=2,
|
|
59
|
+
vtype: str="bipolar",
|
|
60
|
+
) -> "ClassificationModel":
|
|
61
|
+
"""Initialize a ClassificationModel object.
|
|
62
|
+
|
|
63
|
+
Parameters
|
|
64
|
+
----------
|
|
65
|
+
size : int, default 10000
|
|
66
|
+
The size of vectors used to create a Space and define Vector objects.
|
|
67
|
+
levels : int, default 2
|
|
68
|
+
The number of level vectors used to represent numerical data. It is 2 by default.
|
|
69
|
+
vtype : {'binary', 'bipolar'}, default 'bipolar'
|
|
70
|
+
The vector type in space, which is bipolar by default.
|
|
71
|
+
|
|
72
|
+
Raises
|
|
73
|
+
------
|
|
74
|
+
TypeError
|
|
75
|
+
If the vector size or the number of levels are not integer numbers.
|
|
76
|
+
ValueError
|
|
77
|
+
If the number of level vectors is lower than 2.
|
|
78
|
+
|
|
79
|
+
Examples
|
|
80
|
+
--------
|
|
81
|
+
>>> from hdlib.model import ClassificationModel
|
|
82
|
+
>>> model = ClassificationModel(size=10000, levels=100, vtype='bipolar')
|
|
83
|
+
>>> type(model)
|
|
84
|
+
<class 'hdlib.model.ClassificationModel'>
|
|
85
|
+
|
|
86
|
+
This creates a new ClassificationModel object around a Space that can host random bipolar Vector objects with size 10,000.
|
|
87
|
+
It also defines the number of level vectors to 100.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
if not isinstance(size, int):
|
|
91
|
+
raise TypeError("Vectors size must be an integer number")
|
|
92
|
+
|
|
93
|
+
# Register vectors dimensionality
|
|
94
|
+
self.size = size
|
|
95
|
+
|
|
96
|
+
if not isinstance(levels, int):
|
|
97
|
+
raise TypeError("Levels must be an integer number")
|
|
98
|
+
|
|
99
|
+
if levels < 2:
|
|
100
|
+
raise ValueError("The number of levels must be greater than or equal to 2")
|
|
101
|
+
|
|
102
|
+
# Register the number of levels
|
|
103
|
+
self.levels = levels
|
|
104
|
+
|
|
105
|
+
# Minimum and maximum values in the input dataset
|
|
106
|
+
# This is used to define the level boundaries
|
|
107
|
+
self.min_value = None
|
|
108
|
+
self.max_value = None
|
|
109
|
+
|
|
110
|
+
# List of level boundaries
|
|
111
|
+
self.level_list = list()
|
|
112
|
+
|
|
113
|
+
if vtype not in ("bipolar", "binary"):
|
|
114
|
+
raise ValueError("Vectors type can be binary or bipolar only")
|
|
115
|
+
|
|
116
|
+
# Register vectors type
|
|
117
|
+
self.vtype = vtype.lower()
|
|
118
|
+
|
|
119
|
+
# Hyperdimensional space
|
|
120
|
+
self.space = None
|
|
121
|
+
|
|
122
|
+
# Class labels
|
|
123
|
+
self.classes = set()
|
|
124
|
+
|
|
125
|
+
# Keep track of hdlib version
|
|
126
|
+
self.version = __version__
|
|
127
|
+
|
|
128
|
+
def __str__(self) -> str:
|
|
129
|
+
"""Print the ClassificationModel object properties.
|
|
130
|
+
|
|
131
|
+
Returns
|
|
132
|
+
-------
|
|
133
|
+
str
|
|
134
|
+
A description of the ClassificationModel object. It reports the vectors size, the vector type,
|
|
135
|
+
the number of level vectors, the number of data points, and the number of class labels.
|
|
136
|
+
|
|
137
|
+
Examples
|
|
138
|
+
--------
|
|
139
|
+
>>> from hdlib.model import ClassificationModel
|
|
140
|
+
>>> model = ClassificationModel()
|
|
141
|
+
>>> print(model)
|
|
142
|
+
|
|
143
|
+
Class: hdlib.model.classification.ClassificationModel
|
|
144
|
+
Version: 0.1.17
|
|
145
|
+
Size: 10000
|
|
146
|
+
Type: bipolar
|
|
147
|
+
Levels: 2
|
|
148
|
+
Points: 0
|
|
149
|
+
Classes:
|
|
150
|
+
|
|
151
|
+
[]
|
|
152
|
+
|
|
153
|
+
Print the ClassificationModel object properties. By default, the size of vectors in space is 10,000,
|
|
154
|
+
their type is bipolar, and the number of level vectors is 2. The number of data points
|
|
155
|
+
and the number of class labels are empty here since no dataset has been processed yet.
|
|
156
|
+
"""
|
|
157
|
+
|
|
158
|
+
return f"""
|
|
159
|
+
Class: hdlib.model.classification.ClassificationModel
|
|
160
|
+
Version: {self.version}
|
|
161
|
+
Size: {self.size}
|
|
162
|
+
Type: {self.vtype}
|
|
163
|
+
Levels: {self.levels}
|
|
164
|
+
Points: {len(self.space.memory()) - self.levels if self.space is not None else 0}
|
|
165
|
+
Classes:
|
|
166
|
+
|
|
167
|
+
{np.array(list(self.classes))}
|
|
168
|
+
"""
|
|
169
|
+
|
|
170
|
+
def _init_fit_predict(
|
|
171
|
+
self,
|
|
172
|
+
size: int=10000,
|
|
173
|
+
levels: int=2,
|
|
174
|
+
vtype: str="bipolar",
|
|
175
|
+
points: Optional[List[List[float]]]=None,
|
|
176
|
+
labels: Optional[List[str]]=None,
|
|
177
|
+
cv: int=5,
|
|
178
|
+
distance_method: str="cosine",
|
|
179
|
+
retrain: int=0,
|
|
180
|
+
n_jobs: int=1,
|
|
181
|
+
metric: str="accuracy"
|
|
182
|
+
) -> Tuple[int, int, float]:
|
|
183
|
+
"""Initialize a new ClassificationModel, then fit and cross-validate it. Used for size and levels hyperparameters tuning.
|
|
184
|
+
|
|
185
|
+
Parameters
|
|
186
|
+
----------
|
|
187
|
+
size : int, default 10000
|
|
188
|
+
The size of vectors used to create a Space and define Vector objects.
|
|
189
|
+
levels : int, default 2
|
|
190
|
+
The number of level vectors used to represent numerical data. It is 2 by default.
|
|
191
|
+
vtype : {'binary', 'bipolar'}, default 'bipolar'
|
|
192
|
+
The vector type in space, which is bipolar by default.
|
|
193
|
+
points : list
|
|
194
|
+
List of lists with numerical data (floats).
|
|
195
|
+
labels : list
|
|
196
|
+
List with class labels. It has the same size of `points`.
|
|
197
|
+
cv : int, default 5
|
|
198
|
+
Number of folds for cross-validating the model.
|
|
199
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
200
|
+
Method used to compute the distance between vectors in space.
|
|
201
|
+
retrain : int, default 0
|
|
202
|
+
Number of retraining iterations.
|
|
203
|
+
n_jobs : int, default 1,
|
|
204
|
+
Number of jobs for processing folds in parallel.
|
|
205
|
+
metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
|
|
206
|
+
Metric used to evaluate the model.
|
|
207
|
+
|
|
208
|
+
Returns
|
|
209
|
+
-------
|
|
210
|
+
tuple
|
|
211
|
+
A tuple with the input size, the number of level vectors, and the model score
|
|
212
|
+
according to the input metric.
|
|
213
|
+
|
|
214
|
+
Raises
|
|
215
|
+
------
|
|
216
|
+
ValueError
|
|
217
|
+
If the provided metric is not supported.
|
|
218
|
+
"""
|
|
219
|
+
|
|
220
|
+
# Available score metrics
|
|
221
|
+
score_metrics = {
|
|
222
|
+
"accuracy": accuracy_score,
|
|
223
|
+
"f1": f1_score,
|
|
224
|
+
"precision": precision_score,
|
|
225
|
+
"recall": recall_score
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
metric = metric.lower()
|
|
229
|
+
|
|
230
|
+
if metric not in score_metrics:
|
|
231
|
+
raise ValueError("Score metric {} is not supported".format(metric))
|
|
232
|
+
|
|
233
|
+
# Generate a new Model
|
|
234
|
+
model = ClassificationModel(size=size, levels=levels, vtype=vtype)
|
|
235
|
+
|
|
236
|
+
# Fit the model
|
|
237
|
+
model.fit(points, labels=labels)
|
|
238
|
+
|
|
239
|
+
# Cross-validate the model
|
|
240
|
+
predictions = model.cross_val_predict(
|
|
241
|
+
points,
|
|
242
|
+
labels,
|
|
243
|
+
cv=cv,
|
|
244
|
+
distance_method=distance_method,
|
|
245
|
+
retrain=retrain,
|
|
246
|
+
n_jobs=n_jobs
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
# For each prediction, compute the score and return the average
|
|
250
|
+
scores = list()
|
|
251
|
+
|
|
252
|
+
for y_indices, y_pred, _, _, _, _ in predictions:
|
|
253
|
+
y_true = [label for position, label in enumerate(labels) if position in y_indices]
|
|
254
|
+
|
|
255
|
+
if metric == "accuracy":
|
|
256
|
+
scores.append(score_metrics[metric](y_true, y_pred))
|
|
257
|
+
|
|
258
|
+
else:
|
|
259
|
+
# Use the average weighted to account for label imbalance
|
|
260
|
+
scores.append(score_metrics[metric](y_true, y_pred, average="weighted"))
|
|
261
|
+
|
|
262
|
+
return size, levels, statistics.mean(scores)
|
|
263
|
+
|
|
264
|
+
def fit(
|
|
265
|
+
self,
|
|
266
|
+
points: List[List[float]],
|
|
267
|
+
labels: List[str],
|
|
268
|
+
seed: Optional[int]=None,
|
|
269
|
+
) -> None:
|
|
270
|
+
"""Build a vector-symbolic architecture. Define level vectors and encode samples.
|
|
271
|
+
|
|
272
|
+
Parameters
|
|
273
|
+
----------
|
|
274
|
+
points : list
|
|
275
|
+
List of lists with numerical data (floats).
|
|
276
|
+
labels : list
|
|
277
|
+
List with class labels. It has the same size of `points`.
|
|
278
|
+
seed : int, optional
|
|
279
|
+
An optional seed for reproducibly generating the vectors numpy.ndarray randomly.
|
|
280
|
+
|
|
281
|
+
Raises
|
|
282
|
+
------
|
|
283
|
+
Exception
|
|
284
|
+
- if there are not enough data points (the length of `points` is < 3);
|
|
285
|
+
- if the length of `points` does not match the length of `labels`;
|
|
286
|
+
- if there is only one class label.
|
|
287
|
+
"""
|
|
288
|
+
|
|
289
|
+
if len(points) < 3:
|
|
290
|
+
# This is based on the assumption that the minimum number of data points for training
|
|
291
|
+
# the classification model is 2, while 1 data point is enough for the test set
|
|
292
|
+
raise Exception("Not enough data points")
|
|
293
|
+
|
|
294
|
+
if len(points) != len(labels):
|
|
295
|
+
raise Exception("The number of data points does not match with the number of class labels")
|
|
296
|
+
|
|
297
|
+
if len(set(labels)) < 2:
|
|
298
|
+
raise Exception("The number of unique class labels must be > 1")
|
|
299
|
+
|
|
300
|
+
self.classes = set(labels)
|
|
301
|
+
|
|
302
|
+
# Initialize the hyperdimensional space so that it overwrites any existing space in Model
|
|
303
|
+
self.space = Space(size=self.size, vtype=self.vtype)
|
|
304
|
+
|
|
305
|
+
index_vector = range(self.size)
|
|
306
|
+
|
|
307
|
+
change = int(self.size / 2)
|
|
308
|
+
next_level = int((self.size / 2 / self.levels))
|
|
309
|
+
|
|
310
|
+
# Also define the interval level list
|
|
311
|
+
self.level_list = list()
|
|
312
|
+
|
|
313
|
+
# Get the minimum and maximum value in the input dataset
|
|
314
|
+
self.min_value = np.inf
|
|
315
|
+
self.max_value = -np.inf
|
|
316
|
+
|
|
317
|
+
for point in points:
|
|
318
|
+
min_point = min(point)
|
|
319
|
+
max_point = max(point)
|
|
320
|
+
|
|
321
|
+
if min_point < self.min_value:
|
|
322
|
+
self.min_value = min_point
|
|
323
|
+
|
|
324
|
+
if max_point > self.max_value:
|
|
325
|
+
self.max_value = max_point
|
|
326
|
+
|
|
327
|
+
gap = (self.max_value - self.min_value) / self.levels
|
|
328
|
+
|
|
329
|
+
if seed is None:
|
|
330
|
+
rand = np.random.default_rng()
|
|
331
|
+
|
|
332
|
+
else:
|
|
333
|
+
# Conditions on random seed for reproducibility
|
|
334
|
+
# numpy allows integers as random seeds
|
|
335
|
+
if not isinstance(seed, int):
|
|
336
|
+
raise TypeError("Seed must be an integer number")
|
|
337
|
+
|
|
338
|
+
rand = np.random.default_rng(seed=seed)
|
|
339
|
+
|
|
340
|
+
# Create level vectors
|
|
341
|
+
for level_count in range(self.levels):
|
|
342
|
+
level = "level_{}".format(level_count)
|
|
343
|
+
|
|
344
|
+
if level_count == 0:
|
|
345
|
+
base = np.full(self.size, -1 if self.vtype == "bipolar" else 0)
|
|
346
|
+
to_one = rand.permutation(index_vector)[:change]
|
|
347
|
+
|
|
348
|
+
else:
|
|
349
|
+
to_one = rand.permutation(index_vector)[:next_level]
|
|
350
|
+
|
|
351
|
+
for index in to_one:
|
|
352
|
+
base[index] = base[index] * -1 if self.vtype == "bipolar" else base[index] + 1
|
|
353
|
+
|
|
354
|
+
vector = Vector(
|
|
355
|
+
name=level,
|
|
356
|
+
size=self.size,
|
|
357
|
+
vtype=self.vtype,
|
|
358
|
+
vector=copy.deepcopy(base)
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
self.space.insert(vector)
|
|
362
|
+
|
|
363
|
+
right_bound = self.min_value + level_count * gap
|
|
364
|
+
|
|
365
|
+
if level_count == 0:
|
|
366
|
+
left_bound = right_bound
|
|
367
|
+
|
|
368
|
+
else:
|
|
369
|
+
left_bound = self.min_value + (level_count - 1) * gap
|
|
370
|
+
|
|
371
|
+
self.level_list.append((left_bound, right_bound))
|
|
372
|
+
|
|
373
|
+
# Encode all data points
|
|
374
|
+
for point_position, point in enumerate(points):
|
|
375
|
+
point_vector = self._encode_point(point)
|
|
376
|
+
|
|
377
|
+
# Add the hyperdimensional representation of the data point to the space
|
|
378
|
+
point_vector.name = "point_{}".format(point_position)
|
|
379
|
+
self.space.insert(point_vector)
|
|
380
|
+
|
|
381
|
+
# Tag vector with its class label
|
|
382
|
+
self.space.add_tag(name=point_vector.name, tag=labels[point_position])
|
|
383
|
+
|
|
384
|
+
def _encode_point(self, point: List[float]) -> Vector:
|
|
385
|
+
"""Encode a single data point. It must be used after `fit()`.
|
|
386
|
+
|
|
387
|
+
Parameters
|
|
388
|
+
----------
|
|
389
|
+
point : list
|
|
390
|
+
A data point.
|
|
391
|
+
|
|
392
|
+
Returns
|
|
393
|
+
-------
|
|
394
|
+
Vector
|
|
395
|
+
The encoded data point.
|
|
396
|
+
"""
|
|
397
|
+
|
|
398
|
+
sum_vector = None
|
|
399
|
+
|
|
400
|
+
for value_position, value in enumerate(point):
|
|
401
|
+
level_count = 0
|
|
402
|
+
|
|
403
|
+
if value == self.min_value:
|
|
404
|
+
level_count = 0
|
|
405
|
+
|
|
406
|
+
elif value == self.max_value:
|
|
407
|
+
level_count = self.levels - 1
|
|
408
|
+
|
|
409
|
+
else:
|
|
410
|
+
for level_position in range(len(self.level_list)):
|
|
411
|
+
left_bound, right_bound = self.level_list[level_position]
|
|
412
|
+
|
|
413
|
+
if left_bound <= value and right_bound > value:
|
|
414
|
+
level_count = level_position
|
|
415
|
+
|
|
416
|
+
break
|
|
417
|
+
|
|
418
|
+
level_vector = self.space.get(names=["level_{}".format(level_count)])[0]
|
|
419
|
+
|
|
420
|
+
roll_vector = permute(level_vector, rotate_by=value_position)
|
|
421
|
+
|
|
422
|
+
if sum_vector is None:
|
|
423
|
+
sum_vector = roll_vector
|
|
424
|
+
|
|
425
|
+
else:
|
|
426
|
+
sum_vector = bundle(sum_vector, roll_vector)
|
|
427
|
+
|
|
428
|
+
return sum_vector
|
|
429
|
+
|
|
430
|
+
def error_rate(
|
|
431
|
+
self,
|
|
432
|
+
training_vectors: List[Vector],
|
|
433
|
+
class_vectors: List[Vector],
|
|
434
|
+
distance_method: str="cosine"
|
|
435
|
+
) -> Tuple[float, List[Vector], List[str]]:
|
|
436
|
+
"""Compute the error rate.
|
|
437
|
+
|
|
438
|
+
Parameters
|
|
439
|
+
----------
|
|
440
|
+
training_vectors : list
|
|
441
|
+
List with Vector objects used for training the classification model.
|
|
442
|
+
class_vectors : list
|
|
443
|
+
List with the Vector representation of classes.
|
|
444
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
445
|
+
Method used to compute the distance between vectors in the space.
|
|
446
|
+
|
|
447
|
+
Returns
|
|
448
|
+
-------
|
|
449
|
+
tuple
|
|
450
|
+
A tuple with the error rate, the list of wrongly predicted Vector objects, and the list
|
|
451
|
+
of wrong predictions with the same length of the list with wrongly predicted Vector objects.
|
|
452
|
+
|
|
453
|
+
Raises
|
|
454
|
+
------
|
|
455
|
+
ValueError
|
|
456
|
+
- if the input `training_vectors` does not contain Vector objects;
|
|
457
|
+
- if the input `class_vectors` does not contain Vector objects.
|
|
458
|
+
Exception
|
|
459
|
+
- if no training vectors have been provided;
|
|
460
|
+
- if no class vectors have been provided.
|
|
461
|
+
"""
|
|
462
|
+
|
|
463
|
+
if not training_vectors:
|
|
464
|
+
raise Exception("No training vectors have been provided")
|
|
465
|
+
|
|
466
|
+
if not class_vectors:
|
|
467
|
+
raise Exception("No class vectors have been provided")
|
|
468
|
+
|
|
469
|
+
wrongly_predicted_training_vectors = list()
|
|
470
|
+
|
|
471
|
+
wrong_predictions = list()
|
|
472
|
+
|
|
473
|
+
for class_vector in class_vectors:
|
|
474
|
+
if not isinstance(class_vector, Vector):
|
|
475
|
+
raise ValueError("The list of class vectors does not contain Vector objects")
|
|
476
|
+
|
|
477
|
+
for training_vector in training_vectors:
|
|
478
|
+
if not isinstance(training_vector, Vector):
|
|
479
|
+
raise ValueError("The list of training vectors does not contain Vector objects")
|
|
480
|
+
|
|
481
|
+
# Vectors contain only their class info in tags
|
|
482
|
+
true_class = list(training_vector.tags)[0]
|
|
483
|
+
|
|
484
|
+
if true_class != None:
|
|
485
|
+
closest_class = None
|
|
486
|
+
closest_dist = -np.inf
|
|
487
|
+
|
|
488
|
+
for class_vector in class_vectors:
|
|
489
|
+
# Compute the distance between the training points and the hyperdimensional representations of classes
|
|
490
|
+
with np.errstate(invalid="ignore", divide="ignore"):
|
|
491
|
+
distance = training_vector.dist(class_vector, method=distance_method)
|
|
492
|
+
|
|
493
|
+
if closest_class is None:
|
|
494
|
+
closest_class = list(class_vector.tags)[0]
|
|
495
|
+
closest_dist = distance
|
|
496
|
+
|
|
497
|
+
else:
|
|
498
|
+
if distance < closest_dist:
|
|
499
|
+
closest_class = list(class_vector.tags)[0]
|
|
500
|
+
closest_dist = distance
|
|
501
|
+
|
|
502
|
+
if closest_class != true_class:
|
|
503
|
+
wrongly_predicted_training_vectors.append(training_vector)
|
|
504
|
+
|
|
505
|
+
wrong_predictions.append(closest_class)
|
|
506
|
+
|
|
507
|
+
model_error_rate = len(wrongly_predicted_training_vectors) / len(training_vectors)
|
|
508
|
+
|
|
509
|
+
return model_error_rate, wrongly_predicted_training_vectors, wrong_predictions
|
|
510
|
+
|
|
511
|
+
def predict(
|
|
512
|
+
self,
|
|
513
|
+
test_indices: List[int],
|
|
514
|
+
distance_method: str="cosine",
|
|
515
|
+
retrain: int=0
|
|
516
|
+
) -> Tuple[List[int], List[str], List[List[float]], int, float, List[Vector]]:
|
|
517
|
+
"""Supervised Learning. Predict the class labels of the data points in the test set.
|
|
518
|
+
|
|
519
|
+
Parameters
|
|
520
|
+
----------
|
|
521
|
+
test_indices : list
|
|
522
|
+
Indices of data points in the list of points used with fit() to be used for testing the classification model.
|
|
523
|
+
Note that all the other points will be used for training the model.
|
|
524
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
525
|
+
Method used to compute the distance between vectors in the space.
|
|
526
|
+
retrain : int, default 0
|
|
527
|
+
Maximum number of retraining iterations.
|
|
528
|
+
|
|
529
|
+
Returns
|
|
530
|
+
-------
|
|
531
|
+
tuple
|
|
532
|
+
A tuple with the input list `test_indices` in addition to a list with the predicted class labels with the
|
|
533
|
+
same size of `test_indices`, the distances between the test vectors and classes, the total number of
|
|
534
|
+
retraining iterations used to retrain the classification model, the model error rate, and the retrained
|
|
535
|
+
class vectors (i.e., the actual model).
|
|
536
|
+
|
|
537
|
+
Raises
|
|
538
|
+
------
|
|
539
|
+
ValueError
|
|
540
|
+
If the number of retraining iterations is <0.
|
|
541
|
+
Exception
|
|
542
|
+
- if no test indices have been provided;
|
|
543
|
+
- if no class labels have been provided while fitting the model;
|
|
544
|
+
- if the number of test indices does not match the number of points retrieved from the space.
|
|
545
|
+
|
|
546
|
+
Notes
|
|
547
|
+
-----
|
|
548
|
+
The supervised classification model based on the hyperdimensional computing paradigm has been originally described in [1]_.
|
|
549
|
+
|
|
550
|
+
.. [1] Cumbo, Fabio, Eleonora Cappelli, and Emanuel Weitschek. "A brain-inspired hyperdimensional computing approach
|
|
551
|
+
for classifying massive dna methylation data of cancer." Algorithms 13.9 (2020): 233.
|
|
552
|
+
"""
|
|
553
|
+
|
|
554
|
+
if not test_indices:
|
|
555
|
+
raise Exception("No test indices have been provided")
|
|
556
|
+
|
|
557
|
+
if retrain < 0:
|
|
558
|
+
raise ValueError("The number of retraining iterations must be >=0")
|
|
559
|
+
|
|
560
|
+
if len(self.classes) == 0:
|
|
561
|
+
raise Exception("No class labels found")
|
|
562
|
+
|
|
563
|
+
# List with test vectors
|
|
564
|
+
test_vectors = list()
|
|
565
|
+
|
|
566
|
+
# List with training vectors
|
|
567
|
+
training_vectors = list()
|
|
568
|
+
|
|
569
|
+
# Retrieve test and training vectors from the space
|
|
570
|
+
for vector_name in self.space.space:
|
|
571
|
+
if vector_name.startswith("point_"):
|
|
572
|
+
vector_id = int(vector_name.split("_")[-1])
|
|
573
|
+
|
|
574
|
+
vector = self.space.space[vector_name]
|
|
575
|
+
|
|
576
|
+
if vector_id in test_indices:
|
|
577
|
+
test_vectors.append(vector)
|
|
578
|
+
|
|
579
|
+
else:
|
|
580
|
+
training_vectors.append(vector)
|
|
581
|
+
|
|
582
|
+
if len(test_vectors) != len(test_indices):
|
|
583
|
+
raise Exception("Unable to retrieve all the test vectors from the space")
|
|
584
|
+
|
|
585
|
+
class_vectors = list()
|
|
586
|
+
|
|
587
|
+
for class_pos, class_label in enumerate(self.classes):
|
|
588
|
+
# Get training vectors for the current class label
|
|
589
|
+
class_points = [vector for vector in training_vectors if class_label in vector.tags]
|
|
590
|
+
|
|
591
|
+
# Build the vector representations of the current class
|
|
592
|
+
class_vector = None
|
|
593
|
+
|
|
594
|
+
for vector in class_points:
|
|
595
|
+
if class_vector is None:
|
|
596
|
+
class_vector = vector
|
|
597
|
+
|
|
598
|
+
else:
|
|
599
|
+
class_vector = bundle(class_vector, vector)
|
|
600
|
+
|
|
601
|
+
class_vector.name = "class_{}".format(class_pos)
|
|
602
|
+
|
|
603
|
+
class_vector.tags.add(class_label)
|
|
604
|
+
|
|
605
|
+
class_vectors.append(class_vector)
|
|
606
|
+
|
|
607
|
+
# Make a copy of the vector representation of classes for retraining the model
|
|
608
|
+
retraining_class_vectors = copy.deepcopy(class_vectors) if retrain > 0 else class_vectors
|
|
609
|
+
|
|
610
|
+
# Take track of the error rate in case of retraining the model
|
|
611
|
+
model_error_rate, wrongly_predicted_training_vectors, wrong_predictions = self.error_rate(
|
|
612
|
+
training_vectors,
|
|
613
|
+
retraining_class_vectors,
|
|
614
|
+
distance_method=distance_method
|
|
615
|
+
)
|
|
616
|
+
|
|
617
|
+
# Count retraining iterations
|
|
618
|
+
retraining_iterations = 0
|
|
619
|
+
|
|
620
|
+
if retrain > 0:
|
|
621
|
+
for _ in range(retrain):
|
|
622
|
+
retraining_class_vectors_iter = copy.deepcopy(retraining_class_vectors)
|
|
623
|
+
|
|
624
|
+
for vector_position, training_vector in enumerate(wrongly_predicted_training_vectors):
|
|
625
|
+
true_class = list(training_vector.tags)[0]
|
|
626
|
+
|
|
627
|
+
# Error mitigation
|
|
628
|
+
for class_vector in retraining_class_vectors_iter:
|
|
629
|
+
if true_class in class_vector.tags:
|
|
630
|
+
class_vector.vector = class_vector.vector + training_vector.vector
|
|
631
|
+
|
|
632
|
+
elif wrong_predictions[vector_position] in class_vector.tags:
|
|
633
|
+
class_vector.vector = class_vector.vector - training_vector.vector
|
|
634
|
+
|
|
635
|
+
retraining_error_rate, wrongly_predicted_training_vectors, wrong_predictions = self.error_rate(
|
|
636
|
+
training_vectors,
|
|
637
|
+
retraining_class_vectors_iter,
|
|
638
|
+
distance_method=distance_method
|
|
639
|
+
)
|
|
640
|
+
|
|
641
|
+
if model_error_rate < retraining_error_rate:
|
|
642
|
+
# Does not make sense to keep retraining if the error rate increases compared to the previous iteration
|
|
643
|
+
break
|
|
644
|
+
|
|
645
|
+
# Take track of the error rate
|
|
646
|
+
model_error_rate = retraining_error_rate
|
|
647
|
+
|
|
648
|
+
# Use the retrained class vectors
|
|
649
|
+
retraining_class_vectors = retraining_class_vectors_iter
|
|
650
|
+
|
|
651
|
+
# Also take track of the number of retraining iterations
|
|
652
|
+
retraining_iterations += 1
|
|
653
|
+
|
|
654
|
+
prediction = list()
|
|
655
|
+
distances = list()
|
|
656
|
+
|
|
657
|
+
for test_vector in sorted(test_vectors, key=lambda vector: test_indices.index(int(vector.name.split("_")[-1]))):
|
|
658
|
+
pred, dist = self._predict_vector(test_vector, retraining_class_vectors, distance_method=distance_method)
|
|
659
|
+
|
|
660
|
+
prediction.append(pred)
|
|
661
|
+
distances.append(dist)
|
|
662
|
+
|
|
663
|
+
return test_indices, prediction, distances, retraining_iterations, model_error_rate, retraining_class_vectors
|
|
664
|
+
|
|
665
|
+
def _predict_vector(
|
|
666
|
+
self,
|
|
667
|
+
vector: Vector,
|
|
668
|
+
training_class_vectors: List[Vector],
|
|
669
|
+
distance_method: str="cosine"
|
|
670
|
+
) -> Tuple[str, List[float]]:
|
|
671
|
+
"""Predict the class of an input vector.
|
|
672
|
+
|
|
673
|
+
Parameters
|
|
674
|
+
----------
|
|
675
|
+
vector : Vector
|
|
676
|
+
The input vector for prediction.
|
|
677
|
+
training_class_vectors : list
|
|
678
|
+
List of eventually retrained class vectors representing the classification model.
|
|
679
|
+
|
|
680
|
+
Returns
|
|
681
|
+
-------
|
|
682
|
+
tuple
|
|
683
|
+
The closest class as the prediction and a list of vector to classes distances.
|
|
684
|
+
"""
|
|
685
|
+
|
|
686
|
+
closest_class = None
|
|
687
|
+
closest_dist = -np.inf
|
|
688
|
+
|
|
689
|
+
distances = list()
|
|
690
|
+
|
|
691
|
+
for class_vector in training_class_vectors:
|
|
692
|
+
# Compute the distance between the input vector and the hyperdimensional representations of classes
|
|
693
|
+
with np.errstate(invalid="ignore", divide="ignore"):
|
|
694
|
+
distance = vector.dist(class_vector, method=distance_method)
|
|
695
|
+
|
|
696
|
+
distances.append(distance)
|
|
697
|
+
|
|
698
|
+
if closest_class is None:
|
|
699
|
+
closest_class = list(class_vector.tags)[0]
|
|
700
|
+
closest_dist = distance
|
|
701
|
+
|
|
702
|
+
else:
|
|
703
|
+
if distance < closest_dist:
|
|
704
|
+
closest_class = list(class_vector.tags)[0]
|
|
705
|
+
closest_dist = distance
|
|
706
|
+
|
|
707
|
+
return closest_class, distances
|
|
708
|
+
|
|
709
|
+
def cross_val_predict(
|
|
710
|
+
self,
|
|
711
|
+
points: List[List[float]],
|
|
712
|
+
labels: List[str],
|
|
713
|
+
cv: int=5,
|
|
714
|
+
distance_method: str="cosine",
|
|
715
|
+
retrain: int=0,
|
|
716
|
+
n_jobs: int=1
|
|
717
|
+
) -> List[Tuple[List[int], List[str], int]]:
|
|
718
|
+
"""Run `predict()` in cross validation.
|
|
719
|
+
|
|
720
|
+
Parameters
|
|
721
|
+
----------
|
|
722
|
+
points : list
|
|
723
|
+
List of lists with numerical data (floats).
|
|
724
|
+
labels : list
|
|
725
|
+
List with class labels. It has the same size of `points`.
|
|
726
|
+
cv : int, default 5
|
|
727
|
+
Number of folds for cross-validating the model.
|
|
728
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
729
|
+
Method used to compute the distance between vectors in space.
|
|
730
|
+
retrain : int, default 0
|
|
731
|
+
Number of retraining iterations.
|
|
732
|
+
n_jobs : int, default 1,
|
|
733
|
+
Number of jobs for processing folds in parallel.
|
|
734
|
+
|
|
735
|
+
Returns
|
|
736
|
+
-------
|
|
737
|
+
list
|
|
738
|
+
A list with the results of `predict()` for each fold.
|
|
739
|
+
|
|
740
|
+
Raises
|
|
741
|
+
------
|
|
742
|
+
Exception
|
|
743
|
+
- if the number of data points does not match with the number of class labels;
|
|
744
|
+
- if there is only one class label.
|
|
745
|
+
ValueError
|
|
746
|
+
- if the number of folds is a number < 2;
|
|
747
|
+
- if the number of folds exceeds the number of data points;
|
|
748
|
+
- if the number of retraining iterations is <0.
|
|
749
|
+
"""
|
|
750
|
+
|
|
751
|
+
if len(points) != len(labels):
|
|
752
|
+
raise Exception("The number of data points does not match with the number of class labels")
|
|
753
|
+
|
|
754
|
+
if len(set(labels)) < 2:
|
|
755
|
+
raise Exception("The number of unique class labels must be > 1")
|
|
756
|
+
|
|
757
|
+
if cv < 2:
|
|
758
|
+
raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
|
|
759
|
+
|
|
760
|
+
if cv > len(points):
|
|
761
|
+
raise ValueError("The number of folds cannot exceed the number of data points")
|
|
762
|
+
|
|
763
|
+
if retrain < 0:
|
|
764
|
+
raise ValueError("The number of retraining iterations must be >=0")
|
|
765
|
+
|
|
766
|
+
# Use all the available resources if n_job < 1
|
|
767
|
+
n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
|
|
768
|
+
|
|
769
|
+
kf = StratifiedKFold(n_splits=cv, shuffle=True, random_state=0)
|
|
770
|
+
|
|
771
|
+
# Collect results from every self.predict call
|
|
772
|
+
predictions = list()
|
|
773
|
+
|
|
774
|
+
if n_jobs == 1:
|
|
775
|
+
for _, test_indices in kf.split(points, labels):
|
|
776
|
+
test_indices = test_indices.tolist()
|
|
777
|
+
|
|
778
|
+
_, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors = self.predict(
|
|
779
|
+
test_indices,
|
|
780
|
+
distance_method=distance_method,
|
|
781
|
+
retrain=retrain
|
|
782
|
+
)
|
|
783
|
+
|
|
784
|
+
predictions.append((test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors))
|
|
785
|
+
|
|
786
|
+
else:
|
|
787
|
+
predict_partial = partial(
|
|
788
|
+
self.predict,
|
|
789
|
+
distance_method=distance_method,
|
|
790
|
+
retrain=retrain
|
|
791
|
+
)
|
|
792
|
+
|
|
793
|
+
# Run prediction on folds in parallel
|
|
794
|
+
with mp.Pool(processes=n_jobs) as pool:
|
|
795
|
+
jobs = [
|
|
796
|
+
pool.apply_async(
|
|
797
|
+
predict_partial,
|
|
798
|
+
args=(test_indices.tolist(),)
|
|
799
|
+
)
|
|
800
|
+
for _, test_indices in kf.split(points, labels)
|
|
801
|
+
]
|
|
802
|
+
|
|
803
|
+
# Get results from jobs
|
|
804
|
+
for job in jobs:
|
|
805
|
+
test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors = job.get()
|
|
806
|
+
|
|
807
|
+
predictions.append((test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors))
|
|
808
|
+
|
|
809
|
+
return predictions
|
|
810
|
+
|
|
811
|
+
def auto_tune(
|
|
812
|
+
self,
|
|
813
|
+
points: List[List[float]],
|
|
814
|
+
labels: List[str],
|
|
815
|
+
size_range: range,
|
|
816
|
+
levels_range: range,
|
|
817
|
+
cv: int=5,
|
|
818
|
+
distance_method: str="cosine",
|
|
819
|
+
retrain: int=0,
|
|
820
|
+
n_jobs: int=1,
|
|
821
|
+
metric: str="accuracy"
|
|
822
|
+
) -> Tuple[int, int, float]:
|
|
823
|
+
"""Automated hyperparameters tuning. Perform a Parameter Sweep Analysis (PSA) on space dimensionality and number of levels.
|
|
824
|
+
|
|
825
|
+
Parameters
|
|
826
|
+
----------
|
|
827
|
+
points : list
|
|
828
|
+
List of lists with numerical data (floats).
|
|
829
|
+
labels : list
|
|
830
|
+
List with class labels. It has the same size of `points`.
|
|
831
|
+
size_range : range
|
|
832
|
+
Range of dimensionalities for performing PSA.
|
|
833
|
+
levels_range : range
|
|
834
|
+
Range of number of levels for performing PSA.
|
|
835
|
+
cv : int, default 5
|
|
836
|
+
Number of folds for cross-validating the model.
|
|
837
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
838
|
+
Method used to compute the distance between vectors in space.
|
|
839
|
+
retrain : int, default 0
|
|
840
|
+
Number of retraining iterations.
|
|
841
|
+
n_jobs : int, default 1,
|
|
842
|
+
Number of jobs for processing folds in parallel.
|
|
843
|
+
metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
|
|
844
|
+
Metric used to evaluate the model.
|
|
845
|
+
|
|
846
|
+
Returns
|
|
847
|
+
-------
|
|
848
|
+
tuple
|
|
849
|
+
A tuple with the best size and levels according to the accuracies of the cross-validated models.
|
|
850
|
+
|
|
851
|
+
Raises
|
|
852
|
+
------
|
|
853
|
+
ValueError
|
|
854
|
+
- if the number of class labels does not match with the number of data points;
|
|
855
|
+
- if the number of specified folds for cross-validating the model is lower than 2;
|
|
856
|
+
- if the number of folds exceeds the number of data points;
|
|
857
|
+
- if the number of retraining iterations is <0.
|
|
858
|
+
Exception
|
|
859
|
+
- if no data points have been provided in input;
|
|
860
|
+
- if no class labels have been provided in input.
|
|
861
|
+
"""
|
|
862
|
+
|
|
863
|
+
if not points:
|
|
864
|
+
raise Exception("No data points have been provided")
|
|
865
|
+
|
|
866
|
+
if not labels:
|
|
867
|
+
raise Exception("No class labels have been provided")
|
|
868
|
+
|
|
869
|
+
if len(points) != len(labels):
|
|
870
|
+
raise ValueError("The number of class labels must match with the number of data points")
|
|
871
|
+
|
|
872
|
+
if cv < 2:
|
|
873
|
+
raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
|
|
874
|
+
|
|
875
|
+
if cv > len(points):
|
|
876
|
+
raise ValueError("The number of folds cannot exceed the number of data points")
|
|
877
|
+
|
|
878
|
+
if retrain < 0:
|
|
879
|
+
raise ValueError("The number of retraining iterations must be >=0")
|
|
880
|
+
|
|
881
|
+
# Use all the available resources if n_job < 1
|
|
882
|
+
n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
|
|
883
|
+
|
|
884
|
+
partial_init_fit_predict = partial(
|
|
885
|
+
self._init_fit_predict,
|
|
886
|
+
vtype=self.vtype,
|
|
887
|
+
points=points,
|
|
888
|
+
labels=labels,
|
|
889
|
+
cv=cv,
|
|
890
|
+
distance_method=distance_method,
|
|
891
|
+
retrain=retrain,
|
|
892
|
+
n_jobs=1,
|
|
893
|
+
metric=metric
|
|
894
|
+
)
|
|
895
|
+
|
|
896
|
+
best_metric = None
|
|
897
|
+
best_size = None
|
|
898
|
+
best_levels = None
|
|
899
|
+
|
|
900
|
+
with mp.Pool(processes=n_jobs) as pool:
|
|
901
|
+
jobs = [
|
|
902
|
+
pool.apply_async(
|
|
903
|
+
partial_init_fit_predict,
|
|
904
|
+
args=(size, levels,)
|
|
905
|
+
)
|
|
906
|
+
for size, levels in list(itertools.product(size_range, levels_range)) \
|
|
907
|
+
if size > len(points) and levels > 1
|
|
908
|
+
]
|
|
909
|
+
|
|
910
|
+
# Get results from jobs
|
|
911
|
+
for job in jobs:
|
|
912
|
+
job_size, job_levels, job_metric = job.get()
|
|
913
|
+
|
|
914
|
+
if best_metric is None:
|
|
915
|
+
best_metric = job_metric
|
|
916
|
+
best_size = job_size
|
|
917
|
+
best_levels = job_levels
|
|
918
|
+
|
|
919
|
+
else:
|
|
920
|
+
if job_metric > best_metric:
|
|
921
|
+
# Get the size and levels of the classification model with the best score metric
|
|
922
|
+
best_metric = job_metric
|
|
923
|
+
best_size = job_size
|
|
924
|
+
best_levels = job_levels
|
|
925
|
+
|
|
926
|
+
elif job_metric == best_metric:
|
|
927
|
+
# Minimize the number of levels in this case
|
|
928
|
+
if job_levels < best_levels:
|
|
929
|
+
best_size = job_size
|
|
930
|
+
best_levels = job_levels
|
|
931
|
+
|
|
932
|
+
elif job_levels == best_levels:
|
|
933
|
+
# Minimize the size in this case
|
|
934
|
+
if job_size < best_size:
|
|
935
|
+
best_size = job_size
|
|
936
|
+
|
|
937
|
+
return best_size, best_levels, best_metric
|
|
938
|
+
|
|
939
|
+
def _stepwise_regression_iter(
|
|
940
|
+
self,
|
|
941
|
+
features_indices: Set[int],
|
|
942
|
+
points: List[List[float]],
|
|
943
|
+
labels: List[str],
|
|
944
|
+
cv: int=5,
|
|
945
|
+
distance_method: str="cosine",
|
|
946
|
+
retrain: int=0,
|
|
947
|
+
metric: str="accuracy"
|
|
948
|
+
) -> Tuple[Set[float], float]:
|
|
949
|
+
"""Just a single iteration of the feature selection method.
|
|
950
|
+
|
|
951
|
+
Parameters
|
|
952
|
+
----------
|
|
953
|
+
features_indices : set
|
|
954
|
+
Indices of features for shaping points.
|
|
955
|
+
points : list
|
|
956
|
+
List of lists with numerical data (floats).
|
|
957
|
+
labels : list
|
|
958
|
+
List with class labels. It has the same size of `points`.
|
|
959
|
+
cv : int, default 5
|
|
960
|
+
Number of folds for cross-validating the model.
|
|
961
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
962
|
+
Method used to compute the distance between vectors in space.
|
|
963
|
+
retrain : int, default 0
|
|
964
|
+
Number of retraining iterations.
|
|
965
|
+
metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
|
|
966
|
+
Metric used to evaluate the model.
|
|
967
|
+
|
|
968
|
+
Returns
|
|
969
|
+
-------
|
|
970
|
+
tuple
|
|
971
|
+
A tuple with the considered features and the score of the classification model based on the provided metric.
|
|
972
|
+
"""
|
|
973
|
+
|
|
974
|
+
data_points = [[point[i] for i in range(len(point)) if i in features_indices] for point in points]
|
|
975
|
+
|
|
976
|
+
_, _, score = self._init_fit_predict(
|
|
977
|
+
size=self.size,
|
|
978
|
+
levels=self.levels,
|
|
979
|
+
vtype=self.vtype,
|
|
980
|
+
points=data_points,
|
|
981
|
+
labels=labels,
|
|
982
|
+
cv=cv,
|
|
983
|
+
distance_method=distance_method,
|
|
984
|
+
retrain=retrain,
|
|
985
|
+
n_jobs=1,
|
|
986
|
+
metric=metric
|
|
987
|
+
)
|
|
988
|
+
|
|
989
|
+
return features_indices, score
|
|
990
|
+
|
|
991
|
+
def stepwise_regression(
|
|
992
|
+
self,
|
|
993
|
+
points: List[List[float]],
|
|
994
|
+
features: List[str],
|
|
995
|
+
labels: List[str],
|
|
996
|
+
method: str="backward",
|
|
997
|
+
cv: int=5,
|
|
998
|
+
distance_method: str="cosine",
|
|
999
|
+
retrain: int=0,
|
|
1000
|
+
n_jobs: int=1,
|
|
1001
|
+
metric: str="accuracy",
|
|
1002
|
+
threshold: float=0.6,
|
|
1003
|
+
uncertainty: float=5.0,
|
|
1004
|
+
stop_if_worse: bool=False
|
|
1005
|
+
) -> Tuple[Dict[str, int], Dict[int, float], int, int]:
|
|
1006
|
+
"""Stepwise regression as backward variable elimination or forward variable selection.
|
|
1007
|
+
|
|
1008
|
+
Parameters
|
|
1009
|
+
----------
|
|
1010
|
+
points : list
|
|
1011
|
+
List of lists with numerical data (floats).
|
|
1012
|
+
features : list
|
|
1013
|
+
List of features.
|
|
1014
|
+
labels : list
|
|
1015
|
+
List with class labels. It has the same size of `points`.
|
|
1016
|
+
method : {'backward', 'forward'}, default 'backward'
|
|
1017
|
+
Feature selection method.
|
|
1018
|
+
cv : int, default 5
|
|
1019
|
+
Number of folds for cross-validating the model.
|
|
1020
|
+
distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
|
|
1021
|
+
Method used to compute the distance between vectors in space.
|
|
1022
|
+
retrain : int, default 0
|
|
1023
|
+
Number of retraining iterations.
|
|
1024
|
+
n_jobs : int, default 1,
|
|
1025
|
+
Number of jobs for processing models in parallel.
|
|
1026
|
+
metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
|
|
1027
|
+
Metric used to evaluate the model.
|
|
1028
|
+
threshold : float, default 0.6
|
|
1029
|
+
Threshold on the model score metric. Stop running the feature selection if the best
|
|
1030
|
+
reached score is lower than this threshold.
|
|
1031
|
+
uncertainty : float, default 5.0
|
|
1032
|
+
Uncertainty percentage threshold for comparing models metrics.
|
|
1033
|
+
stop_if_worse : bool, default False
|
|
1034
|
+
Stop running the feature selection if the accuracy reached at the iteration i is lower than the accuracy reached at i-1.
|
|
1035
|
+
|
|
1036
|
+
Returns
|
|
1037
|
+
-------
|
|
1038
|
+
tuple
|
|
1039
|
+
A tuple with a dictionary with features and their importance in addition to the best score for each importance rank,
|
|
1040
|
+
the best importance, and the total mount of ML models built and evaluated. For what concerns the importance, in case of
|
|
1041
|
+
`method='backward'`, the lower the better. In case of `method='forward'`, the higher the better.
|
|
1042
|
+
|
|
1043
|
+
Raises
|
|
1044
|
+
------
|
|
1045
|
+
ValueError
|
|
1046
|
+
- if there are not enough features for running the feature selection;
|
|
1047
|
+
- if the number of class labels does not match with the number of data points;
|
|
1048
|
+
- if the specified feature selection method is not supported;
|
|
1049
|
+
- if the number of specified folds for cross-validating the model is lower than 2;
|
|
1050
|
+
- if the number of folds exceeds the number of data points;
|
|
1051
|
+
- if the number of retraining iterations is <0;
|
|
1052
|
+
- if the threshold is negative or greater than 1.0;
|
|
1053
|
+
- if the uncertainty percentage is negative or greater than 100.0.
|
|
1054
|
+
Exception
|
|
1055
|
+
- if no data points have been provided in input;
|
|
1056
|
+
- if no class labels have been provided in input.
|
|
1057
|
+
"""
|
|
1058
|
+
|
|
1059
|
+
if not points:
|
|
1060
|
+
raise Exception("No data points have been provided")
|
|
1061
|
+
|
|
1062
|
+
if len(features) < 2:
|
|
1063
|
+
raise ValueError("Not enough features for running a feature selection")
|
|
1064
|
+
|
|
1065
|
+
if not labels:
|
|
1066
|
+
raise Exception("No class labels have been provided")
|
|
1067
|
+
|
|
1068
|
+
if len(points) != len(labels):
|
|
1069
|
+
raise ValueError("The number of class labels must match with the number of data points")
|
|
1070
|
+
|
|
1071
|
+
method = method.lower()
|
|
1072
|
+
|
|
1073
|
+
if method not in ("backward", "forward"):
|
|
1074
|
+
raise ValueError("Stepwise method {} is not supported".format(method))
|
|
1075
|
+
|
|
1076
|
+
if cv < 2:
|
|
1077
|
+
raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
|
|
1078
|
+
|
|
1079
|
+
if cv > len(points):
|
|
1080
|
+
raise ValueError("The number of folds cannot exceed the number of data points")
|
|
1081
|
+
|
|
1082
|
+
if retrain < 0:
|
|
1083
|
+
raise ValueError("The number of retraining iterations must be >=0")
|
|
1084
|
+
|
|
1085
|
+
if threshold < 0.0 or threshold > 1.0:
|
|
1086
|
+
raise ValueError("Invalid threshold! It must be >= 0.0 and <= 1.0")
|
|
1087
|
+
|
|
1088
|
+
if uncertainty < 0.0 or uncertainty > 100.0:
|
|
1089
|
+
raise ValueError("Invalid uncertainty percentage! It must be >= 0.0 and <= 100.0")
|
|
1090
|
+
|
|
1091
|
+
# Use all the available resources if n_job < 1
|
|
1092
|
+
n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
|
|
1093
|
+
|
|
1094
|
+
# Initialize the importance of features to 0
|
|
1095
|
+
features_importance = {feature: {"importance": 0, "score": 0.0} for feature in features}
|
|
1096
|
+
|
|
1097
|
+
features_indices = set(range(len(features)))
|
|
1098
|
+
|
|
1099
|
+
# Take track of the last feature selection
|
|
1100
|
+
# Only in case of forward variable selection
|
|
1101
|
+
last_selection = set()
|
|
1102
|
+
|
|
1103
|
+
prev_score = 0.0
|
|
1104
|
+
|
|
1105
|
+
count_iter = 1
|
|
1106
|
+
|
|
1107
|
+
# Count the total amount of ML models built and evaluated
|
|
1108
|
+
count_models = 0
|
|
1109
|
+
|
|
1110
|
+
while features_indices:
|
|
1111
|
+
if method == "backward":
|
|
1112
|
+
features_set_size = len(features_indices) - 1
|
|
1113
|
+
|
|
1114
|
+
if len(features_indices) == 1:
|
|
1115
|
+
break
|
|
1116
|
+
|
|
1117
|
+
elif method == "forward":
|
|
1118
|
+
features_set_size = count_iter
|
|
1119
|
+
|
|
1120
|
+
if features_set_size >= len(features_indices):
|
|
1121
|
+
break
|
|
1122
|
+
|
|
1123
|
+
if features_set_size > 0:
|
|
1124
|
+
best_score = 0.0
|
|
1125
|
+
classification_results = list()
|
|
1126
|
+
|
|
1127
|
+
partial_stepwise_regression_iter = partial(
|
|
1128
|
+
self._stepwise_regression_iter,
|
|
1129
|
+
points=points,
|
|
1130
|
+
labels=labels,
|
|
1131
|
+
cv=cv,
|
|
1132
|
+
distance_method=distance_method,
|
|
1133
|
+
retrain=retrain,
|
|
1134
|
+
metric=metric
|
|
1135
|
+
)
|
|
1136
|
+
|
|
1137
|
+
with mp.Pool(processes=n_jobs) as pool:
|
|
1138
|
+
# Get all combinations of features of a given size
|
|
1139
|
+
jobs = [
|
|
1140
|
+
pool.apply_async(
|
|
1141
|
+
partial_stepwise_regression_iter,
|
|
1142
|
+
args=(features_set,)
|
|
1143
|
+
)
|
|
1144
|
+
for features_set in itertools.combinations(features_indices, features_set_size)
|
|
1145
|
+
]
|
|
1146
|
+
|
|
1147
|
+
# Get results from jobs
|
|
1148
|
+
for job in jobs:
|
|
1149
|
+
job_features_set, job_score = job.get()
|
|
1150
|
+
|
|
1151
|
+
if job_score >= best_score:
|
|
1152
|
+
# Keep track of the best score
|
|
1153
|
+
best_score = job_score
|
|
1154
|
+
|
|
1155
|
+
classification_results.append((job_features_set, job_score))
|
|
1156
|
+
|
|
1157
|
+
count_models += 1
|
|
1158
|
+
|
|
1159
|
+
selection = set()
|
|
1160
|
+
|
|
1161
|
+
best_scores = list()
|
|
1162
|
+
|
|
1163
|
+
for features_set, score in classification_results:
|
|
1164
|
+
if score >= best_score - (best_score * uncertainty / 100.0):
|
|
1165
|
+
# Keep track of the missing features in models that reached the best score
|
|
1166
|
+
if method == "backward":
|
|
1167
|
+
selection.update(features_indices.difference(features_set))
|
|
1168
|
+
|
|
1169
|
+
elif method == "forward":
|
|
1170
|
+
selection.update(features_set)
|
|
1171
|
+
|
|
1172
|
+
best_scores.append(score)
|
|
1173
|
+
|
|
1174
|
+
if method == "forward" and last_selection:
|
|
1175
|
+
if len(last_selection) == len(selection) and len(last_selection.difference(selection)) == 0:
|
|
1176
|
+
break
|
|
1177
|
+
|
|
1178
|
+
last_selection = selection
|
|
1179
|
+
|
|
1180
|
+
avg_score = statistics.mean(best_scores)
|
|
1181
|
+
|
|
1182
|
+
if method == "backward":
|
|
1183
|
+
# Keep decreasing the importance of worst features detected in previous iterations
|
|
1184
|
+
for feature in features_importance:
|
|
1185
|
+
if features_importance[feature]["importance"] >= 1:
|
|
1186
|
+
features_importance[feature]["importance"] += 1
|
|
1187
|
+
|
|
1188
|
+
# Set the importance of the selected features
|
|
1189
|
+
for feature_index in selection:
|
|
1190
|
+
features_importance[features[feature_index]]["importance"] += 1
|
|
1191
|
+
features_importance[features[feature_index]]["score"] = avg_score
|
|
1192
|
+
|
|
1193
|
+
if method == "backward":
|
|
1194
|
+
features_indices = features_indices.difference(selection)
|
|
1195
|
+
|
|
1196
|
+
elif method == "forward":
|
|
1197
|
+
features_indices = selection
|
|
1198
|
+
|
|
1199
|
+
if stop_if_worse:
|
|
1200
|
+
if best_score < prev_score - (prev_score * uncertainty / 100.0):
|
|
1201
|
+
prev_score = avg_score
|
|
1202
|
+
|
|
1203
|
+
break
|
|
1204
|
+
|
|
1205
|
+
prev_score = avg_score
|
|
1206
|
+
|
|
1207
|
+
count_iter += 1
|
|
1208
|
+
|
|
1209
|
+
if best_score < threshold:
|
|
1210
|
+
break
|
|
1211
|
+
|
|
1212
|
+
importances = dict()
|
|
1213
|
+
|
|
1214
|
+
scores = dict()
|
|
1215
|
+
|
|
1216
|
+
for feature in features_importance:
|
|
1217
|
+
importances[feature] = features_importance[feature]["importance"]
|
|
1218
|
+
|
|
1219
|
+
scores[features_importance[feature]["importance"]] = features_importance[feature]["score"]
|
|
1220
|
+
|
|
1221
|
+
best_importance = sorted(scores.keys(), key=lambda imp: scores[imp])[-1]
|
|
1222
|
+
|
|
1223
|
+
return importances, scores, best_importance, count_models
|
|
1224
|
+
|
|
1225
|
+
|
|
1226
|
+
class QuantumClassificationModel(object):
|
|
1227
|
+
"""Supervised Quantum Classification Model."""
|
|
1228
|
+
|
|
1229
|
+
def __init__(
|
|
1230
|
+
self,
|
|
1231
|
+
size: int=64,
|
|
1232
|
+
levels: int=2,
|
|
1233
|
+
seed: int=42,
|
|
1234
|
+
shots: int=1024,
|
|
1235
|
+
channel: Optional[str]=None,
|
|
1236
|
+
instance: Optional[str]=None,
|
|
1237
|
+
backend: Optional[str]=None,
|
|
1238
|
+
api_key: Optional[str]=None,
|
|
1239
|
+
noise_model_from: Optional[str]=None
|
|
1240
|
+
) -> "QuantumClassificationModel":
|
|
1241
|
+
"""Initialize a QuantumClassificationModel object.
|
|
1242
|
+
Run the classification model on a simulator with Qiskit by default.
|
|
1243
|
+
It can interact with specific IBM channels, instances, and backends if specified (it requires an IBM account).
|
|
1244
|
+
|
|
1245
|
+
Parameters
|
|
1246
|
+
----------
|
|
1247
|
+
size : int, default 64
|
|
1248
|
+
The vectors dimensionality as power of 2.
|
|
1249
|
+
levels : int, default 2
|
|
1250
|
+
Number of level vectors.
|
|
1251
|
+
seed : int, default 42
|
|
1252
|
+
Seed for reproducibility.
|
|
1253
|
+
shots : int, default 1024
|
|
1254
|
+
The number of times to run the quantum circuit for the Hadamard test.
|
|
1255
|
+
channel : str, default None, optional
|
|
1256
|
+
IBM channel.
|
|
1257
|
+
instance : str, default None, optional
|
|
1258
|
+
IBM instance. Required in case of specific channel only.
|
|
1259
|
+
backend : str, default None, optional
|
|
1260
|
+
IBM backend (e.g., "ibm_cleveland"). Required in case of specific instance only.
|
|
1261
|
+
If `instance` is not None, this is "least_busy" by default.
|
|
1262
|
+
api_key : str, default None, optional
|
|
1263
|
+
IBM API key. Required in case of specific backend only.
|
|
1264
|
+
noise_model_from : str, default None, optional
|
|
1265
|
+
The name of a real IBM backend (e.g., "ibm_cleveland") to build a noise model from the simulation.
|
|
1266
|
+
If provided, `api_key` is required. This parameter is ignored if `channel`, `instance`, and `backend` are provided for hardware execution.
|
|
1267
|
+
Noise models are retrieved from the "ibm_quantum_platform" channel.
|
|
1268
|
+
|
|
1269
|
+
Raises
|
|
1270
|
+
------
|
|
1271
|
+
ValueError
|
|
1272
|
+
If the vector dimensionality `size` is not a power of 2.
|
|
1273
|
+
TypeError
|
|
1274
|
+
If seed is not an integer.
|
|
1275
|
+
|
|
1276
|
+
Examples
|
|
1277
|
+
--------
|
|
1278
|
+
>>> from hdlib.model import QuantumClassificationModel
|
|
1279
|
+
>>> model = QuantumClassificationModel(size=32, levels=2)
|
|
1280
|
+
>>> type(model)
|
|
1281
|
+
<class 'hdlib.model.QuantumClassificationModel'>
|
|
1282
|
+
|
|
1283
|
+
This creates a new QuantumClassificationModel object with random bipolar vectors with size 32 and 2 level vectors.
|
|
1284
|
+
"""
|
|
1285
|
+
|
|
1286
|
+
if not ((size > 0) and ((size & (size - 1)) == 0)):
|
|
1287
|
+
# Check if a the vector dimensionality is a power of 2.
|
|
1288
|
+
raise ValueError("The vector dimensionality must be a power of 2.")
|
|
1289
|
+
|
|
1290
|
+
self.size = size
|
|
1291
|
+
self.levels = levels
|
|
1292
|
+
self.shots = shots
|
|
1293
|
+
|
|
1294
|
+
# Vectors must be bipolar here
|
|
1295
|
+
self.vtype = "bipolar"
|
|
1296
|
+
|
|
1297
|
+
# Keep track of the level vectors
|
|
1298
|
+
self.level_hvs = list()
|
|
1299
|
+
|
|
1300
|
+
# Keep track of class prototype vectors
|
|
1301
|
+
# This is filled up during `fit`
|
|
1302
|
+
self.prototypes = list()
|
|
1303
|
+
|
|
1304
|
+
if channel is None:
|
|
1305
|
+
# Use a simulator if no channel is specified
|
|
1306
|
+
noise_model = None
|
|
1307
|
+
|
|
1308
|
+
if noise_model_from:
|
|
1309
|
+
if not api_key:
|
|
1310
|
+
raise ValueError("`api_key` must be provided to fetch backend properties for a noise model.")
|
|
1311
|
+
|
|
1312
|
+
# Initialize a temporary service connection
|
|
1313
|
+
# Always use "ibm_quantum_platform" to fetch the backend properties
|
|
1314
|
+
noise_service = QiskitRuntimeService(channel="ibm_quantum_platform", token=api_key)
|
|
1315
|
+
|
|
1316
|
+
# Retrieve a backend
|
|
1317
|
+
# We only need its noise model
|
|
1318
|
+
backend_for_noise = noise_service.backend(noise_model_from)
|
|
1319
|
+
|
|
1320
|
+
# Finally, define the noise model
|
|
1321
|
+
noise_model = NoiseModel.from_backend(backend_for_noise)
|
|
1322
|
+
|
|
1323
|
+
# Use a simulator if no channel is specified
|
|
1324
|
+
# This can be noise-free or use a specific noise model
|
|
1325
|
+
try:
|
|
1326
|
+
# Attempt to initialize with GPU acceleration
|
|
1327
|
+
self.backend = AerSimulator(device="GPU", noise_model=noise_model)
|
|
1328
|
+
|
|
1329
|
+
except Exception as e:
|
|
1330
|
+
# Fallback to the default CPU device
|
|
1331
|
+
print(f"GPU not available, falling back to CPU.")
|
|
1332
|
+
|
|
1333
|
+
self.backend = AerSimulator(noise_model=noise_model)
|
|
1334
|
+
|
|
1335
|
+
else:
|
|
1336
|
+
# Initialize a quantum runtime service for a specific IBM QC channel, instance, and backend
|
|
1337
|
+
service = QiskitRuntimeService(channel=channel, token=api_key, instance=instance)
|
|
1338
|
+
|
|
1339
|
+
# The backend is the "least_busy" by default
|
|
1340
|
+
if backend is None:
|
|
1341
|
+
self.backend = service.least_busy(operational=True, simulator=False)
|
|
1342
|
+
|
|
1343
|
+
else:
|
|
1344
|
+
self.backend = service.backend(backend)
|
|
1345
|
+
|
|
1346
|
+
# Conditions on random seed for reproducibility
|
|
1347
|
+
# numpy allows integers as random seeds
|
|
1348
|
+
if not isinstance(seed, int):
|
|
1349
|
+
raise TypeError("Seed must be an integer number")
|
|
1350
|
+
|
|
1351
|
+
self.seed = seed
|
|
1352
|
+
|
|
1353
|
+
# Keep track of hdlib version
|
|
1354
|
+
self.version = __version__
|
|
1355
|
+
|
|
1356
|
+
def __str__(self) -> str:
|
|
1357
|
+
"""Print the QuantumClassificationModel object properties.
|
|
1358
|
+
|
|
1359
|
+
Returns
|
|
1360
|
+
-------
|
|
1361
|
+
str
|
|
1362
|
+
A description of the QuantumClassificationModel object. It reports the vectors size, the vector type,
|
|
1363
|
+
the number of level vectors, and the number of shots.
|
|
1364
|
+
|
|
1365
|
+
Examples
|
|
1366
|
+
--------
|
|
1367
|
+
>>> from hdlib.model import QuantumClassificationModel
|
|
1368
|
+
>>> model = QuantumClassificationModel()
|
|
1369
|
+
>>> print(model)
|
|
1370
|
+
|
|
1371
|
+
Class: hdlib.model.classification.QuantumClassificationModel
|
|
1372
|
+
Version: 2.0.0
|
|
1373
|
+
Size: 64
|
|
1374
|
+
Type: bipolar
|
|
1375
|
+
Levels: 2
|
|
1376
|
+
Shots: 1024
|
|
1377
|
+
|
|
1378
|
+
Print the QuantumClassificationModel object properties.
|
|
1379
|
+
"""
|
|
1380
|
+
|
|
1381
|
+
return f"""
|
|
1382
|
+
Class: hdlib.model.classification.QuantumClassificationModel
|
|
1383
|
+
Version: {self.version}
|
|
1384
|
+
Size: {self.size}
|
|
1385
|
+
Type: {self.vtype}
|
|
1386
|
+
Levels: {self.levels}
|
|
1387
|
+
Shots: {self.shots}
|
|
1388
|
+
"""
|
|
1389
|
+
|
|
1390
|
+
def _build_quantum_sample_encoder(self, sample_row: List[float], level_vectors: List[np.ndarray], D: int) -> List[QuantumCircuit]:
|
|
1391
|
+
"""Creates a single quantum circuit that encodes one real-valued sample by quantumly permuting its feature vectors.
|
|
1392
|
+
|
|
1393
|
+
Parameters
|
|
1394
|
+
----------
|
|
1395
|
+
sample_row : list
|
|
1396
|
+
Single sample as list of numerical values (float).
|
|
1397
|
+
level_vectors : list
|
|
1398
|
+
List of level vector.
|
|
1399
|
+
D : int
|
|
1400
|
+
Vector dimensionality.
|
|
1401
|
+
|
|
1402
|
+
Returns
|
|
1403
|
+
-------
|
|
1404
|
+
List[QuantumCircuit]
|
|
1405
|
+
The list of quantum feature circuits for encoding the input sample.
|
|
1406
|
+
"""
|
|
1407
|
+
|
|
1408
|
+
num_qubits = int(log2(D))
|
|
1409
|
+
feature_circuits = list()
|
|
1410
|
+
|
|
1411
|
+
for i, value in enumerate(sample_row):
|
|
1412
|
+
level_index = max(0, min(len(level_vectors) - 1, int(value * (len(level_vectors) - 1))))
|
|
1413
|
+
level_vec = level_vectors[level_index]
|
|
1414
|
+
|
|
1415
|
+
# Create a quantum circuit for this single feature
|
|
1416
|
+
feature_qc = quantum_encode(level_vec, label=f"Feature_{i}")
|
|
1417
|
+
|
|
1418
|
+
# Apply the permutation based in the feature index
|
|
1419
|
+
feature_qc = quantum_permute(feature_qc, num_qubits, shift=i)
|
|
1420
|
+
|
|
1421
|
+
# Report circuit metrics
|
|
1422
|
+
#print(f"Positional permutation metrics: {get_circuit_metrics(feature_qc, num_qubits, self.backend, optimization_level=3)}")
|
|
1423
|
+
|
|
1424
|
+
feature_circuits.append(feature_qc)
|
|
1425
|
+
|
|
1426
|
+
# The feature circuits are bundled all together with the feature circuits of the other samples
|
|
1427
|
+
# belonging to the same class to build the circuit representation of that class.
|
|
1428
|
+
return feature_circuits
|
|
1429
|
+
|
|
1430
|
+
def _scale_circuit_phases(self, circuit: Any, factor: float) -> Any:
|
|
1431
|
+
"""Helper to safely scale the precise phase angles of a compressed circuit.
|
|
1432
|
+
"""
|
|
1433
|
+
|
|
1434
|
+
new_circ = QuantumCircuit(circuit.num_qubits, name=circuit.name)
|
|
1435
|
+
|
|
1436
|
+
for instr in circuit.data:
|
|
1437
|
+
if instr.operation.name == "diagonal":
|
|
1438
|
+
# Extract the exact angles and multiply by the factor (negates if factor is negative)
|
|
1439
|
+
phases = np.angle(np.array(instr.operation.params, dtype=complex))
|
|
1440
|
+
new_phases = np.exp(1j * phases * factor)
|
|
1441
|
+
|
|
1442
|
+
GateClass = instr.operation.__class__
|
|
1443
|
+
new_circ.append(GateClass(new_phases.tolist()), instr.qubits)
|
|
1444
|
+
|
|
1445
|
+
else:
|
|
1446
|
+
new_circ.append(instr)
|
|
1447
|
+
|
|
1448
|
+
return new_circ
|
|
1449
|
+
|
|
1450
|
+
def fit(self, train_points: List[List[float]], train_labels: List[str]) -> None:
|
|
1451
|
+
"""Build a vector-symbolic architecture. Define level vectors, encode samples, and build prototypes.
|
|
1452
|
+
|
|
1453
|
+
Parameters
|
|
1454
|
+
----------
|
|
1455
|
+
train_points : list
|
|
1456
|
+
List of lists with numerical data points (floats).
|
|
1457
|
+
train_labels : list
|
|
1458
|
+
List with class labels. It has the same size of `points`.
|
|
1459
|
+
"""
|
|
1460
|
+
|
|
1461
|
+
def _generate_level_vectors(D, num_levels, rng):
|
|
1462
|
+
"""Generates a set of bipolar vectors for discretizing real numbers.
|
|
1463
|
+
"""
|
|
1464
|
+
|
|
1465
|
+
level_vectors = [rng.choice([-1, 1], size=D)]
|
|
1466
|
+
|
|
1467
|
+
change = int(D / 2)
|
|
1468
|
+
next_level = int((D / 2 / num_levels))
|
|
1469
|
+
|
|
1470
|
+
for i in range(1, num_levels):
|
|
1471
|
+
prev_vec = level_vectors[i-1].copy()
|
|
1472
|
+
|
|
1473
|
+
if i-1 == 0:
|
|
1474
|
+
flip_indices = rng.choice(D, size=change, replace=False)
|
|
1475
|
+
|
|
1476
|
+
else:
|
|
1477
|
+
flip_indices = rng.choice(D, size=next_level, replace=False)
|
|
1478
|
+
|
|
1479
|
+
prev_vec[flip_indices] *= -1
|
|
1480
|
+
level_vectors.append(prev_vec)
|
|
1481
|
+
|
|
1482
|
+
return level_vectors
|
|
1483
|
+
|
|
1484
|
+
rand = np.random.default_rng(seed=self.seed)
|
|
1485
|
+
|
|
1486
|
+
# Create level vectors
|
|
1487
|
+
self.level_hvs = _generate_level_vectors(self.size, self.levels, rand)
|
|
1488
|
+
|
|
1489
|
+
self.classes_ = sorted(list(set(train_labels)))
|
|
1490
|
+
|
|
1491
|
+
# Hold class prototypes
|
|
1492
|
+
self.prototypes = dict()
|
|
1493
|
+
|
|
1494
|
+
# Track the mass of each class so we can normalize updates proportionally later
|
|
1495
|
+
self.class_counts_ = dict()
|
|
1496
|
+
|
|
1497
|
+
for c in self.classes_:
|
|
1498
|
+
# Building quantum prototype for Class `c`
|
|
1499
|
+
class_samples = [sample for pos, sample in enumerate(train_points) if train_labels[pos] == c]
|
|
1500
|
+
|
|
1501
|
+
# Building quantum samples' feature encoder circuits
|
|
1502
|
+
sample_encoders = [encoder for sample in class_samples for encoder in self._build_quantum_sample_encoder(sample, self.level_hvs, self.size)]
|
|
1503
|
+
|
|
1504
|
+
# Bundling
|
|
1505
|
+
prototype_circuit = compress_circuit(quantum_bundle(sample_encoders))
|
|
1506
|
+
prototype_circuit.name = f"Prototype_{c}"
|
|
1507
|
+
|
|
1508
|
+
# Report circuit metrics
|
|
1509
|
+
#print(f"Class prototype bundling metrics: {get_circuit_metrics(prototype_circuit, int(log2(self.size)), self.backend, optimization_level=3)}")
|
|
1510
|
+
|
|
1511
|
+
# The prototype is the circuit that prepares the state
|
|
1512
|
+
self.prototypes[c] = prototype_circuit
|
|
1513
|
+
self.class_counts_[c] = float(len(class_samples))
|
|
1514
|
+
|
|
1515
|
+
def predict(self, test_points: List[List[float]]) -> Tuple[List[str], List[List[float]]]:
|
|
1516
|
+
"""Predict the class labels of the data points in the test set.
|
|
1517
|
+
|
|
1518
|
+
Parameters
|
|
1519
|
+
----------
|
|
1520
|
+
test_points : list
|
|
1521
|
+
Test data points.
|
|
1522
|
+
|
|
1523
|
+
Returns
|
|
1524
|
+
-------
|
|
1525
|
+
Tuple
|
|
1526
|
+
A list with the predicted class labels in the same order of data points in `test_points`,
|
|
1527
|
+
and a list of similarities between the test samples and the class prototypes.
|
|
1528
|
+
|
|
1529
|
+
Raises
|
|
1530
|
+
------
|
|
1531
|
+
RuntimeError
|
|
1532
|
+
If `predict` is called before `fit`.
|
|
1533
|
+
"""
|
|
1534
|
+
|
|
1535
|
+
if not hasattr(self, "classes_"):
|
|
1536
|
+
raise RuntimeError("You must call fit before calling predict.")
|
|
1537
|
+
|
|
1538
|
+
# Check whether the test should be performed on a simulator or on the quantum hardware
|
|
1539
|
+
is_simulated = isinstance(self.backend, AerSimulator)
|
|
1540
|
+
|
|
1541
|
+
predictions = list()
|
|
1542
|
+
similarities = list()
|
|
1543
|
+
|
|
1544
|
+
# For hardware, manage a single session for all prediction tasks
|
|
1545
|
+
with (Session(backend=self.backend) if not is_simulated else nullcontext()) as session:
|
|
1546
|
+
sampler = None
|
|
1547
|
+
|
|
1548
|
+
# Use the session-based Sampler if on hardware
|
|
1549
|
+
if session:
|
|
1550
|
+
options = SamplerOptions()
|
|
1551
|
+
|
|
1552
|
+
# Set the default number of shots
|
|
1553
|
+
options.default_shots = self.shots
|
|
1554
|
+
|
|
1555
|
+
# Enable dynamical decoupling
|
|
1556
|
+
options.dynamical_decoupling.enable = True
|
|
1557
|
+
options.dynamical_decoupling.sequence_type = "XpXm"
|
|
1558
|
+
|
|
1559
|
+
# Enable gate twirling
|
|
1560
|
+
options.twirling.enable_gates = True
|
|
1561
|
+
|
|
1562
|
+
sampler = Sampler(mode=session, options=options)
|
|
1563
|
+
|
|
1564
|
+
# Build all query circuits
|
|
1565
|
+
queries = list()
|
|
1566
|
+
|
|
1567
|
+
for sample in test_points:
|
|
1568
|
+
# Get the list of feature circuits for the test sample
|
|
1569
|
+
sample_features_circuits = self._build_quantum_sample_encoder(sample, self.level_hvs, self.size)
|
|
1570
|
+
|
|
1571
|
+
# Bundle them into a single query circuit
|
|
1572
|
+
query_circuit = compress_circuit(quantum_bundle(sample_features_circuits))
|
|
1573
|
+
|
|
1574
|
+
# Report circuit metrics
|
|
1575
|
+
#print(f"Query bundling metrics: {get_circuit_metrics(query_circuit, int(log2(self.size)), self.backend, optimization_level=3)}")
|
|
1576
|
+
|
|
1577
|
+
queries.append(query_circuit)
|
|
1578
|
+
|
|
1579
|
+
# Evaluate all queries against class prototypes in safe hardware batch mode to prevent IBM memory overflow
|
|
1580
|
+
# https://ibm.biz/error_codes#6073
|
|
1581
|
+
max_queries_per_job = 5
|
|
1582
|
+
|
|
1583
|
+
# Sort prototype circuits based on self.classes_
|
|
1584
|
+
prototype_circuits = [self.prototypes[c] for c in self.classes_]
|
|
1585
|
+
|
|
1586
|
+
for i in range(0, len(queries), max_queries_per_job):
|
|
1587
|
+
query_batch = queries[i : i + max_queries_per_job]
|
|
1588
|
+
|
|
1589
|
+
sims, _ = run_compute_uncompute_test(query_batch, prototype_circuits, self.backend, shots=self.shots, seed=self.seed, sampler=sampler)
|
|
1590
|
+
|
|
1591
|
+
for sim in sims:
|
|
1592
|
+
predictions.append(self.classes_[int(np.argmax(sim))])
|
|
1593
|
+
similarities.append(sim)
|
|
1594
|
+
|
|
1595
|
+
return predictions, similarities
|
|
1596
|
+
|
|
1597
|
+
def retrain(self, train_points: List[List[float]], train_labels: List[str], epochs: int=10, lr: float=1.0, verbose: bool=True, test_points: Optional[List[List[float]]]=None, test_labels: Optional[List[str]]=None, early_stop: bool=True) -> Tuple[float, int]:
|
|
1598
|
+
"""Retrain the model by adjusting class prototypes based on misclassified samples.
|
|
1599
|
+
|
|
1600
|
+
Per-epoch error rates are recorded on ``self.retrain_history_`` as a list
|
|
1601
|
+
of ``{"epoch": int, "error": float}`` entries (starting at epoch 0), so
|
|
1602
|
+
callers can access the training curve without parsing stdout. Setting
|
|
1603
|
+
``verbose=False`` suppresses the per-epoch prints.
|
|
1604
|
+
|
|
1605
|
+
If ``test_points`` (and ``test_labels``) are provided, each history entry
|
|
1606
|
+
also carries a ``"test_error"`` field measured on that held-out set. This
|
|
1607
|
+
lets callers plot a genuine test-error-vs-epoch curve, but note it costs
|
|
1608
|
+
one extra ``predict`` over the test set per epoch (the expensive step for
|
|
1609
|
+
the quantum model), so it is opt-in.
|
|
1610
|
+
|
|
1611
|
+
``early_stop`` (default ``True``) preserves the original behaviour: as soon
|
|
1612
|
+
as an epoch fails to reduce the training error the best-known prototypes are
|
|
1613
|
+
restored and the loop exits. Set ``early_stop=False`` to run the full
|
|
1614
|
+
``epochs`` budget and record the complete curve; the best-known prototypes
|
|
1615
|
+
are still restored before returning, so the returned model is unchanged by
|
|
1616
|
+
this flag.
|
|
1617
|
+
"""
|
|
1618
|
+
|
|
1619
|
+
if not hasattr(self, "classes_"):
|
|
1620
|
+
raise RuntimeError("You must call fit before calling retrain.")
|
|
1621
|
+
|
|
1622
|
+
if test_points is not None and test_labels is None:
|
|
1623
|
+
raise ValueError("test_labels must be provided when test_points is given.")
|
|
1624
|
+
|
|
1625
|
+
def _error(points, labels):
|
|
1626
|
+
preds, _ = self.predict(points)
|
|
1627
|
+
return sum(1 for p, t in zip(preds, labels) if p != t) / len(labels)
|
|
1628
|
+
|
|
1629
|
+
# 1. Evaluate the base model before any retraining to establish a baseline
|
|
1630
|
+
predictions, _ = self.predict(train_points)
|
|
1631
|
+
|
|
1632
|
+
best_error = sum(1 for p, t in zip(predictions, train_labels) if p != t) / len(train_labels)
|
|
1633
|
+
epoch_0 = {"epoch": 0, "error": float(best_error)}
|
|
1634
|
+
if test_points is not None:
|
|
1635
|
+
epoch_0["test_error"] = float(_error(test_points, test_labels))
|
|
1636
|
+
self.retrain_history_ = [epoch_0]
|
|
1637
|
+
if verbose:
|
|
1638
|
+
print(f"\tepoch 0: {best_error}")
|
|
1639
|
+
|
|
1640
|
+
if best_error == 0.0:
|
|
1641
|
+
return best_error, 0
|
|
1642
|
+
|
|
1643
|
+
# Save a backup of the best circuits
|
|
1644
|
+
best_prototypes = {c: qc.copy() for c, qc in self.prototypes.items()}
|
|
1645
|
+
|
|
1646
|
+
final_epoch = 0
|
|
1647
|
+
|
|
1648
|
+
for epoch in range(1, epochs + 1):
|
|
1649
|
+
final_epoch = epoch
|
|
1650
|
+
|
|
1651
|
+
# Queue to store correctly phase-scaled updates
|
|
1652
|
+
epoch_updates = {c: [] for c in self.classes_}
|
|
1653
|
+
updates_made = False
|
|
1654
|
+
|
|
1655
|
+
# 2. Adjust circuits based on misclassifications
|
|
1656
|
+
for i, (pred, true_label) in enumerate(zip(predictions, train_labels)):
|
|
1657
|
+
if pred != true_label:
|
|
1658
|
+
# Encode the misclassified sample
|
|
1659
|
+
raw_features = self._build_quantum_sample_encoder(train_points[i], self.level_hvs, self.size)
|
|
1660
|
+
sample_update = compress_circuit(quantum_bundle(raw_features))
|
|
1661
|
+
|
|
1662
|
+
# Calculate exact scale factor to match the prototype's mass denominator.
|
|
1663
|
+
# lr defaults to 1.0 (pure Perceptron rule), but is adjusted based on class population
|
|
1664
|
+
factor_true = lr / self.class_counts_[true_label]
|
|
1665
|
+
factor_pred = lr / self.class_counts_[pred]
|
|
1666
|
+
|
|
1667
|
+
# Add to the true class
|
|
1668
|
+
epoch_updates[true_label].append(self._scale_circuit_phases(sample_update, factor=factor_true))
|
|
1669
|
+
|
|
1670
|
+
# Remove from the incorrectly predicted class (negative factor)
|
|
1671
|
+
epoch_updates[pred].append(self._scale_circuit_phases(sample_update, factor=-factor_pred))
|
|
1672
|
+
|
|
1673
|
+
updates_made = True
|
|
1674
|
+
|
|
1675
|
+
if updates_made:
|
|
1676
|
+
for label, updates in epoch_updates.items():
|
|
1677
|
+
if updates:
|
|
1678
|
+
# 3. Apply the update to the prototype
|
|
1679
|
+
new_proto = self.prototypes[label].copy()
|
|
1680
|
+
|
|
1681
|
+
for upd in updates:
|
|
1682
|
+
new_proto.compose(upd, inplace=True)
|
|
1683
|
+
|
|
1684
|
+
self.prototypes[label] = compress_circuit(new_proto)
|
|
1685
|
+
self.prototypes[label].name = f"Prototype_{label}"
|
|
1686
|
+
|
|
1687
|
+
# 4. Evaluate the newly updated model
|
|
1688
|
+
predictions, _ = self.predict(train_points)
|
|
1689
|
+
|
|
1690
|
+
current_error = sum(1 for p, t in zip(predictions, train_labels) if p != t) / len(train_labels)
|
|
1691
|
+
epoch_entry = {"epoch": epoch, "error": float(current_error)}
|
|
1692
|
+
if test_points is not None:
|
|
1693
|
+
epoch_entry["test_error"] = float(_error(test_points, test_labels))
|
|
1694
|
+
self.retrain_history_.append(epoch_entry)
|
|
1695
|
+
if verbose:
|
|
1696
|
+
print(f"\tepoch {epoch}: {current_error}")
|
|
1697
|
+
|
|
1698
|
+
# 5. Track the best-known state; optionally stop early
|
|
1699
|
+
if current_error < best_error:
|
|
1700
|
+
# Improvement found: update the best error and save the new state
|
|
1701
|
+
best_error = current_error
|
|
1702
|
+
best_prototypes = {c: qc.copy() for c, qc in self.prototypes.items()}
|
|
1703
|
+
|
|
1704
|
+
if best_error == 0.0:
|
|
1705
|
+
break
|
|
1706
|
+
elif early_stop:
|
|
1707
|
+
# The updates made the model worse (or it stopped improving).
|
|
1708
|
+
# Revert to the best known state and exit!
|
|
1709
|
+
self.prototypes = best_prototypes
|
|
1710
|
+
|
|
1711
|
+
break
|
|
1712
|
+
# When early_stop is False the loop keeps running from the current
|
|
1713
|
+
# (possibly worse) prototypes so the full curve is recorded; the best
|
|
1714
|
+
# state is restored just before returning below.
|
|
1715
|
+
|
|
1716
|
+
# Always return the best-known prototypes, regardless of early_stop.
|
|
1717
|
+
self.prototypes = best_prototypes
|
|
1718
|
+
|
|
1719
|
+
return best_error, final_epoch
|