hdlib 2.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1719 @@
1
+ """Classification with hdlib.
2
+
3
+ It implements the __hdlib.model.classification.ClassificationModel__ class object which allows to generate, fit, and test a classification model
4
+ built according to the Hyperdimensional Computing (HDC) paradigm as described in _Cumbo et al. 2020_ https://doi.org/10.3390/a13090233.
5
+
6
+ It also implements a stepwise regression model as backward and forward variable elimination techniques for selecting
7
+ relevant features in a dataset according to the same HDC paradigm.
8
+
9
+ The quantum version of this classification model is also provided here in __hdlib.model.classification.QuantumClassificationModel__
10
+ as described in _Cumbo et al. 2025_ https://doi.org/10.48550/arXiv.2511.12664."""
11
+
12
+ import copy
13
+ import itertools
14
+ import multiprocessing as mp
15
+ import os
16
+ import statistics
17
+ from math import log2
18
+ from functools import partial
19
+ from typing import Any, Dict, List, Optional, Set, Tuple
20
+ from contextlib import nullcontext
21
+
22
+ import numpy as np
23
+ from sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score
24
+ from sklearn.model_selection import StratifiedKFold
25
+ from qiskit import QuantumCircuit, QuantumRegister, transpile
26
+ from qiskit.quantum_info import Statevector
27
+ from qiskit_aer import AerSimulator
28
+ from qiskit_aer.noise import NoiseModel
29
+ from qiskit_ibm_runtime import (
30
+ QiskitRuntimeService,
31
+ Sampler,
32
+ SamplerOptions,
33
+ Session
34
+ )
35
+
36
+ from hdlib import __version__
37
+ from hdlib.space import Space
38
+ from hdlib.vector import Vector
39
+ from hdlib.arithmetic import bundle, permute
40
+
41
+ # Quantum functions
42
+ from hdlib.arithmetic.quantum import (
43
+ encode as quantum_encode,
44
+ bundle as quantum_bundle,
45
+ permute as quantum_permute,
46
+ compress_circuit,
47
+ run_compute_uncompute_test,
48
+ get_circuit_metrics,
49
+ )
50
+
51
+
52
+ class ClassificationModel(object):
53
+ """Supervised Classification Model."""
54
+
55
+ def __init__(
56
+ self,
57
+ size: int=10000,
58
+ levels: int=2,
59
+ vtype: str="bipolar",
60
+ ) -> "ClassificationModel":
61
+ """Initialize a ClassificationModel object.
62
+
63
+ Parameters
64
+ ----------
65
+ size : int, default 10000
66
+ The size of vectors used to create a Space and define Vector objects.
67
+ levels : int, default 2
68
+ The number of level vectors used to represent numerical data. It is 2 by default.
69
+ vtype : {'binary', 'bipolar'}, default 'bipolar'
70
+ The vector type in space, which is bipolar by default.
71
+
72
+ Raises
73
+ ------
74
+ TypeError
75
+ If the vector size or the number of levels are not integer numbers.
76
+ ValueError
77
+ If the number of level vectors is lower than 2.
78
+
79
+ Examples
80
+ --------
81
+ >>> from hdlib.model import ClassificationModel
82
+ >>> model = ClassificationModel(size=10000, levels=100, vtype='bipolar')
83
+ >>> type(model)
84
+ <class 'hdlib.model.ClassificationModel'>
85
+
86
+ This creates a new ClassificationModel object around a Space that can host random bipolar Vector objects with size 10,000.
87
+ It also defines the number of level vectors to 100.
88
+ """
89
+
90
+ if not isinstance(size, int):
91
+ raise TypeError("Vectors size must be an integer number")
92
+
93
+ # Register vectors dimensionality
94
+ self.size = size
95
+
96
+ if not isinstance(levels, int):
97
+ raise TypeError("Levels must be an integer number")
98
+
99
+ if levels < 2:
100
+ raise ValueError("The number of levels must be greater than or equal to 2")
101
+
102
+ # Register the number of levels
103
+ self.levels = levels
104
+
105
+ # Minimum and maximum values in the input dataset
106
+ # This is used to define the level boundaries
107
+ self.min_value = None
108
+ self.max_value = None
109
+
110
+ # List of level boundaries
111
+ self.level_list = list()
112
+
113
+ if vtype not in ("bipolar", "binary"):
114
+ raise ValueError("Vectors type can be binary or bipolar only")
115
+
116
+ # Register vectors type
117
+ self.vtype = vtype.lower()
118
+
119
+ # Hyperdimensional space
120
+ self.space = None
121
+
122
+ # Class labels
123
+ self.classes = set()
124
+
125
+ # Keep track of hdlib version
126
+ self.version = __version__
127
+
128
+ def __str__(self) -> str:
129
+ """Print the ClassificationModel object properties.
130
+
131
+ Returns
132
+ -------
133
+ str
134
+ A description of the ClassificationModel object. It reports the vectors size, the vector type,
135
+ the number of level vectors, the number of data points, and the number of class labels.
136
+
137
+ Examples
138
+ --------
139
+ >>> from hdlib.model import ClassificationModel
140
+ >>> model = ClassificationModel()
141
+ >>> print(model)
142
+
143
+ Class: hdlib.model.classification.ClassificationModel
144
+ Version: 0.1.17
145
+ Size: 10000
146
+ Type: bipolar
147
+ Levels: 2
148
+ Points: 0
149
+ Classes:
150
+
151
+ []
152
+
153
+ Print the ClassificationModel object properties. By default, the size of vectors in space is 10,000,
154
+ their type is bipolar, and the number of level vectors is 2. The number of data points
155
+ and the number of class labels are empty here since no dataset has been processed yet.
156
+ """
157
+
158
+ return f"""
159
+ Class: hdlib.model.classification.ClassificationModel
160
+ Version: {self.version}
161
+ Size: {self.size}
162
+ Type: {self.vtype}
163
+ Levels: {self.levels}
164
+ Points: {len(self.space.memory()) - self.levels if self.space is not None else 0}
165
+ Classes:
166
+
167
+ {np.array(list(self.classes))}
168
+ """
169
+
170
+ def _init_fit_predict(
171
+ self,
172
+ size: int=10000,
173
+ levels: int=2,
174
+ vtype: str="bipolar",
175
+ points: Optional[List[List[float]]]=None,
176
+ labels: Optional[List[str]]=None,
177
+ cv: int=5,
178
+ distance_method: str="cosine",
179
+ retrain: int=0,
180
+ n_jobs: int=1,
181
+ metric: str="accuracy"
182
+ ) -> Tuple[int, int, float]:
183
+ """Initialize a new ClassificationModel, then fit and cross-validate it. Used for size and levels hyperparameters tuning.
184
+
185
+ Parameters
186
+ ----------
187
+ size : int, default 10000
188
+ The size of vectors used to create a Space and define Vector objects.
189
+ levels : int, default 2
190
+ The number of level vectors used to represent numerical data. It is 2 by default.
191
+ vtype : {'binary', 'bipolar'}, default 'bipolar'
192
+ The vector type in space, which is bipolar by default.
193
+ points : list
194
+ List of lists with numerical data (floats).
195
+ labels : list
196
+ List with class labels. It has the same size of `points`.
197
+ cv : int, default 5
198
+ Number of folds for cross-validating the model.
199
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
200
+ Method used to compute the distance between vectors in space.
201
+ retrain : int, default 0
202
+ Number of retraining iterations.
203
+ n_jobs : int, default 1,
204
+ Number of jobs for processing folds in parallel.
205
+ metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
206
+ Metric used to evaluate the model.
207
+
208
+ Returns
209
+ -------
210
+ tuple
211
+ A tuple with the input size, the number of level vectors, and the model score
212
+ according to the input metric.
213
+
214
+ Raises
215
+ ------
216
+ ValueError
217
+ If the provided metric is not supported.
218
+ """
219
+
220
+ # Available score metrics
221
+ score_metrics = {
222
+ "accuracy": accuracy_score,
223
+ "f1": f1_score,
224
+ "precision": precision_score,
225
+ "recall": recall_score
226
+ }
227
+
228
+ metric = metric.lower()
229
+
230
+ if metric not in score_metrics:
231
+ raise ValueError("Score metric {} is not supported".format(metric))
232
+
233
+ # Generate a new Model
234
+ model = ClassificationModel(size=size, levels=levels, vtype=vtype)
235
+
236
+ # Fit the model
237
+ model.fit(points, labels=labels)
238
+
239
+ # Cross-validate the model
240
+ predictions = model.cross_val_predict(
241
+ points,
242
+ labels,
243
+ cv=cv,
244
+ distance_method=distance_method,
245
+ retrain=retrain,
246
+ n_jobs=n_jobs
247
+ )
248
+
249
+ # For each prediction, compute the score and return the average
250
+ scores = list()
251
+
252
+ for y_indices, y_pred, _, _, _, _ in predictions:
253
+ y_true = [label for position, label in enumerate(labels) if position in y_indices]
254
+
255
+ if metric == "accuracy":
256
+ scores.append(score_metrics[metric](y_true, y_pred))
257
+
258
+ else:
259
+ # Use the average weighted to account for label imbalance
260
+ scores.append(score_metrics[metric](y_true, y_pred, average="weighted"))
261
+
262
+ return size, levels, statistics.mean(scores)
263
+
264
+ def fit(
265
+ self,
266
+ points: List[List[float]],
267
+ labels: List[str],
268
+ seed: Optional[int]=None,
269
+ ) -> None:
270
+ """Build a vector-symbolic architecture. Define level vectors and encode samples.
271
+
272
+ Parameters
273
+ ----------
274
+ points : list
275
+ List of lists with numerical data (floats).
276
+ labels : list
277
+ List with class labels. It has the same size of `points`.
278
+ seed : int, optional
279
+ An optional seed for reproducibly generating the vectors numpy.ndarray randomly.
280
+
281
+ Raises
282
+ ------
283
+ Exception
284
+ - if there are not enough data points (the length of `points` is < 3);
285
+ - if the length of `points` does not match the length of `labels`;
286
+ - if there is only one class label.
287
+ """
288
+
289
+ if len(points) < 3:
290
+ # This is based on the assumption that the minimum number of data points for training
291
+ # the classification model is 2, while 1 data point is enough for the test set
292
+ raise Exception("Not enough data points")
293
+
294
+ if len(points) != len(labels):
295
+ raise Exception("The number of data points does not match with the number of class labels")
296
+
297
+ if len(set(labels)) < 2:
298
+ raise Exception("The number of unique class labels must be > 1")
299
+
300
+ self.classes = set(labels)
301
+
302
+ # Initialize the hyperdimensional space so that it overwrites any existing space in Model
303
+ self.space = Space(size=self.size, vtype=self.vtype)
304
+
305
+ index_vector = range(self.size)
306
+
307
+ change = int(self.size / 2)
308
+ next_level = int((self.size / 2 / self.levels))
309
+
310
+ # Also define the interval level list
311
+ self.level_list = list()
312
+
313
+ # Get the minimum and maximum value in the input dataset
314
+ self.min_value = np.inf
315
+ self.max_value = -np.inf
316
+
317
+ for point in points:
318
+ min_point = min(point)
319
+ max_point = max(point)
320
+
321
+ if min_point < self.min_value:
322
+ self.min_value = min_point
323
+
324
+ if max_point > self.max_value:
325
+ self.max_value = max_point
326
+
327
+ gap = (self.max_value - self.min_value) / self.levels
328
+
329
+ if seed is None:
330
+ rand = np.random.default_rng()
331
+
332
+ else:
333
+ # Conditions on random seed for reproducibility
334
+ # numpy allows integers as random seeds
335
+ if not isinstance(seed, int):
336
+ raise TypeError("Seed must be an integer number")
337
+
338
+ rand = np.random.default_rng(seed=seed)
339
+
340
+ # Create level vectors
341
+ for level_count in range(self.levels):
342
+ level = "level_{}".format(level_count)
343
+
344
+ if level_count == 0:
345
+ base = np.full(self.size, -1 if self.vtype == "bipolar" else 0)
346
+ to_one = rand.permutation(index_vector)[:change]
347
+
348
+ else:
349
+ to_one = rand.permutation(index_vector)[:next_level]
350
+
351
+ for index in to_one:
352
+ base[index] = base[index] * -1 if self.vtype == "bipolar" else base[index] + 1
353
+
354
+ vector = Vector(
355
+ name=level,
356
+ size=self.size,
357
+ vtype=self.vtype,
358
+ vector=copy.deepcopy(base)
359
+ )
360
+
361
+ self.space.insert(vector)
362
+
363
+ right_bound = self.min_value + level_count * gap
364
+
365
+ if level_count == 0:
366
+ left_bound = right_bound
367
+
368
+ else:
369
+ left_bound = self.min_value + (level_count - 1) * gap
370
+
371
+ self.level_list.append((left_bound, right_bound))
372
+
373
+ # Encode all data points
374
+ for point_position, point in enumerate(points):
375
+ point_vector = self._encode_point(point)
376
+
377
+ # Add the hyperdimensional representation of the data point to the space
378
+ point_vector.name = "point_{}".format(point_position)
379
+ self.space.insert(point_vector)
380
+
381
+ # Tag vector with its class label
382
+ self.space.add_tag(name=point_vector.name, tag=labels[point_position])
383
+
384
+ def _encode_point(self, point: List[float]) -> Vector:
385
+ """Encode a single data point. It must be used after `fit()`.
386
+
387
+ Parameters
388
+ ----------
389
+ point : list
390
+ A data point.
391
+
392
+ Returns
393
+ -------
394
+ Vector
395
+ The encoded data point.
396
+ """
397
+
398
+ sum_vector = None
399
+
400
+ for value_position, value in enumerate(point):
401
+ level_count = 0
402
+
403
+ if value == self.min_value:
404
+ level_count = 0
405
+
406
+ elif value == self.max_value:
407
+ level_count = self.levels - 1
408
+
409
+ else:
410
+ for level_position in range(len(self.level_list)):
411
+ left_bound, right_bound = self.level_list[level_position]
412
+
413
+ if left_bound <= value and right_bound > value:
414
+ level_count = level_position
415
+
416
+ break
417
+
418
+ level_vector = self.space.get(names=["level_{}".format(level_count)])[0]
419
+
420
+ roll_vector = permute(level_vector, rotate_by=value_position)
421
+
422
+ if sum_vector is None:
423
+ sum_vector = roll_vector
424
+
425
+ else:
426
+ sum_vector = bundle(sum_vector, roll_vector)
427
+
428
+ return sum_vector
429
+
430
+ def error_rate(
431
+ self,
432
+ training_vectors: List[Vector],
433
+ class_vectors: List[Vector],
434
+ distance_method: str="cosine"
435
+ ) -> Tuple[float, List[Vector], List[str]]:
436
+ """Compute the error rate.
437
+
438
+ Parameters
439
+ ----------
440
+ training_vectors : list
441
+ List with Vector objects used for training the classification model.
442
+ class_vectors : list
443
+ List with the Vector representation of classes.
444
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
445
+ Method used to compute the distance between vectors in the space.
446
+
447
+ Returns
448
+ -------
449
+ tuple
450
+ A tuple with the error rate, the list of wrongly predicted Vector objects, and the list
451
+ of wrong predictions with the same length of the list with wrongly predicted Vector objects.
452
+
453
+ Raises
454
+ ------
455
+ ValueError
456
+ - if the input `training_vectors` does not contain Vector objects;
457
+ - if the input `class_vectors` does not contain Vector objects.
458
+ Exception
459
+ - if no training vectors have been provided;
460
+ - if no class vectors have been provided.
461
+ """
462
+
463
+ if not training_vectors:
464
+ raise Exception("No training vectors have been provided")
465
+
466
+ if not class_vectors:
467
+ raise Exception("No class vectors have been provided")
468
+
469
+ wrongly_predicted_training_vectors = list()
470
+
471
+ wrong_predictions = list()
472
+
473
+ for class_vector in class_vectors:
474
+ if not isinstance(class_vector, Vector):
475
+ raise ValueError("The list of class vectors does not contain Vector objects")
476
+
477
+ for training_vector in training_vectors:
478
+ if not isinstance(training_vector, Vector):
479
+ raise ValueError("The list of training vectors does not contain Vector objects")
480
+
481
+ # Vectors contain only their class info in tags
482
+ true_class = list(training_vector.tags)[0]
483
+
484
+ if true_class != None:
485
+ closest_class = None
486
+ closest_dist = -np.inf
487
+
488
+ for class_vector in class_vectors:
489
+ # Compute the distance between the training points and the hyperdimensional representations of classes
490
+ with np.errstate(invalid="ignore", divide="ignore"):
491
+ distance = training_vector.dist(class_vector, method=distance_method)
492
+
493
+ if closest_class is None:
494
+ closest_class = list(class_vector.tags)[0]
495
+ closest_dist = distance
496
+
497
+ else:
498
+ if distance < closest_dist:
499
+ closest_class = list(class_vector.tags)[0]
500
+ closest_dist = distance
501
+
502
+ if closest_class != true_class:
503
+ wrongly_predicted_training_vectors.append(training_vector)
504
+
505
+ wrong_predictions.append(closest_class)
506
+
507
+ model_error_rate = len(wrongly_predicted_training_vectors) / len(training_vectors)
508
+
509
+ return model_error_rate, wrongly_predicted_training_vectors, wrong_predictions
510
+
511
+ def predict(
512
+ self,
513
+ test_indices: List[int],
514
+ distance_method: str="cosine",
515
+ retrain: int=0
516
+ ) -> Tuple[List[int], List[str], List[List[float]], int, float, List[Vector]]:
517
+ """Supervised Learning. Predict the class labels of the data points in the test set.
518
+
519
+ Parameters
520
+ ----------
521
+ test_indices : list
522
+ Indices of data points in the list of points used with fit() to be used for testing the classification model.
523
+ Note that all the other points will be used for training the model.
524
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
525
+ Method used to compute the distance between vectors in the space.
526
+ retrain : int, default 0
527
+ Maximum number of retraining iterations.
528
+
529
+ Returns
530
+ -------
531
+ tuple
532
+ A tuple with the input list `test_indices` in addition to a list with the predicted class labels with the
533
+ same size of `test_indices`, the distances between the test vectors and classes, the total number of
534
+ retraining iterations used to retrain the classification model, the model error rate, and the retrained
535
+ class vectors (i.e., the actual model).
536
+
537
+ Raises
538
+ ------
539
+ ValueError
540
+ If the number of retraining iterations is <0.
541
+ Exception
542
+ - if no test indices have been provided;
543
+ - if no class labels have been provided while fitting the model;
544
+ - if the number of test indices does not match the number of points retrieved from the space.
545
+
546
+ Notes
547
+ -----
548
+ The supervised classification model based on the hyperdimensional computing paradigm has been originally described in [1]_.
549
+
550
+ .. [1] Cumbo, Fabio, Eleonora Cappelli, and Emanuel Weitschek. "A brain-inspired hyperdimensional computing approach
551
+ for classifying massive dna methylation data of cancer." Algorithms 13.9 (2020): 233.
552
+ """
553
+
554
+ if not test_indices:
555
+ raise Exception("No test indices have been provided")
556
+
557
+ if retrain < 0:
558
+ raise ValueError("The number of retraining iterations must be >=0")
559
+
560
+ if len(self.classes) == 0:
561
+ raise Exception("No class labels found")
562
+
563
+ # List with test vectors
564
+ test_vectors = list()
565
+
566
+ # List with training vectors
567
+ training_vectors = list()
568
+
569
+ # Retrieve test and training vectors from the space
570
+ for vector_name in self.space.space:
571
+ if vector_name.startswith("point_"):
572
+ vector_id = int(vector_name.split("_")[-1])
573
+
574
+ vector = self.space.space[vector_name]
575
+
576
+ if vector_id in test_indices:
577
+ test_vectors.append(vector)
578
+
579
+ else:
580
+ training_vectors.append(vector)
581
+
582
+ if len(test_vectors) != len(test_indices):
583
+ raise Exception("Unable to retrieve all the test vectors from the space")
584
+
585
+ class_vectors = list()
586
+
587
+ for class_pos, class_label in enumerate(self.classes):
588
+ # Get training vectors for the current class label
589
+ class_points = [vector for vector in training_vectors if class_label in vector.tags]
590
+
591
+ # Build the vector representations of the current class
592
+ class_vector = None
593
+
594
+ for vector in class_points:
595
+ if class_vector is None:
596
+ class_vector = vector
597
+
598
+ else:
599
+ class_vector = bundle(class_vector, vector)
600
+
601
+ class_vector.name = "class_{}".format(class_pos)
602
+
603
+ class_vector.tags.add(class_label)
604
+
605
+ class_vectors.append(class_vector)
606
+
607
+ # Make a copy of the vector representation of classes for retraining the model
608
+ retraining_class_vectors = copy.deepcopy(class_vectors) if retrain > 0 else class_vectors
609
+
610
+ # Take track of the error rate in case of retraining the model
611
+ model_error_rate, wrongly_predicted_training_vectors, wrong_predictions = self.error_rate(
612
+ training_vectors,
613
+ retraining_class_vectors,
614
+ distance_method=distance_method
615
+ )
616
+
617
+ # Count retraining iterations
618
+ retraining_iterations = 0
619
+
620
+ if retrain > 0:
621
+ for _ in range(retrain):
622
+ retraining_class_vectors_iter = copy.deepcopy(retraining_class_vectors)
623
+
624
+ for vector_position, training_vector in enumerate(wrongly_predicted_training_vectors):
625
+ true_class = list(training_vector.tags)[0]
626
+
627
+ # Error mitigation
628
+ for class_vector in retraining_class_vectors_iter:
629
+ if true_class in class_vector.tags:
630
+ class_vector.vector = class_vector.vector + training_vector.vector
631
+
632
+ elif wrong_predictions[vector_position] in class_vector.tags:
633
+ class_vector.vector = class_vector.vector - training_vector.vector
634
+
635
+ retraining_error_rate, wrongly_predicted_training_vectors, wrong_predictions = self.error_rate(
636
+ training_vectors,
637
+ retraining_class_vectors_iter,
638
+ distance_method=distance_method
639
+ )
640
+
641
+ if model_error_rate < retraining_error_rate:
642
+ # Does not make sense to keep retraining if the error rate increases compared to the previous iteration
643
+ break
644
+
645
+ # Take track of the error rate
646
+ model_error_rate = retraining_error_rate
647
+
648
+ # Use the retrained class vectors
649
+ retraining_class_vectors = retraining_class_vectors_iter
650
+
651
+ # Also take track of the number of retraining iterations
652
+ retraining_iterations += 1
653
+
654
+ prediction = list()
655
+ distances = list()
656
+
657
+ for test_vector in sorted(test_vectors, key=lambda vector: test_indices.index(int(vector.name.split("_")[-1]))):
658
+ pred, dist = self._predict_vector(test_vector, retraining_class_vectors, distance_method=distance_method)
659
+
660
+ prediction.append(pred)
661
+ distances.append(dist)
662
+
663
+ return test_indices, prediction, distances, retraining_iterations, model_error_rate, retraining_class_vectors
664
+
665
+ def _predict_vector(
666
+ self,
667
+ vector: Vector,
668
+ training_class_vectors: List[Vector],
669
+ distance_method: str="cosine"
670
+ ) -> Tuple[str, List[float]]:
671
+ """Predict the class of an input vector.
672
+
673
+ Parameters
674
+ ----------
675
+ vector : Vector
676
+ The input vector for prediction.
677
+ training_class_vectors : list
678
+ List of eventually retrained class vectors representing the classification model.
679
+
680
+ Returns
681
+ -------
682
+ tuple
683
+ The closest class as the prediction and a list of vector to classes distances.
684
+ """
685
+
686
+ closest_class = None
687
+ closest_dist = -np.inf
688
+
689
+ distances = list()
690
+
691
+ for class_vector in training_class_vectors:
692
+ # Compute the distance between the input vector and the hyperdimensional representations of classes
693
+ with np.errstate(invalid="ignore", divide="ignore"):
694
+ distance = vector.dist(class_vector, method=distance_method)
695
+
696
+ distances.append(distance)
697
+
698
+ if closest_class is None:
699
+ closest_class = list(class_vector.tags)[0]
700
+ closest_dist = distance
701
+
702
+ else:
703
+ if distance < closest_dist:
704
+ closest_class = list(class_vector.tags)[0]
705
+ closest_dist = distance
706
+
707
+ return closest_class, distances
708
+
709
+ def cross_val_predict(
710
+ self,
711
+ points: List[List[float]],
712
+ labels: List[str],
713
+ cv: int=5,
714
+ distance_method: str="cosine",
715
+ retrain: int=0,
716
+ n_jobs: int=1
717
+ ) -> List[Tuple[List[int], List[str], int]]:
718
+ """Run `predict()` in cross validation.
719
+
720
+ Parameters
721
+ ----------
722
+ points : list
723
+ List of lists with numerical data (floats).
724
+ labels : list
725
+ List with class labels. It has the same size of `points`.
726
+ cv : int, default 5
727
+ Number of folds for cross-validating the model.
728
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
729
+ Method used to compute the distance between vectors in space.
730
+ retrain : int, default 0
731
+ Number of retraining iterations.
732
+ n_jobs : int, default 1,
733
+ Number of jobs for processing folds in parallel.
734
+
735
+ Returns
736
+ -------
737
+ list
738
+ A list with the results of `predict()` for each fold.
739
+
740
+ Raises
741
+ ------
742
+ Exception
743
+ - if the number of data points does not match with the number of class labels;
744
+ - if there is only one class label.
745
+ ValueError
746
+ - if the number of folds is a number < 2;
747
+ - if the number of folds exceeds the number of data points;
748
+ - if the number of retraining iterations is <0.
749
+ """
750
+
751
+ if len(points) != len(labels):
752
+ raise Exception("The number of data points does not match with the number of class labels")
753
+
754
+ if len(set(labels)) < 2:
755
+ raise Exception("The number of unique class labels must be > 1")
756
+
757
+ if cv < 2:
758
+ raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
759
+
760
+ if cv > len(points):
761
+ raise ValueError("The number of folds cannot exceed the number of data points")
762
+
763
+ if retrain < 0:
764
+ raise ValueError("The number of retraining iterations must be >=0")
765
+
766
+ # Use all the available resources if n_job < 1
767
+ n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
768
+
769
+ kf = StratifiedKFold(n_splits=cv, shuffle=True, random_state=0)
770
+
771
+ # Collect results from every self.predict call
772
+ predictions = list()
773
+
774
+ if n_jobs == 1:
775
+ for _, test_indices in kf.split(points, labels):
776
+ test_indices = test_indices.tolist()
777
+
778
+ _, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors = self.predict(
779
+ test_indices,
780
+ distance_method=distance_method,
781
+ retrain=retrain
782
+ )
783
+
784
+ predictions.append((test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors))
785
+
786
+ else:
787
+ predict_partial = partial(
788
+ self.predict,
789
+ distance_method=distance_method,
790
+ retrain=retrain
791
+ )
792
+
793
+ # Run prediction on folds in parallel
794
+ with mp.Pool(processes=n_jobs) as pool:
795
+ jobs = [
796
+ pool.apply_async(
797
+ predict_partial,
798
+ args=(test_indices.tolist(),)
799
+ )
800
+ for _, test_indices in kf.split(points, labels)
801
+ ]
802
+
803
+ # Get results from jobs
804
+ for job in jobs:
805
+ test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors = job.get()
806
+
807
+ predictions.append((test_indices, test_predictions, test_distances, retraining_iterations, model_error_rate, training_class_vectors))
808
+
809
+ return predictions
810
+
811
+ def auto_tune(
812
+ self,
813
+ points: List[List[float]],
814
+ labels: List[str],
815
+ size_range: range,
816
+ levels_range: range,
817
+ cv: int=5,
818
+ distance_method: str="cosine",
819
+ retrain: int=0,
820
+ n_jobs: int=1,
821
+ metric: str="accuracy"
822
+ ) -> Tuple[int, int, float]:
823
+ """Automated hyperparameters tuning. Perform a Parameter Sweep Analysis (PSA) on space dimensionality and number of levels.
824
+
825
+ Parameters
826
+ ----------
827
+ points : list
828
+ List of lists with numerical data (floats).
829
+ labels : list
830
+ List with class labels. It has the same size of `points`.
831
+ size_range : range
832
+ Range of dimensionalities for performing PSA.
833
+ levels_range : range
834
+ Range of number of levels for performing PSA.
835
+ cv : int, default 5
836
+ Number of folds for cross-validating the model.
837
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
838
+ Method used to compute the distance between vectors in space.
839
+ retrain : int, default 0
840
+ Number of retraining iterations.
841
+ n_jobs : int, default 1,
842
+ Number of jobs for processing folds in parallel.
843
+ metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
844
+ Metric used to evaluate the model.
845
+
846
+ Returns
847
+ -------
848
+ tuple
849
+ A tuple with the best size and levels according to the accuracies of the cross-validated models.
850
+
851
+ Raises
852
+ ------
853
+ ValueError
854
+ - if the number of class labels does not match with the number of data points;
855
+ - if the number of specified folds for cross-validating the model is lower than 2;
856
+ - if the number of folds exceeds the number of data points;
857
+ - if the number of retraining iterations is <0.
858
+ Exception
859
+ - if no data points have been provided in input;
860
+ - if no class labels have been provided in input.
861
+ """
862
+
863
+ if not points:
864
+ raise Exception("No data points have been provided")
865
+
866
+ if not labels:
867
+ raise Exception("No class labels have been provided")
868
+
869
+ if len(points) != len(labels):
870
+ raise ValueError("The number of class labels must match with the number of data points")
871
+
872
+ if cv < 2:
873
+ raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
874
+
875
+ if cv > len(points):
876
+ raise ValueError("The number of folds cannot exceed the number of data points")
877
+
878
+ if retrain < 0:
879
+ raise ValueError("The number of retraining iterations must be >=0")
880
+
881
+ # Use all the available resources if n_job < 1
882
+ n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
883
+
884
+ partial_init_fit_predict = partial(
885
+ self._init_fit_predict,
886
+ vtype=self.vtype,
887
+ points=points,
888
+ labels=labels,
889
+ cv=cv,
890
+ distance_method=distance_method,
891
+ retrain=retrain,
892
+ n_jobs=1,
893
+ metric=metric
894
+ )
895
+
896
+ best_metric = None
897
+ best_size = None
898
+ best_levels = None
899
+
900
+ with mp.Pool(processes=n_jobs) as pool:
901
+ jobs = [
902
+ pool.apply_async(
903
+ partial_init_fit_predict,
904
+ args=(size, levels,)
905
+ )
906
+ for size, levels in list(itertools.product(size_range, levels_range)) \
907
+ if size > len(points) and levels > 1
908
+ ]
909
+
910
+ # Get results from jobs
911
+ for job in jobs:
912
+ job_size, job_levels, job_metric = job.get()
913
+
914
+ if best_metric is None:
915
+ best_metric = job_metric
916
+ best_size = job_size
917
+ best_levels = job_levels
918
+
919
+ else:
920
+ if job_metric > best_metric:
921
+ # Get the size and levels of the classification model with the best score metric
922
+ best_metric = job_metric
923
+ best_size = job_size
924
+ best_levels = job_levels
925
+
926
+ elif job_metric == best_metric:
927
+ # Minimize the number of levels in this case
928
+ if job_levels < best_levels:
929
+ best_size = job_size
930
+ best_levels = job_levels
931
+
932
+ elif job_levels == best_levels:
933
+ # Minimize the size in this case
934
+ if job_size < best_size:
935
+ best_size = job_size
936
+
937
+ return best_size, best_levels, best_metric
938
+
939
+ def _stepwise_regression_iter(
940
+ self,
941
+ features_indices: Set[int],
942
+ points: List[List[float]],
943
+ labels: List[str],
944
+ cv: int=5,
945
+ distance_method: str="cosine",
946
+ retrain: int=0,
947
+ metric: str="accuracy"
948
+ ) -> Tuple[Set[float], float]:
949
+ """Just a single iteration of the feature selection method.
950
+
951
+ Parameters
952
+ ----------
953
+ features_indices : set
954
+ Indices of features for shaping points.
955
+ points : list
956
+ List of lists with numerical data (floats).
957
+ labels : list
958
+ List with class labels. It has the same size of `points`.
959
+ cv : int, default 5
960
+ Number of folds for cross-validating the model.
961
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
962
+ Method used to compute the distance between vectors in space.
963
+ retrain : int, default 0
964
+ Number of retraining iterations.
965
+ metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
966
+ Metric used to evaluate the model.
967
+
968
+ Returns
969
+ -------
970
+ tuple
971
+ A tuple with the considered features and the score of the classification model based on the provided metric.
972
+ """
973
+
974
+ data_points = [[point[i] for i in range(len(point)) if i in features_indices] for point in points]
975
+
976
+ _, _, score = self._init_fit_predict(
977
+ size=self.size,
978
+ levels=self.levels,
979
+ vtype=self.vtype,
980
+ points=data_points,
981
+ labels=labels,
982
+ cv=cv,
983
+ distance_method=distance_method,
984
+ retrain=retrain,
985
+ n_jobs=1,
986
+ metric=metric
987
+ )
988
+
989
+ return features_indices, score
990
+
991
+ def stepwise_regression(
992
+ self,
993
+ points: List[List[float]],
994
+ features: List[str],
995
+ labels: List[str],
996
+ method: str="backward",
997
+ cv: int=5,
998
+ distance_method: str="cosine",
999
+ retrain: int=0,
1000
+ n_jobs: int=1,
1001
+ metric: str="accuracy",
1002
+ threshold: float=0.6,
1003
+ uncertainty: float=5.0,
1004
+ stop_if_worse: bool=False
1005
+ ) -> Tuple[Dict[str, int], Dict[int, float], int, int]:
1006
+ """Stepwise regression as backward variable elimination or forward variable selection.
1007
+
1008
+ Parameters
1009
+ ----------
1010
+ points : list
1011
+ List of lists with numerical data (floats).
1012
+ features : list
1013
+ List of features.
1014
+ labels : list
1015
+ List with class labels. It has the same size of `points`.
1016
+ method : {'backward', 'forward'}, default 'backward'
1017
+ Feature selection method.
1018
+ cv : int, default 5
1019
+ Number of folds for cross-validating the model.
1020
+ distance_method : {'cosine', 'euclidean', 'hamming'}, default 'cosine'
1021
+ Method used to compute the distance between vectors in space.
1022
+ retrain : int, default 0
1023
+ Number of retraining iterations.
1024
+ n_jobs : int, default 1,
1025
+ Number of jobs for processing models in parallel.
1026
+ metric: {'accuracy', 'f1', 'precision', 'recall'}, default 'accuracy'
1027
+ Metric used to evaluate the model.
1028
+ threshold : float, default 0.6
1029
+ Threshold on the model score metric. Stop running the feature selection if the best
1030
+ reached score is lower than this threshold.
1031
+ uncertainty : float, default 5.0
1032
+ Uncertainty percentage threshold for comparing models metrics.
1033
+ stop_if_worse : bool, default False
1034
+ Stop running the feature selection if the accuracy reached at the iteration i is lower than the accuracy reached at i-1.
1035
+
1036
+ Returns
1037
+ -------
1038
+ tuple
1039
+ A tuple with a dictionary with features and their importance in addition to the best score for each importance rank,
1040
+ the best importance, and the total mount of ML models built and evaluated. For what concerns the importance, in case of
1041
+ `method='backward'`, the lower the better. In case of `method='forward'`, the higher the better.
1042
+
1043
+ Raises
1044
+ ------
1045
+ ValueError
1046
+ - if there are not enough features for running the feature selection;
1047
+ - if the number of class labels does not match with the number of data points;
1048
+ - if the specified feature selection method is not supported;
1049
+ - if the number of specified folds for cross-validating the model is lower than 2;
1050
+ - if the number of folds exceeds the number of data points;
1051
+ - if the number of retraining iterations is <0;
1052
+ - if the threshold is negative or greater than 1.0;
1053
+ - if the uncertainty percentage is negative or greater than 100.0.
1054
+ Exception
1055
+ - if no data points have been provided in input;
1056
+ - if no class labels have been provided in input.
1057
+ """
1058
+
1059
+ if not points:
1060
+ raise Exception("No data points have been provided")
1061
+
1062
+ if len(features) < 2:
1063
+ raise ValueError("Not enough features for running a feature selection")
1064
+
1065
+ if not labels:
1066
+ raise Exception("No class labels have been provided")
1067
+
1068
+ if len(points) != len(labels):
1069
+ raise ValueError("The number of class labels must match with the number of data points")
1070
+
1071
+ method = method.lower()
1072
+
1073
+ if method not in ("backward", "forward"):
1074
+ raise ValueError("Stepwise method {} is not supported".format(method))
1075
+
1076
+ if cv < 2:
1077
+ raise ValueError("Not enough folds for cross-validating the model. Please use a minimum of 2 folds")
1078
+
1079
+ if cv > len(points):
1080
+ raise ValueError("The number of folds cannot exceed the number of data points")
1081
+
1082
+ if retrain < 0:
1083
+ raise ValueError("The number of retraining iterations must be >=0")
1084
+
1085
+ if threshold < 0.0 or threshold > 1.0:
1086
+ raise ValueError("Invalid threshold! It must be >= 0.0 and <= 1.0")
1087
+
1088
+ if uncertainty < 0.0 or uncertainty > 100.0:
1089
+ raise ValueError("Invalid uncertainty percentage! It must be >= 0.0 and <= 100.0")
1090
+
1091
+ # Use all the available resources if n_job < 1
1092
+ n_jobs = os.cpu_count() if n_jobs < 1 else n_jobs
1093
+
1094
+ # Initialize the importance of features to 0
1095
+ features_importance = {feature: {"importance": 0, "score": 0.0} for feature in features}
1096
+
1097
+ features_indices = set(range(len(features)))
1098
+
1099
+ # Take track of the last feature selection
1100
+ # Only in case of forward variable selection
1101
+ last_selection = set()
1102
+
1103
+ prev_score = 0.0
1104
+
1105
+ count_iter = 1
1106
+
1107
+ # Count the total amount of ML models built and evaluated
1108
+ count_models = 0
1109
+
1110
+ while features_indices:
1111
+ if method == "backward":
1112
+ features_set_size = len(features_indices) - 1
1113
+
1114
+ if len(features_indices) == 1:
1115
+ break
1116
+
1117
+ elif method == "forward":
1118
+ features_set_size = count_iter
1119
+
1120
+ if features_set_size >= len(features_indices):
1121
+ break
1122
+
1123
+ if features_set_size > 0:
1124
+ best_score = 0.0
1125
+ classification_results = list()
1126
+
1127
+ partial_stepwise_regression_iter = partial(
1128
+ self._stepwise_regression_iter,
1129
+ points=points,
1130
+ labels=labels,
1131
+ cv=cv,
1132
+ distance_method=distance_method,
1133
+ retrain=retrain,
1134
+ metric=metric
1135
+ )
1136
+
1137
+ with mp.Pool(processes=n_jobs) as pool:
1138
+ # Get all combinations of features of a given size
1139
+ jobs = [
1140
+ pool.apply_async(
1141
+ partial_stepwise_regression_iter,
1142
+ args=(features_set,)
1143
+ )
1144
+ for features_set in itertools.combinations(features_indices, features_set_size)
1145
+ ]
1146
+
1147
+ # Get results from jobs
1148
+ for job in jobs:
1149
+ job_features_set, job_score = job.get()
1150
+
1151
+ if job_score >= best_score:
1152
+ # Keep track of the best score
1153
+ best_score = job_score
1154
+
1155
+ classification_results.append((job_features_set, job_score))
1156
+
1157
+ count_models += 1
1158
+
1159
+ selection = set()
1160
+
1161
+ best_scores = list()
1162
+
1163
+ for features_set, score in classification_results:
1164
+ if score >= best_score - (best_score * uncertainty / 100.0):
1165
+ # Keep track of the missing features in models that reached the best score
1166
+ if method == "backward":
1167
+ selection.update(features_indices.difference(features_set))
1168
+
1169
+ elif method == "forward":
1170
+ selection.update(features_set)
1171
+
1172
+ best_scores.append(score)
1173
+
1174
+ if method == "forward" and last_selection:
1175
+ if len(last_selection) == len(selection) and len(last_selection.difference(selection)) == 0:
1176
+ break
1177
+
1178
+ last_selection = selection
1179
+
1180
+ avg_score = statistics.mean(best_scores)
1181
+
1182
+ if method == "backward":
1183
+ # Keep decreasing the importance of worst features detected in previous iterations
1184
+ for feature in features_importance:
1185
+ if features_importance[feature]["importance"] >= 1:
1186
+ features_importance[feature]["importance"] += 1
1187
+
1188
+ # Set the importance of the selected features
1189
+ for feature_index in selection:
1190
+ features_importance[features[feature_index]]["importance"] += 1
1191
+ features_importance[features[feature_index]]["score"] = avg_score
1192
+
1193
+ if method == "backward":
1194
+ features_indices = features_indices.difference(selection)
1195
+
1196
+ elif method == "forward":
1197
+ features_indices = selection
1198
+
1199
+ if stop_if_worse:
1200
+ if best_score < prev_score - (prev_score * uncertainty / 100.0):
1201
+ prev_score = avg_score
1202
+
1203
+ break
1204
+
1205
+ prev_score = avg_score
1206
+
1207
+ count_iter += 1
1208
+
1209
+ if best_score < threshold:
1210
+ break
1211
+
1212
+ importances = dict()
1213
+
1214
+ scores = dict()
1215
+
1216
+ for feature in features_importance:
1217
+ importances[feature] = features_importance[feature]["importance"]
1218
+
1219
+ scores[features_importance[feature]["importance"]] = features_importance[feature]["score"]
1220
+
1221
+ best_importance = sorted(scores.keys(), key=lambda imp: scores[imp])[-1]
1222
+
1223
+ return importances, scores, best_importance, count_models
1224
+
1225
+
1226
+ class QuantumClassificationModel(object):
1227
+ """Supervised Quantum Classification Model."""
1228
+
1229
+ def __init__(
1230
+ self,
1231
+ size: int=64,
1232
+ levels: int=2,
1233
+ seed: int=42,
1234
+ shots: int=1024,
1235
+ channel: Optional[str]=None,
1236
+ instance: Optional[str]=None,
1237
+ backend: Optional[str]=None,
1238
+ api_key: Optional[str]=None,
1239
+ noise_model_from: Optional[str]=None
1240
+ ) -> "QuantumClassificationModel":
1241
+ """Initialize a QuantumClassificationModel object.
1242
+ Run the classification model on a simulator with Qiskit by default.
1243
+ It can interact with specific IBM channels, instances, and backends if specified (it requires an IBM account).
1244
+
1245
+ Parameters
1246
+ ----------
1247
+ size : int, default 64
1248
+ The vectors dimensionality as power of 2.
1249
+ levels : int, default 2
1250
+ Number of level vectors.
1251
+ seed : int, default 42
1252
+ Seed for reproducibility.
1253
+ shots : int, default 1024
1254
+ The number of times to run the quantum circuit for the Hadamard test.
1255
+ channel : str, default None, optional
1256
+ IBM channel.
1257
+ instance : str, default None, optional
1258
+ IBM instance. Required in case of specific channel only.
1259
+ backend : str, default None, optional
1260
+ IBM backend (e.g., "ibm_cleveland"). Required in case of specific instance only.
1261
+ If `instance` is not None, this is "least_busy" by default.
1262
+ api_key : str, default None, optional
1263
+ IBM API key. Required in case of specific backend only.
1264
+ noise_model_from : str, default None, optional
1265
+ The name of a real IBM backend (e.g., "ibm_cleveland") to build a noise model from the simulation.
1266
+ If provided, `api_key` is required. This parameter is ignored if `channel`, `instance`, and `backend` are provided for hardware execution.
1267
+ Noise models are retrieved from the "ibm_quantum_platform" channel.
1268
+
1269
+ Raises
1270
+ ------
1271
+ ValueError
1272
+ If the vector dimensionality `size` is not a power of 2.
1273
+ TypeError
1274
+ If seed is not an integer.
1275
+
1276
+ Examples
1277
+ --------
1278
+ >>> from hdlib.model import QuantumClassificationModel
1279
+ >>> model = QuantumClassificationModel(size=32, levels=2)
1280
+ >>> type(model)
1281
+ <class 'hdlib.model.QuantumClassificationModel'>
1282
+
1283
+ This creates a new QuantumClassificationModel object with random bipolar vectors with size 32 and 2 level vectors.
1284
+ """
1285
+
1286
+ if not ((size > 0) and ((size & (size - 1)) == 0)):
1287
+ # Check if a the vector dimensionality is a power of 2.
1288
+ raise ValueError("The vector dimensionality must be a power of 2.")
1289
+
1290
+ self.size = size
1291
+ self.levels = levels
1292
+ self.shots = shots
1293
+
1294
+ # Vectors must be bipolar here
1295
+ self.vtype = "bipolar"
1296
+
1297
+ # Keep track of the level vectors
1298
+ self.level_hvs = list()
1299
+
1300
+ # Keep track of class prototype vectors
1301
+ # This is filled up during `fit`
1302
+ self.prototypes = list()
1303
+
1304
+ if channel is None:
1305
+ # Use a simulator if no channel is specified
1306
+ noise_model = None
1307
+
1308
+ if noise_model_from:
1309
+ if not api_key:
1310
+ raise ValueError("`api_key` must be provided to fetch backend properties for a noise model.")
1311
+
1312
+ # Initialize a temporary service connection
1313
+ # Always use "ibm_quantum_platform" to fetch the backend properties
1314
+ noise_service = QiskitRuntimeService(channel="ibm_quantum_platform", token=api_key)
1315
+
1316
+ # Retrieve a backend
1317
+ # We only need its noise model
1318
+ backend_for_noise = noise_service.backend(noise_model_from)
1319
+
1320
+ # Finally, define the noise model
1321
+ noise_model = NoiseModel.from_backend(backend_for_noise)
1322
+
1323
+ # Use a simulator if no channel is specified
1324
+ # This can be noise-free or use a specific noise model
1325
+ try:
1326
+ # Attempt to initialize with GPU acceleration
1327
+ self.backend = AerSimulator(device="GPU", noise_model=noise_model)
1328
+
1329
+ except Exception as e:
1330
+ # Fallback to the default CPU device
1331
+ print(f"GPU not available, falling back to CPU.")
1332
+
1333
+ self.backend = AerSimulator(noise_model=noise_model)
1334
+
1335
+ else:
1336
+ # Initialize a quantum runtime service for a specific IBM QC channel, instance, and backend
1337
+ service = QiskitRuntimeService(channel=channel, token=api_key, instance=instance)
1338
+
1339
+ # The backend is the "least_busy" by default
1340
+ if backend is None:
1341
+ self.backend = service.least_busy(operational=True, simulator=False)
1342
+
1343
+ else:
1344
+ self.backend = service.backend(backend)
1345
+
1346
+ # Conditions on random seed for reproducibility
1347
+ # numpy allows integers as random seeds
1348
+ if not isinstance(seed, int):
1349
+ raise TypeError("Seed must be an integer number")
1350
+
1351
+ self.seed = seed
1352
+
1353
+ # Keep track of hdlib version
1354
+ self.version = __version__
1355
+
1356
+ def __str__(self) -> str:
1357
+ """Print the QuantumClassificationModel object properties.
1358
+
1359
+ Returns
1360
+ -------
1361
+ str
1362
+ A description of the QuantumClassificationModel object. It reports the vectors size, the vector type,
1363
+ the number of level vectors, and the number of shots.
1364
+
1365
+ Examples
1366
+ --------
1367
+ >>> from hdlib.model import QuantumClassificationModel
1368
+ >>> model = QuantumClassificationModel()
1369
+ >>> print(model)
1370
+
1371
+ Class: hdlib.model.classification.QuantumClassificationModel
1372
+ Version: 2.0.0
1373
+ Size: 64
1374
+ Type: bipolar
1375
+ Levels: 2
1376
+ Shots: 1024
1377
+
1378
+ Print the QuantumClassificationModel object properties.
1379
+ """
1380
+
1381
+ return f"""
1382
+ Class: hdlib.model.classification.QuantumClassificationModel
1383
+ Version: {self.version}
1384
+ Size: {self.size}
1385
+ Type: {self.vtype}
1386
+ Levels: {self.levels}
1387
+ Shots: {self.shots}
1388
+ """
1389
+
1390
+ def _build_quantum_sample_encoder(self, sample_row: List[float], level_vectors: List[np.ndarray], D: int) -> List[QuantumCircuit]:
1391
+ """Creates a single quantum circuit that encodes one real-valued sample by quantumly permuting its feature vectors.
1392
+
1393
+ Parameters
1394
+ ----------
1395
+ sample_row : list
1396
+ Single sample as list of numerical values (float).
1397
+ level_vectors : list
1398
+ List of level vector.
1399
+ D : int
1400
+ Vector dimensionality.
1401
+
1402
+ Returns
1403
+ -------
1404
+ List[QuantumCircuit]
1405
+ The list of quantum feature circuits for encoding the input sample.
1406
+ """
1407
+
1408
+ num_qubits = int(log2(D))
1409
+ feature_circuits = list()
1410
+
1411
+ for i, value in enumerate(sample_row):
1412
+ level_index = max(0, min(len(level_vectors) - 1, int(value * (len(level_vectors) - 1))))
1413
+ level_vec = level_vectors[level_index]
1414
+
1415
+ # Create a quantum circuit for this single feature
1416
+ feature_qc = quantum_encode(level_vec, label=f"Feature_{i}")
1417
+
1418
+ # Apply the permutation based in the feature index
1419
+ feature_qc = quantum_permute(feature_qc, num_qubits, shift=i)
1420
+
1421
+ # Report circuit metrics
1422
+ #print(f"Positional permutation metrics: {get_circuit_metrics(feature_qc, num_qubits, self.backend, optimization_level=3)}")
1423
+
1424
+ feature_circuits.append(feature_qc)
1425
+
1426
+ # The feature circuits are bundled all together with the feature circuits of the other samples
1427
+ # belonging to the same class to build the circuit representation of that class.
1428
+ return feature_circuits
1429
+
1430
+ def _scale_circuit_phases(self, circuit: Any, factor: float) -> Any:
1431
+ """Helper to safely scale the precise phase angles of a compressed circuit.
1432
+ """
1433
+
1434
+ new_circ = QuantumCircuit(circuit.num_qubits, name=circuit.name)
1435
+
1436
+ for instr in circuit.data:
1437
+ if instr.operation.name == "diagonal":
1438
+ # Extract the exact angles and multiply by the factor (negates if factor is negative)
1439
+ phases = np.angle(np.array(instr.operation.params, dtype=complex))
1440
+ new_phases = np.exp(1j * phases * factor)
1441
+
1442
+ GateClass = instr.operation.__class__
1443
+ new_circ.append(GateClass(new_phases.tolist()), instr.qubits)
1444
+
1445
+ else:
1446
+ new_circ.append(instr)
1447
+
1448
+ return new_circ
1449
+
1450
+ def fit(self, train_points: List[List[float]], train_labels: List[str]) -> None:
1451
+ """Build a vector-symbolic architecture. Define level vectors, encode samples, and build prototypes.
1452
+
1453
+ Parameters
1454
+ ----------
1455
+ train_points : list
1456
+ List of lists with numerical data points (floats).
1457
+ train_labels : list
1458
+ List with class labels. It has the same size of `points`.
1459
+ """
1460
+
1461
+ def _generate_level_vectors(D, num_levels, rng):
1462
+ """Generates a set of bipolar vectors for discretizing real numbers.
1463
+ """
1464
+
1465
+ level_vectors = [rng.choice([-1, 1], size=D)]
1466
+
1467
+ change = int(D / 2)
1468
+ next_level = int((D / 2 / num_levels))
1469
+
1470
+ for i in range(1, num_levels):
1471
+ prev_vec = level_vectors[i-1].copy()
1472
+
1473
+ if i-1 == 0:
1474
+ flip_indices = rng.choice(D, size=change, replace=False)
1475
+
1476
+ else:
1477
+ flip_indices = rng.choice(D, size=next_level, replace=False)
1478
+
1479
+ prev_vec[flip_indices] *= -1
1480
+ level_vectors.append(prev_vec)
1481
+
1482
+ return level_vectors
1483
+
1484
+ rand = np.random.default_rng(seed=self.seed)
1485
+
1486
+ # Create level vectors
1487
+ self.level_hvs = _generate_level_vectors(self.size, self.levels, rand)
1488
+
1489
+ self.classes_ = sorted(list(set(train_labels)))
1490
+
1491
+ # Hold class prototypes
1492
+ self.prototypes = dict()
1493
+
1494
+ # Track the mass of each class so we can normalize updates proportionally later
1495
+ self.class_counts_ = dict()
1496
+
1497
+ for c in self.classes_:
1498
+ # Building quantum prototype for Class `c`
1499
+ class_samples = [sample for pos, sample in enumerate(train_points) if train_labels[pos] == c]
1500
+
1501
+ # Building quantum samples' feature encoder circuits
1502
+ sample_encoders = [encoder for sample in class_samples for encoder in self._build_quantum_sample_encoder(sample, self.level_hvs, self.size)]
1503
+
1504
+ # Bundling
1505
+ prototype_circuit = compress_circuit(quantum_bundle(sample_encoders))
1506
+ prototype_circuit.name = f"Prototype_{c}"
1507
+
1508
+ # Report circuit metrics
1509
+ #print(f"Class prototype bundling metrics: {get_circuit_metrics(prototype_circuit, int(log2(self.size)), self.backend, optimization_level=3)}")
1510
+
1511
+ # The prototype is the circuit that prepares the state
1512
+ self.prototypes[c] = prototype_circuit
1513
+ self.class_counts_[c] = float(len(class_samples))
1514
+
1515
+ def predict(self, test_points: List[List[float]]) -> Tuple[List[str], List[List[float]]]:
1516
+ """Predict the class labels of the data points in the test set.
1517
+
1518
+ Parameters
1519
+ ----------
1520
+ test_points : list
1521
+ Test data points.
1522
+
1523
+ Returns
1524
+ -------
1525
+ Tuple
1526
+ A list with the predicted class labels in the same order of data points in `test_points`,
1527
+ and a list of similarities between the test samples and the class prototypes.
1528
+
1529
+ Raises
1530
+ ------
1531
+ RuntimeError
1532
+ If `predict` is called before `fit`.
1533
+ """
1534
+
1535
+ if not hasattr(self, "classes_"):
1536
+ raise RuntimeError("You must call fit before calling predict.")
1537
+
1538
+ # Check whether the test should be performed on a simulator or on the quantum hardware
1539
+ is_simulated = isinstance(self.backend, AerSimulator)
1540
+
1541
+ predictions = list()
1542
+ similarities = list()
1543
+
1544
+ # For hardware, manage a single session for all prediction tasks
1545
+ with (Session(backend=self.backend) if not is_simulated else nullcontext()) as session:
1546
+ sampler = None
1547
+
1548
+ # Use the session-based Sampler if on hardware
1549
+ if session:
1550
+ options = SamplerOptions()
1551
+
1552
+ # Set the default number of shots
1553
+ options.default_shots = self.shots
1554
+
1555
+ # Enable dynamical decoupling
1556
+ options.dynamical_decoupling.enable = True
1557
+ options.dynamical_decoupling.sequence_type = "XpXm"
1558
+
1559
+ # Enable gate twirling
1560
+ options.twirling.enable_gates = True
1561
+
1562
+ sampler = Sampler(mode=session, options=options)
1563
+
1564
+ # Build all query circuits
1565
+ queries = list()
1566
+
1567
+ for sample in test_points:
1568
+ # Get the list of feature circuits for the test sample
1569
+ sample_features_circuits = self._build_quantum_sample_encoder(sample, self.level_hvs, self.size)
1570
+
1571
+ # Bundle them into a single query circuit
1572
+ query_circuit = compress_circuit(quantum_bundle(sample_features_circuits))
1573
+
1574
+ # Report circuit metrics
1575
+ #print(f"Query bundling metrics: {get_circuit_metrics(query_circuit, int(log2(self.size)), self.backend, optimization_level=3)}")
1576
+
1577
+ queries.append(query_circuit)
1578
+
1579
+ # Evaluate all queries against class prototypes in safe hardware batch mode to prevent IBM memory overflow
1580
+ # https://ibm.biz/error_codes#6073
1581
+ max_queries_per_job = 5
1582
+
1583
+ # Sort prototype circuits based on self.classes_
1584
+ prototype_circuits = [self.prototypes[c] for c in self.classes_]
1585
+
1586
+ for i in range(0, len(queries), max_queries_per_job):
1587
+ query_batch = queries[i : i + max_queries_per_job]
1588
+
1589
+ sims, _ = run_compute_uncompute_test(query_batch, prototype_circuits, self.backend, shots=self.shots, seed=self.seed, sampler=sampler)
1590
+
1591
+ for sim in sims:
1592
+ predictions.append(self.classes_[int(np.argmax(sim))])
1593
+ similarities.append(sim)
1594
+
1595
+ return predictions, similarities
1596
+
1597
+ def retrain(self, train_points: List[List[float]], train_labels: List[str], epochs: int=10, lr: float=1.0, verbose: bool=True, test_points: Optional[List[List[float]]]=None, test_labels: Optional[List[str]]=None, early_stop: bool=True) -> Tuple[float, int]:
1598
+ """Retrain the model by adjusting class prototypes based on misclassified samples.
1599
+
1600
+ Per-epoch error rates are recorded on ``self.retrain_history_`` as a list
1601
+ of ``{"epoch": int, "error": float}`` entries (starting at epoch 0), so
1602
+ callers can access the training curve without parsing stdout. Setting
1603
+ ``verbose=False`` suppresses the per-epoch prints.
1604
+
1605
+ If ``test_points`` (and ``test_labels``) are provided, each history entry
1606
+ also carries a ``"test_error"`` field measured on that held-out set. This
1607
+ lets callers plot a genuine test-error-vs-epoch curve, but note it costs
1608
+ one extra ``predict`` over the test set per epoch (the expensive step for
1609
+ the quantum model), so it is opt-in.
1610
+
1611
+ ``early_stop`` (default ``True``) preserves the original behaviour: as soon
1612
+ as an epoch fails to reduce the training error the best-known prototypes are
1613
+ restored and the loop exits. Set ``early_stop=False`` to run the full
1614
+ ``epochs`` budget and record the complete curve; the best-known prototypes
1615
+ are still restored before returning, so the returned model is unchanged by
1616
+ this flag.
1617
+ """
1618
+
1619
+ if not hasattr(self, "classes_"):
1620
+ raise RuntimeError("You must call fit before calling retrain.")
1621
+
1622
+ if test_points is not None and test_labels is None:
1623
+ raise ValueError("test_labels must be provided when test_points is given.")
1624
+
1625
+ def _error(points, labels):
1626
+ preds, _ = self.predict(points)
1627
+ return sum(1 for p, t in zip(preds, labels) if p != t) / len(labels)
1628
+
1629
+ # 1. Evaluate the base model before any retraining to establish a baseline
1630
+ predictions, _ = self.predict(train_points)
1631
+
1632
+ best_error = sum(1 for p, t in zip(predictions, train_labels) if p != t) / len(train_labels)
1633
+ epoch_0 = {"epoch": 0, "error": float(best_error)}
1634
+ if test_points is not None:
1635
+ epoch_0["test_error"] = float(_error(test_points, test_labels))
1636
+ self.retrain_history_ = [epoch_0]
1637
+ if verbose:
1638
+ print(f"\tepoch 0: {best_error}")
1639
+
1640
+ if best_error == 0.0:
1641
+ return best_error, 0
1642
+
1643
+ # Save a backup of the best circuits
1644
+ best_prototypes = {c: qc.copy() for c, qc in self.prototypes.items()}
1645
+
1646
+ final_epoch = 0
1647
+
1648
+ for epoch in range(1, epochs + 1):
1649
+ final_epoch = epoch
1650
+
1651
+ # Queue to store correctly phase-scaled updates
1652
+ epoch_updates = {c: [] for c in self.classes_}
1653
+ updates_made = False
1654
+
1655
+ # 2. Adjust circuits based on misclassifications
1656
+ for i, (pred, true_label) in enumerate(zip(predictions, train_labels)):
1657
+ if pred != true_label:
1658
+ # Encode the misclassified sample
1659
+ raw_features = self._build_quantum_sample_encoder(train_points[i], self.level_hvs, self.size)
1660
+ sample_update = compress_circuit(quantum_bundle(raw_features))
1661
+
1662
+ # Calculate exact scale factor to match the prototype's mass denominator.
1663
+ # lr defaults to 1.0 (pure Perceptron rule), but is adjusted based on class population
1664
+ factor_true = lr / self.class_counts_[true_label]
1665
+ factor_pred = lr / self.class_counts_[pred]
1666
+
1667
+ # Add to the true class
1668
+ epoch_updates[true_label].append(self._scale_circuit_phases(sample_update, factor=factor_true))
1669
+
1670
+ # Remove from the incorrectly predicted class (negative factor)
1671
+ epoch_updates[pred].append(self._scale_circuit_phases(sample_update, factor=-factor_pred))
1672
+
1673
+ updates_made = True
1674
+
1675
+ if updates_made:
1676
+ for label, updates in epoch_updates.items():
1677
+ if updates:
1678
+ # 3. Apply the update to the prototype
1679
+ new_proto = self.prototypes[label].copy()
1680
+
1681
+ for upd in updates:
1682
+ new_proto.compose(upd, inplace=True)
1683
+
1684
+ self.prototypes[label] = compress_circuit(new_proto)
1685
+ self.prototypes[label].name = f"Prototype_{label}"
1686
+
1687
+ # 4. Evaluate the newly updated model
1688
+ predictions, _ = self.predict(train_points)
1689
+
1690
+ current_error = sum(1 for p, t in zip(predictions, train_labels) if p != t) / len(train_labels)
1691
+ epoch_entry = {"epoch": epoch, "error": float(current_error)}
1692
+ if test_points is not None:
1693
+ epoch_entry["test_error"] = float(_error(test_points, test_labels))
1694
+ self.retrain_history_.append(epoch_entry)
1695
+ if verbose:
1696
+ print(f"\tepoch {epoch}: {current_error}")
1697
+
1698
+ # 5. Track the best-known state; optionally stop early
1699
+ if current_error < best_error:
1700
+ # Improvement found: update the best error and save the new state
1701
+ best_error = current_error
1702
+ best_prototypes = {c: qc.copy() for c, qc in self.prototypes.items()}
1703
+
1704
+ if best_error == 0.0:
1705
+ break
1706
+ elif early_stop:
1707
+ # The updates made the model worse (or it stopped improving).
1708
+ # Revert to the best known state and exit!
1709
+ self.prototypes = best_prototypes
1710
+
1711
+ break
1712
+ # When early_stop is False the loop keeps running from the current
1713
+ # (possibly worse) prototypes so the full curve is recorded; the best
1714
+ # state is restored just before returning below.
1715
+
1716
+ # Always return the best-known prototypes, regardless of early_stop.
1717
+ self.prototypes = best_prototypes
1718
+
1719
+ return best_error, final_epoch