hdlib 2.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hdlib/__init__.py +16 -0
- hdlib/arithmetic/__init__.py +265 -0
- hdlib/arithmetic/quantum.py +1257 -0
- hdlib/model/__init__.py +12 -0
- hdlib/model/classification.py +1719 -0
- hdlib/model/clustering.py +172 -0
- hdlib/model/graph.py +764 -0
- hdlib/model/regression.py +306 -0
- hdlib/space.py +864 -0
- hdlib/vector.py +608 -0
- hdlib-2.1.0.data/scripts/chopin2.py +473 -0
- hdlib-2.1.0.dist-info/METADATA +137 -0
- hdlib-2.1.0.dist-info/RECORD +16 -0
- hdlib-2.1.0.dist-info/WHEEL +5 -0
- hdlib-2.1.0.dist-info/licenses/LICENSE +21 -0
- hdlib-2.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Clustering model.
|
|
2
|
+
|
|
3
|
+
It implements the __hdlib.model.clustering.ClusteringModel__ class object built according to the
|
|
4
|
+
Hyperdimensional Computing (HDC) paradigm as described in _Gupta et al. 2022_ https://doi.org/10.1145/3503541."""
|
|
5
|
+
|
|
6
|
+
from typing import List, Optional
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
from hdlib.vector import Vector
|
|
11
|
+
from hdlib.arithmetic import bundle
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ClusteringModel:
|
|
15
|
+
"""Hyperdimensional ClusteringModel representation."""
|
|
16
|
+
|
|
17
|
+
def __init__(
|
|
18
|
+
self,
|
|
19
|
+
k: int,
|
|
20
|
+
n_features: int,
|
|
21
|
+
size: int=10000,
|
|
22
|
+
vtype: str="bipolar",
|
|
23
|
+
max_iter: int=100,
|
|
24
|
+
seed: Optional[int]=None
|
|
25
|
+
) -> "ClusteringModel":
|
|
26
|
+
"""Initialize a ClusteringModel object.
|
|
27
|
+
|
|
28
|
+
Parameters
|
|
29
|
+
----------
|
|
30
|
+
k : int
|
|
31
|
+
Number of clusters.
|
|
32
|
+
n_features : int
|
|
33
|
+
Number of features.
|
|
34
|
+
size : int, default 10000
|
|
35
|
+
Vector dimensionality.
|
|
36
|
+
vtype : str, deafult 'bipolar'
|
|
37
|
+
Vector type.
|
|
38
|
+
max_iter : int, default 100
|
|
39
|
+
Maximum number of iterations.
|
|
40
|
+
seed : int, default None
|
|
41
|
+
Seed for reproducibility.
|
|
42
|
+
|
|
43
|
+
Returns
|
|
44
|
+
-------
|
|
45
|
+
ClusteringModel
|
|
46
|
+
The clustering model object.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
if not isinstance(k, int) or k <= 0:
|
|
50
|
+
raise ValueError("The number of clusters `k` must be a positive integer")
|
|
51
|
+
|
|
52
|
+
if not isinstance(n_features, int) or n_features <= 0:
|
|
53
|
+
raise ValueError("The number of features `n_features` must be a positive integer")
|
|
54
|
+
|
|
55
|
+
if not isinstance(size, int):
|
|
56
|
+
raise ValueError("Vectors size must be an integer")
|
|
57
|
+
|
|
58
|
+
if not isinstance(max_iter, int) or max_iter <= 0:
|
|
59
|
+
raise ValueError("The number of iterations `max_iter` must be a positive integer")
|
|
60
|
+
|
|
61
|
+
if seed is not None and not isinstance(seed, int):
|
|
62
|
+
raise TypeError("Seed must be an integer number")
|
|
63
|
+
|
|
64
|
+
self.k = k
|
|
65
|
+
self.n_features = n_features
|
|
66
|
+
self.size = size
|
|
67
|
+
self.vtype = vtype
|
|
68
|
+
self.max_iter = max_iter
|
|
69
|
+
self.seed = seed
|
|
70
|
+
|
|
71
|
+
if self.seed is not None:
|
|
72
|
+
np.random.seed(self.seed)
|
|
73
|
+
|
|
74
|
+
# Internal projection matrix for encoding
|
|
75
|
+
self.projection_matrix_ = np.random.randn(self.n_features, self.size)
|
|
76
|
+
|
|
77
|
+
self.centroids_: List[Vector] = []
|
|
78
|
+
self.labels_: np.ndarray = np.array([])
|
|
79
|
+
|
|
80
|
+
def _encode(self, X: np.ndarray) -> List[Vector]:
|
|
81
|
+
"""Encodes a low-dimensional dataset into a list of hypervectors."""
|
|
82
|
+
|
|
83
|
+
encoded_vectors = []
|
|
84
|
+
|
|
85
|
+
for point in X:
|
|
86
|
+
hd_vector_float = np.dot(point, self.projection_matrix_)
|
|
87
|
+
hd_vector_bipolar = np.sign(hd_vector_float)
|
|
88
|
+
encoded_vectors.append(Vector(vtype=self.vtype, vector=hd_vector_bipolar))
|
|
89
|
+
|
|
90
|
+
return encoded_vectors
|
|
91
|
+
|
|
92
|
+
def fit(self, X: np.ndarray) -> "ClusteringModel":
|
|
93
|
+
"""Compute HDC-based k-means clustering from a raw data matrix."""
|
|
94
|
+
|
|
95
|
+
if not isinstance(X, np.ndarray) or X.ndim != 2:
|
|
96
|
+
raise TypeError("Input `X` must be a 2D NumPy array.")
|
|
97
|
+
|
|
98
|
+
if X.shape[1] != self.n_features:
|
|
99
|
+
raise ValueError(f"Input data has {X.shape[1]} features, but model was initialized with {self.n_features}.")
|
|
100
|
+
|
|
101
|
+
# Step 1: Encode the input data
|
|
102
|
+
points = self._encode(X)
|
|
103
|
+
|
|
104
|
+
num_points = len(points)
|
|
105
|
+
self.labels_ = np.zeros(num_points, dtype=int)
|
|
106
|
+
|
|
107
|
+
initial_indices = np.random.choice(num_points, size=self.k, replace=False)
|
|
108
|
+
self.centroids_ = [Vector(vtype=points[i].vtype, vector=points[i].vector) for i in initial_indices]
|
|
109
|
+
|
|
110
|
+
for i, centroid in enumerate(self.centroids_):
|
|
111
|
+
centroid.name = f"centroid_{i}"
|
|
112
|
+
|
|
113
|
+
for iteration in range(self.max_iter):
|
|
114
|
+
new_labels = np.zeros(num_points, dtype=int)
|
|
115
|
+
|
|
116
|
+
for i, point in enumerate(points):
|
|
117
|
+
distances = [point.dist(centroid, method="cosine") for centroid in self.centroids_]
|
|
118
|
+
new_labels[i] = np.argmin(distances)
|
|
119
|
+
|
|
120
|
+
new_centroids = []
|
|
121
|
+
|
|
122
|
+
for j in range(self.k):
|
|
123
|
+
cluster_points = [points[i] for i, label in enumerate(new_labels) if label == j]
|
|
124
|
+
|
|
125
|
+
if not cluster_points:
|
|
126
|
+
random_idx = np.random.randint(0, num_points)
|
|
127
|
+
new_centroids.append(Vector(vtype=points[random_idx].vtype, vector=points[random_idx].vector))
|
|
128
|
+
|
|
129
|
+
else:
|
|
130
|
+
new_centroid_vector = cluster_points[0]
|
|
131
|
+
if len(cluster_points) > 1:
|
|
132
|
+
for point in cluster_points[1:]:
|
|
133
|
+
new_centroid_vector = bundle(new_centroid_vector, point)
|
|
134
|
+
|
|
135
|
+
new_centroid_vector.vector /= len(cluster_points)
|
|
136
|
+
new_centroids.append(new_centroid_vector)
|
|
137
|
+
|
|
138
|
+
for i, centroid in enumerate(new_centroids):
|
|
139
|
+
centroid.name = f"centroid_{i}"
|
|
140
|
+
|
|
141
|
+
self.centroids_ = new_centroids
|
|
142
|
+
|
|
143
|
+
if np.array_equal(self.labels_, new_labels):
|
|
144
|
+
print(f"Converged after {iteration + 1} iterations.")
|
|
145
|
+
break
|
|
146
|
+
|
|
147
|
+
self.labels_ = new_labels
|
|
148
|
+
|
|
149
|
+
return self
|
|
150
|
+
|
|
151
|
+
def predict(self, X: np.ndarray) -> np.ndarray:
|
|
152
|
+
"""Predict the closest cluster for each sample in a raw data matrix."""
|
|
153
|
+
|
|
154
|
+
if not self.centroids_:
|
|
155
|
+
raise RuntimeError("The model has not been fitted yet. Call .fit() first.")
|
|
156
|
+
|
|
157
|
+
if not isinstance(X, np.ndarray) or X.ndim != 2:
|
|
158
|
+
raise TypeError("Input `X` must be a 2D NumPy array.")
|
|
159
|
+
|
|
160
|
+
if X.shape[1] != self.n_features:
|
|
161
|
+
raise ValueError(f"Input data has {X.shape[1]} features, but model was initialized with {self.n_features}.")
|
|
162
|
+
|
|
163
|
+
# Encode the new data points
|
|
164
|
+
points = self._encode(X)
|
|
165
|
+
|
|
166
|
+
predictions = np.zeros(len(points), dtype=int)
|
|
167
|
+
|
|
168
|
+
for i, point in enumerate(points):
|
|
169
|
+
distances = [point.dist(centroid, method="cosine") for centroid in self.centroids_]
|
|
170
|
+
predictions[i] = np.argmin(distances)
|
|
171
|
+
|
|
172
|
+
return predictions
|