sparsekmeans 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/PKG-INFO +1 -1
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/setup.py +1 -1
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans/sparse_kmeans.py +12 -14
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans.egg-info/PKG-INFO +1 -1
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/LICENSE +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/README.md +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/setup.cfg +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans/__init__.py +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans.egg-info/SOURCES.txt +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans.egg-info/dependency_links.txt +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans.egg-info/requires.txt +0 -0
- {sparsekmeans-0.2.2 → sparsekmeans-0.2.3}/sparsekmeans.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sparsekmeans
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
|
|
5
5
|
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
6
6
|
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
@@ -361,22 +361,20 @@ class SparseKmeans:
|
|
|
361
361
|
# See Algorithm E.1 in supplementary materials of Dang et al., CIKM 2025.
|
|
362
362
|
if n_empty > 0:
|
|
363
363
|
far_samples_idx = np.argsort(self.sample_centroids_closest_distance)[:: -1]
|
|
364
|
-
|
|
364
|
+
k, j = 0, 0
|
|
365
365
|
|
|
366
|
-
while
|
|
367
|
-
origin_cluster = self.labels[far_samples_idx[
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
self.labels[far_samples_idx[i]] = target_empty_cluster
|
|
377
|
-
j += 1
|
|
366
|
+
while k < n_samples - n_empty and j < n_empty:
|
|
367
|
+
origin_cluster = self.labels[far_samples_idx[k]]
|
|
368
|
+
|
|
369
|
+
if cluster_sizes[origin_cluster] > 1:
|
|
370
|
+
target_empty_cluster = empty_clusters[j]
|
|
371
|
+
cluster_sizes[origin_cluster] -= 1
|
|
372
|
+
cluster_sizes[target_empty_cluster] = 1
|
|
373
|
+
self.labels[far_samples_idx[k]] = target_empty_cluster
|
|
374
|
+
j += 1
|
|
375
|
+
k +=1
|
|
378
376
|
|
|
379
|
-
if
|
|
377
|
+
if k == n_samples - n_empty:
|
|
380
378
|
raise ValueError("Not enough distinct points to assign to empty clusters.")
|
|
381
379
|
|
|
382
380
|
# Get the weight of each sample in its corresponding cluster
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sparsekmeans
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
|
|
5
5
|
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
6
6
|
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|