sparsekmeans 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sparsekmeans
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
5
5
  Home-page: https://github.com/cjlin1/sparsekmeans
6
6
  Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
@@ -2,7 +2,7 @@ from setuptools import setup
2
2
 
3
3
  PACKAGE_DIR = ["sparsekmeans"]
4
4
  PACKAGE_NAME = "sparsekmeans"
5
- VERSION = "0.2.2"
5
+ VERSION = "0.2.3"
6
6
 
7
7
 
8
8
  # license parameters
@@ -361,22 +361,20 @@ class SparseKmeans:
361
361
  # See Algorithm E.1 in supplementary materials of Dang et al., CIKM 2025.
362
362
  if n_empty > 0:
363
363
  far_samples_idx = np.argsort(self.sample_centroids_closest_distance)[:: -1]
364
- i, j = 0, 0
364
+ k, j = 0, 0
365
365
 
366
- while i < n_samples - n_empty and j < n_empty:
367
- origin_cluster = self.labels[far_samples_idx[i]]
368
- i += 1
369
-
370
- if cluster_sizes[origin_cluster] == 1:
371
- continue
372
-
373
- target_empty_cluster = empty_clusters[j]
374
- cluster_sizes[origin_cluster] -= 1
375
- cluster_sizes[target_empty_cluster] = 1
376
- self.labels[far_samples_idx[i]] = target_empty_cluster
377
- j += 1
366
+ while k < n_samples - n_empty and j < n_empty:
367
+ origin_cluster = self.labels[far_samples_idx[k]]
368
+
369
+ if cluster_sizes[origin_cluster] > 1:
370
+ target_empty_cluster = empty_clusters[j]
371
+ cluster_sizes[origin_cluster] -= 1
372
+ cluster_sizes[target_empty_cluster] = 1
373
+ self.labels[far_samples_idx[k]] = target_empty_cluster
374
+ j += 1
375
+ k +=1
378
376
 
379
- if i == n_samples - n_empty:
377
+ if k == n_samples - n_empty:
380
378
  raise ValueError("Not enough distinct points to assign to empty clusters.")
381
379
 
382
380
  # Get the weight of each sample in its corresponding cluster
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sparsekmeans
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
5
5
  Home-page: https://github.com/cjlin1/sparsekmeans
6
6
  Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
File without changes
File without changes
File without changes