sparsekmeans 0.2__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: sparsekmeans
3
+ Version: 0.2.2
4
+ Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
5
+ Home-page: https://github.com/cjlin1/sparsekmeans
6
+ Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
7
+ Author-email: cjlin@csie.ntu.edu.tw
8
+ License: MIT
9
+ Requires-Python: >=3.10
10
+ License-File: LICENSE
11
+ Requires-Dist: numpy
12
+ Requires-Dist: python-graphblas
13
+ Requires-Dist: scipy
14
+ Dynamic: author
15
+ Dynamic: author-email
16
+ Dynamic: description
17
+ Dynamic: home-page
18
+ Dynamic: license
19
+ Dynamic: license-file
20
+ Dynamic: requires-dist
21
+ Dynamic: requires-python
22
+ Dynamic: summary
23
+
24
+ See documentation here: https://github.com/cjlin1/sparsekmeans
@@ -2,6 +2,8 @@
2
2
 
3
3
  The sparsekmeans package provides an efficient implementation of the K-means clustering algorithm optimized for sparse data sets. It is designed to handle high-dimensional and sparse data commonly found in text mining, recommender systems, and bioinformatics. By leveraging appropriate storage format and sparse matrix multiplication operations, our package ensures significant speedup in running time while maintaining consistency for clustering results compared with scikit-learn. Besides, the design of the package allows users to easily extend and customize, making it suitable for research or integrating into large-scale machine learning systems.
4
4
 
5
+ Implementation details are in the paper ["SparseKmeans: Efficient K-means Clustering for Sparse Data"](https://www.csie.ntu.edu.tw/~cjlin/papers/sparse_kmeans/sparsekmeans-cikm.pdf), CIKM 2025.
6
+
5
7
  ## Installation
6
8
 
7
9
  Use the following command to install sparsekmeans with python >= 3.10.
@@ -2,7 +2,7 @@ from setuptools import setup
2
2
 
3
3
  PACKAGE_DIR = ["sparsekmeans"]
4
4
  PACKAGE_NAME = "sparsekmeans"
5
- VERSION = "0.2"
5
+ VERSION = "0.2.2"
6
6
 
7
7
 
8
8
  # license parameters
@@ -17,8 +17,8 @@ setup(
17
17
  version=VERSION,
18
18
  python_requires=">=3.10",
19
19
  install_requires=["numpy","python-graphblas","scipy"],
20
- description="",
21
- long_description_content_type="",
20
+ description="A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets",
21
+ long_description="See documentation here: https://github.com/cjlin1/sparsekmeans",
22
22
  author="Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang",
23
23
  author_email="cjlin@csie.ntu.edu.tw",
24
24
  url="https://github.com/cjlin1/sparsekmeans",
@@ -33,7 +33,7 @@ def squared_row_norms(X: gb.Matrix, n_threads: int=default_threads):
33
33
 
34
34
 
35
35
  def check_centroids_density(centroids: gb.Matrix):
36
- gamma = 0.1
36
+ gamma = 0.05
37
37
 
38
38
  if centroids.ss.format == "fullr":
39
39
  is_centroid_dense = True
@@ -52,6 +52,10 @@ def check_centroids_density(centroids: gb.Matrix):
52
52
 
53
53
 
54
54
  def predict_labels(X: gb.Matrix, centroids: gb.Matrix, is_centroid_dense: bool, n_threads: int=default_threads):
55
+ if X.dtype != dtypes.FP64:
56
+ X = X.dup(dtype="FP64")
57
+ if centroids.dtype != dtypes.FP64:
58
+ centroids = centroids.dup(dtype="FP64")
55
59
  n_samples = X.shape[0]
56
60
  n_clusters = centroids.shape[0]
57
61
 
@@ -65,16 +69,13 @@ def predict_labels(X: gb.Matrix, centroids: gb.Matrix, is_centroid_dense: bool,
65
69
  centroids = centroids.ss.export("fullc")
66
70
  centroids = gb.Matrix.ss.import_fullc(**centroids)
67
71
 
68
- XCt << 0
69
- XCt(accum=gb.binary.plus, nthreads=n_threads) << X.mxm(centroids.T)
70
- XCt = 2 * XCt.to_dense()
71
-
72
72
  else:
73
73
  centroids = centroids.ss.export("csc")
74
74
  centroids = gb.Matrix.ss.import_csc(**centroids)
75
75
 
76
- XCt(nthreads=n_threads) << X.mxm(centroids.T)
77
- XCt = 2 * XCt.to_dense(fill_value=0)
76
+ XCt << 0
77
+ XCt(accum=gb.binary.plus, nthreads=n_threads) << X.mxm(centroids.T)
78
+ XCt = 2 * XCt.to_dense()
78
79
 
79
80
  # x_squared_norms is not needed
80
81
  c_squared_norms = squared_row_norms(centroids, n_threads)
@@ -298,6 +299,8 @@ class SparseKmeans:
298
299
 
299
300
  if not isinstance(X, gb.Matrix):
300
301
  X = gb.io.from_scipy_sparse(X)
302
+ if X.dtype != dtypes.FP64:
303
+ X = X.dup(dtype="FP64")
301
304
 
302
305
  start_init_centroids = time.time()
303
306
  self.centroids = self._initialize_centroids(X)
@@ -355,6 +358,7 @@ class SparseKmeans:
355
358
  # If a selected point belongs to a cluster with only one point, we skip it and move on to the next furthest point.
356
359
  # Each empty cluster gets one point
357
360
  # We also need to update cluster_sizes
361
+ # See Algorithm E.1 in supplementary materials of Dang et al., CIKM 2025.
358
362
  if n_empty > 0:
359
363
  far_samples_idx = np.argsort(self.sample_centroids_closest_distance)[:: -1]
360
364
  i, j = 0, 0
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: sparsekmeans
3
+ Version: 0.2.2
4
+ Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
5
+ Home-page: https://github.com/cjlin1/sparsekmeans
6
+ Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
7
+ Author-email: cjlin@csie.ntu.edu.tw
8
+ License: MIT
9
+ Requires-Python: >=3.10
10
+ License-File: LICENSE
11
+ Requires-Dist: numpy
12
+ Requires-Dist: python-graphblas
13
+ Requires-Dist: scipy
14
+ Dynamic: author
15
+ Dynamic: author-email
16
+ Dynamic: description
17
+ Dynamic: home-page
18
+ Dynamic: license
19
+ Dynamic: license-file
20
+ Dynamic: requires-dist
21
+ Dynamic: requires-python
22
+ Dynamic: summary
23
+
24
+ See documentation here: https://github.com/cjlin1/sparsekmeans
sparsekmeans-0.2/PKG-INFO DELETED
@@ -1,12 +0,0 @@
1
- Metadata-Version: 2.1
2
- Name: sparsekmeans
3
- Version: 0.2
4
- Home-page: https://github.com/cjlin1/sparsekmeans
5
- Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
6
- Author-email: cjlin@csie.ntu.edu.tw
7
- License: MIT
8
- Requires-Python: >=3.10
9
- License-File: LICENSE
10
- Requires-Dist: numpy
11
- Requires-Dist: python-graphblas
12
- Requires-Dist: scipy
@@ -1,12 +0,0 @@
1
- Metadata-Version: 2.1
2
- Name: sparsekmeans
3
- Version: 0.2
4
- Home-page: https://github.com/cjlin1/sparsekmeans
5
- Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
6
- Author-email: cjlin@csie.ntu.edu.tw
7
- License: MIT
8
- Requires-Python: >=3.10
9
- License-File: LICENSE
10
- Requires-Dist: numpy
11
- Requires-Dist: python-graphblas
12
- Requires-Dist: scipy
File without changes
File without changes