sparsekmeans 0.2__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sparsekmeans-0.2.2/PKG-INFO +24 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/README.md +2 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/setup.py +3 -3
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans/sparse_kmeans.py +11 -7
- sparsekmeans-0.2.2/sparsekmeans.egg-info/PKG-INFO +24 -0
- sparsekmeans-0.2/PKG-INFO +0 -12
- sparsekmeans-0.2/sparsekmeans.egg-info/PKG-INFO +0 -12
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/LICENSE +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/setup.cfg +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans/__init__.py +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans.egg-info/SOURCES.txt +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans.egg-info/dependency_links.txt +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans.egg-info/requires.txt +0 -0
- {sparsekmeans-0.2 → sparsekmeans-0.2.2}/sparsekmeans.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sparsekmeans
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
|
|
5
|
+
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
6
|
+
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
7
|
+
Author-email: cjlin@csie.ntu.edu.tw
|
|
8
|
+
License: MIT
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy
|
|
12
|
+
Requires-Dist: python-graphblas
|
|
13
|
+
Requires-Dist: scipy
|
|
14
|
+
Dynamic: author
|
|
15
|
+
Dynamic: author-email
|
|
16
|
+
Dynamic: description
|
|
17
|
+
Dynamic: home-page
|
|
18
|
+
Dynamic: license
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
Dynamic: requires-dist
|
|
21
|
+
Dynamic: requires-python
|
|
22
|
+
Dynamic: summary
|
|
23
|
+
|
|
24
|
+
See documentation here: https://github.com/cjlin1/sparsekmeans
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
The sparsekmeans package provides an efficient implementation of the K-means clustering algorithm optimized for sparse data sets. It is designed to handle high-dimensional and sparse data commonly found in text mining, recommender systems, and bioinformatics. By leveraging appropriate storage format and sparse matrix multiplication operations, our package ensures significant speedup in running time while maintaining consistency for clustering results compared with scikit-learn. Besides, the design of the package allows users to easily extend and customize, making it suitable for research or integrating into large-scale machine learning systems.
|
|
4
4
|
|
|
5
|
+
Implementation details are in the paper ["SparseKmeans: Efficient K-means Clustering for Sparse Data"](https://www.csie.ntu.edu.tw/~cjlin/papers/sparse_kmeans/sparsekmeans-cikm.pdf), CIKM 2025.
|
|
6
|
+
|
|
5
7
|
## Installation
|
|
6
8
|
|
|
7
9
|
Use the following command to install sparsekmeans with python >= 3.10.
|
|
@@ -2,7 +2,7 @@ from setuptools import setup
|
|
|
2
2
|
|
|
3
3
|
PACKAGE_DIR = ["sparsekmeans"]
|
|
4
4
|
PACKAGE_NAME = "sparsekmeans"
|
|
5
|
-
VERSION = "0.2"
|
|
5
|
+
VERSION = "0.2.2"
|
|
6
6
|
|
|
7
7
|
|
|
8
8
|
# license parameters
|
|
@@ -17,8 +17,8 @@ setup(
|
|
|
17
17
|
version=VERSION,
|
|
18
18
|
python_requires=">=3.10",
|
|
19
19
|
install_requires=["numpy","python-graphblas","scipy"],
|
|
20
|
-
description="",
|
|
21
|
-
|
|
20
|
+
description="A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets",
|
|
21
|
+
long_description="See documentation here: https://github.com/cjlin1/sparsekmeans",
|
|
22
22
|
author="Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang",
|
|
23
23
|
author_email="cjlin@csie.ntu.edu.tw",
|
|
24
24
|
url="https://github.com/cjlin1/sparsekmeans",
|
|
@@ -33,7 +33,7 @@ def squared_row_norms(X: gb.Matrix, n_threads: int=default_threads):
|
|
|
33
33
|
|
|
34
34
|
|
|
35
35
|
def check_centroids_density(centroids: gb.Matrix):
|
|
36
|
-
gamma = 0.
|
|
36
|
+
gamma = 0.05
|
|
37
37
|
|
|
38
38
|
if centroids.ss.format == "fullr":
|
|
39
39
|
is_centroid_dense = True
|
|
@@ -52,6 +52,10 @@ def check_centroids_density(centroids: gb.Matrix):
|
|
|
52
52
|
|
|
53
53
|
|
|
54
54
|
def predict_labels(X: gb.Matrix, centroids: gb.Matrix, is_centroid_dense: bool, n_threads: int=default_threads):
|
|
55
|
+
if X.dtype != dtypes.FP64:
|
|
56
|
+
X = X.dup(dtype="FP64")
|
|
57
|
+
if centroids.dtype != dtypes.FP64:
|
|
58
|
+
centroids = centroids.dup(dtype="FP64")
|
|
55
59
|
n_samples = X.shape[0]
|
|
56
60
|
n_clusters = centroids.shape[0]
|
|
57
61
|
|
|
@@ -65,16 +69,13 @@ def predict_labels(X: gb.Matrix, centroids: gb.Matrix, is_centroid_dense: bool,
|
|
|
65
69
|
centroids = centroids.ss.export("fullc")
|
|
66
70
|
centroids = gb.Matrix.ss.import_fullc(**centroids)
|
|
67
71
|
|
|
68
|
-
XCt << 0
|
|
69
|
-
XCt(accum=gb.binary.plus, nthreads=n_threads) << X.mxm(centroids.T)
|
|
70
|
-
XCt = 2 * XCt.to_dense()
|
|
71
|
-
|
|
72
72
|
else:
|
|
73
73
|
centroids = centroids.ss.export("csc")
|
|
74
74
|
centroids = gb.Matrix.ss.import_csc(**centroids)
|
|
75
75
|
|
|
76
|
-
|
|
77
|
-
|
|
76
|
+
XCt << 0
|
|
77
|
+
XCt(accum=gb.binary.plus, nthreads=n_threads) << X.mxm(centroids.T)
|
|
78
|
+
XCt = 2 * XCt.to_dense()
|
|
78
79
|
|
|
79
80
|
# x_squared_norms is not needed
|
|
80
81
|
c_squared_norms = squared_row_norms(centroids, n_threads)
|
|
@@ -298,6 +299,8 @@ class SparseKmeans:
|
|
|
298
299
|
|
|
299
300
|
if not isinstance(X, gb.Matrix):
|
|
300
301
|
X = gb.io.from_scipy_sparse(X)
|
|
302
|
+
if X.dtype != dtypes.FP64:
|
|
303
|
+
X = X.dup(dtype="FP64")
|
|
301
304
|
|
|
302
305
|
start_init_centroids = time.time()
|
|
303
306
|
self.centroids = self._initialize_centroids(X)
|
|
@@ -355,6 +358,7 @@ class SparseKmeans:
|
|
|
355
358
|
# If a selected point belongs to a cluster with only one point, we skip it and move on to the next furthest point.
|
|
356
359
|
# Each empty cluster gets one point
|
|
357
360
|
# We also need to update cluster_sizes
|
|
361
|
+
# See Algorithm E.1 in supplementary materials of Dang et al., CIKM 2025.
|
|
358
362
|
if n_empty > 0:
|
|
359
363
|
far_samples_idx = np.argsort(self.sample_centroids_closest_distance)[:: -1]
|
|
360
364
|
i, j = 0, 0
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sparsekmeans
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: A package for efficient implementation of the K-means clustering on high-dimensional sparse data sets
|
|
5
|
+
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
6
|
+
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
7
|
+
Author-email: cjlin@csie.ntu.edu.tw
|
|
8
|
+
License: MIT
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy
|
|
12
|
+
Requires-Dist: python-graphblas
|
|
13
|
+
Requires-Dist: scipy
|
|
14
|
+
Dynamic: author
|
|
15
|
+
Dynamic: author-email
|
|
16
|
+
Dynamic: description
|
|
17
|
+
Dynamic: home-page
|
|
18
|
+
Dynamic: license
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
Dynamic: requires-dist
|
|
21
|
+
Dynamic: requires-python
|
|
22
|
+
Dynamic: summary
|
|
23
|
+
|
|
24
|
+
See documentation here: https://github.com/cjlin1/sparsekmeans
|
sparsekmeans-0.2/PKG-INFO
DELETED
|
@@ -1,12 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.1
|
|
2
|
-
Name: sparsekmeans
|
|
3
|
-
Version: 0.2
|
|
4
|
-
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
5
|
-
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
6
|
-
Author-email: cjlin@csie.ntu.edu.tw
|
|
7
|
-
License: MIT
|
|
8
|
-
Requires-Python: >=3.10
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: numpy
|
|
11
|
-
Requires-Dist: python-graphblas
|
|
12
|
-
Requires-Dist: scipy
|
|
@@ -1,12 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.1
|
|
2
|
-
Name: sparsekmeans
|
|
3
|
-
Version: 0.2
|
|
4
|
-
Home-page: https://github.com/cjlin1/sparsekmeans
|
|
5
|
-
Author: Chih-Jen Lin, He-Zhe Lin, Khoi Nguyen Pham Dang
|
|
6
|
-
Author-email: cjlin@csie.ntu.edu.tw
|
|
7
|
-
License: MIT
|
|
8
|
-
Requires-Python: >=3.10
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: numpy
|
|
11
|
-
Requires-Dist: python-graphblas
|
|
12
|
-
Requires-Dist: scipy
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|