gmcluster 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gmcluster-0.3.0/LICENSE +29 -0
- gmcluster-0.3.0/PKG-INFO +82 -0
- gmcluster-0.3.0/README.rst +58 -0
- gmcluster-0.3.0/gmcluster/__init__.py +4 -0
- gmcluster-0.3.0/gmcluster/gmcluster.py +732 -0
- gmcluster-0.3.0/gmcluster.egg-info/PKG-INFO +82 -0
- gmcluster-0.3.0/gmcluster.egg-info/SOURCES.txt +15 -0
- gmcluster-0.3.0/gmcluster.egg-info/dependency_links.txt +1 -0
- gmcluster-0.3.0/gmcluster.egg-info/requires.txt +12 -0
- gmcluster-0.3.0/gmcluster.egg-info/top_level.txt +1 -0
- gmcluster-0.3.0/pyproject.toml +34 -0
- gmcluster-0.3.0/setup.cfg +4 -0
- gmcluster-0.3.0/tests/test_api.py +115 -0
- gmcluster-0.3.0/tests/test_equivalence.py +58 -0
- gmcluster-0.3.0/tests/test_order_selection.py +26 -0
- gmcluster-0.3.0/tests/test_recover_params.py +31 -0
- gmcluster-0.3.0/tests/test_smoke.py +17 -0
gmcluster-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2022, Charles A Bouman
|
|
4
|
+
All rights reserved.
|
|
5
|
+
|
|
6
|
+
Redistribution and use in source and binary forms, with or without
|
|
7
|
+
modification, are permitted provided that the following conditions are met:
|
|
8
|
+
|
|
9
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
10
|
+
list of conditions and the following disclaimer.
|
|
11
|
+
|
|
12
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
13
|
+
this list of conditions and the following disclaimer in the documentation
|
|
14
|
+
and/or other materials provided with the distribution.
|
|
15
|
+
|
|
16
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
17
|
+
contributors may be used to endorse or promote products derived from
|
|
18
|
+
this software without specific prior written permission.
|
|
19
|
+
|
|
20
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
21
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
22
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
23
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
24
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
25
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
26
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
27
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
28
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
29
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
gmcluster-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gmcluster
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: EM Gaussian-mixture clustering with automatic MDL order selection
|
|
5
|
+
Author: Charles A. Bouman, Mohammad Samin Nur Chowdhury
|
|
6
|
+
License: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/cabouman/gmcluster
|
|
8
|
+
Project-URL: Documentation, https://gmcluster.readthedocs.io
|
|
9
|
+
Project-URL: Repository, https://github.com/cabouman/gmcluster
|
|
10
|
+
Requires-Python: >=3.11
|
|
11
|
+
Description-Content-Type: text/x-rst
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: numpy
|
|
14
|
+
Requires-Dist: matplotlib
|
|
15
|
+
Provides-Extra: test
|
|
16
|
+
Requires-Dist: pytest; extra == "test"
|
|
17
|
+
Provides-Extra: docs
|
|
18
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
19
|
+
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
20
|
+
Requires-Dist: sphinx-design; extra == "docs"
|
|
21
|
+
Requires-Dist: sphinx-copybutton; extra == "docs"
|
|
22
|
+
Requires-Dist: sphinxcontrib-bibtex; extra == "docs"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
GMCluster
|
|
26
|
+
=========
|
|
27
|
+
|
|
28
|
+
GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
|
|
29
|
+
|
|
30
|
+
It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
|
|
31
|
+
|
|
32
|
+
Installing
|
|
33
|
+
----------
|
|
34
|
+
|
|
35
|
+
Install the latest release from PyPI::
|
|
36
|
+
|
|
37
|
+
pip install gmcluster
|
|
38
|
+
|
|
39
|
+
To install from source (for development), clone the repository and do an editable install::
|
|
40
|
+
|
|
41
|
+
git clone https://github.com/cabouman/gmcluster.git
|
|
42
|
+
cd gmcluster
|
|
43
|
+
pip install -e .
|
|
44
|
+
|
|
45
|
+
Quick Start
|
|
46
|
+
-----------
|
|
47
|
+
|
|
48
|
+
The package provides one class, ``GaussianMixture``. Fit it to your data, read the
|
|
49
|
+
estimated parameters, then classify points or draw new samples.
|
|
50
|
+
|
|
51
|
+
.. code-block:: python
|
|
52
|
+
|
|
53
|
+
import numpy as np
|
|
54
|
+
from gmcluster import GaussianMixture
|
|
55
|
+
|
|
56
|
+
X = np.random.default_rng(0).standard_normal((500, 2))
|
|
57
|
+
|
|
58
|
+
# Fit the mixture; "auto" selects the number of clusters by MDL.
|
|
59
|
+
gm = GaussianMixture(num_clusters="auto").fit(X)
|
|
60
|
+
|
|
61
|
+
print(gm.estimated_num_clusters) # number of clusters found
|
|
62
|
+
print(gm.estimated_weights) # shape (K,)
|
|
63
|
+
print(gm.estimated_means) # shape (K, M)
|
|
64
|
+
print(gm.estimated_covariances) # shape (K, M, M)
|
|
65
|
+
|
|
66
|
+
labels = gm.classify(X) # most-likely cluster per point, shape (N,)
|
|
67
|
+
new_points = gm.sample(100) # draw 100 samples from the fitted mixture
|
|
68
|
+
|
|
69
|
+
Running the demos
|
|
70
|
+
-----------------
|
|
71
|
+
|
|
72
|
+
Validate the installation by running a demo::
|
|
73
|
+
|
|
74
|
+
cd demo
|
|
75
|
+
python demo_1.py
|
|
76
|
+
|
|
77
|
+
Citation
|
|
78
|
+
--------
|
|
79
|
+
|
|
80
|
+
Please cite this software when you use it. The BibTeX entry is in
|
|
81
|
+
``docs/source/credits.rst`` and in the online documentation at
|
|
82
|
+
https://gmcluster.readthedocs.io/ .
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
GMCluster
|
|
2
|
+
=========
|
|
3
|
+
|
|
4
|
+
GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
|
|
5
|
+
|
|
6
|
+
It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
|
|
7
|
+
|
|
8
|
+
Installing
|
|
9
|
+
----------
|
|
10
|
+
|
|
11
|
+
Install the latest release from PyPI::
|
|
12
|
+
|
|
13
|
+
pip install gmcluster
|
|
14
|
+
|
|
15
|
+
To install from source (for development), clone the repository and do an editable install::
|
|
16
|
+
|
|
17
|
+
git clone https://github.com/cabouman/gmcluster.git
|
|
18
|
+
cd gmcluster
|
|
19
|
+
pip install -e .
|
|
20
|
+
|
|
21
|
+
Quick Start
|
|
22
|
+
-----------
|
|
23
|
+
|
|
24
|
+
The package provides one class, ``GaussianMixture``. Fit it to your data, read the
|
|
25
|
+
estimated parameters, then classify points or draw new samples.
|
|
26
|
+
|
|
27
|
+
.. code-block:: python
|
|
28
|
+
|
|
29
|
+
import numpy as np
|
|
30
|
+
from gmcluster import GaussianMixture
|
|
31
|
+
|
|
32
|
+
X = np.random.default_rng(0).standard_normal((500, 2))
|
|
33
|
+
|
|
34
|
+
# Fit the mixture; "auto" selects the number of clusters by MDL.
|
|
35
|
+
gm = GaussianMixture(num_clusters="auto").fit(X)
|
|
36
|
+
|
|
37
|
+
print(gm.estimated_num_clusters) # number of clusters found
|
|
38
|
+
print(gm.estimated_weights) # shape (K,)
|
|
39
|
+
print(gm.estimated_means) # shape (K, M)
|
|
40
|
+
print(gm.estimated_covariances) # shape (K, M, M)
|
|
41
|
+
|
|
42
|
+
labels = gm.classify(X) # most-likely cluster per point, shape (N,)
|
|
43
|
+
new_points = gm.sample(100) # draw 100 samples from the fitted mixture
|
|
44
|
+
|
|
45
|
+
Running the demos
|
|
46
|
+
-----------------
|
|
47
|
+
|
|
48
|
+
Validate the installation by running a demo::
|
|
49
|
+
|
|
50
|
+
cd demo
|
|
51
|
+
python demo_1.py
|
|
52
|
+
|
|
53
|
+
Citation
|
|
54
|
+
--------
|
|
55
|
+
|
|
56
|
+
Please cite this software when you use it. The BibTeX entry is in
|
|
57
|
+
``docs/source/credits.rst`` and in the online documentation at
|
|
58
|
+
https://gmcluster.readthedocs.io/ .
|
|
@@ -0,0 +1,732 @@
|
|
|
1
|
+
# EM Clustering Library
|
|
2
|
+
# Copyright (C) 2022, Charles A Bouman.
|
|
3
|
+
# All rights reserved.
|
|
4
|
+
|
|
5
|
+
import copy
|
|
6
|
+
import logging
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger("gmcluster")
|
|
11
|
+
|
|
12
|
+
# Names guarded before fit. Reading any of these on an unfitted model raises.
|
|
13
|
+
_ESTIMATE_NAMES = (
|
|
14
|
+
"estimated_num_clusters",
|
|
15
|
+
"estimated_weights",
|
|
16
|
+
"estimated_means",
|
|
17
|
+
"estimated_covariances",
|
|
18
|
+
"mdl",
|
|
19
|
+
"mdl_path",
|
|
20
|
+
"converged",
|
|
21
|
+
"num_iterations",
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class GaussianMixture:
|
|
26
|
+
"""Gaussian mixture model fit by EM with MDL order selection.
|
|
27
|
+
|
|
28
|
+
The constructor holds settings. fit(X) runs EM and stores the results in
|
|
29
|
+
the estimated_* attributes and the diagnostics (mdl, mdl_path, converged,
|
|
30
|
+
num_iterations). Reading any of those before fit raises.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __init__(self, num_clusters="auto", max_clusters=20, covariance_type="full",
|
|
34
|
+
alpha=0.1, whiten=False, verbose=False):
|
|
35
|
+
"""Store settings after validating them.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
num_clusters: "auto" to select the order by MDL, or a positive int to fix it.
|
|
39
|
+
max_clusters: positive int, the ceiling for the "auto" search.
|
|
40
|
+
covariance_type: "full" or "diagonal".
|
|
41
|
+
alpha: covariance regularization, 0 < alpha <= 1 (1 spherical, ->0 elliptical).
|
|
42
|
+
whiten: decorrelate coordinates before clustering.
|
|
43
|
+
verbose: report progress through the logging module.
|
|
44
|
+
"""
|
|
45
|
+
# num_clusters: "auto" or a positive int (bool is not a count).
|
|
46
|
+
if isinstance(num_clusters, str):
|
|
47
|
+
if num_clusters != "auto":
|
|
48
|
+
raise ValueError('num_clusters must be "auto" or a positive int')
|
|
49
|
+
elif isinstance(num_clusters, bool) or not isinstance(num_clusters, (int, np.integer)):
|
|
50
|
+
raise TypeError('num_clusters must be "auto" or a positive int')
|
|
51
|
+
elif num_clusters <= 0:
|
|
52
|
+
raise ValueError("num_clusters must be a positive int")
|
|
53
|
+
|
|
54
|
+
# max_clusters: positive int.
|
|
55
|
+
if isinstance(max_clusters, bool) or not isinstance(max_clusters, (int, np.integer)):
|
|
56
|
+
raise TypeError("max_clusters must be a positive int")
|
|
57
|
+
if max_clusters <= 0:
|
|
58
|
+
raise ValueError("max_clusters must be a positive int")
|
|
59
|
+
|
|
60
|
+
# covariance_type: "full" or "diagonal" (mapped to the internal "diag").
|
|
61
|
+
if covariance_type == "full":
|
|
62
|
+
est_kind = "full"
|
|
63
|
+
elif covariance_type == "diagonal":
|
|
64
|
+
est_kind = "diag"
|
|
65
|
+
else:
|
|
66
|
+
raise ValueError('covariance_type must be "full" or "diagonal"')
|
|
67
|
+
|
|
68
|
+
# alpha: 0 < alpha <= 1.
|
|
69
|
+
if isinstance(alpha, bool) or not isinstance(alpha, (int, float, np.integer, np.floating)):
|
|
70
|
+
raise TypeError("alpha must be a number in (0, 1]")
|
|
71
|
+
if not (0 < alpha <= 1):
|
|
72
|
+
raise ValueError("alpha must satisfy 0 < alpha <= 1")
|
|
73
|
+
|
|
74
|
+
if not isinstance(whiten, bool):
|
|
75
|
+
raise TypeError("whiten must be a bool")
|
|
76
|
+
if not isinstance(verbose, bool):
|
|
77
|
+
raise TypeError("verbose must be a bool")
|
|
78
|
+
|
|
79
|
+
self.num_clusters = num_clusters
|
|
80
|
+
self.max_clusters = int(max_clusters)
|
|
81
|
+
self.covariance_type = covariance_type
|
|
82
|
+
self.alpha = float(alpha)
|
|
83
|
+
self.whiten = whiten
|
|
84
|
+
self.verbose = verbose
|
|
85
|
+
|
|
86
|
+
# Internal estimator kind ("full" or "diag") and the fitted engine mixture.
|
|
87
|
+
self._est_kind = est_kind
|
|
88
|
+
self._mixture = None
|
|
89
|
+
self._fitted = False
|
|
90
|
+
|
|
91
|
+
def __getattr__(self, name):
|
|
92
|
+
"""Raise a clear error when an estimate is read before fit."""
|
|
93
|
+
# __getattr__ runs only when normal lookup fails, i.e. before fit sets these.
|
|
94
|
+
if name in _ESTIMATE_NAMES:
|
|
95
|
+
raise RuntimeError("GaussianMixture is not fitted; call fit(X) first")
|
|
96
|
+
raise AttributeError(name)
|
|
97
|
+
|
|
98
|
+
def fit(self, X):
|
|
99
|
+
"""Fit the mixture to X by EM and store the estimates. Returns self.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
X: (num_points, num_features) 2D float array of observations.
|
|
103
|
+
"""
|
|
104
|
+
X = _check_data(X)
|
|
105
|
+
|
|
106
|
+
if self.num_clusters == "auto":
|
|
107
|
+
init_K = self.max_clusters
|
|
108
|
+
final_K = 0
|
|
109
|
+
else:
|
|
110
|
+
final_K = int(self.num_clusters)
|
|
111
|
+
init_K = max(self.max_clusters, final_K)
|
|
112
|
+
|
|
113
|
+
mixture, mdl_path = _fit_mixture(X, init_K, final_K, self._est_kind,
|
|
114
|
+
self.alpha, self.whiten, self.verbose)
|
|
115
|
+
|
|
116
|
+
self._populate(mixture, mdl_path)
|
|
117
|
+
return self
|
|
118
|
+
|
|
119
|
+
def _populate(self, mixture, mdl_path):
|
|
120
|
+
"""Set the estimated_* attributes and diagnostics from an engine mixture."""
|
|
121
|
+
clusters = mixture.cluster
|
|
122
|
+
self._mixture = mixture
|
|
123
|
+
self.estimated_num_clusters = int(mixture.K)
|
|
124
|
+
self.estimated_weights = np.array([float(c.pb) for c in clusters])
|
|
125
|
+
self.estimated_means = np.array([c.mu.ravel() for c in clusters])
|
|
126
|
+
self.estimated_covariances = np.array([np.asarray(c.R) for c in clusters])
|
|
127
|
+
self.mdl = mixture.rissanen
|
|
128
|
+
self.mdl_path = mdl_path
|
|
129
|
+
self.converged = True
|
|
130
|
+
self.num_iterations = getattr(mixture, "num_iterations", None)
|
|
131
|
+
self._fitted = True
|
|
132
|
+
|
|
133
|
+
def _require_fitted(self):
|
|
134
|
+
if not self._fitted:
|
|
135
|
+
raise RuntimeError("GaussianMixture is not fitted; call fit(X) first")
|
|
136
|
+
|
|
137
|
+
def posterior(self, X):
|
|
138
|
+
"""Return P(cluster | x), shape (N, K), rows summing to 1."""
|
|
139
|
+
self._require_fitted()
|
|
140
|
+
X = _check_data(X, n_features=self._mixture.M)
|
|
141
|
+
_, _ = E_step(self._mixture, X)
|
|
142
|
+
return np.array(self._mixture.pnk)
|
|
143
|
+
|
|
144
|
+
def classify(self, X):
|
|
145
|
+
"""Return the most-likely cluster index per point, shape (N,)."""
|
|
146
|
+
return np.argmax(self.posterior(X), axis=1)
|
|
147
|
+
|
|
148
|
+
def log_likelihood(self, X):
|
|
149
|
+
"""Return the per-point log density log p(x), shape (N,)."""
|
|
150
|
+
self._require_fitted()
|
|
151
|
+
X = _check_data(X, n_features=self._mixture.M)
|
|
152
|
+
return _class_log_likelihood(self._mixture, X).ravel()
|
|
153
|
+
|
|
154
|
+
def sample(self, num_samples=1, rng=None, with_labels=False):
|
|
155
|
+
"""Draw samples from the fitted mixture.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
num_samples: number of samples to draw.
|
|
159
|
+
rng: a numpy Generator or seed for reproducibility.
|
|
160
|
+
with_labels: also return the component index of each sample.
|
|
161
|
+
|
|
162
|
+
Returns:
|
|
163
|
+
X of shape (num_samples, M), or (X, labels) with labels of shape
|
|
164
|
+
(num_samples,) if with_labels is True.
|
|
165
|
+
"""
|
|
166
|
+
self._require_fitted()
|
|
167
|
+
rng = np.random.default_rng(rng)
|
|
168
|
+
weights = self.estimated_weights
|
|
169
|
+
means = self.estimated_means
|
|
170
|
+
covs = self.estimated_covariances
|
|
171
|
+
M = means.shape[1]
|
|
172
|
+
|
|
173
|
+
labels = rng.choice(len(weights), size=num_samples, p=weights)
|
|
174
|
+
samples = np.empty((num_samples, M))
|
|
175
|
+
|
|
176
|
+
# Draw each component's points from its Gaussian via a symmetric-eigen factor.
|
|
177
|
+
for k in range(len(weights)):
|
|
178
|
+
idx = np.nonzero(labels == k)[0]
|
|
179
|
+
if idx.size == 0:
|
|
180
|
+
continue
|
|
181
|
+
eigvals, eigvecs = np.linalg.eigh(covs[k])
|
|
182
|
+
eigvals = np.clip(eigvals, 0.0, None)
|
|
183
|
+
factor = eigvecs * np.sqrt(eigvals)
|
|
184
|
+
z = rng.standard_normal((idx.size, M))
|
|
185
|
+
samples[idx] = means[k] + z @ factor.T
|
|
186
|
+
|
|
187
|
+
if with_labels:
|
|
188
|
+
return samples, labels
|
|
189
|
+
return samples
|
|
190
|
+
|
|
191
|
+
def split_clusters(self):
|
|
192
|
+
"""Return a list of single-cluster GaussianMixture models, one per component.
|
|
193
|
+
|
|
194
|
+
Each returned model is fitted with one cluster (weight 1) and is usable
|
|
195
|
+
with classify, posterior, log_likelihood, and sample. Kept for use
|
|
196
|
+
alongside other segmentation packages.
|
|
197
|
+
"""
|
|
198
|
+
self._require_fitted()
|
|
199
|
+
parts = []
|
|
200
|
+
for k in range(self._mixture.K):
|
|
201
|
+
single = MixtureObj()
|
|
202
|
+
single.K = 1
|
|
203
|
+
single.M = self._mixture.M
|
|
204
|
+
single.cluster = [copy.deepcopy(self._mixture.cluster[k])]
|
|
205
|
+
single.D_reg = self._mixture.D_reg
|
|
206
|
+
# Renormalize so the single component is a proper order-1 density (weight 1).
|
|
207
|
+
single = cluster_normalize(single)
|
|
208
|
+
single.rissanen = None
|
|
209
|
+
single.loglikelihood = None
|
|
210
|
+
single.num_iterations = None
|
|
211
|
+
|
|
212
|
+
child = GaussianMixture(num_clusters=1, max_clusters=self.max_clusters,
|
|
213
|
+
covariance_type=self.covariance_type, alpha=self.alpha,
|
|
214
|
+
whiten=self.whiten, verbose=self.verbose)
|
|
215
|
+
child._populate(single, mdl_path=[(1, None)])
|
|
216
|
+
child.mdl = None
|
|
217
|
+
child.converged = True
|
|
218
|
+
parts.append(child)
|
|
219
|
+
return parts
|
|
220
|
+
|
|
221
|
+
def __repr__(self):
|
|
222
|
+
if self._fitted:
|
|
223
|
+
return "GaussianMixture(clusters={}, dims={}, mdl={})".format(
|
|
224
|
+
self.estimated_num_clusters, self._mixture.M, self.mdl)
|
|
225
|
+
return "GaussianMixture(num_clusters={!r}, covariance_type={!r}, unfitted)".format(
|
|
226
|
+
self.num_clusters, self.covariance_type)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _check_data(X, n_features=None):
|
|
230
|
+
"""Validate X as a 2D float array and return it. Raise on bad input."""
|
|
231
|
+
X = np.asarray(X)
|
|
232
|
+
if X.ndim != 2:
|
|
233
|
+
raise ValueError("X must be a 2D array of shape (num_points, num_features)")
|
|
234
|
+
if not np.issubdtype(X.dtype, np.number):
|
|
235
|
+
raise ValueError("X must be a numeric array")
|
|
236
|
+
if X.dtype != float:
|
|
237
|
+
X = X.astype(float)
|
|
238
|
+
if n_features is not None and X.shape[1] != n_features:
|
|
239
|
+
raise ValueError("X has {} features; the model was fit on {}".format(
|
|
240
|
+
X.shape[1], n_features))
|
|
241
|
+
return X
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
class MixtureObj:
|
|
245
|
+
"""Bag of parameters for a Gaussian mixture (the engine's mixture record)."""
|
|
246
|
+
|
|
247
|
+
def __init__(self):
|
|
248
|
+
"""Initialize the fields to None."""
|
|
249
|
+
self.K = None
|
|
250
|
+
self.M = None
|
|
251
|
+
self.cluster = None
|
|
252
|
+
self.rissanen = None
|
|
253
|
+
self.loglikelihood = None
|
|
254
|
+
self.pnk = None
|
|
255
|
+
self.D_reg = None
|
|
256
|
+
self.num_iterations = None
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class ClusterObj:
|
|
260
|
+
"""Bag of parameters for one cluster (the engine's cluster record)."""
|
|
261
|
+
|
|
262
|
+
def __init__(self):
|
|
263
|
+
"""Initialize the fields to None."""
|
|
264
|
+
self.N = None
|
|
265
|
+
self.pb = None
|
|
266
|
+
self.mu = None
|
|
267
|
+
self.R = None
|
|
268
|
+
self.invR = None
|
|
269
|
+
self.const = None
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _fit_mixture(data, init_K, final_K, est_kind, alpha, whiten, verbose):
|
|
273
|
+
"""Run EM order selection and return (opt_mixture, mdl_path).
|
|
274
|
+
|
|
275
|
+
Starts at init_K clusters, merges down one order at a time, and either
|
|
276
|
+
returns the model at final_K (fixed order) or the minimum-MDL model
|
|
277
|
+
(final_K == 0). mdl_path lists (K, MDL) for every order visited.
|
|
278
|
+
|
|
279
|
+
Args:
|
|
280
|
+
data: (N, M) 2D float array of observations.
|
|
281
|
+
init_K: number of clusters to start from.
|
|
282
|
+
final_K: fixed final order, or 0 to select by MDL.
|
|
283
|
+
est_kind: "full" or "diag".
|
|
284
|
+
alpha: covariance regularization in (0, 1].
|
|
285
|
+
whiten: decorrelate coordinates before clustering.
|
|
286
|
+
verbose: log progress if True.
|
|
287
|
+
|
|
288
|
+
Returns:
|
|
289
|
+
(opt_mixture, mdl_path) where opt_mixture is a MixtureObj and mdl_path
|
|
290
|
+
is a list of (K, MDL) tuples in ascending K.
|
|
291
|
+
"""
|
|
292
|
+
if whiten:
|
|
293
|
+
data, T, smean = decorrelate_and_normalize(data)
|
|
294
|
+
|
|
295
|
+
[N, M] = np.shape(data)
|
|
296
|
+
|
|
297
|
+
# Number of parameters per cluster.
|
|
298
|
+
if est_kind == 'full':
|
|
299
|
+
nparams_clust = 1 + M + 0.5 * M * (M + 1)
|
|
300
|
+
else:
|
|
301
|
+
nparams_clust = 1 + M + M
|
|
302
|
+
|
|
303
|
+
ndata_points = np.size(data)
|
|
304
|
+
|
|
305
|
+
# Cap the starting order to the amount of data.
|
|
306
|
+
max_params = (ndata_points + 1) / nparams_clust - 1
|
|
307
|
+
if init_K > (max_params / 2):
|
|
308
|
+
init_K = int(max_params / 2)
|
|
309
|
+
if verbose:
|
|
310
|
+
logger.warning("Too many clusters for the given data; init_K set to %d", init_K)
|
|
311
|
+
|
|
312
|
+
mtr = init_mixture(data, init_K, est_kind, alpha)
|
|
313
|
+
mtr = EM_iterate(mtr, data, est_kind, alpha)
|
|
314
|
+
if verbose:
|
|
315
|
+
logger.info("K: %d MDL: %g", mtr.K, mtr.rissanen)
|
|
316
|
+
|
|
317
|
+
mixture = [None] * (mtr.K - max(1, final_K) + 1)
|
|
318
|
+
mixture[mtr.K - max(1, final_K)] = copy.deepcopy(mtr)
|
|
319
|
+
while mtr.K > max(1, final_K):
|
|
320
|
+
mtr = MDL_reduce_order(mtr, False)
|
|
321
|
+
mtr = EM_iterate(mtr, data, est_kind, alpha)
|
|
322
|
+
if verbose:
|
|
323
|
+
logger.info("K: %d MDL: %g", mtr.K, mtr.rissanen)
|
|
324
|
+
mixture[mtr.K - max(1, final_K)] = copy.deepcopy(mtr)
|
|
325
|
+
|
|
326
|
+
if final_K > 0:
|
|
327
|
+
opt_mixture = mixture[0]
|
|
328
|
+
else:
|
|
329
|
+
min_riss = mixture[-1].rissanen
|
|
330
|
+
opt_l = len(mixture) - 1
|
|
331
|
+
for l in range(len(mixture) - 2, -1, -1):
|
|
332
|
+
if mixture[l].rissanen < min_riss:
|
|
333
|
+
min_riss = mixture[l].rissanen
|
|
334
|
+
opt_l = l
|
|
335
|
+
opt_mixture = copy.deepcopy(mixture[opt_l])
|
|
336
|
+
|
|
337
|
+
# MDL for every order visited, in ascending K.
|
|
338
|
+
mdl_path = [(m.K, m.rissanen) for m in mixture if m is not None]
|
|
339
|
+
|
|
340
|
+
if whiten:
|
|
341
|
+
opt_mixture = transform_back_to_original_coordinates(opt_mixture, T, smean)
|
|
342
|
+
|
|
343
|
+
return opt_mixture, mdl_path
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _class_log_likelihood(mixture, data):
|
|
347
|
+
"""Return per-point log density log p(x) as an (N, 1) array."""
|
|
348
|
+
[N, M] = np.shape(data)
|
|
349
|
+
pnk = np.zeros((N, mixture.K))
|
|
350
|
+
pb_mat = np.zeros((1, mixture.K))
|
|
351
|
+
|
|
352
|
+
for k in range(mixture.K):
|
|
353
|
+
cluster_obj = mixture.cluster[k]
|
|
354
|
+
Y1 = data - cluster_obj.mu.T
|
|
355
|
+
Y2 = -0.5 * Y1 @ cluster_obj.invR
|
|
356
|
+
pnk[:, k] = np.sum(Y1 * Y2, axis=1) + mixture.cluster[k].const
|
|
357
|
+
pb_mat[0, k] = cluster_obj.pb
|
|
358
|
+
|
|
359
|
+
llmax = np.expand_dims(np.max(pnk, axis=1), axis=1)
|
|
360
|
+
pnk = np.exp(pnk - llmax)
|
|
361
|
+
pnk = pnk * pb_mat
|
|
362
|
+
ss = np.expand_dims(np.sum(pnk, axis=1), axis=1)
|
|
363
|
+
ll = np.log(ss) + llmax
|
|
364
|
+
|
|
365
|
+
return ll
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def cluster_normalize(mixture):
|
|
369
|
+
"""Normalize cluster weights to sum to 1 and refresh invR and const.
|
|
370
|
+
|
|
371
|
+
Args:
|
|
372
|
+
mixture(class): a Gaussian mixture record.
|
|
373
|
+
|
|
374
|
+
Returns:
|
|
375
|
+
class object: the mixture with normalized weights and updated invR/const.
|
|
376
|
+
"""
|
|
377
|
+
cluster = mixture.cluster
|
|
378
|
+
|
|
379
|
+
s = 0
|
|
380
|
+
for k in range(mixture.K):
|
|
381
|
+
cluster_obj = cluster[k]
|
|
382
|
+
s = s + np.sum(cluster_obj.pb)
|
|
383
|
+
|
|
384
|
+
for k in range(mixture.K):
|
|
385
|
+
cluster_obj = cluster[k]
|
|
386
|
+
cluster_obj.pb = cluster_obj.pb / s
|
|
387
|
+
cluster_obj.invR = np.linalg.inv(cluster_obj.R)
|
|
388
|
+
cluster_obj.const = -(mixture.M * np.log(2 * np.pi) + np.log(np.linalg.det(cluster_obj.R))) / 2
|
|
389
|
+
cluster[k] = cluster_obj
|
|
390
|
+
mixture.cluster = cluster
|
|
391
|
+
|
|
392
|
+
return mixture
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def ridge_regression(R, est_kind, alpha, D_reg=None):
|
|
396
|
+
"""Regularize and constrain a class covariance matrix.
|
|
397
|
+
|
|
398
|
+
Args:
|
|
399
|
+
R(ndarray): the initial class covariance matrix
|
|
400
|
+
est_kind(str):
|
|
401
|
+
- est_kind = 'diag' constrains the class covariance matrices to be diagonal
|
|
402
|
+
- est_kind = 'full' allows the class covariance matrices to be full matrices
|
|
403
|
+
alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
|
|
404
|
+
matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
|
|
405
|
+
The default value is 0.1
|
|
406
|
+
D_reg(ndarray,optional): a diagonal matrix used as the regularization term in the class covariance matrix update
|
|
407
|
+
equation. The function will compute it from the given R if set to default
|
|
408
|
+
|
|
409
|
+
Returns:
|
|
410
|
+
ndarray: the regularized and constrained class covariance matrix
|
|
411
|
+
tuple/ndarray: (R, D_reg) or just R (if return_D_reg is false), where
|
|
412
|
+
- R(ndarray): the regularized and constrained class covariance matrix
|
|
413
|
+
- D_reg(ndarray): diagonal matrix used as the regularization term
|
|
414
|
+
"""
|
|
415
|
+
if est_kind == 'diag':
|
|
416
|
+
R = np.diag(np.diag(R))
|
|
417
|
+
|
|
418
|
+
if D_reg is None:
|
|
419
|
+
return_D_reg = True
|
|
420
|
+
D_reg = np.mean(np.diag(R)) * np.eye(R.shape[0])
|
|
421
|
+
else:
|
|
422
|
+
return_D_reg = False
|
|
423
|
+
|
|
424
|
+
# Ensure that the alpha of R is <= alpha
|
|
425
|
+
R = (1.0 - (alpha ** 2)) * R + (alpha ** 2) * D_reg
|
|
426
|
+
|
|
427
|
+
if return_D_reg:
|
|
428
|
+
return R, D_reg
|
|
429
|
+
else:
|
|
430
|
+
return R
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def init_mixture(data, K, est_kind, alpha):
|
|
434
|
+
"""Initialize a Gaussian mixture record of a given order.
|
|
435
|
+
|
|
436
|
+
Args:
|
|
437
|
+
data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
|
|
438
|
+
totally N observations
|
|
439
|
+
K(int): order of the mixture
|
|
440
|
+
est_kind(str):
|
|
441
|
+
- est_kind = 'diag' constrains the class covariance matrices to be diagonal
|
|
442
|
+
- est_kind = 'full' allows the class covariance matrices to be full matrices
|
|
443
|
+
alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
|
|
444
|
+
matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
|
|
445
|
+
The default value is 0.1
|
|
446
|
+
|
|
447
|
+
Returns:
|
|
448
|
+
class object: a structure containing the initial parameter values for the Gaussian mixture of a given order
|
|
449
|
+
"""
|
|
450
|
+
[N, M] = np.shape(data)
|
|
451
|
+
|
|
452
|
+
mixture = MixtureObj()
|
|
453
|
+
mixture.K = K
|
|
454
|
+
mixture.M = M
|
|
455
|
+
|
|
456
|
+
# Compute sample covariance for entire data set
|
|
457
|
+
R = (N - 1) * np.cov(data, rowvar=False) / N
|
|
458
|
+
|
|
459
|
+
# Regularize the covariance matrix and impose constrains
|
|
460
|
+
R, D_reg = ridge_regression(R, est_kind, alpha)
|
|
461
|
+
|
|
462
|
+
# Allocate and array of K clusters
|
|
463
|
+
cluster = [None] * K
|
|
464
|
+
|
|
465
|
+
# Initalize first element of cluster
|
|
466
|
+
cluster_obj = ClusterObj()
|
|
467
|
+
cluster_obj.N = 0
|
|
468
|
+
cluster_obj.pb = 1 / K
|
|
469
|
+
cluster_obj.mu = np.expand_dims(data[0, :], 1)
|
|
470
|
+
cluster_obj.R = R
|
|
471
|
+
cluster[0] = cluster_obj
|
|
472
|
+
|
|
473
|
+
# Initialize remaining clusters in array
|
|
474
|
+
if K > 1:
|
|
475
|
+
period = (N - 1) / (K - 1)
|
|
476
|
+
for k in range(1, K):
|
|
477
|
+
cluster_obj = ClusterObj()
|
|
478
|
+
cluster_obj.N = 0
|
|
479
|
+
cluster_obj.pb = 1 / K
|
|
480
|
+
cluster_obj.mu = np.expand_dims(data[int((k - 1) * period + 1), :], 1)
|
|
481
|
+
cluster_obj.R = R
|
|
482
|
+
cluster[k] = cluster_obj
|
|
483
|
+
|
|
484
|
+
mixture.cluster = cluster
|
|
485
|
+
mixture.D_reg = D_reg
|
|
486
|
+
mixture = cluster_normalize(mixture)
|
|
487
|
+
|
|
488
|
+
return mixture
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def E_step(mixture, data):
|
|
492
|
+
"""Perform the E-step: compute responsibilities pnk and the log-likelihood.
|
|
493
|
+
|
|
494
|
+
Args:
|
|
495
|
+
mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
|
|
496
|
+
data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
|
|
497
|
+
totally N observations
|
|
498
|
+
Returns:
|
|
499
|
+
tuple: (mixture, likelihood), where
|
|
500
|
+
- mixture(class): a structure containing the Gaussian mixture parameters for the same order with updated pnk
|
|
501
|
+
- likelihood(float): log ( prob(Y=y|theta) )
|
|
502
|
+
"""
|
|
503
|
+
[N, M] = np.shape(data)
|
|
504
|
+
pnk = np.zeros((N, mixture.K))
|
|
505
|
+
pb_mat = np.zeros((1, mixture.K))
|
|
506
|
+
|
|
507
|
+
for k in range(mixture.K):
|
|
508
|
+
cluster_obj = mixture.cluster[k]
|
|
509
|
+
Y1 = data - cluster_obj.mu.T
|
|
510
|
+
Y2 = -0.5 * Y1 @ cluster_obj.invR
|
|
511
|
+
pnk[:, k] = np.sum(Y1 * Y2, axis=1) + cluster_obj.const
|
|
512
|
+
pb_mat[0, k] = cluster_obj.pb
|
|
513
|
+
|
|
514
|
+
llmax = np.expand_dims(np.max(pnk, axis=1), axis=1)
|
|
515
|
+
pnk = np.exp(pnk - llmax)
|
|
516
|
+
pnk = pnk * pb_mat
|
|
517
|
+
ss = np.expand_dims(np.sum(pnk, axis=1), axis=1)
|
|
518
|
+
likelihood = np.sum(np.log(ss) + llmax)
|
|
519
|
+
pnk = pnk / ss
|
|
520
|
+
mixture.pnk = pnk
|
|
521
|
+
|
|
522
|
+
return mixture, likelihood
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def M_step(mixture, data, est_kind, alpha):
|
|
526
|
+
"""Perform the M-step: update each cluster's weight, mean, and covariance.
|
|
527
|
+
|
|
528
|
+
Args:
|
|
529
|
+
mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
|
|
530
|
+
data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
|
|
531
|
+
totally N observations
|
|
532
|
+
est_kind(str):
|
|
533
|
+
- est_kind = 'diag' constrains the class covariance matrices to be diagonal
|
|
534
|
+
- est_kind = 'full' allows the class covariance matrices to be full matrices
|
|
535
|
+
alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
|
|
536
|
+
matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
|
|
537
|
+
The default value is 0.1
|
|
538
|
+
|
|
539
|
+
Returns:
|
|
540
|
+
class object: a structure containing the parameters for a Gaussian mixture of the same order with updated
|
|
541
|
+
cluster parameters
|
|
542
|
+
"""
|
|
543
|
+
for k in range(mixture.K):
|
|
544
|
+
cluster_obj = mixture.cluster[k]
|
|
545
|
+
cluster_obj.N = np.sum(mixture.pnk[:, k])
|
|
546
|
+
cluster_obj.pb = cluster_obj.N
|
|
547
|
+
cluster_obj.mu = np.expand_dims((data.T @ mixture.pnk[:, k]) / cluster_obj.N, axis=1)
|
|
548
|
+
|
|
549
|
+
# Weighted covariance about the cluster mean
|
|
550
|
+
w = mixture.pnk[:, k]
|
|
551
|
+
Xc = data - cluster_obj.mu.T
|
|
552
|
+
R = (Xc.T * w) @ Xc / cluster_obj.N
|
|
553
|
+
|
|
554
|
+
# Regularize the covariance matrix and impose constrains
|
|
555
|
+
R = ridge_regression(R, est_kind, alpha, mixture.D_reg)
|
|
556
|
+
|
|
557
|
+
cluster_obj.R = R
|
|
558
|
+
mixture.cluster[k] = cluster_obj
|
|
559
|
+
|
|
560
|
+
mixture = cluster_normalize(mixture)
|
|
561
|
+
|
|
562
|
+
return mixture
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def EM_iterate(mixture, data, est_kind, alpha):
|
|
566
|
+
"""Run EM to convergence at a fixed order K.
|
|
567
|
+
|
|
568
|
+
Records the number of EM iterations on mixture.num_iterations. The counter
|
|
569
|
+
is bookkeeping only; the update equations and the convergence test are
|
|
570
|
+
unchanged.
|
|
571
|
+
|
|
572
|
+
Args:
|
|
573
|
+
mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
|
|
574
|
+
data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
|
|
575
|
+
totally N observations
|
|
576
|
+
est_kind(str):
|
|
577
|
+
- est_kind = 'diag' constrains the class covariance matrices to be diagonal
|
|
578
|
+
- est_kind = 'full' allows the class covariance matrices to be full matrices
|
|
579
|
+
alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
|
|
580
|
+
matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
|
|
581
|
+
The default value is 0.1
|
|
582
|
+
|
|
583
|
+
Returns:
|
|
584
|
+
class object: a structure containing the parameters for the converged Gaussian mixture of order K
|
|
585
|
+
"""
|
|
586
|
+
[N, M] = np.shape(data)
|
|
587
|
+
|
|
588
|
+
if est_kind == 'full':
|
|
589
|
+
Lc = 1 + M + 0.5 * M * (M + 1)
|
|
590
|
+
else:
|
|
591
|
+
Lc = 1 + M + M
|
|
592
|
+
|
|
593
|
+
epsilon = 0.01 * Lc * np.log(N * M)
|
|
594
|
+
[mixture, ll_new] = E_step(mixture, data)
|
|
595
|
+
|
|
596
|
+
n_iter = 0
|
|
597
|
+
while True:
|
|
598
|
+
ll_old = ll_new
|
|
599
|
+
mixture = M_step(mixture, data, est_kind, alpha)
|
|
600
|
+
[mixture, ll_new] = E_step(mixture, data)
|
|
601
|
+
n_iter += 1
|
|
602
|
+
if (ll_new - ll_old) <= epsilon:
|
|
603
|
+
break
|
|
604
|
+
|
|
605
|
+
mixture.rissanen = -ll_new + 0.5 * (mixture.K * Lc - 1) * np.log(N * M)
|
|
606
|
+
mixture.loglikelihood = ll_new
|
|
607
|
+
mixture.num_iterations = n_iter
|
|
608
|
+
|
|
609
|
+
return mixture
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
def add_cluster(cluster1, cluster2):
|
|
613
|
+
"""Combine two clusters into one.
|
|
614
|
+
|
|
615
|
+
Args:
|
|
616
|
+
cluster1(class): the first cluster
|
|
617
|
+
cluster2(class): the second cluster
|
|
618
|
+
|
|
619
|
+
Returns:
|
|
620
|
+
class object: the combined cluster
|
|
621
|
+
"""
|
|
622
|
+
wt1 = cluster1.N / (cluster1.N + cluster2.N)
|
|
623
|
+
wt2 = 1 - wt1
|
|
624
|
+
M = np.shape(cluster1.mu)[0]
|
|
625
|
+
|
|
626
|
+
cluster3 = ClusterObj()
|
|
627
|
+
cluster3.mu = wt1 * cluster1.mu + wt2 * cluster2.mu
|
|
628
|
+
cluster3.R = wt1 * (cluster1.R + (cluster3.mu - cluster1.mu) @ (cluster3.mu - cluster1.mu).T) \
|
|
629
|
+
+ wt2 * (cluster2.R + (cluster3.mu - cluster2.mu) @ (cluster3.mu - cluster2.mu).T)
|
|
630
|
+
cluster3.invR = np.linalg.inv(cluster3.R)
|
|
631
|
+
cluster3.pb = cluster1.pb + cluster2.pb
|
|
632
|
+
cluster3.N = cluster1.N + cluster2.N
|
|
633
|
+
cluster3.const = -(M * np.log(2 * np.pi) + np.log(np.linalg.det(cluster3.R))) / 2
|
|
634
|
+
|
|
635
|
+
return cluster3
|
|
636
|
+
|
|
637
|
+
|
|
638
|
+
def distance(cluster1, cluster2):
|
|
639
|
+
"""Return the merge distance between two clusters.
|
|
640
|
+
|
|
641
|
+
Args:
|
|
642
|
+
cluster1(class): the first cluster
|
|
643
|
+
cluster2(class): the second cluster
|
|
644
|
+
|
|
645
|
+
Returns:
|
|
646
|
+
float: distance between the two clusters
|
|
647
|
+
"""
|
|
648
|
+
cluster3 = add_cluster(cluster1, cluster2)
|
|
649
|
+
dist = cluster1.N * cluster1.const + cluster2.N * cluster2.const - cluster3.N * cluster3.const
|
|
650
|
+
|
|
651
|
+
return dist
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def MDL_reduce_order(mixture, verbose):
|
|
655
|
+
"""Reduce the order by one by merging the two closest clusters.
|
|
656
|
+
|
|
657
|
+
Args:
|
|
658
|
+
mixture(class): a structure containing the parameters for the converged Gaussian mixture of a given order K
|
|
659
|
+
verbose(bool): true/false, return clustering information if true
|
|
660
|
+
|
|
661
|
+
Returns:
|
|
662
|
+
class object: a structure containing the parameters for the converged Gaussian mixture of order (K-1)
|
|
663
|
+
"""
|
|
664
|
+
K = mixture.K
|
|
665
|
+
|
|
666
|
+
min_dist = np.inf
|
|
667
|
+
for k1 in range(K):
|
|
668
|
+
for k2 in range(k1 + 1, K):
|
|
669
|
+
dist = distance(mixture.cluster[k1], mixture.cluster[k2])
|
|
670
|
+
if (k1 == 0 and k2 == 1) or (dist < min_dist):
|
|
671
|
+
mink1 = k1
|
|
672
|
+
mink2 = k2
|
|
673
|
+
min_dist = dist
|
|
674
|
+
if verbose:
|
|
675
|
+
logger.info("combining cluster %d and %d", mink1, mink2)
|
|
676
|
+
|
|
677
|
+
mixture.cluster[mink1] = add_cluster(mixture.cluster[mink1], mixture.cluster[mink2])
|
|
678
|
+
mixture.cluster[mink2: (K - 1)] = mixture.cluster[(mink2 + 1): K]
|
|
679
|
+
mixture.cluster = mixture.cluster[:(K - 1)]
|
|
680
|
+
mixture.K = K - 1
|
|
681
|
+
mixture = cluster_normalize(mixture)
|
|
682
|
+
|
|
683
|
+
return mixture
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
def decorrelate_and_normalize(data):
|
|
687
|
+
"""Decorrelate and normalize data, returning the transform for inversion.
|
|
688
|
+
|
|
689
|
+
Args:
|
|
690
|
+
data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
|
|
691
|
+
totally N observations
|
|
692
|
+
|
|
693
|
+
Returns:
|
|
694
|
+
tuple: (data, T, smean), where
|
|
695
|
+
- data(ndarray): decorrelated and normalized observation vectors
|
|
696
|
+
- T(ndarray): transformation 2D array
|
|
697
|
+
- smean(ndarray): mean values
|
|
698
|
+
"""
|
|
699
|
+
# Decorrelate and normalize the data
|
|
700
|
+
smean = np.mean(data, axis=0)
|
|
701
|
+
scov = np.cov(data, rowvar=False)
|
|
702
|
+
D, E = np.linalg.eig(scov)
|
|
703
|
+
D = np.diag(D)
|
|
704
|
+
T = E @ np.linalg.inv(np.sqrt(D))
|
|
705
|
+
data = (data - (np.diag(smean) @ np.ones((np.shape(data)[1], np.shape(data)[0]))).T) @ T
|
|
706
|
+
|
|
707
|
+
return data, T, smean
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
def transform_back_to_original_coordinates(opt_mixture, T, smean):
|
|
711
|
+
"""Map mixture parameters from whitened coordinates back to the original ones.
|
|
712
|
+
|
|
713
|
+
Args:
|
|
714
|
+
opt_mixture(class): a structure representing the optimum Gaussian mixture parameters corresponding to
|
|
715
|
+
decorrelated coordinates
|
|
716
|
+
T(ndarray): transformation 2D array
|
|
717
|
+
smean(ndarray): mean values
|
|
718
|
+
|
|
719
|
+
Returns:
|
|
720
|
+
class object: a structure representing the optimum Gaussian mixture parameters corresponding to original
|
|
721
|
+
coordinates
|
|
722
|
+
"""
|
|
723
|
+
invT = np.linalg.inv(T)
|
|
724
|
+
# Transform the parameters back to original coordinates
|
|
725
|
+
for k in range(opt_mixture.K):
|
|
726
|
+
opt_mixture.cluster[k].mu = (opt_mixture.cluster[k].mu.T @ invT + smean).T
|
|
727
|
+
opt_mixture.cluster[k].R = invT.T @ opt_mixture.cluster[k].R @ invT
|
|
728
|
+
opt_mixture.cluster[k].invR = T @ opt_mixture.cluster[k].invR @ T.T
|
|
729
|
+
opt_mixture.cluster[k].const = opt_mixture.cluster[k].const - np.log(
|
|
730
|
+
np.linalg.det(invT.T @ invT)) / 2
|
|
731
|
+
|
|
732
|
+
return opt_mixture
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gmcluster
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: EM Gaussian-mixture clustering with automatic MDL order selection
|
|
5
|
+
Author: Charles A. Bouman, Mohammad Samin Nur Chowdhury
|
|
6
|
+
License: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/cabouman/gmcluster
|
|
8
|
+
Project-URL: Documentation, https://gmcluster.readthedocs.io
|
|
9
|
+
Project-URL: Repository, https://github.com/cabouman/gmcluster
|
|
10
|
+
Requires-Python: >=3.11
|
|
11
|
+
Description-Content-Type: text/x-rst
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: numpy
|
|
14
|
+
Requires-Dist: matplotlib
|
|
15
|
+
Provides-Extra: test
|
|
16
|
+
Requires-Dist: pytest; extra == "test"
|
|
17
|
+
Provides-Extra: docs
|
|
18
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
19
|
+
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
20
|
+
Requires-Dist: sphinx-design; extra == "docs"
|
|
21
|
+
Requires-Dist: sphinx-copybutton; extra == "docs"
|
|
22
|
+
Requires-Dist: sphinxcontrib-bibtex; extra == "docs"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
GMCluster
|
|
26
|
+
=========
|
|
27
|
+
|
|
28
|
+
GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
|
|
29
|
+
|
|
30
|
+
It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
|
|
31
|
+
|
|
32
|
+
Installing
|
|
33
|
+
----------
|
|
34
|
+
|
|
35
|
+
Install the latest release from PyPI::
|
|
36
|
+
|
|
37
|
+
pip install gmcluster
|
|
38
|
+
|
|
39
|
+
To install from source (for development), clone the repository and do an editable install::
|
|
40
|
+
|
|
41
|
+
git clone https://github.com/cabouman/gmcluster.git
|
|
42
|
+
cd gmcluster
|
|
43
|
+
pip install -e .
|
|
44
|
+
|
|
45
|
+
Quick Start
|
|
46
|
+
-----------
|
|
47
|
+
|
|
48
|
+
The package provides one class, ``GaussianMixture``. Fit it to your data, read the
|
|
49
|
+
estimated parameters, then classify points or draw new samples.
|
|
50
|
+
|
|
51
|
+
.. code-block:: python
|
|
52
|
+
|
|
53
|
+
import numpy as np
|
|
54
|
+
from gmcluster import GaussianMixture
|
|
55
|
+
|
|
56
|
+
X = np.random.default_rng(0).standard_normal((500, 2))
|
|
57
|
+
|
|
58
|
+
# Fit the mixture; "auto" selects the number of clusters by MDL.
|
|
59
|
+
gm = GaussianMixture(num_clusters="auto").fit(X)
|
|
60
|
+
|
|
61
|
+
print(gm.estimated_num_clusters) # number of clusters found
|
|
62
|
+
print(gm.estimated_weights) # shape (K,)
|
|
63
|
+
print(gm.estimated_means) # shape (K, M)
|
|
64
|
+
print(gm.estimated_covariances) # shape (K, M, M)
|
|
65
|
+
|
|
66
|
+
labels = gm.classify(X) # most-likely cluster per point, shape (N,)
|
|
67
|
+
new_points = gm.sample(100) # draw 100 samples from the fitted mixture
|
|
68
|
+
|
|
69
|
+
Running the demos
|
|
70
|
+
-----------------
|
|
71
|
+
|
|
72
|
+
Validate the installation by running a demo::
|
|
73
|
+
|
|
74
|
+
cd demo
|
|
75
|
+
python demo_1.py
|
|
76
|
+
|
|
77
|
+
Citation
|
|
78
|
+
--------
|
|
79
|
+
|
|
80
|
+
Please cite this software when you use it. The BibTeX entry is in
|
|
81
|
+
``docs/source/credits.rst`` and in the online documentation at
|
|
82
|
+
https://gmcluster.readthedocs.io/ .
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.rst
|
|
3
|
+
pyproject.toml
|
|
4
|
+
gmcluster/__init__.py
|
|
5
|
+
gmcluster/gmcluster.py
|
|
6
|
+
gmcluster.egg-info/PKG-INFO
|
|
7
|
+
gmcluster.egg-info/SOURCES.txt
|
|
8
|
+
gmcluster.egg-info/dependency_links.txt
|
|
9
|
+
gmcluster.egg-info/requires.txt
|
|
10
|
+
gmcluster.egg-info/top_level.txt
|
|
11
|
+
tests/test_api.py
|
|
12
|
+
tests/test_equivalence.py
|
|
13
|
+
tests/test_order_selection.py
|
|
14
|
+
tests/test_recover_params.py
|
|
15
|
+
tests/test_smoke.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
gmcluster
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=64"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "gmcluster"
|
|
7
|
+
description = "EM Gaussian-mixture clustering with automatic MDL order selection"
|
|
8
|
+
readme = "README.rst"
|
|
9
|
+
license = { text = "BSD-3-Clause" }
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
dynamic = ["version"]
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Charles A. Bouman" },
|
|
14
|
+
{ name = "Mohammad Samin Nur Chowdhury" },
|
|
15
|
+
]
|
|
16
|
+
dependencies = [
|
|
17
|
+
"numpy",
|
|
18
|
+
"matplotlib",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[project.optional-dependencies]
|
|
22
|
+
test = ["pytest"]
|
|
23
|
+
docs = ["sphinx", "sphinx-book-theme", "sphinx-design", "sphinx-copybutton", "sphinxcontrib-bibtex"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://github.com/cabouman/gmcluster"
|
|
27
|
+
Documentation = "https://gmcluster.readthedocs.io"
|
|
28
|
+
Repository = "https://github.com/cabouman/gmcluster"
|
|
29
|
+
|
|
30
|
+
[tool.setuptools]
|
|
31
|
+
packages = ["gmcluster"]
|
|
32
|
+
|
|
33
|
+
[tool.setuptools.dynamic]
|
|
34
|
+
version = { attr = "gmcluster.__version__" }
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Behavior of the GaussianMixture public surface. Small and fast.
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from gmcluster import GaussianMixture
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def make_data(seed=0, n=200):
|
|
10
|
+
rng = np.random.default_rng(seed)
|
|
11
|
+
centers = np.array([[4.0, 4.0], [-4.0, -4.0], [4.0, -4.0]])
|
|
12
|
+
return np.vstack([rng.standard_normal((n, 2)) + c for c in centers])
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def fitted_model():
|
|
16
|
+
return GaussianMixture(num_clusters="auto", max_clusters=6).fit(make_data())
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_posterior_rows_sum_to_one():
|
|
20
|
+
gm = fitted_model()
|
|
21
|
+
X = make_data(seed=1, n=50)
|
|
22
|
+
P = gm.posterior(X)
|
|
23
|
+
assert P.shape == (X.shape[0], gm.estimated_num_clusters)
|
|
24
|
+
assert np.allclose(P.sum(axis=1), 1.0)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_classify_is_argmax_of_posterior():
|
|
28
|
+
gm = fitted_model()
|
|
29
|
+
X = make_data(seed=1, n=50)
|
|
30
|
+
assert np.array_equal(gm.classify(X), np.argmax(gm.posterior(X), axis=1))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_log_likelihood_shape():
|
|
34
|
+
gm = fitted_model()
|
|
35
|
+
X = make_data(seed=1, n=50)
|
|
36
|
+
ll = gm.log_likelihood(X)
|
|
37
|
+
assert ll.shape == (X.shape[0],)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_sample_shapes_and_reproducibility():
|
|
41
|
+
gm = fitted_model()
|
|
42
|
+
M = gm.estimated_means.shape[1]
|
|
43
|
+
|
|
44
|
+
X = gm.sample(30, rng=np.random.default_rng(7))
|
|
45
|
+
assert X.shape == (30, M)
|
|
46
|
+
|
|
47
|
+
X2, labels = gm.sample(30, rng=np.random.default_rng(7), with_labels=True)
|
|
48
|
+
assert X2.shape == (30, M)
|
|
49
|
+
assert labels.shape == (30,)
|
|
50
|
+
# Same seed reproduces the same draw.
|
|
51
|
+
assert np.array_equal(X, X2)
|
|
52
|
+
assert np.all((labels >= 0) & (labels < gm.estimated_num_clusters))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_split_clusters():
|
|
56
|
+
gm = fitted_model()
|
|
57
|
+
parts = gm.split_clusters()
|
|
58
|
+
assert len(parts) == gm.estimated_num_clusters
|
|
59
|
+
X = make_data(seed=2, n=20)
|
|
60
|
+
for part in parts:
|
|
61
|
+
assert part.estimated_num_clusters == 1
|
|
62
|
+
assert np.isclose(part.estimated_weights.sum(), 1.0)
|
|
63
|
+
# Each split model is usable as its own density.
|
|
64
|
+
assert part.log_likelihood(X).shape == (X.shape[0],)
|
|
65
|
+
assert part.classify(X).shape == (X.shape[0],)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_estimate_shapes_and_weights():
|
|
69
|
+
gm = fitted_model()
|
|
70
|
+
K = gm.estimated_num_clusters
|
|
71
|
+
M = gm.estimated_means.shape[1]
|
|
72
|
+
assert gm.estimated_means.shape == (K, M)
|
|
73
|
+
assert gm.estimated_covariances.shape == (K, M, M)
|
|
74
|
+
assert gm.estimated_weights.shape == (K,)
|
|
75
|
+
assert np.isclose(gm.estimated_weights.sum(), 1.0)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_access_before_fit_raises():
|
|
79
|
+
gm = GaussianMixture()
|
|
80
|
+
for name in ["estimated_num_clusters", "estimated_weights", "estimated_means",
|
|
81
|
+
"estimated_covariances", "mdl", "mdl_path", "converged", "num_iterations"]:
|
|
82
|
+
with pytest.raises(RuntimeError):
|
|
83
|
+
getattr(gm, name)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def test_query_before_fit_raises():
|
|
87
|
+
gm = GaussianMixture()
|
|
88
|
+
with pytest.raises(RuntimeError):
|
|
89
|
+
gm.posterior(make_data(n=5))
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@pytest.mark.parametrize("kwargs", [
|
|
93
|
+
{"num_clusters": 0},
|
|
94
|
+
{"num_clusters": -3},
|
|
95
|
+
{"alpha": 2},
|
|
96
|
+
{"alpha": 0},
|
|
97
|
+
{"covariance_type": "bogus"},
|
|
98
|
+
{"max_clusters": 0},
|
|
99
|
+
])
|
|
100
|
+
def test_bad_constructor_args_raise(kwargs):
|
|
101
|
+
with pytest.raises((ValueError, TypeError)):
|
|
102
|
+
GaussianMixture(**kwargs)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_non_2d_input_raises():
|
|
106
|
+
gm = GaussianMixture(max_clusters=4)
|
|
107
|
+
with pytest.raises(ValueError):
|
|
108
|
+
gm.fit(np.zeros(10))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def test_repr():
|
|
112
|
+
gm = GaussianMixture()
|
|
113
|
+
assert "unfitted" in repr(gm)
|
|
114
|
+
gm = fitted_model()
|
|
115
|
+
assert "clusters=" in repr(gm) and "dims=" in repr(gm)
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# In-process equivalence: the GaussianMixture wrapper reproduces the private
|
|
2
|
+
# EM engine exactly. This proves the wrapper does not alter the math.
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from gmcluster import GaussianMixture
|
|
8
|
+
from gmcluster.gmcluster import _fit_mixture
|
|
9
|
+
|
|
10
|
+
ATOL = 1e-12
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def make_data(seed=0, n=300):
|
|
14
|
+
"""Seeded three-cluster 2D data."""
|
|
15
|
+
rng = np.random.default_rng(seed)
|
|
16
|
+
centers = np.array([[4.0, 4.0], [-4.0, -4.0], [4.0, -4.0]])
|
|
17
|
+
parts = [rng.standard_normal((n, 2)) + c for c in centers]
|
|
18
|
+
return np.vstack(parts)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def engine_params(mixture):
|
|
22
|
+
"""Pull (K, weights, means, covariances, mdl) out of an engine mixture."""
|
|
23
|
+
K = int(mixture.K)
|
|
24
|
+
weights = np.array([float(c.pb) for c in mixture.cluster])
|
|
25
|
+
means = np.array([c.mu.ravel() for c in mixture.cluster])
|
|
26
|
+
covs = np.array([np.asarray(c.R) for c in mixture.cluster])
|
|
27
|
+
return K, weights, means, covs, mixture.rissanen
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@pytest.mark.parametrize("covariance_type,est_kind", [("full", "full"), ("diagonal", "diag")])
|
|
31
|
+
@pytest.mark.parametrize("num_clusters", ["auto", 2])
|
|
32
|
+
@pytest.mark.parametrize("whiten", [False, True])
|
|
33
|
+
def test_wrapper_matches_engine(covariance_type, est_kind, num_clusters, whiten):
|
|
34
|
+
X = make_data()
|
|
35
|
+
max_clusters = 6
|
|
36
|
+
alpha = 0.1
|
|
37
|
+
|
|
38
|
+
# Wrapper.
|
|
39
|
+
gm = GaussianMixture(num_clusters=num_clusters, max_clusters=max_clusters,
|
|
40
|
+
covariance_type=covariance_type, alpha=alpha,
|
|
41
|
+
whiten=whiten, verbose=False).fit(X)
|
|
42
|
+
|
|
43
|
+
# Private engine with the same init_K / final_K mapping fit() uses.
|
|
44
|
+
if num_clusters == "auto":
|
|
45
|
+
init_K, final_K = max_clusters, 0
|
|
46
|
+
else:
|
|
47
|
+
final_K = int(num_clusters)
|
|
48
|
+
init_K = max(max_clusters, final_K)
|
|
49
|
+
mixture, mdl_path = _fit_mixture(X, init_K, final_K, est_kind, alpha, whiten, False)
|
|
50
|
+
|
|
51
|
+
K, weights, means, covs, mdl = engine_params(mixture)
|
|
52
|
+
|
|
53
|
+
assert gm.estimated_num_clusters == K
|
|
54
|
+
assert np.allclose(gm.estimated_weights, weights, atol=ATOL)
|
|
55
|
+
assert np.allclose(gm.estimated_means, means, atol=ATOL)
|
|
56
|
+
assert np.allclose(gm.estimated_covariances, covs, atol=ATOL)
|
|
57
|
+
assert np.allclose(gm.mdl, mdl, atol=ATOL)
|
|
58
|
+
assert gm.mdl_path == mdl_path
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# MDL order selection finds 3 clusters; a fixed order returns exactly that many.
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
from gmcluster import GaussianMixture
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def make_data(seed=0, n_per=300):
|
|
9
|
+
"""Three well-separated 2-D Gaussians."""
|
|
10
|
+
rng = np.random.default_rng(seed)
|
|
11
|
+
centers = np.array([[6.0, 6.0], [-6.0, -6.0], [6.0, -6.0]])
|
|
12
|
+
return np.vstack([rng.standard_normal((n_per, 2)) + c for c in centers])
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_auto_selects_three():
|
|
16
|
+
data = make_data(seed=0)
|
|
17
|
+
gm = GaussianMixture(num_clusters="auto", max_clusters=6).fit(data)
|
|
18
|
+
assert gm.estimated_num_clusters == 3
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_fixed_order_returns_two():
|
|
22
|
+
data = make_data(seed=0)
|
|
23
|
+
gm = GaussianMixture(num_clusters=2).fit(data)
|
|
24
|
+
assert gm.estimated_num_clusters == 2
|
|
25
|
+
assert gm.estimated_means.shape[0] == 2
|
|
26
|
+
assert gm.estimated_weights.shape[0] == 2
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# The fit recovers the true cluster means on well-separated data.
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
from gmcluster import GaussianMixture
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def make_mixture(seed=0, n_per=300):
|
|
9
|
+
"""Three well-separated 2-D Gaussians. Returns (data, true_means)."""
|
|
10
|
+
rng = np.random.default_rng(seed)
|
|
11
|
+
true_means = np.array([[6.0, 6.0], [-6.0, -6.0], [6.0, -6.0]])
|
|
12
|
+
data = np.vstack([rng.standard_normal((n_per, 2)) + m for m in true_means])
|
|
13
|
+
return data, true_means
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def match_nearest(true_means, estimated_means):
|
|
17
|
+
"""For each true mean, the distance to its nearest estimated mean."""
|
|
18
|
+
dists = []
|
|
19
|
+
for t in true_means:
|
|
20
|
+
d = np.linalg.norm(estimated_means - t, axis=1)
|
|
21
|
+
dists.append(d.min())
|
|
22
|
+
return np.array(dists)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_recover_means():
|
|
26
|
+
data, true_means = make_mixture(seed=0)
|
|
27
|
+
gm = GaussianMixture(num_clusters="auto", max_clusters=6).fit(data)
|
|
28
|
+
|
|
29
|
+
assert gm.estimated_num_clusters == 3
|
|
30
|
+
# Every true center has an estimated center close to it.
|
|
31
|
+
assert np.all(match_nearest(true_means, gm.estimated_means) < 0.5)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Smoke test: a tiny fit returns a well-formed model.
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from gmcluster import GaussianMixture
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_smoke_tiny_fit():
|
|
8
|
+
np.random.seed(0)
|
|
9
|
+
a = np.random.randn(200, 2) + np.array([5.0, 5.0])
|
|
10
|
+
b = np.random.randn(200, 2) + np.array([-5.0, -5.0])
|
|
11
|
+
data = np.vstack([a, b])
|
|
12
|
+
|
|
13
|
+
gm = GaussianMixture(num_clusters="auto", max_clusters=5, verbose=False).fit(data)
|
|
14
|
+
|
|
15
|
+
assert isinstance(gm.estimated_num_clusters, int)
|
|
16
|
+
assert gm.estimated_num_clusters >= 1
|
|
17
|
+
assert gm.estimated_weights.shape == (gm.estimated_num_clusters,)
|