gmcluster 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2022, Charles A Bouman
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ 1. Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ 2. Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ 3. Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.4
2
+ Name: gmcluster
3
+ Version: 0.3.0
4
+ Summary: EM Gaussian-mixture clustering with automatic MDL order selection
5
+ Author: Charles A. Bouman, Mohammad Samin Nur Chowdhury
6
+ License: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/cabouman/gmcluster
8
+ Project-URL: Documentation, https://gmcluster.readthedocs.io
9
+ Project-URL: Repository, https://github.com/cabouman/gmcluster
10
+ Requires-Python: >=3.11
11
+ Description-Content-Type: text/x-rst
12
+ License-File: LICENSE
13
+ Requires-Dist: numpy
14
+ Requires-Dist: matplotlib
15
+ Provides-Extra: test
16
+ Requires-Dist: pytest; extra == "test"
17
+ Provides-Extra: docs
18
+ Requires-Dist: sphinx; extra == "docs"
19
+ Requires-Dist: sphinx-book-theme; extra == "docs"
20
+ Requires-Dist: sphinx-design; extra == "docs"
21
+ Requires-Dist: sphinx-copybutton; extra == "docs"
22
+ Requires-Dist: sphinxcontrib-bibtex; extra == "docs"
23
+ Dynamic: license-file
24
+
25
+ GMCluster
26
+ =========
27
+
28
+ GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
29
+
30
+ It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
31
+
32
+ Installing
33
+ ----------
34
+
35
+ Install the latest release from PyPI::
36
+
37
+ pip install gmcluster
38
+
39
+ To install from source (for development), clone the repository and do an editable install::
40
+
41
+ git clone https://github.com/cabouman/gmcluster.git
42
+ cd gmcluster
43
+ pip install -e .
44
+
45
+ Quick Start
46
+ -----------
47
+
48
+ The package provides one class, ``GaussianMixture``. Fit it to your data, read the
49
+ estimated parameters, then classify points or draw new samples.
50
+
51
+ .. code-block:: python
52
+
53
+ import numpy as np
54
+ from gmcluster import GaussianMixture
55
+
56
+ X = np.random.default_rng(0).standard_normal((500, 2))
57
+
58
+ # Fit the mixture; "auto" selects the number of clusters by MDL.
59
+ gm = GaussianMixture(num_clusters="auto").fit(X)
60
+
61
+ print(gm.estimated_num_clusters) # number of clusters found
62
+ print(gm.estimated_weights) # shape (K,)
63
+ print(gm.estimated_means) # shape (K, M)
64
+ print(gm.estimated_covariances) # shape (K, M, M)
65
+
66
+ labels = gm.classify(X) # most-likely cluster per point, shape (N,)
67
+ new_points = gm.sample(100) # draw 100 samples from the fitted mixture
68
+
69
+ Running the demos
70
+ -----------------
71
+
72
+ Validate the installation by running a demo::
73
+
74
+ cd demo
75
+ python demo_1.py
76
+
77
+ Citation
78
+ --------
79
+
80
+ Please cite this software when you use it. The BibTeX entry is in
81
+ ``docs/source/credits.rst`` and in the online documentation at
82
+ https://gmcluster.readthedocs.io/ .
@@ -0,0 +1,58 @@
1
+ GMCluster
2
+ =========
3
+
4
+ GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
5
+
6
+ It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
7
+
8
+ Installing
9
+ ----------
10
+
11
+ Install the latest release from PyPI::
12
+
13
+ pip install gmcluster
14
+
15
+ To install from source (for development), clone the repository and do an editable install::
16
+
17
+ git clone https://github.com/cabouman/gmcluster.git
18
+ cd gmcluster
19
+ pip install -e .
20
+
21
+ Quick Start
22
+ -----------
23
+
24
+ The package provides one class, ``GaussianMixture``. Fit it to your data, read the
25
+ estimated parameters, then classify points or draw new samples.
26
+
27
+ .. code-block:: python
28
+
29
+ import numpy as np
30
+ from gmcluster import GaussianMixture
31
+
32
+ X = np.random.default_rng(0).standard_normal((500, 2))
33
+
34
+ # Fit the mixture; "auto" selects the number of clusters by MDL.
35
+ gm = GaussianMixture(num_clusters="auto").fit(X)
36
+
37
+ print(gm.estimated_num_clusters) # number of clusters found
38
+ print(gm.estimated_weights) # shape (K,)
39
+ print(gm.estimated_means) # shape (K, M)
40
+ print(gm.estimated_covariances) # shape (K, M, M)
41
+
42
+ labels = gm.classify(X) # most-likely cluster per point, shape (N,)
43
+ new_points = gm.sample(100) # draw 100 samples from the fitted mixture
44
+
45
+ Running the demos
46
+ -----------------
47
+
48
+ Validate the installation by running a demo::
49
+
50
+ cd demo
51
+ python demo_1.py
52
+
53
+ Citation
54
+ --------
55
+
56
+ Please cite this software when you use it. The BibTeX entry is in
57
+ ``docs/source/credits.rst`` and in the online documentation at
58
+ https://gmcluster.readthedocs.io/ .
@@ -0,0 +1,4 @@
1
+ __version__ = '0.3.0'
2
+ from .gmcluster import GaussianMixture
3
+
4
+ __all__ = ["GaussianMixture"]
@@ -0,0 +1,732 @@
1
+ # EM Clustering Library
2
+ # Copyright (C) 2022, Charles A Bouman.
3
+ # All rights reserved.
4
+
5
+ import copy
6
+ import logging
7
+
8
+ import numpy as np
9
+
10
+ logger = logging.getLogger("gmcluster")
11
+
12
+ # Names guarded before fit. Reading any of these on an unfitted model raises.
13
+ _ESTIMATE_NAMES = (
14
+ "estimated_num_clusters",
15
+ "estimated_weights",
16
+ "estimated_means",
17
+ "estimated_covariances",
18
+ "mdl",
19
+ "mdl_path",
20
+ "converged",
21
+ "num_iterations",
22
+ )
23
+
24
+
25
+ class GaussianMixture:
26
+ """Gaussian mixture model fit by EM with MDL order selection.
27
+
28
+ The constructor holds settings. fit(X) runs EM and stores the results in
29
+ the estimated_* attributes and the diagnostics (mdl, mdl_path, converged,
30
+ num_iterations). Reading any of those before fit raises.
31
+ """
32
+
33
+ def __init__(self, num_clusters="auto", max_clusters=20, covariance_type="full",
34
+ alpha=0.1, whiten=False, verbose=False):
35
+ """Store settings after validating them.
36
+
37
+ Args:
38
+ num_clusters: "auto" to select the order by MDL, or a positive int to fix it.
39
+ max_clusters: positive int, the ceiling for the "auto" search.
40
+ covariance_type: "full" or "diagonal".
41
+ alpha: covariance regularization, 0 < alpha <= 1 (1 spherical, ->0 elliptical).
42
+ whiten: decorrelate coordinates before clustering.
43
+ verbose: report progress through the logging module.
44
+ """
45
+ # num_clusters: "auto" or a positive int (bool is not a count).
46
+ if isinstance(num_clusters, str):
47
+ if num_clusters != "auto":
48
+ raise ValueError('num_clusters must be "auto" or a positive int')
49
+ elif isinstance(num_clusters, bool) or not isinstance(num_clusters, (int, np.integer)):
50
+ raise TypeError('num_clusters must be "auto" or a positive int')
51
+ elif num_clusters <= 0:
52
+ raise ValueError("num_clusters must be a positive int")
53
+
54
+ # max_clusters: positive int.
55
+ if isinstance(max_clusters, bool) or not isinstance(max_clusters, (int, np.integer)):
56
+ raise TypeError("max_clusters must be a positive int")
57
+ if max_clusters <= 0:
58
+ raise ValueError("max_clusters must be a positive int")
59
+
60
+ # covariance_type: "full" or "diagonal" (mapped to the internal "diag").
61
+ if covariance_type == "full":
62
+ est_kind = "full"
63
+ elif covariance_type == "diagonal":
64
+ est_kind = "diag"
65
+ else:
66
+ raise ValueError('covariance_type must be "full" or "diagonal"')
67
+
68
+ # alpha: 0 < alpha <= 1.
69
+ if isinstance(alpha, bool) or not isinstance(alpha, (int, float, np.integer, np.floating)):
70
+ raise TypeError("alpha must be a number in (0, 1]")
71
+ if not (0 < alpha <= 1):
72
+ raise ValueError("alpha must satisfy 0 < alpha <= 1")
73
+
74
+ if not isinstance(whiten, bool):
75
+ raise TypeError("whiten must be a bool")
76
+ if not isinstance(verbose, bool):
77
+ raise TypeError("verbose must be a bool")
78
+
79
+ self.num_clusters = num_clusters
80
+ self.max_clusters = int(max_clusters)
81
+ self.covariance_type = covariance_type
82
+ self.alpha = float(alpha)
83
+ self.whiten = whiten
84
+ self.verbose = verbose
85
+
86
+ # Internal estimator kind ("full" or "diag") and the fitted engine mixture.
87
+ self._est_kind = est_kind
88
+ self._mixture = None
89
+ self._fitted = False
90
+
91
+ def __getattr__(self, name):
92
+ """Raise a clear error when an estimate is read before fit."""
93
+ # __getattr__ runs only when normal lookup fails, i.e. before fit sets these.
94
+ if name in _ESTIMATE_NAMES:
95
+ raise RuntimeError("GaussianMixture is not fitted; call fit(X) first")
96
+ raise AttributeError(name)
97
+
98
+ def fit(self, X):
99
+ """Fit the mixture to X by EM and store the estimates. Returns self.
100
+
101
+ Args:
102
+ X: (num_points, num_features) 2D float array of observations.
103
+ """
104
+ X = _check_data(X)
105
+
106
+ if self.num_clusters == "auto":
107
+ init_K = self.max_clusters
108
+ final_K = 0
109
+ else:
110
+ final_K = int(self.num_clusters)
111
+ init_K = max(self.max_clusters, final_K)
112
+
113
+ mixture, mdl_path = _fit_mixture(X, init_K, final_K, self._est_kind,
114
+ self.alpha, self.whiten, self.verbose)
115
+
116
+ self._populate(mixture, mdl_path)
117
+ return self
118
+
119
+ def _populate(self, mixture, mdl_path):
120
+ """Set the estimated_* attributes and diagnostics from an engine mixture."""
121
+ clusters = mixture.cluster
122
+ self._mixture = mixture
123
+ self.estimated_num_clusters = int(mixture.K)
124
+ self.estimated_weights = np.array([float(c.pb) for c in clusters])
125
+ self.estimated_means = np.array([c.mu.ravel() for c in clusters])
126
+ self.estimated_covariances = np.array([np.asarray(c.R) for c in clusters])
127
+ self.mdl = mixture.rissanen
128
+ self.mdl_path = mdl_path
129
+ self.converged = True
130
+ self.num_iterations = getattr(mixture, "num_iterations", None)
131
+ self._fitted = True
132
+
133
+ def _require_fitted(self):
134
+ if not self._fitted:
135
+ raise RuntimeError("GaussianMixture is not fitted; call fit(X) first")
136
+
137
+ def posterior(self, X):
138
+ """Return P(cluster | x), shape (N, K), rows summing to 1."""
139
+ self._require_fitted()
140
+ X = _check_data(X, n_features=self._mixture.M)
141
+ _, _ = E_step(self._mixture, X)
142
+ return np.array(self._mixture.pnk)
143
+
144
+ def classify(self, X):
145
+ """Return the most-likely cluster index per point, shape (N,)."""
146
+ return np.argmax(self.posterior(X), axis=1)
147
+
148
+ def log_likelihood(self, X):
149
+ """Return the per-point log density log p(x), shape (N,)."""
150
+ self._require_fitted()
151
+ X = _check_data(X, n_features=self._mixture.M)
152
+ return _class_log_likelihood(self._mixture, X).ravel()
153
+
154
+ def sample(self, num_samples=1, rng=None, with_labels=False):
155
+ """Draw samples from the fitted mixture.
156
+
157
+ Args:
158
+ num_samples: number of samples to draw.
159
+ rng: a numpy Generator or seed for reproducibility.
160
+ with_labels: also return the component index of each sample.
161
+
162
+ Returns:
163
+ X of shape (num_samples, M), or (X, labels) with labels of shape
164
+ (num_samples,) if with_labels is True.
165
+ """
166
+ self._require_fitted()
167
+ rng = np.random.default_rng(rng)
168
+ weights = self.estimated_weights
169
+ means = self.estimated_means
170
+ covs = self.estimated_covariances
171
+ M = means.shape[1]
172
+
173
+ labels = rng.choice(len(weights), size=num_samples, p=weights)
174
+ samples = np.empty((num_samples, M))
175
+
176
+ # Draw each component's points from its Gaussian via a symmetric-eigen factor.
177
+ for k in range(len(weights)):
178
+ idx = np.nonzero(labels == k)[0]
179
+ if idx.size == 0:
180
+ continue
181
+ eigvals, eigvecs = np.linalg.eigh(covs[k])
182
+ eigvals = np.clip(eigvals, 0.0, None)
183
+ factor = eigvecs * np.sqrt(eigvals)
184
+ z = rng.standard_normal((idx.size, M))
185
+ samples[idx] = means[k] + z @ factor.T
186
+
187
+ if with_labels:
188
+ return samples, labels
189
+ return samples
190
+
191
+ def split_clusters(self):
192
+ """Return a list of single-cluster GaussianMixture models, one per component.
193
+
194
+ Each returned model is fitted with one cluster (weight 1) and is usable
195
+ with classify, posterior, log_likelihood, and sample. Kept for use
196
+ alongside other segmentation packages.
197
+ """
198
+ self._require_fitted()
199
+ parts = []
200
+ for k in range(self._mixture.K):
201
+ single = MixtureObj()
202
+ single.K = 1
203
+ single.M = self._mixture.M
204
+ single.cluster = [copy.deepcopy(self._mixture.cluster[k])]
205
+ single.D_reg = self._mixture.D_reg
206
+ # Renormalize so the single component is a proper order-1 density (weight 1).
207
+ single = cluster_normalize(single)
208
+ single.rissanen = None
209
+ single.loglikelihood = None
210
+ single.num_iterations = None
211
+
212
+ child = GaussianMixture(num_clusters=1, max_clusters=self.max_clusters,
213
+ covariance_type=self.covariance_type, alpha=self.alpha,
214
+ whiten=self.whiten, verbose=self.verbose)
215
+ child._populate(single, mdl_path=[(1, None)])
216
+ child.mdl = None
217
+ child.converged = True
218
+ parts.append(child)
219
+ return parts
220
+
221
+ def __repr__(self):
222
+ if self._fitted:
223
+ return "GaussianMixture(clusters={}, dims={}, mdl={})".format(
224
+ self.estimated_num_clusters, self._mixture.M, self.mdl)
225
+ return "GaussianMixture(num_clusters={!r}, covariance_type={!r}, unfitted)".format(
226
+ self.num_clusters, self.covariance_type)
227
+
228
+
229
+ def _check_data(X, n_features=None):
230
+ """Validate X as a 2D float array and return it. Raise on bad input."""
231
+ X = np.asarray(X)
232
+ if X.ndim != 2:
233
+ raise ValueError("X must be a 2D array of shape (num_points, num_features)")
234
+ if not np.issubdtype(X.dtype, np.number):
235
+ raise ValueError("X must be a numeric array")
236
+ if X.dtype != float:
237
+ X = X.astype(float)
238
+ if n_features is not None and X.shape[1] != n_features:
239
+ raise ValueError("X has {} features; the model was fit on {}".format(
240
+ X.shape[1], n_features))
241
+ return X
242
+
243
+
244
+ class MixtureObj:
245
+ """Bag of parameters for a Gaussian mixture (the engine's mixture record)."""
246
+
247
+ def __init__(self):
248
+ """Initialize the fields to None."""
249
+ self.K = None
250
+ self.M = None
251
+ self.cluster = None
252
+ self.rissanen = None
253
+ self.loglikelihood = None
254
+ self.pnk = None
255
+ self.D_reg = None
256
+ self.num_iterations = None
257
+
258
+
259
+ class ClusterObj:
260
+ """Bag of parameters for one cluster (the engine's cluster record)."""
261
+
262
+ def __init__(self):
263
+ """Initialize the fields to None."""
264
+ self.N = None
265
+ self.pb = None
266
+ self.mu = None
267
+ self.R = None
268
+ self.invR = None
269
+ self.const = None
270
+
271
+
272
+ def _fit_mixture(data, init_K, final_K, est_kind, alpha, whiten, verbose):
273
+ """Run EM order selection and return (opt_mixture, mdl_path).
274
+
275
+ Starts at init_K clusters, merges down one order at a time, and either
276
+ returns the model at final_K (fixed order) or the minimum-MDL model
277
+ (final_K == 0). mdl_path lists (K, MDL) for every order visited.
278
+
279
+ Args:
280
+ data: (N, M) 2D float array of observations.
281
+ init_K: number of clusters to start from.
282
+ final_K: fixed final order, or 0 to select by MDL.
283
+ est_kind: "full" or "diag".
284
+ alpha: covariance regularization in (0, 1].
285
+ whiten: decorrelate coordinates before clustering.
286
+ verbose: log progress if True.
287
+
288
+ Returns:
289
+ (opt_mixture, mdl_path) where opt_mixture is a MixtureObj and mdl_path
290
+ is a list of (K, MDL) tuples in ascending K.
291
+ """
292
+ if whiten:
293
+ data, T, smean = decorrelate_and_normalize(data)
294
+
295
+ [N, M] = np.shape(data)
296
+
297
+ # Number of parameters per cluster.
298
+ if est_kind == 'full':
299
+ nparams_clust = 1 + M + 0.5 * M * (M + 1)
300
+ else:
301
+ nparams_clust = 1 + M + M
302
+
303
+ ndata_points = np.size(data)
304
+
305
+ # Cap the starting order to the amount of data.
306
+ max_params = (ndata_points + 1) / nparams_clust - 1
307
+ if init_K > (max_params / 2):
308
+ init_K = int(max_params / 2)
309
+ if verbose:
310
+ logger.warning("Too many clusters for the given data; init_K set to %d", init_K)
311
+
312
+ mtr = init_mixture(data, init_K, est_kind, alpha)
313
+ mtr = EM_iterate(mtr, data, est_kind, alpha)
314
+ if verbose:
315
+ logger.info("K: %d MDL: %g", mtr.K, mtr.rissanen)
316
+
317
+ mixture = [None] * (mtr.K - max(1, final_K) + 1)
318
+ mixture[mtr.K - max(1, final_K)] = copy.deepcopy(mtr)
319
+ while mtr.K > max(1, final_K):
320
+ mtr = MDL_reduce_order(mtr, False)
321
+ mtr = EM_iterate(mtr, data, est_kind, alpha)
322
+ if verbose:
323
+ logger.info("K: %d MDL: %g", mtr.K, mtr.rissanen)
324
+ mixture[mtr.K - max(1, final_K)] = copy.deepcopy(mtr)
325
+
326
+ if final_K > 0:
327
+ opt_mixture = mixture[0]
328
+ else:
329
+ min_riss = mixture[-1].rissanen
330
+ opt_l = len(mixture) - 1
331
+ for l in range(len(mixture) - 2, -1, -1):
332
+ if mixture[l].rissanen < min_riss:
333
+ min_riss = mixture[l].rissanen
334
+ opt_l = l
335
+ opt_mixture = copy.deepcopy(mixture[opt_l])
336
+
337
+ # MDL for every order visited, in ascending K.
338
+ mdl_path = [(m.K, m.rissanen) for m in mixture if m is not None]
339
+
340
+ if whiten:
341
+ opt_mixture = transform_back_to_original_coordinates(opt_mixture, T, smean)
342
+
343
+ return opt_mixture, mdl_path
344
+
345
+
346
+ def _class_log_likelihood(mixture, data):
347
+ """Return per-point log density log p(x) as an (N, 1) array."""
348
+ [N, M] = np.shape(data)
349
+ pnk = np.zeros((N, mixture.K))
350
+ pb_mat = np.zeros((1, mixture.K))
351
+
352
+ for k in range(mixture.K):
353
+ cluster_obj = mixture.cluster[k]
354
+ Y1 = data - cluster_obj.mu.T
355
+ Y2 = -0.5 * Y1 @ cluster_obj.invR
356
+ pnk[:, k] = np.sum(Y1 * Y2, axis=1) + mixture.cluster[k].const
357
+ pb_mat[0, k] = cluster_obj.pb
358
+
359
+ llmax = np.expand_dims(np.max(pnk, axis=1), axis=1)
360
+ pnk = np.exp(pnk - llmax)
361
+ pnk = pnk * pb_mat
362
+ ss = np.expand_dims(np.sum(pnk, axis=1), axis=1)
363
+ ll = np.log(ss) + llmax
364
+
365
+ return ll
366
+
367
+
368
+ def cluster_normalize(mixture):
369
+ """Normalize cluster weights to sum to 1 and refresh invR and const.
370
+
371
+ Args:
372
+ mixture(class): a Gaussian mixture record.
373
+
374
+ Returns:
375
+ class object: the mixture with normalized weights and updated invR/const.
376
+ """
377
+ cluster = mixture.cluster
378
+
379
+ s = 0
380
+ for k in range(mixture.K):
381
+ cluster_obj = cluster[k]
382
+ s = s + np.sum(cluster_obj.pb)
383
+
384
+ for k in range(mixture.K):
385
+ cluster_obj = cluster[k]
386
+ cluster_obj.pb = cluster_obj.pb / s
387
+ cluster_obj.invR = np.linalg.inv(cluster_obj.R)
388
+ cluster_obj.const = -(mixture.M * np.log(2 * np.pi) + np.log(np.linalg.det(cluster_obj.R))) / 2
389
+ cluster[k] = cluster_obj
390
+ mixture.cluster = cluster
391
+
392
+ return mixture
393
+
394
+
395
+ def ridge_regression(R, est_kind, alpha, D_reg=None):
396
+ """Regularize and constrain a class covariance matrix.
397
+
398
+ Args:
399
+ R(ndarray): the initial class covariance matrix
400
+ est_kind(str):
401
+ - est_kind = 'diag' constrains the class covariance matrices to be diagonal
402
+ - est_kind = 'full' allows the class covariance matrices to be full matrices
403
+ alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
404
+ matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
405
+ The default value is 0.1
406
+ D_reg(ndarray,optional): a diagonal matrix used as the regularization term in the class covariance matrix update
407
+ equation. The function will compute it from the given R if set to default
408
+
409
+ Returns:
410
+ ndarray: the regularized and constrained class covariance matrix
411
+ tuple/ndarray: (R, D_reg) or just R (if return_D_reg is false), where
412
+ - R(ndarray): the regularized and constrained class covariance matrix
413
+ - D_reg(ndarray): diagonal matrix used as the regularization term
414
+ """
415
+ if est_kind == 'diag':
416
+ R = np.diag(np.diag(R))
417
+
418
+ if D_reg is None:
419
+ return_D_reg = True
420
+ D_reg = np.mean(np.diag(R)) * np.eye(R.shape[0])
421
+ else:
422
+ return_D_reg = False
423
+
424
+ # Ensure that the alpha of R is <= alpha
425
+ R = (1.0 - (alpha ** 2)) * R + (alpha ** 2) * D_reg
426
+
427
+ if return_D_reg:
428
+ return R, D_reg
429
+ else:
430
+ return R
431
+
432
+
433
+ def init_mixture(data, K, est_kind, alpha):
434
+ """Initialize a Gaussian mixture record of a given order.
435
+
436
+ Args:
437
+ data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
438
+ totally N observations
439
+ K(int): order of the mixture
440
+ est_kind(str):
441
+ - est_kind = 'diag' constrains the class covariance matrices to be diagonal
442
+ - est_kind = 'full' allows the class covariance matrices to be full matrices
443
+ alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
444
+ matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
445
+ The default value is 0.1
446
+
447
+ Returns:
448
+ class object: a structure containing the initial parameter values for the Gaussian mixture of a given order
449
+ """
450
+ [N, M] = np.shape(data)
451
+
452
+ mixture = MixtureObj()
453
+ mixture.K = K
454
+ mixture.M = M
455
+
456
+ # Compute sample covariance for entire data set
457
+ R = (N - 1) * np.cov(data, rowvar=False) / N
458
+
459
+ # Regularize the covariance matrix and impose constrains
460
+ R, D_reg = ridge_regression(R, est_kind, alpha)
461
+
462
+ # Allocate and array of K clusters
463
+ cluster = [None] * K
464
+
465
+ # Initalize first element of cluster
466
+ cluster_obj = ClusterObj()
467
+ cluster_obj.N = 0
468
+ cluster_obj.pb = 1 / K
469
+ cluster_obj.mu = np.expand_dims(data[0, :], 1)
470
+ cluster_obj.R = R
471
+ cluster[0] = cluster_obj
472
+
473
+ # Initialize remaining clusters in array
474
+ if K > 1:
475
+ period = (N - 1) / (K - 1)
476
+ for k in range(1, K):
477
+ cluster_obj = ClusterObj()
478
+ cluster_obj.N = 0
479
+ cluster_obj.pb = 1 / K
480
+ cluster_obj.mu = np.expand_dims(data[int((k - 1) * period + 1), :], 1)
481
+ cluster_obj.R = R
482
+ cluster[k] = cluster_obj
483
+
484
+ mixture.cluster = cluster
485
+ mixture.D_reg = D_reg
486
+ mixture = cluster_normalize(mixture)
487
+
488
+ return mixture
489
+
490
+
491
+ def E_step(mixture, data):
492
+ """Perform the E-step: compute responsibilities pnk and the log-likelihood.
493
+
494
+ Args:
495
+ mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
496
+ data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
497
+ totally N observations
498
+ Returns:
499
+ tuple: (mixture, likelihood), where
500
+ - mixture(class): a structure containing the Gaussian mixture parameters for the same order with updated pnk
501
+ - likelihood(float): log ( prob(Y=y|theta) )
502
+ """
503
+ [N, M] = np.shape(data)
504
+ pnk = np.zeros((N, mixture.K))
505
+ pb_mat = np.zeros((1, mixture.K))
506
+
507
+ for k in range(mixture.K):
508
+ cluster_obj = mixture.cluster[k]
509
+ Y1 = data - cluster_obj.mu.T
510
+ Y2 = -0.5 * Y1 @ cluster_obj.invR
511
+ pnk[:, k] = np.sum(Y1 * Y2, axis=1) + cluster_obj.const
512
+ pb_mat[0, k] = cluster_obj.pb
513
+
514
+ llmax = np.expand_dims(np.max(pnk, axis=1), axis=1)
515
+ pnk = np.exp(pnk - llmax)
516
+ pnk = pnk * pb_mat
517
+ ss = np.expand_dims(np.sum(pnk, axis=1), axis=1)
518
+ likelihood = np.sum(np.log(ss) + llmax)
519
+ pnk = pnk / ss
520
+ mixture.pnk = pnk
521
+
522
+ return mixture, likelihood
523
+
524
+
525
+ def M_step(mixture, data, est_kind, alpha):
526
+ """Perform the M-step: update each cluster's weight, mean, and covariance.
527
+
528
+ Args:
529
+ mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
530
+ data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
531
+ totally N observations
532
+ est_kind(str):
533
+ - est_kind = 'diag' constrains the class covariance matrices to be diagonal
534
+ - est_kind = 'full' allows the class covariance matrices to be full matrices
535
+ alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
536
+ matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
537
+ The default value is 0.1
538
+
539
+ Returns:
540
+ class object: a structure containing the parameters for a Gaussian mixture of the same order with updated
541
+ cluster parameters
542
+ """
543
+ for k in range(mixture.K):
544
+ cluster_obj = mixture.cluster[k]
545
+ cluster_obj.N = np.sum(mixture.pnk[:, k])
546
+ cluster_obj.pb = cluster_obj.N
547
+ cluster_obj.mu = np.expand_dims((data.T @ mixture.pnk[:, k]) / cluster_obj.N, axis=1)
548
+
549
+ # Weighted covariance about the cluster mean
550
+ w = mixture.pnk[:, k]
551
+ Xc = data - cluster_obj.mu.T
552
+ R = (Xc.T * w) @ Xc / cluster_obj.N
553
+
554
+ # Regularize the covariance matrix and impose constrains
555
+ R = ridge_regression(R, est_kind, alpha, mixture.D_reg)
556
+
557
+ cluster_obj.R = R
558
+ mixture.cluster[k] = cluster_obj
559
+
560
+ mixture = cluster_normalize(mixture)
561
+
562
+ return mixture
563
+
564
+
565
+ def EM_iterate(mixture, data, est_kind, alpha):
566
+ """Run EM to convergence at a fixed order K.
567
+
568
+ Records the number of EM iterations on mixture.num_iterations. The counter
569
+ is bookkeeping only; the update equations and the convergence test are
570
+ unchanged.
571
+
572
+ Args:
573
+ mixture(class): a structure representing the parameters for a Gaussian mixture of a given order
574
+ data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
575
+ totally N observations
576
+ est_kind(str):
577
+ - est_kind = 'diag' constrains the class covariance matrices to be diagonal
578
+ - est_kind = 'full' allows the class covariance matrices to be full matrices
579
+ alpha(float): a constant (0 < alpha <= 1) that controls the shape of the cluster by regularizing the covariance
580
+ matrices. alpha = 1 gives the cluster a spherical shape and alpha = 0 gives the cluster an elliptical shape.
581
+ The default value is 0.1
582
+
583
+ Returns:
584
+ class object: a structure containing the parameters for the converged Gaussian mixture of order K
585
+ """
586
+ [N, M] = np.shape(data)
587
+
588
+ if est_kind == 'full':
589
+ Lc = 1 + M + 0.5 * M * (M + 1)
590
+ else:
591
+ Lc = 1 + M + M
592
+
593
+ epsilon = 0.01 * Lc * np.log(N * M)
594
+ [mixture, ll_new] = E_step(mixture, data)
595
+
596
+ n_iter = 0
597
+ while True:
598
+ ll_old = ll_new
599
+ mixture = M_step(mixture, data, est_kind, alpha)
600
+ [mixture, ll_new] = E_step(mixture, data)
601
+ n_iter += 1
602
+ if (ll_new - ll_old) <= epsilon:
603
+ break
604
+
605
+ mixture.rissanen = -ll_new + 0.5 * (mixture.K * Lc - 1) * np.log(N * M)
606
+ mixture.loglikelihood = ll_new
607
+ mixture.num_iterations = n_iter
608
+
609
+ return mixture
610
+
611
+
612
+ def add_cluster(cluster1, cluster2):
613
+ """Combine two clusters into one.
614
+
615
+ Args:
616
+ cluster1(class): the first cluster
617
+ cluster2(class): the second cluster
618
+
619
+ Returns:
620
+ class object: the combined cluster
621
+ """
622
+ wt1 = cluster1.N / (cluster1.N + cluster2.N)
623
+ wt2 = 1 - wt1
624
+ M = np.shape(cluster1.mu)[0]
625
+
626
+ cluster3 = ClusterObj()
627
+ cluster3.mu = wt1 * cluster1.mu + wt2 * cluster2.mu
628
+ cluster3.R = wt1 * (cluster1.R + (cluster3.mu - cluster1.mu) @ (cluster3.mu - cluster1.mu).T) \
629
+ + wt2 * (cluster2.R + (cluster3.mu - cluster2.mu) @ (cluster3.mu - cluster2.mu).T)
630
+ cluster3.invR = np.linalg.inv(cluster3.R)
631
+ cluster3.pb = cluster1.pb + cluster2.pb
632
+ cluster3.N = cluster1.N + cluster2.N
633
+ cluster3.const = -(M * np.log(2 * np.pi) + np.log(np.linalg.det(cluster3.R))) / 2
634
+
635
+ return cluster3
636
+
637
+
638
+ def distance(cluster1, cluster2):
639
+ """Return the merge distance between two clusters.
640
+
641
+ Args:
642
+ cluster1(class): the first cluster
643
+ cluster2(class): the second cluster
644
+
645
+ Returns:
646
+ float: distance between the two clusters
647
+ """
648
+ cluster3 = add_cluster(cluster1, cluster2)
649
+ dist = cluster1.N * cluster1.const + cluster2.N * cluster2.const - cluster3.N * cluster3.const
650
+
651
+ return dist
652
+
653
+
654
+ def MDL_reduce_order(mixture, verbose):
655
+ """Reduce the order by one by merging the two closest clusters.
656
+
657
+ Args:
658
+ mixture(class): a structure containing the parameters for the converged Gaussian mixture of a given order K
659
+ verbose(bool): true/false, return clustering information if true
660
+
661
+ Returns:
662
+ class object: a structure containing the parameters for the converged Gaussian mixture of order (K-1)
663
+ """
664
+ K = mixture.K
665
+
666
+ min_dist = np.inf
667
+ for k1 in range(K):
668
+ for k2 in range(k1 + 1, K):
669
+ dist = distance(mixture.cluster[k1], mixture.cluster[k2])
670
+ if (k1 == 0 and k2 == 1) or (dist < min_dist):
671
+ mink1 = k1
672
+ mink2 = k2
673
+ min_dist = dist
674
+ if verbose:
675
+ logger.info("combining cluster %d and %d", mink1, mink2)
676
+
677
+ mixture.cluster[mink1] = add_cluster(mixture.cluster[mink1], mixture.cluster[mink2])
678
+ mixture.cluster[mink2: (K - 1)] = mixture.cluster[(mink2 + 1): K]
679
+ mixture.cluster = mixture.cluster[:(K - 1)]
680
+ mixture.K = K - 1
681
+ mixture = cluster_normalize(mixture)
682
+
683
+ return mixture
684
+
685
+
686
+ def decorrelate_and_normalize(data):
687
+ """Decorrelate and normalize data, returning the transform for inversion.
688
+
689
+ Args:
690
+ data(ndarray): an N x M 2D array of observation vectors with each row being an M-dimensional observation vector,
691
+ totally N observations
692
+
693
+ Returns:
694
+ tuple: (data, T, smean), where
695
+ - data(ndarray): decorrelated and normalized observation vectors
696
+ - T(ndarray): transformation 2D array
697
+ - smean(ndarray): mean values
698
+ """
699
+ # Decorrelate and normalize the data
700
+ smean = np.mean(data, axis=0)
701
+ scov = np.cov(data, rowvar=False)
702
+ D, E = np.linalg.eig(scov)
703
+ D = np.diag(D)
704
+ T = E @ np.linalg.inv(np.sqrt(D))
705
+ data = (data - (np.diag(smean) @ np.ones((np.shape(data)[1], np.shape(data)[0]))).T) @ T
706
+
707
+ return data, T, smean
708
+
709
+
710
+ def transform_back_to_original_coordinates(opt_mixture, T, smean):
711
+ """Map mixture parameters from whitened coordinates back to the original ones.
712
+
713
+ Args:
714
+ opt_mixture(class): a structure representing the optimum Gaussian mixture parameters corresponding to
715
+ decorrelated coordinates
716
+ T(ndarray): transformation 2D array
717
+ smean(ndarray): mean values
718
+
719
+ Returns:
720
+ class object: a structure representing the optimum Gaussian mixture parameters corresponding to original
721
+ coordinates
722
+ """
723
+ invT = np.linalg.inv(T)
724
+ # Transform the parameters back to original coordinates
725
+ for k in range(opt_mixture.K):
726
+ opt_mixture.cluster[k].mu = (opt_mixture.cluster[k].mu.T @ invT + smean).T
727
+ opt_mixture.cluster[k].R = invT.T @ opt_mixture.cluster[k].R @ invT
728
+ opt_mixture.cluster[k].invR = T @ opt_mixture.cluster[k].invR @ T.T
729
+ opt_mixture.cluster[k].const = opt_mixture.cluster[k].const - np.log(
730
+ np.linalg.det(invT.T @ invT)) / 2
731
+
732
+ return opt_mixture
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.4
2
+ Name: gmcluster
3
+ Version: 0.3.0
4
+ Summary: EM Gaussian-mixture clustering with automatic MDL order selection
5
+ Author: Charles A. Bouman, Mohammad Samin Nur Chowdhury
6
+ License: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/cabouman/gmcluster
8
+ Project-URL: Documentation, https://gmcluster.readthedocs.io
9
+ Project-URL: Repository, https://github.com/cabouman/gmcluster
10
+ Requires-Python: >=3.11
11
+ Description-Content-Type: text/x-rst
12
+ License-File: LICENSE
13
+ Requires-Dist: numpy
14
+ Requires-Dist: matplotlib
15
+ Provides-Extra: test
16
+ Requires-Dist: pytest; extra == "test"
17
+ Provides-Extra: docs
18
+ Requires-Dist: sphinx; extra == "docs"
19
+ Requires-Dist: sphinx-book-theme; extra == "docs"
20
+ Requires-Dist: sphinx-design; extra == "docs"
21
+ Requires-Dist: sphinx-copybutton; extra == "docs"
22
+ Requires-Dist: sphinxcontrib-bibtex; extra == "docs"
23
+ Dynamic: license-file
24
+
25
+ GMCluster
26
+ =========
27
+
28
+ GMCluster fits a Gaussian mixture model to data by EM and selects the number of clusters automatically using the minimum description length (MDL) criterion.
29
+
30
+ It is a Python rewrite of the C package `Cluster <https://engineering.purdue.edu/~bouman/software/cluster/>`_. Full documentation is at https://gmcluster.readthedocs.io/ .
31
+
32
+ Installing
33
+ ----------
34
+
35
+ Install the latest release from PyPI::
36
+
37
+ pip install gmcluster
38
+
39
+ To install from source (for development), clone the repository and do an editable install::
40
+
41
+ git clone https://github.com/cabouman/gmcluster.git
42
+ cd gmcluster
43
+ pip install -e .
44
+
45
+ Quick Start
46
+ -----------
47
+
48
+ The package provides one class, ``GaussianMixture``. Fit it to your data, read the
49
+ estimated parameters, then classify points or draw new samples.
50
+
51
+ .. code-block:: python
52
+
53
+ import numpy as np
54
+ from gmcluster import GaussianMixture
55
+
56
+ X = np.random.default_rng(0).standard_normal((500, 2))
57
+
58
+ # Fit the mixture; "auto" selects the number of clusters by MDL.
59
+ gm = GaussianMixture(num_clusters="auto").fit(X)
60
+
61
+ print(gm.estimated_num_clusters) # number of clusters found
62
+ print(gm.estimated_weights) # shape (K,)
63
+ print(gm.estimated_means) # shape (K, M)
64
+ print(gm.estimated_covariances) # shape (K, M, M)
65
+
66
+ labels = gm.classify(X) # most-likely cluster per point, shape (N,)
67
+ new_points = gm.sample(100) # draw 100 samples from the fitted mixture
68
+
69
+ Running the demos
70
+ -----------------
71
+
72
+ Validate the installation by running a demo::
73
+
74
+ cd demo
75
+ python demo_1.py
76
+
77
+ Citation
78
+ --------
79
+
80
+ Please cite this software when you use it. The BibTeX entry is in
81
+ ``docs/source/credits.rst`` and in the online documentation at
82
+ https://gmcluster.readthedocs.io/ .
@@ -0,0 +1,15 @@
1
+ LICENSE
2
+ README.rst
3
+ pyproject.toml
4
+ gmcluster/__init__.py
5
+ gmcluster/gmcluster.py
6
+ gmcluster.egg-info/PKG-INFO
7
+ gmcluster.egg-info/SOURCES.txt
8
+ gmcluster.egg-info/dependency_links.txt
9
+ gmcluster.egg-info/requires.txt
10
+ gmcluster.egg-info/top_level.txt
11
+ tests/test_api.py
12
+ tests/test_equivalence.py
13
+ tests/test_order_selection.py
14
+ tests/test_recover_params.py
15
+ tests/test_smoke.py
@@ -0,0 +1,12 @@
1
+ numpy
2
+ matplotlib
3
+
4
+ [docs]
5
+ sphinx
6
+ sphinx-book-theme
7
+ sphinx-design
8
+ sphinx-copybutton
9
+ sphinxcontrib-bibtex
10
+
11
+ [test]
12
+ pytest
@@ -0,0 +1 @@
1
+ gmcluster
@@ -0,0 +1,34 @@
1
+ [build-system]
2
+ requires = ["setuptools>=64"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "gmcluster"
7
+ description = "EM Gaussian-mixture clustering with automatic MDL order selection"
8
+ readme = "README.rst"
9
+ license = { text = "BSD-3-Clause" }
10
+ requires-python = ">=3.11"
11
+ dynamic = ["version"]
12
+ authors = [
13
+ { name = "Charles A. Bouman" },
14
+ { name = "Mohammad Samin Nur Chowdhury" },
15
+ ]
16
+ dependencies = [
17
+ "numpy",
18
+ "matplotlib",
19
+ ]
20
+
21
+ [project.optional-dependencies]
22
+ test = ["pytest"]
23
+ docs = ["sphinx", "sphinx-book-theme", "sphinx-design", "sphinx-copybutton", "sphinxcontrib-bibtex"]
24
+
25
+ [project.urls]
26
+ Homepage = "https://github.com/cabouman/gmcluster"
27
+ Documentation = "https://gmcluster.readthedocs.io"
28
+ Repository = "https://github.com/cabouman/gmcluster"
29
+
30
+ [tool.setuptools]
31
+ packages = ["gmcluster"]
32
+
33
+ [tool.setuptools.dynamic]
34
+ version = { attr = "gmcluster.__version__" }
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,115 @@
1
+ # Behavior of the GaussianMixture public surface. Small and fast.
2
+
3
+ import numpy as np
4
+ import pytest
5
+
6
+ from gmcluster import GaussianMixture
7
+
8
+
9
+ def make_data(seed=0, n=200):
10
+ rng = np.random.default_rng(seed)
11
+ centers = np.array([[4.0, 4.0], [-4.0, -4.0], [4.0, -4.0]])
12
+ return np.vstack([rng.standard_normal((n, 2)) + c for c in centers])
13
+
14
+
15
+ def fitted_model():
16
+ return GaussianMixture(num_clusters="auto", max_clusters=6).fit(make_data())
17
+
18
+
19
+ def test_posterior_rows_sum_to_one():
20
+ gm = fitted_model()
21
+ X = make_data(seed=1, n=50)
22
+ P = gm.posterior(X)
23
+ assert P.shape == (X.shape[0], gm.estimated_num_clusters)
24
+ assert np.allclose(P.sum(axis=1), 1.0)
25
+
26
+
27
+ def test_classify_is_argmax_of_posterior():
28
+ gm = fitted_model()
29
+ X = make_data(seed=1, n=50)
30
+ assert np.array_equal(gm.classify(X), np.argmax(gm.posterior(X), axis=1))
31
+
32
+
33
+ def test_log_likelihood_shape():
34
+ gm = fitted_model()
35
+ X = make_data(seed=1, n=50)
36
+ ll = gm.log_likelihood(X)
37
+ assert ll.shape == (X.shape[0],)
38
+
39
+
40
+ def test_sample_shapes_and_reproducibility():
41
+ gm = fitted_model()
42
+ M = gm.estimated_means.shape[1]
43
+
44
+ X = gm.sample(30, rng=np.random.default_rng(7))
45
+ assert X.shape == (30, M)
46
+
47
+ X2, labels = gm.sample(30, rng=np.random.default_rng(7), with_labels=True)
48
+ assert X2.shape == (30, M)
49
+ assert labels.shape == (30,)
50
+ # Same seed reproduces the same draw.
51
+ assert np.array_equal(X, X2)
52
+ assert np.all((labels >= 0) & (labels < gm.estimated_num_clusters))
53
+
54
+
55
+ def test_split_clusters():
56
+ gm = fitted_model()
57
+ parts = gm.split_clusters()
58
+ assert len(parts) == gm.estimated_num_clusters
59
+ X = make_data(seed=2, n=20)
60
+ for part in parts:
61
+ assert part.estimated_num_clusters == 1
62
+ assert np.isclose(part.estimated_weights.sum(), 1.0)
63
+ # Each split model is usable as its own density.
64
+ assert part.log_likelihood(X).shape == (X.shape[0],)
65
+ assert part.classify(X).shape == (X.shape[0],)
66
+
67
+
68
+ def test_estimate_shapes_and_weights():
69
+ gm = fitted_model()
70
+ K = gm.estimated_num_clusters
71
+ M = gm.estimated_means.shape[1]
72
+ assert gm.estimated_means.shape == (K, M)
73
+ assert gm.estimated_covariances.shape == (K, M, M)
74
+ assert gm.estimated_weights.shape == (K,)
75
+ assert np.isclose(gm.estimated_weights.sum(), 1.0)
76
+
77
+
78
+ def test_access_before_fit_raises():
79
+ gm = GaussianMixture()
80
+ for name in ["estimated_num_clusters", "estimated_weights", "estimated_means",
81
+ "estimated_covariances", "mdl", "mdl_path", "converged", "num_iterations"]:
82
+ with pytest.raises(RuntimeError):
83
+ getattr(gm, name)
84
+
85
+
86
+ def test_query_before_fit_raises():
87
+ gm = GaussianMixture()
88
+ with pytest.raises(RuntimeError):
89
+ gm.posterior(make_data(n=5))
90
+
91
+
92
+ @pytest.mark.parametrize("kwargs", [
93
+ {"num_clusters": 0},
94
+ {"num_clusters": -3},
95
+ {"alpha": 2},
96
+ {"alpha": 0},
97
+ {"covariance_type": "bogus"},
98
+ {"max_clusters": 0},
99
+ ])
100
+ def test_bad_constructor_args_raise(kwargs):
101
+ with pytest.raises((ValueError, TypeError)):
102
+ GaussianMixture(**kwargs)
103
+
104
+
105
+ def test_non_2d_input_raises():
106
+ gm = GaussianMixture(max_clusters=4)
107
+ with pytest.raises(ValueError):
108
+ gm.fit(np.zeros(10))
109
+
110
+
111
+ def test_repr():
112
+ gm = GaussianMixture()
113
+ assert "unfitted" in repr(gm)
114
+ gm = fitted_model()
115
+ assert "clusters=" in repr(gm) and "dims=" in repr(gm)
@@ -0,0 +1,58 @@
1
+ # In-process equivalence: the GaussianMixture wrapper reproduces the private
2
+ # EM engine exactly. This proves the wrapper does not alter the math.
3
+
4
+ import numpy as np
5
+ import pytest
6
+
7
+ from gmcluster import GaussianMixture
8
+ from gmcluster.gmcluster import _fit_mixture
9
+
10
+ ATOL = 1e-12
11
+
12
+
13
+ def make_data(seed=0, n=300):
14
+ """Seeded three-cluster 2D data."""
15
+ rng = np.random.default_rng(seed)
16
+ centers = np.array([[4.0, 4.0], [-4.0, -4.0], [4.0, -4.0]])
17
+ parts = [rng.standard_normal((n, 2)) + c for c in centers]
18
+ return np.vstack(parts)
19
+
20
+
21
+ def engine_params(mixture):
22
+ """Pull (K, weights, means, covariances, mdl) out of an engine mixture."""
23
+ K = int(mixture.K)
24
+ weights = np.array([float(c.pb) for c in mixture.cluster])
25
+ means = np.array([c.mu.ravel() for c in mixture.cluster])
26
+ covs = np.array([np.asarray(c.R) for c in mixture.cluster])
27
+ return K, weights, means, covs, mixture.rissanen
28
+
29
+
30
+ @pytest.mark.parametrize("covariance_type,est_kind", [("full", "full"), ("diagonal", "diag")])
31
+ @pytest.mark.parametrize("num_clusters", ["auto", 2])
32
+ @pytest.mark.parametrize("whiten", [False, True])
33
+ def test_wrapper_matches_engine(covariance_type, est_kind, num_clusters, whiten):
34
+ X = make_data()
35
+ max_clusters = 6
36
+ alpha = 0.1
37
+
38
+ # Wrapper.
39
+ gm = GaussianMixture(num_clusters=num_clusters, max_clusters=max_clusters,
40
+ covariance_type=covariance_type, alpha=alpha,
41
+ whiten=whiten, verbose=False).fit(X)
42
+
43
+ # Private engine with the same init_K / final_K mapping fit() uses.
44
+ if num_clusters == "auto":
45
+ init_K, final_K = max_clusters, 0
46
+ else:
47
+ final_K = int(num_clusters)
48
+ init_K = max(max_clusters, final_K)
49
+ mixture, mdl_path = _fit_mixture(X, init_K, final_K, est_kind, alpha, whiten, False)
50
+
51
+ K, weights, means, covs, mdl = engine_params(mixture)
52
+
53
+ assert gm.estimated_num_clusters == K
54
+ assert np.allclose(gm.estimated_weights, weights, atol=ATOL)
55
+ assert np.allclose(gm.estimated_means, means, atol=ATOL)
56
+ assert np.allclose(gm.estimated_covariances, covs, atol=ATOL)
57
+ assert np.allclose(gm.mdl, mdl, atol=ATOL)
58
+ assert gm.mdl_path == mdl_path
@@ -0,0 +1,26 @@
1
+ # MDL order selection finds 3 clusters; a fixed order returns exactly that many.
2
+
3
+ import numpy as np
4
+
5
+ from gmcluster import GaussianMixture
6
+
7
+
8
+ def make_data(seed=0, n_per=300):
9
+ """Three well-separated 2-D Gaussians."""
10
+ rng = np.random.default_rng(seed)
11
+ centers = np.array([[6.0, 6.0], [-6.0, -6.0], [6.0, -6.0]])
12
+ return np.vstack([rng.standard_normal((n_per, 2)) + c for c in centers])
13
+
14
+
15
+ def test_auto_selects_three():
16
+ data = make_data(seed=0)
17
+ gm = GaussianMixture(num_clusters="auto", max_clusters=6).fit(data)
18
+ assert gm.estimated_num_clusters == 3
19
+
20
+
21
+ def test_fixed_order_returns_two():
22
+ data = make_data(seed=0)
23
+ gm = GaussianMixture(num_clusters=2).fit(data)
24
+ assert gm.estimated_num_clusters == 2
25
+ assert gm.estimated_means.shape[0] == 2
26
+ assert gm.estimated_weights.shape[0] == 2
@@ -0,0 +1,31 @@
1
+ # The fit recovers the true cluster means on well-separated data.
2
+
3
+ import numpy as np
4
+
5
+ from gmcluster import GaussianMixture
6
+
7
+
8
+ def make_mixture(seed=0, n_per=300):
9
+ """Three well-separated 2-D Gaussians. Returns (data, true_means)."""
10
+ rng = np.random.default_rng(seed)
11
+ true_means = np.array([[6.0, 6.0], [-6.0, -6.0], [6.0, -6.0]])
12
+ data = np.vstack([rng.standard_normal((n_per, 2)) + m for m in true_means])
13
+ return data, true_means
14
+
15
+
16
+ def match_nearest(true_means, estimated_means):
17
+ """For each true mean, the distance to its nearest estimated mean."""
18
+ dists = []
19
+ for t in true_means:
20
+ d = np.linalg.norm(estimated_means - t, axis=1)
21
+ dists.append(d.min())
22
+ return np.array(dists)
23
+
24
+
25
+ def test_recover_means():
26
+ data, true_means = make_mixture(seed=0)
27
+ gm = GaussianMixture(num_clusters="auto", max_clusters=6).fit(data)
28
+
29
+ assert gm.estimated_num_clusters == 3
30
+ # Every true center has an estimated center close to it.
31
+ assert np.all(match_nearest(true_means, gm.estimated_means) < 0.5)
@@ -0,0 +1,17 @@
1
+ # Smoke test: a tiny fit returns a well-formed model.
2
+
3
+ import numpy as np
4
+ from gmcluster import GaussianMixture
5
+
6
+
7
+ def test_smoke_tiny_fit():
8
+ np.random.seed(0)
9
+ a = np.random.randn(200, 2) + np.array([5.0, 5.0])
10
+ b = np.random.randn(200, 2) + np.array([-5.0, -5.0])
11
+ data = np.vstack([a, b])
12
+
13
+ gm = GaussianMixture(num_clusters="auto", max_clusters=5, verbose=False).fit(data)
14
+
15
+ assert isinstance(gm.estimated_num_clusters, int)
16
+ assert gm.estimated_num_clusters >= 1
17
+ assert gm.estimated_weights.shape == (gm.estimated_num_clusters,)