clustvartools 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. clustvartools/__init__.py +6 -0
  2. clustvartools/clustering/__init__.py +24 -0
  3. clustvartools/clustering/_catclv.py +630 -0
  4. clustvartools/clustering/_cathcav.py +400 -0
  5. clustvartools/clustering/_catvarhca.py +416 -0
  6. clustvartools/clustering/_clv.py +631 -0
  7. clustvartools/clustering/_clvmix.py +707 -0
  8. clustvartools/clustering/_corclv.py +552 -0
  9. clustvartools/clustering/_dclv.py +580 -0
  10. clustvartools/clustering/_hcav.py +384 -0
  11. clustvartools/clustering/_hcavmix.py +401 -0
  12. clustvartools/clustering/functions/__init__.py +0 -0
  13. clustvartools/clustering/functions/cathca.py +64 -0
  14. clustvartools/clustering/functions/concat_empty.py +25 -0
  15. clustvartools/clustering/functions/cov2corr.py +29 -0
  16. clustvartools/clustering/functions/func_dclv.py +134 -0
  17. clustvartools/clustering/functions/func_fillna.py +40 -0
  18. clustvartools/clustering/functions/get_indices.py +43 -0
  19. clustvartools/clustering/functions/get_sup_label.py +61 -0
  20. clustvartools/clustering/functions/gsvd.py +107 -0
  21. clustvartools/clustering/functions/preprocessing.py +69 -0
  22. clustvartools/clustering/functions/revalue.py +63 -0
  23. clustvartools/clustering/functions/statistics.py +354 -0
  24. clustvartools/clustering/functions/tests.py +259 -0
  25. clustvartools/clustering/functions/utils.py +175 -0
  26. clustvartools/datasets/__init__.py +447 -0
  27. clustvartools/datasets/data/__init__.py +0 -0
  28. clustvartools/datasets/data/apples.xlsx +0 -0
  29. clustvartools/datasets/data/autos2005.xlsx +0 -0
  30. clustvartools/datasets/data/burger.xlsx +0 -0
  31. clustvartools/datasets/data/canines.xlsx +0 -0
  32. clustvartools/datasets/data/cars.xlsx +0 -0
  33. clustvartools/datasets/data/congressvotingrecords.xlsx +0 -0
  34. clustvartools/datasets/data/decathlon.xlsx +0 -0
  35. clustvartools/datasets/data/jobrate.xlsx +0 -0
  36. clustvartools/datasets/data/olympic.xlsx +0 -0
  37. clustvartools/datasets/data/poison.xlsx +0 -0
  38. clustvartools/datasets/data/uscrime.xlsx +0 -0
  39. clustvartools/datasets/data/wine.xlsx +0 -0
  40. clustvartools/graphics/__init__.py +12 -0
  41. clustvartools/graphics/_fviz_dend.py +562 -0
  42. clustvartools/graphics/_fviz_dend2.py +600 -0
  43. clustvartools/graphics/_fviz_height.py +167 -0
  44. clustvartools/others/__init__.py +29 -0
  45. clustvartools/others/_clust_diss.py +82 -0
  46. clustvartools/others/_clust_dist.py +72 -0
  47. clustvartools/others/_clust_member.py +68 -0
  48. clustvartools/others/_clust_score.py +55 -0
  49. clustvartools/others/_coeffsim.py +108 -0
  50. clustvartools/others/_disjunctive.py +70 -0
  51. clustvartools/others/_getnnsvar.py +50 -0
  52. clustvartools/others/_save.py +82 -0
  53. clustvartools/others/_splitmix.py +69 -0
  54. clustvartools/others/_sprintf.py +63 -0
  55. clustvartools/others/_summary.py +189 -0
  56. clustvartools-0.0.1.dist-info/METADATA +223 -0
  57. clustvartools-0.0.1.dist-info/RECORD +60 -0
  58. clustvartools-0.0.1.dist-info/WHEEL +5 -0
  59. clustvartools-0.0.1.dist-info/licenses/LICENSE +21 -0
  60. clustvartools-0.0.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,6 @@
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ from .clustering import *
5
+ from .graphics import *
6
+ from .others import *
@@ -0,0 +1,24 @@
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ from ._catclv import CatCLV
5
+ from ._cathcav import CatHCAV
6
+ from ._catvarhca import CatVARHCA
7
+ from ._clv import CLV
8
+ from ._clvmix import CLVmix
9
+ from ._corclv import CorCLV
10
+ from ._dclv import DCLV
11
+ from ._hcav import HCAV
12
+ from ._hcavmix import HCAVmix
13
+
14
+ __all__ = [
15
+ "CatCLV",
16
+ "CatHCAV",
17
+ "CatVARHCA",
18
+ "CLV",
19
+ "CLVmix",
20
+ "CorCLV",
21
+ "DCLV",
22
+ "HCAV",
23
+ "HCAVmix"
24
+ ]
@@ -0,0 +1,630 @@
1
+ # -*- coding: utf-8 -*-
2
+ from numpy import zeros,ones, ndarray, repeat, array, where, c_, sqrt, diag
3
+ from pandas import Series, DataFrame, concat
4
+ from pandas.api.types import is_numeric_dtype
5
+ from itertools import chain
6
+ from collections import namedtuple
7
+ from scipy.cluster.hierarchy import fcluster
8
+ from scipy.stats import t as sst
9
+ from sklearn.base import BaseEstimator, TransformerMixin
10
+ from sklearn.utils.validation import check_is_fitted
11
+
12
+ #interns functions
13
+ from .functions.preprocessing import preprocessing
14
+ from .functions.get_sup_label import get_sup_label
15
+ from .functions.utils import is_all_object_or_category_dtype
16
+ from .functions.statistics import wcorr
17
+ from .functions.tests import wpearsonr, wcorrtest
18
+ from ..others._disjunctive import disjunctive
19
+ from ..others._clust_diss import clust_diss
20
+ from ..others._getnnsvar import getnnsvar
21
+ from ..others._clust_score import clust_score
22
+ from ..others._coeffsim import coeffsim
23
+ from ..others._clust_member import clust_member
24
+
25
+ class CatCLV(BaseEstimator,TransformerMixin):
26
+ """
27
+ Hierarchical Clustering of Categorical Variables around Latent Variables (CatCLV)
28
+
29
+ Performns ascendant hierarchical clustering of a set of categorical variables. The aggregation criterion is the decrease
30
+ in homogeneity for the clusters being merged. The homogeneity of a cluster is the sum of the correlation ratio
31
+ between the variables and the center of the cluster which is the first principal component of multiple correspondence analysis (MCA).
32
+ Supplementary variables (continuous and/or categorical) may be used.
33
+
34
+ Parameters
35
+ ----------
36
+ ncl : int, default = 2
37
+ If a (positive) integer, the tree is cut with ncl clusters. If None, then the ncl is determine using optimal point.
38
+
39
+ row_w : 1d array-like of shape (n_samples,), default = None
40
+ An optional rows weights. The weights are given only for the active rows.
41
+
42
+ sup_var : int, str, list, tuple or range, default = None
43
+ The indexes or names of the supplementary variables (continuous and/or categorical).
44
+
45
+ tol : float, default = 1e-7
46
+ A tolerance threshold to test whether the distance matrix is Euclidean : an eigenvalue is considered positive if it is larger
47
+ than `-tol*lambda1` where `lambda1` is the largest eigenvalue.
48
+
49
+ Attributes
50
+ ----------
51
+ call_ : call
52
+ An object containing the summary called parameters with the following attributes:
53
+
54
+ Xtot : DataFrame of shape (n_samples, n_columns + n_columns_sup)
55
+ Input data.
56
+ X : DataFrame of shape (n_samples, n_columns)
57
+ Active data.
58
+ dummies : DataFrame of shape (n_samples, n_levels)
59
+ Disjunctive table.
60
+ Z : DataFrame of shape (n_samples, n_levels)
61
+ Standardized data.
62
+ center : Series of shape (n_levels,)
63
+ The proportion of levels.
64
+ scale : Series of shape (n_levels)
65
+ The squared root of center.
66
+ row_w : Series of shape (n_samples,)
67
+ The rows weights.
68
+ ncl : int
69
+ The number of clusters kepted.
70
+ tree : tree
71
+ An object containing the results for the hierarchical agglomerative clustering algorithm, with the following attributes:
72
+
73
+ S : DataFrame of shape (n_columns, n_columns)
74
+ Similarity matrix.
75
+ D : DataFrame of shape (n_columns, n_columns)
76
+ Dissimilarity matrix.
77
+ Z : 2d numpy array of shape (n_columns - 1, 4)
78
+ Linkage matrix.
79
+ height: DataFrame of shape (n_columns - 1, 4)
80
+ Reverse Height of aggregation between ``Z[:,0]`` and ``Z[:,1]``.
81
+ merge : 1d numpy array of shape (n_columns - 1,)
82
+ Height of aggregation between ``Z[:,0]`` and ``Z[:,1]``.
83
+ size : 1d numpy array of shape (n_columns - 1,)
84
+ Number of original observations in the newly formed cluster.
85
+
86
+ sup_var : None, list
87
+ The names of the supplementary variables (continuous and/or categorical).
88
+
89
+ cluster_ : cluster
90
+ An object with the following attributes:
91
+
92
+ summary : DataFrame of shape (ncl, 4)
93
+ Cluster summary.
94
+ cor : DataFrame of shape (ncl, ncl)
95
+ Inter-cluster correlations.
96
+ princomps : DataFrame of shape (n_samples, ncl)
97
+ Synthetic variables , i.e cluster centers or latent variables.
98
+
99
+ coef_ : coef
100
+ An object with the following attributes:
101
+
102
+ coef : DataFrame of shape (n_levels + 1, ncl)
103
+ Coefficients of linear combinations defining the synthetic variable of each cluster.
104
+ coef_std : DataFrame of shape (n_levels, ncl)
105
+ Standardized scoring coefficients.
106
+
107
+ quali_var_ : quali_var
108
+ An object with the following attributes:
109
+
110
+ cluster : Series of shape (n_columns,)
111
+ Labels of the cluster each column belongs to.
112
+ sqload : DataFrame of shape (n_columns, ncl)
113
+ Squared loadings of variables and latent variables.
114
+ member : DataFrame of shape (n_columns, 4)
115
+ Cluster members and R-square values.
116
+ sim : DataFrame of shape (n_columns, n_columns)
117
+ Similarities matrix between variables for each cluster.
118
+ loadings : DataFrame of shape (n_levels, ncl)
119
+ Loadings of variables, i.e. the first eigen vector calculated by MCA.
120
+
121
+ var_sup_ : var_sup, optional
122
+ An object with the following attributes:
123
+
124
+ cluster : Series of shape (n_columns_sup,)
125
+ Labels of the cluster each supplementary column belongs to.
126
+ cor : None or DataFrame of shape (n_quanti_sup, ncl)
127
+ Pearson correlation of supplementary variables and latent variables.
128
+ cortest : None or DataFrame of shape (n_quanti_sup*ncl, 6)
129
+ Pearson correlation test of supplementary variables and latent variables.
130
+ sqload : DataFrame of shape (n_columns_sup, ncl)
131
+ Squared loadings of supplementary variables and latent variables.
132
+ member : DataFrame of shape (n_columns_sup, 4)
133
+ Cluster members and R-square values.
134
+
135
+ References
136
+ ----------
137
+ [1] E. Vigneau, M. Qannari. `Clustering of variables around latent components <https://www.researchgate.net/publication/243044470_Clustering_of_Variables_Around_Latent_Components>`_. in Statistics, Simulation and Computation, 32(4), pp.1131-1150, 2003.
138
+
139
+ [2] J. Saracco, M. Chavent, V. Kuentz. `Clustering of categorical variables around latent variables <https://www.researchgate.net/publication/46448264_Clustering_of_categorical_variables_around_latent_variables>`_. Cahiers du GTREThA, N°2010-02, 2010.
140
+
141
+ [3] M. Chavent, V. Kuentz Simonet, B. Liquet, J. Saracco. `ClustOfVar: An R package for the Clustering of Variables <https://www.jstatsoft.org/article/view/v050i13>`_. in Journal of Statistical Software, 50(13), september 2012.
142
+
143
+ [4] Rakotomalala, R. `Classification de variables : classification autour de variables latentes <https://eric.univ-lyon2.fr/ricco/cours/slides/classification_de_variables.pdf>`_. Tutoriel Tanagra pour le Data Mining
144
+
145
+ Examples
146
+ --------
147
+ >>> from clustvartools.datasets import poison
148
+ >>> from clustvartools import CatCLV
149
+ >>> clf = CatCLV(ncl=4,sup_var=range(4))
150
+ >>> clf.fit(poison.data)
151
+ CatCLV(ncl=4,sup_var=range(4))
152
+ """
153
+ def __init__(
154
+ self,
155
+ ncl = 2,
156
+ row_w = None,
157
+ sup_var = None,
158
+ tol = 1e-7
159
+ ):
160
+ self.ncl = ncl
161
+ self.row_w = row_w
162
+ self.sup_var = sup_var
163
+ self.tol = tol
164
+
165
+ def fit(self,X,y=None):
166
+ """
167
+ Compute CatCLV
168
+
169
+ Parameters
170
+ ----------
171
+ X : DataFrame of shape (n_samples, n_columns)
172
+ Training data, where ``n_samples`` in the number of samples
173
+ and ``n_columns`` is the number of columns.
174
+
175
+ y : Ignored
176
+ Not used, present here for API consistency by convention.
177
+
178
+ Returns
179
+ -------
180
+ self : object
181
+ Fitted estimator.
182
+ """
183
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
184
+ # preprocessing
185
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
186
+ X = preprocessing(X=X)
187
+
188
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
189
+ # get supplementary elements labels
190
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
191
+ sup_var_label = get_sup_label(X=X,indexes=self.sup_var,axis=1)
192
+
193
+ # make a copy of the original data
194
+ Xtot = X.copy()
195
+
196
+ # drop supplementary variables columns
197
+ if self.sup_var is not None:
198
+ X_sup_var, X = X.loc[:,sup_var_label], X.drop(columns=sup_var_label)
199
+
200
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
201
+ # hierarchical clustering analysis of mixed data
202
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
203
+ # check if X contains only numerics columns
204
+ if not is_all_object_or_category_dtype(X):
205
+ raise TypeError("Columns in X must be categorical.")
206
+
207
+ # set number of rows and columns
208
+ n_rows, n_cols = X.shape
209
+
210
+ # set individuals weights
211
+ if self.row_w is None:
212
+ row_w = Series(ones(n_rows)/n_rows,index=X.index,name="weight")
213
+ elif not isinstance(self.row_w,(list,tuple,ndarray,Series)):
214
+ raise TypeError("row_w must be a 1d array-like of individuals weights.")
215
+ elif len(self.row_w) != n_rows:
216
+ raise ValueError(f"row_w must be a 1d array-like of shape ({n_rows},).")
217
+ else:
218
+ row_w = Series(array(self.row_w)/sum(self.row_w),index=X.index,name="weight")
219
+
220
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
221
+ #standardization: z_ik = (y_ik - m_k)/s_k
222
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
223
+ # disjunctive table
224
+ dummies = disjunctive(X=X,prefix=True)
225
+ # proportion of levels (=center)
226
+ center = Series((dummies.T * row_w).sum(axis=1).to_numpy(),index=dummies.columns,name = "center")
227
+ # standard deviation (=scale)
228
+ scale = Series(sqrt(center),index=dummies.columns,name = "scale")
229
+ # standardization: z_ik = (y_ik - m_k)/s_k
230
+ Z = (dummies - center)/scale
231
+
232
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
233
+ # data preparation
234
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
235
+ # compute similarity and dissimilarity matrices
236
+ S = DataFrame(diag(ones(n_cols)),index=X.columns,columns=X.columns)
237
+ D = DataFrame(zeros((n_cols,n_cols)),index=X.columns,columns=X.columns)
238
+ for i in range(n_cols-1):
239
+ for j in range(i+1,n_cols):
240
+ k, l = X.columns[i], X.columns[j]
241
+ A = list(Z.columns[Z.columns.str.startswith(k)])
242
+ B = list(Z.columns[Z.columns.str.startswith(l)])
243
+ # similarity matrix
244
+ S.iloc[i,j] = coeffsim(X=X.iloc[:,i],Y=X.iloc[:,j],w=row_w)
245
+ # dissimilarity matrix
246
+ D.iloc[i,j] = clust_diss(X=Z.loc[:,A],Y=Z.loc[:,B],tol=self.tol)
247
+ S.iloc[j,i], D.iloc[j,i] = S.iloc[i,j], D.iloc[i,j]
248
+
249
+ # make a copy of dissimilarity matrix
250
+ Dprim = D.copy()
251
+
252
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
253
+ # ascendant hierarchical cluster analysis from covariance/correlation matrix
254
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
255
+ # initial values
256
+ init = list(range(n_cols)) # initial partition
257
+ init_crit = repeat(1,n_cols) # initial criteria
258
+ maxv = 1e12 # maximum
259
+ a = zeros((n_cols-1))
260
+ b = zeros((n_cols-1))
261
+ ia = zeros((n_cols-1)) # Le premier cluster ou point d'origine sélectionné pour la fusion.
262
+ ib = zeros((n_cols-1)) # Le deuxième cluster ou point d'origine sélectionné pour la fusion.
263
+ lev = zeros((n_cols-1)) # la hauteur entre les deux clusters fusionnés
264
+ card = ones((n_cols)) # cardinalities
265
+ size = zeros((n_cols-1)) # Le nombre total d'observations (points initiaux) contenues dans ce nouveau cluster ainsi formé.
266
+
267
+ # compute nearest neighbor of variables
268
+ nnsvar = getnnsvar(D=D,crit=init_crit)
269
+ # clust matrix
270
+ clustmat = zeros((n_cols,n_cols))
271
+ for i in range(n_cols):
272
+ clustmat[i,n_cols-1] = i
273
+ for ncl in range(n_cols-2,-1,-1):
274
+ # check for agglomerable pair
275
+ minobs = -1
276
+ mindis = maxv
277
+ for i in range(n_cols):
278
+ if init_crit[i] == 1:
279
+ if nnsvar.nndiss[i] < mindis:
280
+ mindis = nnsvar.nndiss[i]
281
+ minobs = i
282
+ # find agglomerands cl1 and cl2, with former < latter
283
+ if minobs < nnsvar.nn[minobs]:
284
+ cl1 = minobs
285
+ cl2 = nnsvar.nn[minobs]
286
+ elif minobs > nnsvar.nn[minobs]:
287
+ cl2 = minobs
288
+ cl1 = nnsvar.nn[minobs]
289
+ # convert to integer
290
+ cl1, cl2 = int(cl1), int(cl2)
291
+ id1 = [i for i,x in enumerate(clustmat[:,ncl+1]) if x == cl1]
292
+ id2 = [i for i,x in enumerate(clustmat[:,ncl+1]) if x == cl2]
293
+ A = [i for i in init if i in id1]
294
+ B = [i for i in init if i in id2]
295
+ clus = [*A,*B]
296
+ # assign to leaf
297
+ a[ncl], b[ncl], size[ncl] = cl1, cl2, len(clus)
298
+ if card[cl1] == 1:
299
+ ia[ncl] = - cl1
300
+ if card[cl2] == 1:
301
+ ib[ncl] = - cl2
302
+ #
303
+ if card[cl1] > 1:
304
+ last_ind = 0
305
+ for i2 in range(n_cols-2,ncl,-1):
306
+ if a[i2] == cl1:
307
+ last_ind = i2
308
+ ia[ncl] = n_cols - last_ind - 1
309
+ if card[cl2] > 1:
310
+ last_ind = 0
311
+ for i2 in range(n_cols-2,ncl,-1):
312
+ if a[i2] == cl2:
313
+ last_ind = i2
314
+ ib[ncl] = n_cols - last_ind - 1
315
+ #
316
+ if ia[ncl] > 0 or ib[ncl] > 0:
317
+ l = min(ia[ncl],ib[ncl])
318
+ if l > 0:
319
+ l -= 1
320
+ r = max(ia[ncl],ib[ncl])
321
+ if r > 0:
322
+ r -= 1
323
+ ia[ncl], ib[ncl] = l, r
324
+ # add height of aggregation
325
+ lev[ncl] = mindis
326
+ # update cluster matrix
327
+ for i in range(n_cols):
328
+ clustmat[i,ncl] = clustmat[i,ncl+1]
329
+ if clustmat[i,ncl] == cl2:
330
+ clustmat[i,ncl] = cl1
331
+ # update dissimilarity matrix in cluster 1
332
+ for i in range(n_cols):
333
+ if (i != cl1) and (i != cl2) and (init_crit[i] == 1):
334
+ idx = [j for j in init if j in [k for k, x in enumerate(clustmat[:,ncl+1]) if x == i]]
335
+ # final label for cluster
336
+ A = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(k)] for k in [X.columns[j] for j in clus]]))
337
+ B = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(k)] for k in [X.columns[j] for j in idx]]))
338
+ # update dissimilarity
339
+ D.iloc[cl1,i] = clust_diss(X=Z.loc[:,A],Y=Z.loc[:,B],tol=self.tol)
340
+ D.iloc[i,cl1] = D.iloc[cl1,i]
341
+ card[cl1] = card[cl1] + card[cl2]
342
+ init_crit[cl2] = 0
343
+ nnsvar.nndiss[cl2] = maxv
344
+ # update dissimilarity matrix in cluster 2
345
+ for i in range(n_cols):
346
+ D.iloc[cl2,i] = maxv
347
+ D.iloc[i,cl2] = D.iloc[cl2,i]
348
+ nnsvar = getnnsvar(D=D,crit=init_crit)
349
+
350
+ # reverse
351
+ merge = array([ia,ib]).T[::-1]
352
+ # convert to python scipy
353
+ linkage = zeros((n_cols-1,4))
354
+ linkage[:,0] = where(merge[:,0] < 0,-merge[:,0],where(merge[:,0]==0,0,merge[:,0]+n_cols))
355
+ linkage[:,1] = where(merge[:,1] < 0,-merge[:,1],where(merge[:,1]==0,n_cols,merge[:,1]+n_cols))
356
+ linkage[:,2] = lev[::-1]
357
+ linkage[:,3] = size[::-1]
358
+
359
+ # height of aggregation
360
+ height = DataFrame(c_[list(range(1,linkage.shape[0]+1)),linkage[:,2][::-1]],columns=["cluster","height"])
361
+ height["diff_1"] = -1*height["height"].diff(1)
362
+ height["diff_2"] = height["diff_1"].diff(-1)
363
+ height["cluster"] = height["cluster"].astype("int")
364
+
365
+ # convert to dictionary
366
+ tree_ = {"S":S,"D":Dprim,"Z":linkage,"height":height,"merge":linkage[:,:2],"size":linkage[:,3]}
367
+ # convert to namedtuple
368
+ tree = namedtuple("tree",tree_.keys())(*tree_.values())
369
+
370
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
371
+ # set numbers of clusters
372
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
373
+ if self.ncl is None:
374
+ ncl = height[height["diff_2"]==height["diff_2"].max()]["cluster"].values[0]
375
+ elif self.ncl < 0:
376
+ raise TypeError("ncl should be a positive integer.")
377
+ elif not isinstance(self.ncl,int):
378
+ raise TypeError("ncl should be an integer")
379
+ else:
380
+ ncl = self.ncl
381
+
382
+ #convert to dictionary
383
+ call_ = {"Xtot":Xtot,"X":X,"dummies" : dummies,"Z":Z,"center": center,"scale":scale,"row_w":row_w,
384
+ "ncl":ncl,"tree":tree,"sup_var":sup_var_label}
385
+ #convert to namedtuple
386
+ self.call_ = namedtuple("call",call_.keys())(*call_.values())
387
+
388
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
389
+ # Informations for variables
390
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
391
+ # assign cluster
392
+ cluster = Series(fcluster(linkage,t=ncl,criterion="maxclust"), index = D.index, name = "cluster",dtype="category")
393
+ # unique cluster
394
+ uq_cluster = sorted(cluster.unique())
395
+
396
+ # cluster infos
397
+ clust_infos = {}
398
+ for k in uq_cluster:
399
+ clus = cluster[cluster==k].index
400
+ A = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(i)] for i in clus]))
401
+ score = clust_score(X=Z.loc[:,A],tol=self.tol)._asdict()
402
+ cor2 = S.loc[clus,clus]
403
+ score = {**score, **{"cluster": clus, "ncl" : len(clus), "name" : A, "cor2" : cor2,}}
404
+ clust_infos[k] = namedtuple("clust_score",score.keys())(*score.values())
405
+
406
+ # principal components - latent components
407
+ pcs = concat((Series(cl.u,index=X.index).to_frame(c) for c, cl in clust_infos.items()),axis=1)
408
+ pcs.columns = pcs.columns.astype("category")
409
+
410
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
411
+ # informations for clusters
412
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
413
+ # cluster summary
414
+ cluster_summary = DataFrame(columns=["Cluster","Members","Variation Explained","Proportion Explained"]).astype("float")
415
+ i = 0
416
+ for c, cl in clust_infos.items():
417
+ cluster_summary.loc[i] = [c, cl.ncl, cl.d, cl.p]
418
+ i += 1
419
+ # convert to category and integer
420
+ cluster_summary["Cluster"] = cluster_summary["Cluster"].astype("int").astype("category")
421
+ cluster_summary["Members"] = cluster_summary["Members"].astype("int")
422
+ # inter cluster correlations
423
+ icluster_cor = wcorr(X=pcs,w=row_w,ddof=0)
424
+ # correlation test between variables each principal components
425
+ icluster_cortest = wcorrtest(X=pcs,w=row_w).drop(columns=["test"]).rename(columns={"statistic" : "r","pvalue" : "Pr(>|t|)"})
426
+ icluster_cortest[["variable1","variable2"]] = icluster_cortest[["variable1","variable2"]].astype("int").astype("category")
427
+ icluster_cortest["r**2"] = icluster_cortest["r"]**2
428
+ icluster_cortest["t"] = icluster_cortest["r"]*sqrt(((n_rows-2)/(1- icluster_cortest["r**2"])))
429
+ icluster_cortest = icluster_cortest[["variable1","variable2","r","r**2","t","Pr(>|t|)"]]
430
+
431
+ # convert to dictionary
432
+ cluster_ = {"summary" : cluster_summary,"princomps" : pcs, "cor" : icluster_cor, "cortest" : icluster_cortest}
433
+ # convert to namedtuple
434
+ self.cluster_ = namedtuple("cluster",cluster_.keys())(*cluster_.values())
435
+
436
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
437
+ # linear coefficients and standard scoring coefficients
438
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
439
+ # loadings
440
+ V = concat((Series(cl.v,index=cl.name).to_frame(c) for c, cl in clust_infos.items()),axis=1).astype("float").loc[Z.columns,:]
441
+ V.columns = V.columns.astype("category")
442
+ # coefficients of linear combinations
443
+ coeffs = concat((-(V.T * center / scale).T.sum(axis=0).to_frame("const").T,(V.T / scale).T),axis=0)
444
+ # standardized scoring coefficients
445
+ coeffs_std = V/sqrt(cluster_summary["Variation Explained"].values)
446
+ coeffs_std.columns = coeffs_std.columns.astype("category")
447
+ # convert to dictionary
448
+ coef_ = {"coef" : coeffs,"coef_std" : coeffs_std}
449
+ # convert to namedtuple
450
+ self.coef_ = namedtuple("coef",coef_.keys())(*coef_.values())
451
+
452
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
453
+ # variables informations
454
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
455
+ # cluster squared loadings
456
+ cluster_sqload = DataFrame(index=X.columns,columns=uq_cluster).astype("float")
457
+ for k in X.columns:
458
+ for i, l in enumerate(uq_cluster):
459
+ cluster_sqload.loc[k,l] = coeffsim(X=pcs.iloc[:,i],Y=X[k],w=row_w)
460
+
461
+ # cluster's members : r**2 own cluster, r**2 next closest, ratio ((1-r**2 own)/(1 - r**2 next))
462
+ cluster_member = clust_member(D=cluster_sqload,cluster=cluster,method="max")
463
+
464
+ # inter similarity matrix
465
+ isim = concat((cl.cor2 for _, cl in clust_infos.items()),axis=0).astype("float").loc[X.columns,X.columns]
466
+ # convert to dictionary
467
+ quali_var_ = {"cluster" : cluster, "sqload" : cluster_sqload, "member" : cluster_member, "sim" : isim, "loadings" : V}
468
+ # convert to namedtuple
469
+ self.quali_var_ = namedtuple("quali_var",quali_var_.keys())(*quali_var_.values())
470
+
471
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
472
+ # supplementary variables informations
473
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
474
+ if self.sup_var is not None:
475
+ # correlation of supplementary continuous variables to synthetic scores
476
+ if any(is_numeric_dtype(X_sup_var[k]) for k in sup_var_label):
477
+ var_sup_cor = DataFrame(index=sup_var_label,columns=uq_cluster).astype("float")
478
+ for k in sup_var_label:
479
+ for i, l in enumerate(uq_cluster):
480
+ if is_numeric_dtype(X_sup_var[k]):
481
+ var_sup_cor.loc[k,l] = wpearsonr(x=X_sup_var[k].values,y=pcs.iloc[:,i],w=row_w).statistic
482
+ # remove rows where some columns are missings
483
+ var_sup_cor = var_sup_cor.dropna(subset=[1])
484
+ # correlation test between supplementary variables each principal components
485
+ var_sup_cortest = var_sup_cor.stack().reset_index()
486
+ var_sup_cortest.columns = ["variable","cluster","r"]
487
+ var_sup_cortest["r**2"] = var_sup_cortest["r"]**2
488
+ var_sup_cortest["t"] = var_sup_cortest["r"]*sqrt(((n_rows-2)/(1- var_sup_cortest["r**2"])))
489
+ var_sup_cortest["Pr(>|t|)"] = 2*sst.sf(abs(var_sup_cortest["t"]),n_rows - 2)
490
+ else:
491
+ var_sup_cor, var_sup_cortest = None, None
492
+
493
+ # squared loadings for supplementary variables and cluster
494
+ var_sup_sqload = DataFrame(index=sup_var_label,columns=uq_cluster).astype("float")
495
+ for k in sup_var_label:
496
+ for i, l in enumerate(uq_cluster):
497
+ var_sup_sqload.loc[k,l] = coeffsim(X=self.cluster_.princomps.iloc[:,i],Y=X_sup_var[k],w=self.call_.row_w)
498
+
499
+ # assign cluster to supplementary variables
500
+ var_sup_cluster = var_sup_sqload.idxmax(axis=1).astype("category")
501
+ var_sup_cluster.name = "cluster"
502
+
503
+ # cluster members and R-square values for supplementary variables
504
+ var_sup_member = clust_member(D=var_sup_sqload,cluster=var_sup_cluster,method="max")
505
+
506
+ # convert to dictionary
507
+ var_sup_ = {"cluster" : var_sup_cluster, "cor" : var_sup_cor, "cortest" : var_sup_cortest,
508
+ "sqload" : var_sup_sqload, "member" : var_sup_member}
509
+ # convert to namedtuple
510
+ self.var_sup_ = namedtuple("var_sup",var_sup_.keys())(*var_sup_.values())
511
+
512
+ return self
513
+
514
+ def fit_predict(self,X,y=None):
515
+ """Compute squared loadings and predict cluster index for each column.
516
+
517
+ Convenience method; equivalent to calling fit(X) followed by predict(X).
518
+
519
+ Parameters
520
+ ----------
521
+ X : DataFrame of shape (n_samples, n_columns)
522
+ New data to transform.
523
+
524
+ y : Ignored
525
+ Not used, present here for API consistency by convention.
526
+
527
+ Returns
528
+ -------
529
+ labels : Series of shape (n_columns,)
530
+ Index of the cluster each column belongs to.
531
+ """
532
+ self.fit(X)
533
+ return self.quali_var_.cluster
534
+
535
+ def fit_transform(self,X,y=None):
536
+ """Compute clustering and transform X to cluster-squared loadings space.
537
+
538
+ Equivalent to fit(X).transform(X), but more efficiently implemented.
539
+
540
+ Parameters
541
+ ----------
542
+ X : DataFrame of shape (n_samples, n_columns)
543
+ Training data, where ``n_samples`` in the number of samples
544
+ and ``n_columns`` is the number of columns.
545
+
546
+ y : Ignored
547
+ Not used, present here for API consistency by convention.
548
+
549
+ Returns
550
+ -------
551
+ X_new : DataFrame of shape (n_columns, ncl)
552
+ X transformed in the new space.
553
+ """
554
+ self.fit(X)
555
+ return self.quali_var_.sqload
556
+
557
+ def predict(self,X):
558
+ """Predict the closest cluster each column in X belongs to.
559
+
560
+ Parameters
561
+ ----------
562
+ X : DataFrame of shape (n_samples, n_columns)
563
+ New data to predict, where ``n_samples`` is the number of samples
564
+ and ``n_columns`` is the number of columns.
565
+
566
+ Returns
567
+ -------
568
+ labels : Series of shape (n_columns,)
569
+ Labels of the cluster each column belongs to.
570
+ """
571
+ # distance for new data points to cluster centers
572
+ dist = self.transform(X)
573
+ # assign cluster to new data points
574
+ cluster = dist.idxmax(axis=1).astype("category")
575
+ cluster.name = "cluster"
576
+ return cluster
577
+
578
+ def transform(self,X):
579
+ """Transform X to a cluster-distance space.
580
+
581
+ In the new space, each dimension is the squared loadings to the cluster centers
582
+
583
+ Parameters
584
+ ----------
585
+ X : DataFrame of shape (n_samples, n_columns)
586
+ New data, where ``n_samples`` is the number of samples
587
+ and ``n_columns`` is the number of columns.
588
+
589
+ Returns
590
+ -------
591
+ X_new : DataFrame of shape (n_columns, ncl)
592
+ X transformed in the new space.
593
+ """
594
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
595
+ # check if the estimator is fitted by verifying the presence of fitted attributes
596
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
597
+ check_is_fitted(self)
598
+
599
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
600
+ # convert to pd.DataFrame if pd.Series
601
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
602
+ if isinstance(X,Series):
603
+ X = X.to_frame()
604
+
605
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
606
+ # check if X is an object of class pd.DataFrame
607
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
608
+ if not isinstance(X,DataFrame):
609
+ raise TypeError(f"{type(X)} is not supported. Please convert to a DataFrame with pd.DataFrame.",
610
+ "For more information see: https://pandas.pydata.org/docs/reference/api/pandas.DataFrame.html")
611
+
612
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
613
+ # set index name as None
614
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
615
+ X.index.name = None
616
+
617
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
618
+ # drop level if ndim greater than 1 and reset columns name
619
+ #---------------------------------------------------------------------------------------------------------------------------------------------------------------------
620
+ if X.columns.nlevels > 1:
621
+ X.columns = X.columns.droplevel()
622
+
623
+ # cluster labels
624
+ uq_cluster = self.cluster_.cor.columns
625
+ # squared loadings for new variables
626
+ sqload = DataFrame(index=X.columns,columns=uq_cluster).astype("float")
627
+ for k in X.columns:
628
+ for i, l in enumerate(uq_cluster):
629
+ sqload.loc[k,l] = coeffsim(X=self.cluster_.princomps.iloc[:,i],Y=X[k],w=self.call_.row_w)
630
+ return sqload