clustvartools 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- clustvartools/__init__.py +6 -0
- clustvartools/clustering/__init__.py +24 -0
- clustvartools/clustering/_catclv.py +630 -0
- clustvartools/clustering/_cathcav.py +400 -0
- clustvartools/clustering/_catvarhca.py +416 -0
- clustvartools/clustering/_clv.py +631 -0
- clustvartools/clustering/_clvmix.py +707 -0
- clustvartools/clustering/_corclv.py +552 -0
- clustvartools/clustering/_dclv.py +580 -0
- clustvartools/clustering/_hcav.py +384 -0
- clustvartools/clustering/_hcavmix.py +401 -0
- clustvartools/clustering/functions/__init__.py +0 -0
- clustvartools/clustering/functions/cathca.py +64 -0
- clustvartools/clustering/functions/concat_empty.py +25 -0
- clustvartools/clustering/functions/cov2corr.py +29 -0
- clustvartools/clustering/functions/func_dclv.py +134 -0
- clustvartools/clustering/functions/func_fillna.py +40 -0
- clustvartools/clustering/functions/get_indices.py +43 -0
- clustvartools/clustering/functions/get_sup_label.py +61 -0
- clustvartools/clustering/functions/gsvd.py +107 -0
- clustvartools/clustering/functions/preprocessing.py +69 -0
- clustvartools/clustering/functions/revalue.py +63 -0
- clustvartools/clustering/functions/statistics.py +354 -0
- clustvartools/clustering/functions/tests.py +259 -0
- clustvartools/clustering/functions/utils.py +175 -0
- clustvartools/datasets/__init__.py +447 -0
- clustvartools/datasets/data/__init__.py +0 -0
- clustvartools/datasets/data/apples.xlsx +0 -0
- clustvartools/datasets/data/autos2005.xlsx +0 -0
- clustvartools/datasets/data/burger.xlsx +0 -0
- clustvartools/datasets/data/canines.xlsx +0 -0
- clustvartools/datasets/data/cars.xlsx +0 -0
- clustvartools/datasets/data/congressvotingrecords.xlsx +0 -0
- clustvartools/datasets/data/decathlon.xlsx +0 -0
- clustvartools/datasets/data/jobrate.xlsx +0 -0
- clustvartools/datasets/data/olympic.xlsx +0 -0
- clustvartools/datasets/data/poison.xlsx +0 -0
- clustvartools/datasets/data/uscrime.xlsx +0 -0
- clustvartools/datasets/data/wine.xlsx +0 -0
- clustvartools/graphics/__init__.py +12 -0
- clustvartools/graphics/_fviz_dend.py +562 -0
- clustvartools/graphics/_fviz_dend2.py +600 -0
- clustvartools/graphics/_fviz_height.py +167 -0
- clustvartools/others/__init__.py +29 -0
- clustvartools/others/_clust_diss.py +82 -0
- clustvartools/others/_clust_dist.py +72 -0
- clustvartools/others/_clust_member.py +68 -0
- clustvartools/others/_clust_score.py +55 -0
- clustvartools/others/_coeffsim.py +108 -0
- clustvartools/others/_disjunctive.py +70 -0
- clustvartools/others/_getnnsvar.py +50 -0
- clustvartools/others/_save.py +82 -0
- clustvartools/others/_splitmix.py +69 -0
- clustvartools/others/_sprintf.py +63 -0
- clustvartools/others/_summary.py +189 -0
- clustvartools-0.0.1.dist-info/METADATA +223 -0
- clustvartools-0.0.1.dist-info/RECORD +60 -0
- clustvartools-0.0.1.dist-info/WHEEL +5 -0
- clustvartools-0.0.1.dist-info/licenses/LICENSE +21 -0
- clustvartools-0.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from ._catclv import CatCLV
|
|
5
|
+
from ._cathcav import CatHCAV
|
|
6
|
+
from ._catvarhca import CatVARHCA
|
|
7
|
+
from ._clv import CLV
|
|
8
|
+
from ._clvmix import CLVmix
|
|
9
|
+
from ._corclv import CorCLV
|
|
10
|
+
from ._dclv import DCLV
|
|
11
|
+
from ._hcav import HCAV
|
|
12
|
+
from ._hcavmix import HCAVmix
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"CatCLV",
|
|
16
|
+
"CatHCAV",
|
|
17
|
+
"CatVARHCA",
|
|
18
|
+
"CLV",
|
|
19
|
+
"CLVmix",
|
|
20
|
+
"CorCLV",
|
|
21
|
+
"DCLV",
|
|
22
|
+
"HCAV",
|
|
23
|
+
"HCAVmix"
|
|
24
|
+
]
|
|
@@ -0,0 +1,630 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from numpy import zeros,ones, ndarray, repeat, array, where, c_, sqrt, diag
|
|
3
|
+
from pandas import Series, DataFrame, concat
|
|
4
|
+
from pandas.api.types import is_numeric_dtype
|
|
5
|
+
from itertools import chain
|
|
6
|
+
from collections import namedtuple
|
|
7
|
+
from scipy.cluster.hierarchy import fcluster
|
|
8
|
+
from scipy.stats import t as sst
|
|
9
|
+
from sklearn.base import BaseEstimator, TransformerMixin
|
|
10
|
+
from sklearn.utils.validation import check_is_fitted
|
|
11
|
+
|
|
12
|
+
#interns functions
|
|
13
|
+
from .functions.preprocessing import preprocessing
|
|
14
|
+
from .functions.get_sup_label import get_sup_label
|
|
15
|
+
from .functions.utils import is_all_object_or_category_dtype
|
|
16
|
+
from .functions.statistics import wcorr
|
|
17
|
+
from .functions.tests import wpearsonr, wcorrtest
|
|
18
|
+
from ..others._disjunctive import disjunctive
|
|
19
|
+
from ..others._clust_diss import clust_diss
|
|
20
|
+
from ..others._getnnsvar import getnnsvar
|
|
21
|
+
from ..others._clust_score import clust_score
|
|
22
|
+
from ..others._coeffsim import coeffsim
|
|
23
|
+
from ..others._clust_member import clust_member
|
|
24
|
+
|
|
25
|
+
class CatCLV(BaseEstimator,TransformerMixin):
|
|
26
|
+
"""
|
|
27
|
+
Hierarchical Clustering of Categorical Variables around Latent Variables (CatCLV)
|
|
28
|
+
|
|
29
|
+
Performns ascendant hierarchical clustering of a set of categorical variables. The aggregation criterion is the decrease
|
|
30
|
+
in homogeneity for the clusters being merged. The homogeneity of a cluster is the sum of the correlation ratio
|
|
31
|
+
between the variables and the center of the cluster which is the first principal component of multiple correspondence analysis (MCA).
|
|
32
|
+
Supplementary variables (continuous and/or categorical) may be used.
|
|
33
|
+
|
|
34
|
+
Parameters
|
|
35
|
+
----------
|
|
36
|
+
ncl : int, default = 2
|
|
37
|
+
If a (positive) integer, the tree is cut with ncl clusters. If None, then the ncl is determine using optimal point.
|
|
38
|
+
|
|
39
|
+
row_w : 1d array-like of shape (n_samples,), default = None
|
|
40
|
+
An optional rows weights. The weights are given only for the active rows.
|
|
41
|
+
|
|
42
|
+
sup_var : int, str, list, tuple or range, default = None
|
|
43
|
+
The indexes or names of the supplementary variables (continuous and/or categorical).
|
|
44
|
+
|
|
45
|
+
tol : float, default = 1e-7
|
|
46
|
+
A tolerance threshold to test whether the distance matrix is Euclidean : an eigenvalue is considered positive if it is larger
|
|
47
|
+
than `-tol*lambda1` where `lambda1` is the largest eigenvalue.
|
|
48
|
+
|
|
49
|
+
Attributes
|
|
50
|
+
----------
|
|
51
|
+
call_ : call
|
|
52
|
+
An object containing the summary called parameters with the following attributes:
|
|
53
|
+
|
|
54
|
+
Xtot : DataFrame of shape (n_samples, n_columns + n_columns_sup)
|
|
55
|
+
Input data.
|
|
56
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
57
|
+
Active data.
|
|
58
|
+
dummies : DataFrame of shape (n_samples, n_levels)
|
|
59
|
+
Disjunctive table.
|
|
60
|
+
Z : DataFrame of shape (n_samples, n_levels)
|
|
61
|
+
Standardized data.
|
|
62
|
+
center : Series of shape (n_levels,)
|
|
63
|
+
The proportion of levels.
|
|
64
|
+
scale : Series of shape (n_levels)
|
|
65
|
+
The squared root of center.
|
|
66
|
+
row_w : Series of shape (n_samples,)
|
|
67
|
+
The rows weights.
|
|
68
|
+
ncl : int
|
|
69
|
+
The number of clusters kepted.
|
|
70
|
+
tree : tree
|
|
71
|
+
An object containing the results for the hierarchical agglomerative clustering algorithm, with the following attributes:
|
|
72
|
+
|
|
73
|
+
S : DataFrame of shape (n_columns, n_columns)
|
|
74
|
+
Similarity matrix.
|
|
75
|
+
D : DataFrame of shape (n_columns, n_columns)
|
|
76
|
+
Dissimilarity matrix.
|
|
77
|
+
Z : 2d numpy array of shape (n_columns - 1, 4)
|
|
78
|
+
Linkage matrix.
|
|
79
|
+
height: DataFrame of shape (n_columns - 1, 4)
|
|
80
|
+
Reverse Height of aggregation between ``Z[:,0]`` and ``Z[:,1]``.
|
|
81
|
+
merge : 1d numpy array of shape (n_columns - 1,)
|
|
82
|
+
Height of aggregation between ``Z[:,0]`` and ``Z[:,1]``.
|
|
83
|
+
size : 1d numpy array of shape (n_columns - 1,)
|
|
84
|
+
Number of original observations in the newly formed cluster.
|
|
85
|
+
|
|
86
|
+
sup_var : None, list
|
|
87
|
+
The names of the supplementary variables (continuous and/or categorical).
|
|
88
|
+
|
|
89
|
+
cluster_ : cluster
|
|
90
|
+
An object with the following attributes:
|
|
91
|
+
|
|
92
|
+
summary : DataFrame of shape (ncl, 4)
|
|
93
|
+
Cluster summary.
|
|
94
|
+
cor : DataFrame of shape (ncl, ncl)
|
|
95
|
+
Inter-cluster correlations.
|
|
96
|
+
princomps : DataFrame of shape (n_samples, ncl)
|
|
97
|
+
Synthetic variables , i.e cluster centers or latent variables.
|
|
98
|
+
|
|
99
|
+
coef_ : coef
|
|
100
|
+
An object with the following attributes:
|
|
101
|
+
|
|
102
|
+
coef : DataFrame of shape (n_levels + 1, ncl)
|
|
103
|
+
Coefficients of linear combinations defining the synthetic variable of each cluster.
|
|
104
|
+
coef_std : DataFrame of shape (n_levels, ncl)
|
|
105
|
+
Standardized scoring coefficients.
|
|
106
|
+
|
|
107
|
+
quali_var_ : quali_var
|
|
108
|
+
An object with the following attributes:
|
|
109
|
+
|
|
110
|
+
cluster : Series of shape (n_columns,)
|
|
111
|
+
Labels of the cluster each column belongs to.
|
|
112
|
+
sqload : DataFrame of shape (n_columns, ncl)
|
|
113
|
+
Squared loadings of variables and latent variables.
|
|
114
|
+
member : DataFrame of shape (n_columns, 4)
|
|
115
|
+
Cluster members and R-square values.
|
|
116
|
+
sim : DataFrame of shape (n_columns, n_columns)
|
|
117
|
+
Similarities matrix between variables for each cluster.
|
|
118
|
+
loadings : DataFrame of shape (n_levels, ncl)
|
|
119
|
+
Loadings of variables, i.e. the first eigen vector calculated by MCA.
|
|
120
|
+
|
|
121
|
+
var_sup_ : var_sup, optional
|
|
122
|
+
An object with the following attributes:
|
|
123
|
+
|
|
124
|
+
cluster : Series of shape (n_columns_sup,)
|
|
125
|
+
Labels of the cluster each supplementary column belongs to.
|
|
126
|
+
cor : None or DataFrame of shape (n_quanti_sup, ncl)
|
|
127
|
+
Pearson correlation of supplementary variables and latent variables.
|
|
128
|
+
cortest : None or DataFrame of shape (n_quanti_sup*ncl, 6)
|
|
129
|
+
Pearson correlation test of supplementary variables and latent variables.
|
|
130
|
+
sqload : DataFrame of shape (n_columns_sup, ncl)
|
|
131
|
+
Squared loadings of supplementary variables and latent variables.
|
|
132
|
+
member : DataFrame of shape (n_columns_sup, 4)
|
|
133
|
+
Cluster members and R-square values.
|
|
134
|
+
|
|
135
|
+
References
|
|
136
|
+
----------
|
|
137
|
+
[1] E. Vigneau, M. Qannari. `Clustering of variables around latent components <https://www.researchgate.net/publication/243044470_Clustering_of_Variables_Around_Latent_Components>`_. in Statistics, Simulation and Computation, 32(4), pp.1131-1150, 2003.
|
|
138
|
+
|
|
139
|
+
[2] J. Saracco, M. Chavent, V. Kuentz. `Clustering of categorical variables around latent variables <https://www.researchgate.net/publication/46448264_Clustering_of_categorical_variables_around_latent_variables>`_. Cahiers du GTREThA, N°2010-02, 2010.
|
|
140
|
+
|
|
141
|
+
[3] M. Chavent, V. Kuentz Simonet, B. Liquet, J. Saracco. `ClustOfVar: An R package for the Clustering of Variables <https://www.jstatsoft.org/article/view/v050i13>`_. in Journal of Statistical Software, 50(13), september 2012.
|
|
142
|
+
|
|
143
|
+
[4] Rakotomalala, R. `Classification de variables : classification autour de variables latentes <https://eric.univ-lyon2.fr/ricco/cours/slides/classification_de_variables.pdf>`_. Tutoriel Tanagra pour le Data Mining
|
|
144
|
+
|
|
145
|
+
Examples
|
|
146
|
+
--------
|
|
147
|
+
>>> from clustvartools.datasets import poison
|
|
148
|
+
>>> from clustvartools import CatCLV
|
|
149
|
+
>>> clf = CatCLV(ncl=4,sup_var=range(4))
|
|
150
|
+
>>> clf.fit(poison.data)
|
|
151
|
+
CatCLV(ncl=4,sup_var=range(4))
|
|
152
|
+
"""
|
|
153
|
+
def __init__(
|
|
154
|
+
self,
|
|
155
|
+
ncl = 2,
|
|
156
|
+
row_w = None,
|
|
157
|
+
sup_var = None,
|
|
158
|
+
tol = 1e-7
|
|
159
|
+
):
|
|
160
|
+
self.ncl = ncl
|
|
161
|
+
self.row_w = row_w
|
|
162
|
+
self.sup_var = sup_var
|
|
163
|
+
self.tol = tol
|
|
164
|
+
|
|
165
|
+
def fit(self,X,y=None):
|
|
166
|
+
"""
|
|
167
|
+
Compute CatCLV
|
|
168
|
+
|
|
169
|
+
Parameters
|
|
170
|
+
----------
|
|
171
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
172
|
+
Training data, where ``n_samples`` in the number of samples
|
|
173
|
+
and ``n_columns`` is the number of columns.
|
|
174
|
+
|
|
175
|
+
y : Ignored
|
|
176
|
+
Not used, present here for API consistency by convention.
|
|
177
|
+
|
|
178
|
+
Returns
|
|
179
|
+
-------
|
|
180
|
+
self : object
|
|
181
|
+
Fitted estimator.
|
|
182
|
+
"""
|
|
183
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
184
|
+
# preprocessing
|
|
185
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
186
|
+
X = preprocessing(X=X)
|
|
187
|
+
|
|
188
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
189
|
+
# get supplementary elements labels
|
|
190
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
191
|
+
sup_var_label = get_sup_label(X=X,indexes=self.sup_var,axis=1)
|
|
192
|
+
|
|
193
|
+
# make a copy of the original data
|
|
194
|
+
Xtot = X.copy()
|
|
195
|
+
|
|
196
|
+
# drop supplementary variables columns
|
|
197
|
+
if self.sup_var is not None:
|
|
198
|
+
X_sup_var, X = X.loc[:,sup_var_label], X.drop(columns=sup_var_label)
|
|
199
|
+
|
|
200
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
201
|
+
# hierarchical clustering analysis of mixed data
|
|
202
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
203
|
+
# check if X contains only numerics columns
|
|
204
|
+
if not is_all_object_or_category_dtype(X):
|
|
205
|
+
raise TypeError("Columns in X must be categorical.")
|
|
206
|
+
|
|
207
|
+
# set number of rows and columns
|
|
208
|
+
n_rows, n_cols = X.shape
|
|
209
|
+
|
|
210
|
+
# set individuals weights
|
|
211
|
+
if self.row_w is None:
|
|
212
|
+
row_w = Series(ones(n_rows)/n_rows,index=X.index,name="weight")
|
|
213
|
+
elif not isinstance(self.row_w,(list,tuple,ndarray,Series)):
|
|
214
|
+
raise TypeError("row_w must be a 1d array-like of individuals weights.")
|
|
215
|
+
elif len(self.row_w) != n_rows:
|
|
216
|
+
raise ValueError(f"row_w must be a 1d array-like of shape ({n_rows},).")
|
|
217
|
+
else:
|
|
218
|
+
row_w = Series(array(self.row_w)/sum(self.row_w),index=X.index,name="weight")
|
|
219
|
+
|
|
220
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
221
|
+
#standardization: z_ik = (y_ik - m_k)/s_k
|
|
222
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
223
|
+
# disjunctive table
|
|
224
|
+
dummies = disjunctive(X=X,prefix=True)
|
|
225
|
+
# proportion of levels (=center)
|
|
226
|
+
center = Series((dummies.T * row_w).sum(axis=1).to_numpy(),index=dummies.columns,name = "center")
|
|
227
|
+
# standard deviation (=scale)
|
|
228
|
+
scale = Series(sqrt(center),index=dummies.columns,name = "scale")
|
|
229
|
+
# standardization: z_ik = (y_ik - m_k)/s_k
|
|
230
|
+
Z = (dummies - center)/scale
|
|
231
|
+
|
|
232
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
233
|
+
# data preparation
|
|
234
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
235
|
+
# compute similarity and dissimilarity matrices
|
|
236
|
+
S = DataFrame(diag(ones(n_cols)),index=X.columns,columns=X.columns)
|
|
237
|
+
D = DataFrame(zeros((n_cols,n_cols)),index=X.columns,columns=X.columns)
|
|
238
|
+
for i in range(n_cols-1):
|
|
239
|
+
for j in range(i+1,n_cols):
|
|
240
|
+
k, l = X.columns[i], X.columns[j]
|
|
241
|
+
A = list(Z.columns[Z.columns.str.startswith(k)])
|
|
242
|
+
B = list(Z.columns[Z.columns.str.startswith(l)])
|
|
243
|
+
# similarity matrix
|
|
244
|
+
S.iloc[i,j] = coeffsim(X=X.iloc[:,i],Y=X.iloc[:,j],w=row_w)
|
|
245
|
+
# dissimilarity matrix
|
|
246
|
+
D.iloc[i,j] = clust_diss(X=Z.loc[:,A],Y=Z.loc[:,B],tol=self.tol)
|
|
247
|
+
S.iloc[j,i], D.iloc[j,i] = S.iloc[i,j], D.iloc[i,j]
|
|
248
|
+
|
|
249
|
+
# make a copy of dissimilarity matrix
|
|
250
|
+
Dprim = D.copy()
|
|
251
|
+
|
|
252
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
253
|
+
# ascendant hierarchical cluster analysis from covariance/correlation matrix
|
|
254
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
255
|
+
# initial values
|
|
256
|
+
init = list(range(n_cols)) # initial partition
|
|
257
|
+
init_crit = repeat(1,n_cols) # initial criteria
|
|
258
|
+
maxv = 1e12 # maximum
|
|
259
|
+
a = zeros((n_cols-1))
|
|
260
|
+
b = zeros((n_cols-1))
|
|
261
|
+
ia = zeros((n_cols-1)) # Le premier cluster ou point d'origine sélectionné pour la fusion.
|
|
262
|
+
ib = zeros((n_cols-1)) # Le deuxième cluster ou point d'origine sélectionné pour la fusion.
|
|
263
|
+
lev = zeros((n_cols-1)) # la hauteur entre les deux clusters fusionnés
|
|
264
|
+
card = ones((n_cols)) # cardinalities
|
|
265
|
+
size = zeros((n_cols-1)) # Le nombre total d'observations (points initiaux) contenues dans ce nouveau cluster ainsi formé.
|
|
266
|
+
|
|
267
|
+
# compute nearest neighbor of variables
|
|
268
|
+
nnsvar = getnnsvar(D=D,crit=init_crit)
|
|
269
|
+
# clust matrix
|
|
270
|
+
clustmat = zeros((n_cols,n_cols))
|
|
271
|
+
for i in range(n_cols):
|
|
272
|
+
clustmat[i,n_cols-1] = i
|
|
273
|
+
for ncl in range(n_cols-2,-1,-1):
|
|
274
|
+
# check for agglomerable pair
|
|
275
|
+
minobs = -1
|
|
276
|
+
mindis = maxv
|
|
277
|
+
for i in range(n_cols):
|
|
278
|
+
if init_crit[i] == 1:
|
|
279
|
+
if nnsvar.nndiss[i] < mindis:
|
|
280
|
+
mindis = nnsvar.nndiss[i]
|
|
281
|
+
minobs = i
|
|
282
|
+
# find agglomerands cl1 and cl2, with former < latter
|
|
283
|
+
if minobs < nnsvar.nn[minobs]:
|
|
284
|
+
cl1 = minobs
|
|
285
|
+
cl2 = nnsvar.nn[minobs]
|
|
286
|
+
elif minobs > nnsvar.nn[minobs]:
|
|
287
|
+
cl2 = minobs
|
|
288
|
+
cl1 = nnsvar.nn[minobs]
|
|
289
|
+
# convert to integer
|
|
290
|
+
cl1, cl2 = int(cl1), int(cl2)
|
|
291
|
+
id1 = [i for i,x in enumerate(clustmat[:,ncl+1]) if x == cl1]
|
|
292
|
+
id2 = [i for i,x in enumerate(clustmat[:,ncl+1]) if x == cl2]
|
|
293
|
+
A = [i for i in init if i in id1]
|
|
294
|
+
B = [i for i in init if i in id2]
|
|
295
|
+
clus = [*A,*B]
|
|
296
|
+
# assign to leaf
|
|
297
|
+
a[ncl], b[ncl], size[ncl] = cl1, cl2, len(clus)
|
|
298
|
+
if card[cl1] == 1:
|
|
299
|
+
ia[ncl] = - cl1
|
|
300
|
+
if card[cl2] == 1:
|
|
301
|
+
ib[ncl] = - cl2
|
|
302
|
+
#
|
|
303
|
+
if card[cl1] > 1:
|
|
304
|
+
last_ind = 0
|
|
305
|
+
for i2 in range(n_cols-2,ncl,-1):
|
|
306
|
+
if a[i2] == cl1:
|
|
307
|
+
last_ind = i2
|
|
308
|
+
ia[ncl] = n_cols - last_ind - 1
|
|
309
|
+
if card[cl2] > 1:
|
|
310
|
+
last_ind = 0
|
|
311
|
+
for i2 in range(n_cols-2,ncl,-1):
|
|
312
|
+
if a[i2] == cl2:
|
|
313
|
+
last_ind = i2
|
|
314
|
+
ib[ncl] = n_cols - last_ind - 1
|
|
315
|
+
#
|
|
316
|
+
if ia[ncl] > 0 or ib[ncl] > 0:
|
|
317
|
+
l = min(ia[ncl],ib[ncl])
|
|
318
|
+
if l > 0:
|
|
319
|
+
l -= 1
|
|
320
|
+
r = max(ia[ncl],ib[ncl])
|
|
321
|
+
if r > 0:
|
|
322
|
+
r -= 1
|
|
323
|
+
ia[ncl], ib[ncl] = l, r
|
|
324
|
+
# add height of aggregation
|
|
325
|
+
lev[ncl] = mindis
|
|
326
|
+
# update cluster matrix
|
|
327
|
+
for i in range(n_cols):
|
|
328
|
+
clustmat[i,ncl] = clustmat[i,ncl+1]
|
|
329
|
+
if clustmat[i,ncl] == cl2:
|
|
330
|
+
clustmat[i,ncl] = cl1
|
|
331
|
+
# update dissimilarity matrix in cluster 1
|
|
332
|
+
for i in range(n_cols):
|
|
333
|
+
if (i != cl1) and (i != cl2) and (init_crit[i] == 1):
|
|
334
|
+
idx = [j for j in init if j in [k for k, x in enumerate(clustmat[:,ncl+1]) if x == i]]
|
|
335
|
+
# final label for cluster
|
|
336
|
+
A = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(k)] for k in [X.columns[j] for j in clus]]))
|
|
337
|
+
B = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(k)] for k in [X.columns[j] for j in idx]]))
|
|
338
|
+
# update dissimilarity
|
|
339
|
+
D.iloc[cl1,i] = clust_diss(X=Z.loc[:,A],Y=Z.loc[:,B],tol=self.tol)
|
|
340
|
+
D.iloc[i,cl1] = D.iloc[cl1,i]
|
|
341
|
+
card[cl1] = card[cl1] + card[cl2]
|
|
342
|
+
init_crit[cl2] = 0
|
|
343
|
+
nnsvar.nndiss[cl2] = maxv
|
|
344
|
+
# update dissimilarity matrix in cluster 2
|
|
345
|
+
for i in range(n_cols):
|
|
346
|
+
D.iloc[cl2,i] = maxv
|
|
347
|
+
D.iloc[i,cl2] = D.iloc[cl2,i]
|
|
348
|
+
nnsvar = getnnsvar(D=D,crit=init_crit)
|
|
349
|
+
|
|
350
|
+
# reverse
|
|
351
|
+
merge = array([ia,ib]).T[::-1]
|
|
352
|
+
# convert to python scipy
|
|
353
|
+
linkage = zeros((n_cols-1,4))
|
|
354
|
+
linkage[:,0] = where(merge[:,0] < 0,-merge[:,0],where(merge[:,0]==0,0,merge[:,0]+n_cols))
|
|
355
|
+
linkage[:,1] = where(merge[:,1] < 0,-merge[:,1],where(merge[:,1]==0,n_cols,merge[:,1]+n_cols))
|
|
356
|
+
linkage[:,2] = lev[::-1]
|
|
357
|
+
linkage[:,3] = size[::-1]
|
|
358
|
+
|
|
359
|
+
# height of aggregation
|
|
360
|
+
height = DataFrame(c_[list(range(1,linkage.shape[0]+1)),linkage[:,2][::-1]],columns=["cluster","height"])
|
|
361
|
+
height["diff_1"] = -1*height["height"].diff(1)
|
|
362
|
+
height["diff_2"] = height["diff_1"].diff(-1)
|
|
363
|
+
height["cluster"] = height["cluster"].astype("int")
|
|
364
|
+
|
|
365
|
+
# convert to dictionary
|
|
366
|
+
tree_ = {"S":S,"D":Dprim,"Z":linkage,"height":height,"merge":linkage[:,:2],"size":linkage[:,3]}
|
|
367
|
+
# convert to namedtuple
|
|
368
|
+
tree = namedtuple("tree",tree_.keys())(*tree_.values())
|
|
369
|
+
|
|
370
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
371
|
+
# set numbers of clusters
|
|
372
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
373
|
+
if self.ncl is None:
|
|
374
|
+
ncl = height[height["diff_2"]==height["diff_2"].max()]["cluster"].values[0]
|
|
375
|
+
elif self.ncl < 0:
|
|
376
|
+
raise TypeError("ncl should be a positive integer.")
|
|
377
|
+
elif not isinstance(self.ncl,int):
|
|
378
|
+
raise TypeError("ncl should be an integer")
|
|
379
|
+
else:
|
|
380
|
+
ncl = self.ncl
|
|
381
|
+
|
|
382
|
+
#convert to dictionary
|
|
383
|
+
call_ = {"Xtot":Xtot,"X":X,"dummies" : dummies,"Z":Z,"center": center,"scale":scale,"row_w":row_w,
|
|
384
|
+
"ncl":ncl,"tree":tree,"sup_var":sup_var_label}
|
|
385
|
+
#convert to namedtuple
|
|
386
|
+
self.call_ = namedtuple("call",call_.keys())(*call_.values())
|
|
387
|
+
|
|
388
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
389
|
+
# Informations for variables
|
|
390
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
391
|
+
# assign cluster
|
|
392
|
+
cluster = Series(fcluster(linkage,t=ncl,criterion="maxclust"), index = D.index, name = "cluster",dtype="category")
|
|
393
|
+
# unique cluster
|
|
394
|
+
uq_cluster = sorted(cluster.unique())
|
|
395
|
+
|
|
396
|
+
# cluster infos
|
|
397
|
+
clust_infos = {}
|
|
398
|
+
for k in uq_cluster:
|
|
399
|
+
clus = cluster[cluster==k].index
|
|
400
|
+
A = list(chain.from_iterable([Z.columns[Z.columns.str.startswith(i)] for i in clus]))
|
|
401
|
+
score = clust_score(X=Z.loc[:,A],tol=self.tol)._asdict()
|
|
402
|
+
cor2 = S.loc[clus,clus]
|
|
403
|
+
score = {**score, **{"cluster": clus, "ncl" : len(clus), "name" : A, "cor2" : cor2,}}
|
|
404
|
+
clust_infos[k] = namedtuple("clust_score",score.keys())(*score.values())
|
|
405
|
+
|
|
406
|
+
# principal components - latent components
|
|
407
|
+
pcs = concat((Series(cl.u,index=X.index).to_frame(c) for c, cl in clust_infos.items()),axis=1)
|
|
408
|
+
pcs.columns = pcs.columns.astype("category")
|
|
409
|
+
|
|
410
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
411
|
+
# informations for clusters
|
|
412
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
413
|
+
# cluster summary
|
|
414
|
+
cluster_summary = DataFrame(columns=["Cluster","Members","Variation Explained","Proportion Explained"]).astype("float")
|
|
415
|
+
i = 0
|
|
416
|
+
for c, cl in clust_infos.items():
|
|
417
|
+
cluster_summary.loc[i] = [c, cl.ncl, cl.d, cl.p]
|
|
418
|
+
i += 1
|
|
419
|
+
# convert to category and integer
|
|
420
|
+
cluster_summary["Cluster"] = cluster_summary["Cluster"].astype("int").astype("category")
|
|
421
|
+
cluster_summary["Members"] = cluster_summary["Members"].astype("int")
|
|
422
|
+
# inter cluster correlations
|
|
423
|
+
icluster_cor = wcorr(X=pcs,w=row_w,ddof=0)
|
|
424
|
+
# correlation test between variables each principal components
|
|
425
|
+
icluster_cortest = wcorrtest(X=pcs,w=row_w).drop(columns=["test"]).rename(columns={"statistic" : "r","pvalue" : "Pr(>|t|)"})
|
|
426
|
+
icluster_cortest[["variable1","variable2"]] = icluster_cortest[["variable1","variable2"]].astype("int").astype("category")
|
|
427
|
+
icluster_cortest["r**2"] = icluster_cortest["r"]**2
|
|
428
|
+
icluster_cortest["t"] = icluster_cortest["r"]*sqrt(((n_rows-2)/(1- icluster_cortest["r**2"])))
|
|
429
|
+
icluster_cortest = icluster_cortest[["variable1","variable2","r","r**2","t","Pr(>|t|)"]]
|
|
430
|
+
|
|
431
|
+
# convert to dictionary
|
|
432
|
+
cluster_ = {"summary" : cluster_summary,"princomps" : pcs, "cor" : icluster_cor, "cortest" : icluster_cortest}
|
|
433
|
+
# convert to namedtuple
|
|
434
|
+
self.cluster_ = namedtuple("cluster",cluster_.keys())(*cluster_.values())
|
|
435
|
+
|
|
436
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
437
|
+
# linear coefficients and standard scoring coefficients
|
|
438
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
439
|
+
# loadings
|
|
440
|
+
V = concat((Series(cl.v,index=cl.name).to_frame(c) for c, cl in clust_infos.items()),axis=1).astype("float").loc[Z.columns,:]
|
|
441
|
+
V.columns = V.columns.astype("category")
|
|
442
|
+
# coefficients of linear combinations
|
|
443
|
+
coeffs = concat((-(V.T * center / scale).T.sum(axis=0).to_frame("const").T,(V.T / scale).T),axis=0)
|
|
444
|
+
# standardized scoring coefficients
|
|
445
|
+
coeffs_std = V/sqrt(cluster_summary["Variation Explained"].values)
|
|
446
|
+
coeffs_std.columns = coeffs_std.columns.astype("category")
|
|
447
|
+
# convert to dictionary
|
|
448
|
+
coef_ = {"coef" : coeffs,"coef_std" : coeffs_std}
|
|
449
|
+
# convert to namedtuple
|
|
450
|
+
self.coef_ = namedtuple("coef",coef_.keys())(*coef_.values())
|
|
451
|
+
|
|
452
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
453
|
+
# variables informations
|
|
454
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
455
|
+
# cluster squared loadings
|
|
456
|
+
cluster_sqload = DataFrame(index=X.columns,columns=uq_cluster).astype("float")
|
|
457
|
+
for k in X.columns:
|
|
458
|
+
for i, l in enumerate(uq_cluster):
|
|
459
|
+
cluster_sqload.loc[k,l] = coeffsim(X=pcs.iloc[:,i],Y=X[k],w=row_w)
|
|
460
|
+
|
|
461
|
+
# cluster's members : r**2 own cluster, r**2 next closest, ratio ((1-r**2 own)/(1 - r**2 next))
|
|
462
|
+
cluster_member = clust_member(D=cluster_sqload,cluster=cluster,method="max")
|
|
463
|
+
|
|
464
|
+
# inter similarity matrix
|
|
465
|
+
isim = concat((cl.cor2 for _, cl in clust_infos.items()),axis=0).astype("float").loc[X.columns,X.columns]
|
|
466
|
+
# convert to dictionary
|
|
467
|
+
quali_var_ = {"cluster" : cluster, "sqload" : cluster_sqload, "member" : cluster_member, "sim" : isim, "loadings" : V}
|
|
468
|
+
# convert to namedtuple
|
|
469
|
+
self.quali_var_ = namedtuple("quali_var",quali_var_.keys())(*quali_var_.values())
|
|
470
|
+
|
|
471
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
472
|
+
# supplementary variables informations
|
|
473
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
474
|
+
if self.sup_var is not None:
|
|
475
|
+
# correlation of supplementary continuous variables to synthetic scores
|
|
476
|
+
if any(is_numeric_dtype(X_sup_var[k]) for k in sup_var_label):
|
|
477
|
+
var_sup_cor = DataFrame(index=sup_var_label,columns=uq_cluster).astype("float")
|
|
478
|
+
for k in sup_var_label:
|
|
479
|
+
for i, l in enumerate(uq_cluster):
|
|
480
|
+
if is_numeric_dtype(X_sup_var[k]):
|
|
481
|
+
var_sup_cor.loc[k,l] = wpearsonr(x=X_sup_var[k].values,y=pcs.iloc[:,i],w=row_w).statistic
|
|
482
|
+
# remove rows where some columns are missings
|
|
483
|
+
var_sup_cor = var_sup_cor.dropna(subset=[1])
|
|
484
|
+
# correlation test between supplementary variables each principal components
|
|
485
|
+
var_sup_cortest = var_sup_cor.stack().reset_index()
|
|
486
|
+
var_sup_cortest.columns = ["variable","cluster","r"]
|
|
487
|
+
var_sup_cortest["r**2"] = var_sup_cortest["r"]**2
|
|
488
|
+
var_sup_cortest["t"] = var_sup_cortest["r"]*sqrt(((n_rows-2)/(1- var_sup_cortest["r**2"])))
|
|
489
|
+
var_sup_cortest["Pr(>|t|)"] = 2*sst.sf(abs(var_sup_cortest["t"]),n_rows - 2)
|
|
490
|
+
else:
|
|
491
|
+
var_sup_cor, var_sup_cortest = None, None
|
|
492
|
+
|
|
493
|
+
# squared loadings for supplementary variables and cluster
|
|
494
|
+
var_sup_sqload = DataFrame(index=sup_var_label,columns=uq_cluster).astype("float")
|
|
495
|
+
for k in sup_var_label:
|
|
496
|
+
for i, l in enumerate(uq_cluster):
|
|
497
|
+
var_sup_sqload.loc[k,l] = coeffsim(X=self.cluster_.princomps.iloc[:,i],Y=X_sup_var[k],w=self.call_.row_w)
|
|
498
|
+
|
|
499
|
+
# assign cluster to supplementary variables
|
|
500
|
+
var_sup_cluster = var_sup_sqload.idxmax(axis=1).astype("category")
|
|
501
|
+
var_sup_cluster.name = "cluster"
|
|
502
|
+
|
|
503
|
+
# cluster members and R-square values for supplementary variables
|
|
504
|
+
var_sup_member = clust_member(D=var_sup_sqload,cluster=var_sup_cluster,method="max")
|
|
505
|
+
|
|
506
|
+
# convert to dictionary
|
|
507
|
+
var_sup_ = {"cluster" : var_sup_cluster, "cor" : var_sup_cor, "cortest" : var_sup_cortest,
|
|
508
|
+
"sqload" : var_sup_sqload, "member" : var_sup_member}
|
|
509
|
+
# convert to namedtuple
|
|
510
|
+
self.var_sup_ = namedtuple("var_sup",var_sup_.keys())(*var_sup_.values())
|
|
511
|
+
|
|
512
|
+
return self
|
|
513
|
+
|
|
514
|
+
def fit_predict(self,X,y=None):
|
|
515
|
+
"""Compute squared loadings and predict cluster index for each column.
|
|
516
|
+
|
|
517
|
+
Convenience method; equivalent to calling fit(X) followed by predict(X).
|
|
518
|
+
|
|
519
|
+
Parameters
|
|
520
|
+
----------
|
|
521
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
522
|
+
New data to transform.
|
|
523
|
+
|
|
524
|
+
y : Ignored
|
|
525
|
+
Not used, present here for API consistency by convention.
|
|
526
|
+
|
|
527
|
+
Returns
|
|
528
|
+
-------
|
|
529
|
+
labels : Series of shape (n_columns,)
|
|
530
|
+
Index of the cluster each column belongs to.
|
|
531
|
+
"""
|
|
532
|
+
self.fit(X)
|
|
533
|
+
return self.quali_var_.cluster
|
|
534
|
+
|
|
535
|
+
def fit_transform(self,X,y=None):
|
|
536
|
+
"""Compute clustering and transform X to cluster-squared loadings space.
|
|
537
|
+
|
|
538
|
+
Equivalent to fit(X).transform(X), but more efficiently implemented.
|
|
539
|
+
|
|
540
|
+
Parameters
|
|
541
|
+
----------
|
|
542
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
543
|
+
Training data, where ``n_samples`` in the number of samples
|
|
544
|
+
and ``n_columns`` is the number of columns.
|
|
545
|
+
|
|
546
|
+
y : Ignored
|
|
547
|
+
Not used, present here for API consistency by convention.
|
|
548
|
+
|
|
549
|
+
Returns
|
|
550
|
+
-------
|
|
551
|
+
X_new : DataFrame of shape (n_columns, ncl)
|
|
552
|
+
X transformed in the new space.
|
|
553
|
+
"""
|
|
554
|
+
self.fit(X)
|
|
555
|
+
return self.quali_var_.sqload
|
|
556
|
+
|
|
557
|
+
def predict(self,X):
|
|
558
|
+
"""Predict the closest cluster each column in X belongs to.
|
|
559
|
+
|
|
560
|
+
Parameters
|
|
561
|
+
----------
|
|
562
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
563
|
+
New data to predict, where ``n_samples`` is the number of samples
|
|
564
|
+
and ``n_columns`` is the number of columns.
|
|
565
|
+
|
|
566
|
+
Returns
|
|
567
|
+
-------
|
|
568
|
+
labels : Series of shape (n_columns,)
|
|
569
|
+
Labels of the cluster each column belongs to.
|
|
570
|
+
"""
|
|
571
|
+
# distance for new data points to cluster centers
|
|
572
|
+
dist = self.transform(X)
|
|
573
|
+
# assign cluster to new data points
|
|
574
|
+
cluster = dist.idxmax(axis=1).astype("category")
|
|
575
|
+
cluster.name = "cluster"
|
|
576
|
+
return cluster
|
|
577
|
+
|
|
578
|
+
def transform(self,X):
|
|
579
|
+
"""Transform X to a cluster-distance space.
|
|
580
|
+
|
|
581
|
+
In the new space, each dimension is the squared loadings to the cluster centers
|
|
582
|
+
|
|
583
|
+
Parameters
|
|
584
|
+
----------
|
|
585
|
+
X : DataFrame of shape (n_samples, n_columns)
|
|
586
|
+
New data, where ``n_samples`` is the number of samples
|
|
587
|
+
and ``n_columns`` is the number of columns.
|
|
588
|
+
|
|
589
|
+
Returns
|
|
590
|
+
-------
|
|
591
|
+
X_new : DataFrame of shape (n_columns, ncl)
|
|
592
|
+
X transformed in the new space.
|
|
593
|
+
"""
|
|
594
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
595
|
+
# check if the estimator is fitted by verifying the presence of fitted attributes
|
|
596
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
597
|
+
check_is_fitted(self)
|
|
598
|
+
|
|
599
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
600
|
+
# convert to pd.DataFrame if pd.Series
|
|
601
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
602
|
+
if isinstance(X,Series):
|
|
603
|
+
X = X.to_frame()
|
|
604
|
+
|
|
605
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
606
|
+
# check if X is an object of class pd.DataFrame
|
|
607
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
608
|
+
if not isinstance(X,DataFrame):
|
|
609
|
+
raise TypeError(f"{type(X)} is not supported. Please convert to a DataFrame with pd.DataFrame.",
|
|
610
|
+
"For more information see: https://pandas.pydata.org/docs/reference/api/pandas.DataFrame.html")
|
|
611
|
+
|
|
612
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
613
|
+
# set index name as None
|
|
614
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
615
|
+
X.index.name = None
|
|
616
|
+
|
|
617
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
618
|
+
# drop level if ndim greater than 1 and reset columns name
|
|
619
|
+
#---------------------------------------------------------------------------------------------------------------------------------------------------------------------
|
|
620
|
+
if X.columns.nlevels > 1:
|
|
621
|
+
X.columns = X.columns.droplevel()
|
|
622
|
+
|
|
623
|
+
# cluster labels
|
|
624
|
+
uq_cluster = self.cluster_.cor.columns
|
|
625
|
+
# squared loadings for new variables
|
|
626
|
+
sqload = DataFrame(index=X.columns,columns=uq_cluster).astype("float")
|
|
627
|
+
for k in X.columns:
|
|
628
|
+
for i, l in enumerate(uq_cluster):
|
|
629
|
+
sqload.loc[k,l] = coeffsim(X=self.cluster_.princomps.iloc[:,i],Y=X[k],w=self.call_.row_w)
|
|
630
|
+
return sqload
|