lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/corCFA.py
ADDED
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
# corCFA.py — analog of corCFA.R + cor_mimm.R.
|
|
2
|
+
#
|
|
3
|
+
# corCFA(): a Multiple-Indicator Measurement Model (MIMM)
|
|
4
|
+
# confirmatory factor analysis of a correlation matrix, given a
|
|
5
|
+
# measurement model that assigns items to factors. Reports scale
|
|
6
|
+
# composition, reliability (Cronbach's alpha, McDonald's omega),
|
|
7
|
+
# the item-factor pattern loadings with an indicator diagnostic,
|
|
8
|
+
# the item/factor correlations, the residual correlations, and
|
|
9
|
+
# the lavaan specification, plus a heat map. The core is a direct
|
|
10
|
+
# port of lessR's deterministic .mimm() (sum-of-correlations
|
|
11
|
+
# composites, a communality iteration, then standardization) — no
|
|
12
|
+
# optimization, so it reproduces R exactly.
|
|
13
|
+
#
|
|
14
|
+
# The model is given as a lavaan-style string ("F1 =~ X1 + X2")
|
|
15
|
+
# or as F-keyword lists of item names (F1=["X1","X2"]).
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
|
|
22
|
+
from .utils import fmt, get_option
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _mimm(R, cuts, NItems, NF, Iter):
|
|
26
|
+
"""The MIMM computation. R is the (NItems+NF) square matrix
|
|
27
|
+
with the item correlations in the top-left block and zeros in
|
|
28
|
+
the factor rows/cols; cuts[j] = (start, end) item indices of
|
|
29
|
+
factor j. Returns (expanded correlation matrix, alpha,
|
|
30
|
+
omega). Direct port of cor_mimm.R .mimm()."""
|
|
31
|
+
R = R.copy().astype(float)
|
|
32
|
+
n = NItems + NF
|
|
33
|
+
# item-factor and factor-factor covariances (column sums)
|
|
34
|
+
for I in range(n):
|
|
35
|
+
for J in range(NF):
|
|
36
|
+
L = NItems + J
|
|
37
|
+
s = 0.0
|
|
38
|
+
for K in range(cuts[J][0], cuts[J][1] + 1):
|
|
39
|
+
s += R[I, K]
|
|
40
|
+
R[I, L] = s
|
|
41
|
+
R[L, I] = s
|
|
42
|
+
# Cronbach's alpha
|
|
43
|
+
Alpha = np.zeros(NF)
|
|
44
|
+
for J in range(NF):
|
|
45
|
+
L = NItems + J
|
|
46
|
+
N = cuts[J][1] - cuts[J][0] + 1
|
|
47
|
+
if N == 1:
|
|
48
|
+
Alpha[J] = 1.0
|
|
49
|
+
else:
|
|
50
|
+
DSum = sum(R[I, I]
|
|
51
|
+
for I in range(cuts[J][0], cuts[J][1] + 1))
|
|
52
|
+
XN = float(N)
|
|
53
|
+
RMean = (R[L, L] - DSum) / (XN * (XN - 1.0))
|
|
54
|
+
Alpha[J] = (XN * RMean) / ((XN - 1) * RMean + 1.0)
|
|
55
|
+
# communality iteration
|
|
56
|
+
if Iter >= 0:
|
|
57
|
+
for J in range(NF):
|
|
58
|
+
if Alpha[J] > 0:
|
|
59
|
+
L = NItems + J
|
|
60
|
+
for _ in range(Iter):
|
|
61
|
+
OldFF = R[L, L]
|
|
62
|
+
R[L, L] = 0.0
|
|
63
|
+
for I in range(cuts[J][0], cuts[J][1] + 1):
|
|
64
|
+
OldII = R[I, I]
|
|
65
|
+
R[I, I] = R[I, L] ** 2 / OldFF
|
|
66
|
+
R[I, L] = R[I, L] + (R[I, I] - OldII)
|
|
67
|
+
R[L, L] += R[I, L]
|
|
68
|
+
# standardize to correlations
|
|
69
|
+
for J in range(NF):
|
|
70
|
+
L = NItems + J
|
|
71
|
+
FacSD = np.sqrt(R[L, L])
|
|
72
|
+
for I in range(L + 1):
|
|
73
|
+
R[I, L] /= FacSD
|
|
74
|
+
R[L, I] = R[I, L]
|
|
75
|
+
for I in range(J, NF):
|
|
76
|
+
LI = NItems + I
|
|
77
|
+
R[L, LI] /= FacSD
|
|
78
|
+
R[LI, L] = R[L, LI]
|
|
79
|
+
# McDonald's omega
|
|
80
|
+
Omega = np.zeros(NF)
|
|
81
|
+
for IFac in range(NF):
|
|
82
|
+
FI = NItems + IFac
|
|
83
|
+
SumLam = SumUnq = 0.0
|
|
84
|
+
for J in range(cuts[IFac][0], cuts[IFac][1] + 1):
|
|
85
|
+
Lam = R[FI, J]
|
|
86
|
+
SumLam += Lam
|
|
87
|
+
SumUnq += 1 - Lam ** 2
|
|
88
|
+
Omega[IFac] = SumLam ** 2 / (SumLam ** 2 + SumUnq)
|
|
89
|
+
if Iter == 0:
|
|
90
|
+
Omega[IFac] = 1.0
|
|
91
|
+
return R, Alpha, Omega
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class corCFAResults:
|
|
95
|
+
"""Results of corCFA(): the item-factor loadings (if_cor),
|
|
96
|
+
factor correlations (ff_cor), communalities, alpha, omega,
|
|
97
|
+
predicted and residual correlations, and the plotly heat map
|
|
98
|
+
in .plots."""
|
|
99
|
+
|
|
100
|
+
def __init__(self, **kw):
|
|
101
|
+
self.__dict__.update(kw)
|
|
102
|
+
|
|
103
|
+
def __repr__(self):
|
|
104
|
+
return (f"<lessPy corCFA: {len(self.alpha)} factors, "
|
|
105
|
+
f"{self.if_cor.shape[0]} items>")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _parse_model(model, factors, fkw, cols):
|
|
109
|
+
"""Collect the factor -> item-name lists from a lavaan-style
|
|
110
|
+
string, a factors dict, or F1=/F2= keyword lists."""
|
|
111
|
+
fac = {}
|
|
112
|
+
if model is not None:
|
|
113
|
+
for line in model.splitlines():
|
|
114
|
+
line = line.strip()
|
|
115
|
+
if not line or "=~" not in line:
|
|
116
|
+
continue
|
|
117
|
+
name, rhs = line.split("=~")
|
|
118
|
+
fac[name.strip()] = [t.strip()
|
|
119
|
+
for t in rhs.split("+")]
|
|
120
|
+
if factors is not None:
|
|
121
|
+
fac.update({k: list(v) for k, v in factors.items()})
|
|
122
|
+
for k in sorted(fkw, key=lambda s: int(s[1:])):
|
|
123
|
+
fac[k] = list(fkw[k])
|
|
124
|
+
if not fac:
|
|
125
|
+
raise ValueError(
|
|
126
|
+
"specify the measurement model: a lavaan-style "
|
|
127
|
+
'string ("F1 =~ X1 + X2"), factors={...}, or '
|
|
128
|
+
"F1=[...], F2=[...]")
|
|
129
|
+
for name, items in fac.items():
|
|
130
|
+
bad = [it for it in items if it not in cols]
|
|
131
|
+
if bad:
|
|
132
|
+
raise ValueError(
|
|
133
|
+
f"factor {name}: items not in the correlation "
|
|
134
|
+
f"matrix: {', '.join(bad)}")
|
|
135
|
+
return fac
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def corCFA(R, model=None, factors=None, min_cor=0.10,
|
|
139
|
+
min_res=0.05, iter=50, sort=True, resid=True,
|
|
140
|
+
item_cor=True, heat_map=True, **fkw):
|
|
141
|
+
"""Confirmatory factor analysis (MIMM) of a correlation
|
|
142
|
+
matrix R (a DataFrame or array; raw data is correlated
|
|
143
|
+
first). The measurement model assigns items to factors, given
|
|
144
|
+
as a lavaan-style string, factors={...}, or F1=[...] lists.
|
|
145
|
+
Prints reliability, the loadings and diagnostics, residuals,
|
|
146
|
+
and the lavaan code; returns a corCFAResults. R analog:
|
|
147
|
+
corCFA()"""
|
|
148
|
+
Rm = pd.DataFrame(R)
|
|
149
|
+
vals = Rm.to_numpy(dtype=float)
|
|
150
|
+
if (Rm.shape[0] != Rm.shape[1]
|
|
151
|
+
or not np.allclose(vals, vals.T, atol=1e-8)):
|
|
152
|
+
Rm = Rm.select_dtypes("number").corr()
|
|
153
|
+
Rm.index = Rm.columns = [str(c) for c in Rm.columns]
|
|
154
|
+
Cmat = Rm.to_numpy(dtype=float)
|
|
155
|
+
pos = {c: i for i, c in enumerate(Rm.columns)}
|
|
156
|
+
|
|
157
|
+
fac = _parse_model(model, factors, fkw, set(Rm.columns))
|
|
158
|
+
fac_names = list(fac)
|
|
159
|
+
NF = len(fac_names)
|
|
160
|
+
label, cuts = [], []
|
|
161
|
+
for name in fac_names:
|
|
162
|
+
start = len(label)
|
|
163
|
+
for it in fac[name]:
|
|
164
|
+
label.append(pos[it])
|
|
165
|
+
cuts.append((start, len(label) - 1))
|
|
166
|
+
NItems = len(label)
|
|
167
|
+
|
|
168
|
+
def run(lab):
|
|
169
|
+
sub = Cmat[np.ix_(lab, lab)]
|
|
170
|
+
E = np.zeros((NItems + NF, NItems + NF))
|
|
171
|
+
E[:NItems, :NItems] = sub
|
|
172
|
+
return _mimm(E, cuts, NItems, NF, iter)
|
|
173
|
+
|
|
174
|
+
Rout, alpha, omega = run(label)
|
|
175
|
+
if sort:
|
|
176
|
+
newlab = []
|
|
177
|
+
for f in range(NF):
|
|
178
|
+
n1, n2 = cuts[f]
|
|
179
|
+
order = sorted(range(n1, n2 + 1),
|
|
180
|
+
key=lambda j: -Rout[NItems + f, j])
|
|
181
|
+
newlab += [label[j] for j in order]
|
|
182
|
+
label = newlab
|
|
183
|
+
Rout, alpha, omega = run(label)
|
|
184
|
+
|
|
185
|
+
items = [Rm.columns[i] for i in label]
|
|
186
|
+
fcols = fac_names
|
|
187
|
+
if_cor = pd.DataFrame(Rout[:NItems, NItems:], index=items,
|
|
188
|
+
columns=fcols)
|
|
189
|
+
ff_cor = pd.DataFrame(Rout[NItems:, NItems:], index=fcols,
|
|
190
|
+
columns=fcols)
|
|
191
|
+
commun = pd.Series(np.diag(Rout[:NItems, :NItems]),
|
|
192
|
+
index=items)
|
|
193
|
+
|
|
194
|
+
lines = _cfa_text(items, fcols, cuts, alpha, omega, iter,
|
|
195
|
+
if_cor, Rout, NItems, NF)
|
|
196
|
+
pred = res = None
|
|
197
|
+
if resid:
|
|
198
|
+
pred, res = _residuals(if_cor, ff_cor, Cmat, label,
|
|
199
|
+
items, cuts, NItems, NF, lines)
|
|
200
|
+
lines += _lavaan(fcols, cuts, items)
|
|
201
|
+
print("\n".join(lines))
|
|
202
|
+
|
|
203
|
+
plots = {}
|
|
204
|
+
if heat_map:
|
|
205
|
+
plots["heatmap"] = _cfa_heatmap(Rout, items, fcols)
|
|
206
|
+
return corCFAResults(
|
|
207
|
+
if_cor=if_cor, ff_cor=ff_cor, communalities=commun,
|
|
208
|
+
alpha=alpha, omega=omega, pred=pred, resid=res,
|
|
209
|
+
plots=plots)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _cfa_text(items, fcols, cuts, alpha, omega, iter, if_cor,
|
|
213
|
+
Rout, NItems, NF):
|
|
214
|
+
L = ["", " FACTOR / SCALE COMPOSITION", ""]
|
|
215
|
+
for f, name in enumerate(fcols):
|
|
216
|
+
members = items[cuts[f][0]:cuts[f][1] + 1]
|
|
217
|
+
L.append(f"{name}: " + " ".join(members))
|
|
218
|
+
L += ["", "", " RELIABILITY ANALYSIS", ""]
|
|
219
|
+
w = max(6, max(len(s) for s in fcols))
|
|
220
|
+
hdr = f"{'Scale':<{w}}{'Alpha':>9}"
|
|
221
|
+
if iter > 0:
|
|
222
|
+
hdr += f"{'Omega':>9}"
|
|
223
|
+
L.append(hdr)
|
|
224
|
+
for f, name in enumerate(fcols):
|
|
225
|
+
row = f"{name:<{w}}{fmt(alpha[f], 3):>9}"
|
|
226
|
+
if iter > 0:
|
|
227
|
+
row += f"{fmt(omega[f], 3):>9}"
|
|
228
|
+
L.append(row)
|
|
229
|
+
if iter > 0:
|
|
230
|
+
iw = max(10, max(len(s) for s in items) + 1)
|
|
231
|
+
L += ["", "", " SOLUTION", "", "Indicator Analysis", "",
|
|
232
|
+
f"{'Factor':<8}{'Indicator':<{iw}}"
|
|
233
|
+
f"{'Pattern':>9}{'Unique':>9} Diagnostics"]
|
|
234
|
+
for f, name in enumerate(fcols):
|
|
235
|
+
for it_i in range(cuts[f][0], cuts[f][1] + 1):
|
|
236
|
+
item = items[it_i]
|
|
237
|
+
lam = Rout[NItems + f, it_i]
|
|
238
|
+
unique = 1 - lam ** 2
|
|
239
|
+
diag = ""
|
|
240
|
+
if lam <= 0:
|
|
241
|
+
diag = "** Negative loading on own factor **"
|
|
242
|
+
unq = " xxxx"
|
|
243
|
+
elif unique <= 0:
|
|
244
|
+
diag = ("** Improper loading **"
|
|
245
|
+
if cuts[f][1] > cuts[f][0]
|
|
246
|
+
else "** Factor defined by one item **")
|
|
247
|
+
unq = fmt(unique, 3)
|
|
248
|
+
else:
|
|
249
|
+
unq = fmt(unique, 3)
|
|
250
|
+
bad = [g + 1 for g in range(NF)
|
|
251
|
+
if abs(Rout[NItems + g, it_i]) > lam]
|
|
252
|
+
if bad:
|
|
253
|
+
diag = " ".join(f"F{b}" for b in bad)
|
|
254
|
+
L.append(f"F{f + 1:<7}{item:<{iw}}"
|
|
255
|
+
f"{fmt(lam, 3):>9}{unq:>9} {diag}")
|
|
256
|
+
return L
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _residuals(if_cor, ff_cor, Cmat, label, items, cuts, NItems,
|
|
260
|
+
NF, lines):
|
|
261
|
+
lam = np.zeros((NItems, NF))
|
|
262
|
+
for f in range(NF):
|
|
263
|
+
for i in range(cuts[f][0], cuts[f][1] + 1):
|
|
264
|
+
lam[i, f] = if_cor.iloc[i, f]
|
|
265
|
+
phi = ff_cor.to_numpy()
|
|
266
|
+
estR = lam @ phi @ lam.T
|
|
267
|
+
np.fill_diagonal(estR, 1.0)
|
|
268
|
+
Ritems = Cmat[np.ix_(label, label)]
|
|
269
|
+
resR = np.round(Ritems - estR, 5)
|
|
270
|
+
np.fill_diagonal(resR, 0.0)
|
|
271
|
+
pred = pd.DataFrame(np.round(estR, 5), index=items,
|
|
272
|
+
columns=items)
|
|
273
|
+
res = pd.DataFrame(resR, index=items, columns=items)
|
|
274
|
+
|
|
275
|
+
ssq_tot = float((resR ** 2).sum())
|
|
276
|
+
n = NItems
|
|
277
|
+
rmsr = np.sqrt(ssq_tot / (n * n - n))
|
|
278
|
+
avg_abs = float(np.abs(resR).sum()) / (n * n - n)
|
|
279
|
+
lines += ["", "", " RESIDUALS", "",
|
|
280
|
+
f"Total sum of squares for all items: "
|
|
281
|
+
f"{fmt(ssq_tot, 3)}",
|
|
282
|
+
f"Root mean square residual (RMSR): {fmt(rmsr, 3)}",
|
|
283
|
+
f"Average absolute residual off-diagonal: "
|
|
284
|
+
f"{fmt(avg_abs, 3)}"]
|
|
285
|
+
return pred, res
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _lavaan(fcols, cuts, items):
|
|
289
|
+
L = ["", "", " LAVAAN / semopy SPECIFICATION", "",
|
|
290
|
+
"MeasModel = \"\"\""]
|
|
291
|
+
for f, name in enumerate(fcols):
|
|
292
|
+
members = items[cuts[f][0]:cuts[f][1] + 1]
|
|
293
|
+
L.append(f" {name} =~ " + " + ".join(members))
|
|
294
|
+
L += ["\"\"\"",
|
|
295
|
+
"# lavaan (R): cfa(MeasModel, data=d, std.lv=True)",
|
|
296
|
+
"# semopy (Python): Model(MeasModel).fit(d)"]
|
|
297
|
+
return L
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _cfa_heatmap(Rout, items, fcols):
|
|
301
|
+
import plotly.graph_objects as go
|
|
302
|
+
from .plotly_utils import plotly_style, to_hex
|
|
303
|
+
labels = items + fcols
|
|
304
|
+
style = plotly_style()
|
|
305
|
+
fig = go.Figure(go.Heatmap(
|
|
306
|
+
z=Rout, x=labels, y=labels, zmin=-1, zmax=1,
|
|
307
|
+
colorscale="RdBu", reversescale=True,
|
|
308
|
+
colorbar=dict(title="r")))
|
|
309
|
+
fig.update_yaxes(autorange="reversed")
|
|
310
|
+
fig.update_layout(
|
|
311
|
+
template=None, paper_bgcolor=to_hex(style["window_fill"]),
|
|
312
|
+
title=dict(text="Item / Factor Correlations", x=0.5,
|
|
313
|
+
xanchor="center",
|
|
314
|
+
font=dict(size=round(
|
|
315
|
+
16 * get_option("main_size", 1)))))
|
|
316
|
+
return fig
|
lessPy/corEFA.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
# corEFA.py — analog of corEFA.R.
|
|
2
|
+
#
|
|
3
|
+
# corEFA(): exploratory factor analysis of a correlation matrix,
|
|
4
|
+
# maximum-likelihood extraction (a direct port of R's factanal
|
|
5
|
+
# ML algorithm) with promax, varimax, or no rotation (ports of
|
|
6
|
+
# R's promax()/varimax()). Reports the sorted loadings (small
|
|
7
|
+
# ones blanked), the sum-of-squares table, and a generated
|
|
8
|
+
# measurement-model string assigning each item to the factor of
|
|
9
|
+
# its largest loading, plus any items deleted for loading below
|
|
10
|
+
# min_loading. Returns a corEFAResults object.
|
|
11
|
+
#
|
|
12
|
+
# The numerics reproduce R to ~4 decimals: same optimizer
|
|
13
|
+
# objective/gradient, same Kaiser-normalized varimax, same
|
|
14
|
+
# promax power=4 target. Input is a correlation matrix (a
|
|
15
|
+
# DataFrame or array); raw data is correlated first.
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
from .utils import fmt
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _factanal(S, q):
|
|
24
|
+
"""Maximum-likelihood factor extraction: optimize the
|
|
25
|
+
uniquenesses, then form the loadings from the top-q
|
|
26
|
+
eigenvectors of the scaled correlation matrix. Direct port of
|
|
27
|
+
R's factanal.fit.mle (FAfn/FAgr/FAout)."""
|
|
28
|
+
from scipy.optimize import minimize
|
|
29
|
+
p = S.shape[0]
|
|
30
|
+
|
|
31
|
+
def eig(psi):
|
|
32
|
+
sc = 1 / np.sqrt(psi)
|
|
33
|
+
v, V = np.linalg.eigh(sc[:, None] * S * sc[None, :])
|
|
34
|
+
return v[::-1], V[:, ::-1] # descending
|
|
35
|
+
|
|
36
|
+
def out(psi):
|
|
37
|
+
v, V = eig(psi)
|
|
38
|
+
load = V[:, :q] * np.sqrt(np.maximum(v[:q] - 1, 0))
|
|
39
|
+
return np.sqrt(psi)[:, None] * load
|
|
40
|
+
|
|
41
|
+
def fn(psi):
|
|
42
|
+
v, _ = eig(psi)
|
|
43
|
+
e = v[q:]
|
|
44
|
+
return -(np.sum(np.log(e) - e) - q + p)
|
|
45
|
+
|
|
46
|
+
def gr(psi):
|
|
47
|
+
load = out(psi)
|
|
48
|
+
g = load @ load.T + np.diag(psi) - S
|
|
49
|
+
return np.diag(g) / psi ** 2
|
|
50
|
+
|
|
51
|
+
start = (1 - 0.5 * q / p) / np.diag(np.linalg.inv(S))
|
|
52
|
+
res = minimize(fn, start, jac=gr, method="L-BFGS-B",
|
|
53
|
+
bounds=[(0.005, 1)] * p)
|
|
54
|
+
return out(res.x), bool(res.success)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _sort_extract(L):
|
|
58
|
+
"""factanal's sortLoadings: order factors by descending SS,
|
|
59
|
+
flip each column so its loadings sum to a positive value."""
|
|
60
|
+
L = L[:, np.argsort(-(L ** 2).sum(0))]
|
|
61
|
+
neg = L.sum(0) < 0
|
|
62
|
+
L[:, neg] *= -1
|
|
63
|
+
return L
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _varimax(x, normalize=True, eps=1e-5):
|
|
67
|
+
"""Kaiser-normalized varimax. Port of R's stats::varimax."""
|
|
68
|
+
nc = x.shape[1]
|
|
69
|
+
if nc < 2:
|
|
70
|
+
return x
|
|
71
|
+
x = x.copy()
|
|
72
|
+
if normalize:
|
|
73
|
+
sc = np.sqrt((x ** 2).sum(1))
|
|
74
|
+
x = x / sc[:, None]
|
|
75
|
+
p, TT, d = x.shape[0], np.eye(nc), 0.0
|
|
76
|
+
for _ in range(1000):
|
|
77
|
+
z = x @ TT
|
|
78
|
+
B = x.T @ (z ** 3 - z @ np.diag((z ** 2).sum(0)) / p)
|
|
79
|
+
U, s, Vt = np.linalg.svd(B)
|
|
80
|
+
TT = U @ Vt
|
|
81
|
+
d, dpast = s.sum(), d
|
|
82
|
+
if d < dpast * (1 + eps):
|
|
83
|
+
break
|
|
84
|
+
z = x @ TT
|
|
85
|
+
return z * sc[:, None] if normalize else z
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _promax(x, m=4):
|
|
89
|
+
"""Promax oblique rotation, power m=4. Port of R's
|
|
90
|
+
stats::promax (varimax, then oblique target)."""
|
|
91
|
+
if x.shape[1] < 2:
|
|
92
|
+
return x
|
|
93
|
+
xv = _varimax(x)
|
|
94
|
+
Q = xv * np.abs(xv) ** (m - 1)
|
|
95
|
+
U = np.linalg.lstsq(xv, Q, rcond=None)[0]
|
|
96
|
+
U = U @ np.diag(np.sqrt(np.diag(np.linalg.inv(U.T @ U))))
|
|
97
|
+
return xv @ U
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class corEFAResults:
|
|
101
|
+
"""Results of corEFA(): the sorted loadings DataFrame, the
|
|
102
|
+
sum-of-squares table, the measurement model string, and the
|
|
103
|
+
convergence flag."""
|
|
104
|
+
|
|
105
|
+
def __init__(self, **kw):
|
|
106
|
+
self.__dict__.update(kw)
|
|
107
|
+
|
|
108
|
+
def __repr__(self):
|
|
109
|
+
return (f"<lessPy corEFA: {self.n_factors} factors, "
|
|
110
|
+
f"{self.loadings.shape[0]} items>")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def corEFA(R, n_factors, rotate="promax", min_loading=0.2,
|
|
114
|
+
sort=True):
|
|
115
|
+
"""Exploratory factor analysis of a correlation matrix R
|
|
116
|
+
(a DataFrame or array; raw data is correlated first).
|
|
117
|
+
Maximum-likelihood extraction with rotate= "promax" (default),
|
|
118
|
+
"varimax", or "none". Prints the loadings, the sum-of-squares
|
|
119
|
+
table, and the measurement-model code; returns a
|
|
120
|
+
corEFAResults. R analog: corEFA()"""
|
|
121
|
+
if rotate not in ("promax", "varimax", "none"):
|
|
122
|
+
raise ValueError('rotate: "promax", "varimax", "none"')
|
|
123
|
+
Rm = pd.DataFrame(R)
|
|
124
|
+
vals = Rm.to_numpy(dtype=float)
|
|
125
|
+
if (Rm.shape[0] != Rm.shape[1]
|
|
126
|
+
or not np.allclose(vals, vals.T, atol=1e-8)):
|
|
127
|
+
Rm = Rm.select_dtypes("number").corr() # raw data
|
|
128
|
+
names = [str(c) for c in Rm.columns]
|
|
129
|
+
S = Rm.to_numpy(dtype=float)
|
|
130
|
+
|
|
131
|
+
Lam, converged = _factanal(S, n_factors)
|
|
132
|
+
Lam = _sort_extract(Lam)
|
|
133
|
+
if n_factors > 1 and rotate != "none":
|
|
134
|
+
Lam = _promax(Lam) if rotate == "promax" else _varimax(Lam)
|
|
135
|
+
|
|
136
|
+
fac = [f"Factor{i + 1}" for i in range(n_factors)]
|
|
137
|
+
ld = pd.DataFrame(Lam, index=names, columns=fac)
|
|
138
|
+
|
|
139
|
+
if sort and n_factors > 1:
|
|
140
|
+
top = np.argmax(np.abs(ld.to_numpy()), axis=1)
|
|
141
|
+
order = []
|
|
142
|
+
for f in range(n_factors):
|
|
143
|
+
items = [i for i in range(len(ld)) if top[i] == f]
|
|
144
|
+
items.sort(key=lambda i: -abs(ld.iloc[i, f]))
|
|
145
|
+
order += items
|
|
146
|
+
ld = ld.iloc[order]
|
|
147
|
+
|
|
148
|
+
print("\n".join(_efa_output(ld, n_factors, rotate,
|
|
149
|
+
min_loading, converged)))
|
|
150
|
+
model, deleted = _cfa_model(ld, n_factors, min_loading)
|
|
151
|
+
ss = _ss_table(ld, n_factors)
|
|
152
|
+
return corEFAResults(
|
|
153
|
+
loadings=ld, ss=ss, model=model, deleted=deleted,
|
|
154
|
+
converged=converged, n_factors=n_factors)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _ss_table(ld, n_factors):
|
|
158
|
+
v = (ld ** 2).sum(axis=0)
|
|
159
|
+
n = ld.shape[0]
|
|
160
|
+
rows = {"SS loadings": v, "Proportion Var": v / n}
|
|
161
|
+
if n_factors > 1:
|
|
162
|
+
rows["Cumulative Var"] = np.cumsum(v / n)
|
|
163
|
+
return pd.DataFrame(rows).T
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _efa_output(ld, n_factors, rotate, min_loading, converged):
|
|
167
|
+
L = ["", " EXPLORATORY FACTOR ANALYSIS", "",
|
|
168
|
+
"Extraction: maximum likelihood"]
|
|
169
|
+
if n_factors > 1:
|
|
170
|
+
L.append(f"Rotation: {rotate}")
|
|
171
|
+
if not converged:
|
|
172
|
+
L.append(">>> Warning: extraction did not fully converge")
|
|
173
|
+
L += ["", f"Loadings (except -{min_loading} to "
|
|
174
|
+
f"{min_loading})", ""]
|
|
175
|
+
w = max(len(s) for s in ld.index)
|
|
176
|
+
L.append(" " * w + "".join(f"{c:>10}" for c in ld.columns))
|
|
177
|
+
for item, row in ld.iterrows():
|
|
178
|
+
cells = "".join(
|
|
179
|
+
(f"{fmt(v, 3):>10}" if abs(v) >= min_loading
|
|
180
|
+
else " " * 10) for v in row)
|
|
181
|
+
L.append(f"{item:<{w}}{cells}")
|
|
182
|
+
# sum of squares
|
|
183
|
+
ss = _ss_table(ld, n_factors)
|
|
184
|
+
L += ["", "Sum of Squares", ""]
|
|
185
|
+
w2 = max(len(s) for s in ss.index)
|
|
186
|
+
L.append(" " * w2 + "".join(f"{c:>10}" for c in ss.columns))
|
|
187
|
+
for lbl, row in ss.iterrows():
|
|
188
|
+
L.append(f"{lbl:<{w2}}"
|
|
189
|
+
+ "".join(f"{fmt(v, 3):>10}" for v in row))
|
|
190
|
+
return L
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _cfa_model(ld, n_factors, min_loading):
|
|
194
|
+
"""Assign each item to the factor of its largest loading (if
|
|
195
|
+
above min_loading), build the F# =~ items measurement model
|
|
196
|
+
string, and collect deleted items. ~ corEFA MIMM code."""
|
|
197
|
+
lv = ld.to_numpy()
|
|
198
|
+
items = list(ld.index)
|
|
199
|
+
assign = []
|
|
200
|
+
for i in range(len(items)):
|
|
201
|
+
j = int(np.argmax(np.abs(lv[i])))
|
|
202
|
+
assign.append(j if abs(lv[i, j]) > min_loading else -1)
|
|
203
|
+
|
|
204
|
+
lines = []
|
|
205
|
+
for f in range(n_factors):
|
|
206
|
+
members = [items[i] for i in range(len(items))
|
|
207
|
+
if assign[i] == f]
|
|
208
|
+
if members:
|
|
209
|
+
lines.append(f" F{f + 1} =~ " + " + ".join(members))
|
|
210
|
+
model = "\n".join(lines)
|
|
211
|
+
|
|
212
|
+
deleted = [items[i] for i in range(len(items))
|
|
213
|
+
if assign[i] == -1]
|
|
214
|
+
print("\n".join(
|
|
215
|
+
["", " MEASUREMENT MODEL (for confirmatory analysis)",
|
|
216
|
+
"", model]
|
|
217
|
+
+ (["", f"Deletion threshold: min_loading = {min_loading}",
|
|
218
|
+
"Deleted items: " + " ".join(deleted)]
|
|
219
|
+
if deleted else [])))
|
|
220
|
+
return model, deleted
|
lessPy/corPrint.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# corPrint.py — analog of corPrint.R.
|
|
2
|
+
#
|
|
3
|
+
# corPrint(): print a correlation matrix in lessR's compact
|
|
4
|
+
# style — two decimals with the leading "0." stripped (0.85 ->
|
|
5
|
+
# "85", -0.07 -> "-07"), the diagonal 1.00 shown as "100", and
|
|
6
|
+
# any coefficient with |r| < min_value blanked. Prints the
|
|
7
|
+
# formatted matrix and returns it as a text string.
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _compact(v):
|
|
14
|
+
s = f"{v:.2f}".replace("0.", "") # 0.85 -> "85"
|
|
15
|
+
return s.replace("1.00", "100").replace("-1.00", "-100")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def corPrint(R, min_value=0):
|
|
19
|
+
"""Print a correlation matrix R (a DataFrame or array; raw
|
|
20
|
+
data is correlated first) in the compact lessR style, blanking
|
|
21
|
+
coefficients with |r| < min_value. Returns the formatted text.
|
|
22
|
+
R analog: corPrint()"""
|
|
23
|
+
Rm = pd.DataFrame(R)
|
|
24
|
+
vals = Rm.to_numpy(dtype=float)
|
|
25
|
+
if (Rm.shape[0] != Rm.shape[1]
|
|
26
|
+
or not np.allclose(vals, vals.T, atol=1e-8)):
|
|
27
|
+
Rm = Rm.select_dtypes("number").corr()
|
|
28
|
+
vals = Rm.to_numpy(dtype=float)
|
|
29
|
+
names = [str(c) for c in Rm.columns]
|
|
30
|
+
n = len(names)
|
|
31
|
+
lab_w = max(len(s) for s in names) + 2
|
|
32
|
+
col_w = [max(4, len(nm) + 1) for nm in names]
|
|
33
|
+
|
|
34
|
+
lines = [" " * lab_w + "".join(
|
|
35
|
+
nm.rjust(col_w[j]) for j, nm in enumerate(names))]
|
|
36
|
+
for i in range(n):
|
|
37
|
+
row = (" " + names[i]).rjust(lab_w)
|
|
38
|
+
for j in range(n):
|
|
39
|
+
v = vals[i, j]
|
|
40
|
+
cell = "" if abs(v) < min_value else _compact(v)
|
|
41
|
+
row += cell.rjust(col_w[j])
|
|
42
|
+
lines.append(row)
|
|
43
|
+
text = "\n".join(lines)
|
|
44
|
+
print(text)
|
|
45
|
+
return text
|
lessPy/corProp.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# corProp.py — analog of corProp.R.
|
|
2
|
+
#
|
|
3
|
+
# corProp(): the item proportionality matrix — for each pair of
|
|
4
|
+
# items, how similarly they correlate with all the OTHER items
|
|
5
|
+
# (the correlation of their two correlation profiles, excluding
|
|
6
|
+
# the pair itself). A high value means two items are nearly
|
|
7
|
+
# interchangeable indicators, useful in scale construction.
|
|
8
|
+
# Returns the proportionality matrix (rounded to 2 as in R); the
|
|
9
|
+
# heat map is on the returned frame's .attrs["plots"].
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
import pandas as pd
|
|
13
|
+
|
|
14
|
+
from .utils import get_option
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def corProp(R, main=None, heat_map=True):
|
|
18
|
+
"""Item proportionality coefficients of a correlation matrix
|
|
19
|
+
R (a DataFrame or array; raw data is correlated first): the
|
|
20
|
+
profile similarity of each item pair across the other items.
|
|
21
|
+
Returns the proportionality matrix DataFrame (rounded to 2,
|
|
22
|
+
as R); its .attrs["plots"] holds the heat map. R analog:
|
|
23
|
+
corProp()"""
|
|
24
|
+
Rm = pd.DataFrame(R)
|
|
25
|
+
vals = Rm.to_numpy(dtype=float)
|
|
26
|
+
if (Rm.shape[0] != Rm.shape[1]
|
|
27
|
+
or not np.allclose(vals, vals.T, atol=1e-8)):
|
|
28
|
+
Rm = Rm.select_dtypes("number").corr()
|
|
29
|
+
Rm.index = Rm.columns = [str(c) for c in Rm.columns]
|
|
30
|
+
S = Rm.to_numpy(dtype=float)
|
|
31
|
+
|
|
32
|
+
# Diag[i] = sum_k R[k,i]^2 ; RR = R @ R (the cross products).
|
|
33
|
+
# For a pair (i,j) exclude i and j from both the cross product
|
|
34
|
+
# and the two sums of squares, then normalize. ~ corProp.R
|
|
35
|
+
diag = np.diag(S)
|
|
36
|
+
Diag = (S ** 2).sum(axis=0)
|
|
37
|
+
RR = S @ S
|
|
38
|
+
cross = RR - S * (diag[:, None] + diag[None, :])
|
|
39
|
+
D1 = Diag[:, None] - (diag[:, None] ** 2 + S ** 2)
|
|
40
|
+
D2 = Diag[None, :] - (diag[None, :] ** 2 + S ** 2)
|
|
41
|
+
with np.errstate(invalid="ignore", divide="ignore"):
|
|
42
|
+
P = cross / np.sqrt(D1 * D2)
|
|
43
|
+
np.fill_diagonal(P, 1.0)
|
|
44
|
+
P = np.round(P, 2)
|
|
45
|
+
|
|
46
|
+
out = pd.DataFrame(P, index=Rm.index, columns=Rm.columns)
|
|
47
|
+
plots = {}
|
|
48
|
+
if heat_map:
|
|
49
|
+
plots["heatmap"] = _heatmap(out, main)
|
|
50
|
+
out.attrs["plots"] = plots
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _heatmap(df, main):
|
|
55
|
+
import plotly.graph_objects as go
|
|
56
|
+
|
|
57
|
+
from .plotly_utils import plotly_style, to_hex
|
|
58
|
+
labels = list(df.columns)
|
|
59
|
+
Z = df.to_numpy(dtype=float).copy()
|
|
60
|
+
np.fill_diagonal(Z, np.nan) # ignore the diagonal
|
|
61
|
+
style = plotly_style()
|
|
62
|
+
fig = go.Figure(go.Heatmap(
|
|
63
|
+
z=Z, x=labels, y=labels, zmin=-1, zmax=1,
|
|
64
|
+
colorscale="RdBu", reversescale=True,
|
|
65
|
+
colorbar=dict(title="prop")))
|
|
66
|
+
fig.update_yaxes(autorange="reversed")
|
|
67
|
+
fig.update_layout(
|
|
68
|
+
template=None, paper_bgcolor=to_hex(style["window_fill"]),
|
|
69
|
+
title=dict(text=main or "Item Proportionalities", x=0.5,
|
|
70
|
+
xanchor="center",
|
|
71
|
+
font=dict(size=round(
|
|
72
|
+
16 * get_option("main_size", 1)))))
|
|
73
|
+
return fig
|