lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/ANOVA.py
ADDED
|
@@ -0,0 +1,680 @@
|
|
|
1
|
+
# ANOVA.py — analog of ANOVA.R (one- and two-factor designs).
|
|
2
|
+
#
|
|
3
|
+
# ANOVA(): analysis of variance with the lessR output pipeline —
|
|
4
|
+
# background, descriptive statistics per cell, the ANOVA summary
|
|
5
|
+
# table, association and effect sizes, and Tukey multiple
|
|
6
|
+
# comparisons, plus the plotly graphics. Three designs, from the
|
|
7
|
+
# formula:
|
|
8
|
+
# Y ~ X one-way between groups
|
|
9
|
+
# Y ~ X1 * X2 two-way between groups (crossed)
|
|
10
|
+
# Y ~ X + Block one-way randomized blocks (blocking second)
|
|
11
|
+
# For more complex designs use statsmodels directly.
|
|
12
|
+
#
|
|
13
|
+
# The model is a formula string; the factors are treated as
|
|
14
|
+
# categorical. Rmd= writes a Quarto (.qmd) report (anova_rmd.py,
|
|
15
|
+
# ~ av.Rmd.R). Numerics through statsmodels (ols + anova_lm) and
|
|
16
|
+
# scipy (the studentized range for Tukey), imported lazily. As
|
|
17
|
+
# elsewhere, the pipeline is ported, not the lines. Returns an
|
|
18
|
+
# ANOVAResults object; figures in .plots are not auto-shown.
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
import plotly.graph_objects as go
|
|
23
|
+
|
|
24
|
+
from .plotly_utils import (
|
|
25
|
+
BASE_COLORS, axis_cat, axis_format, axis_num, make_trans,
|
|
26
|
+
plot_border, plotly_style, to_hex, x_grid)
|
|
27
|
+
from .Regression import _getdigits, _prntbl
|
|
28
|
+
from .utils import fmt, get_column, get_option, pretty
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ANOVAResults:
|
|
32
|
+
"""Numeric results and figures of ANOVA(): descriptive
|
|
33
|
+
statistics, the ANOVA table, effect sizes, Tukey comparisons
|
|
34
|
+
(DataFrames/dicts), and the plotly figures in .plots."""
|
|
35
|
+
|
|
36
|
+
def __init__(self, **kw):
|
|
37
|
+
self.__dict__.update(kw)
|
|
38
|
+
|
|
39
|
+
def __repr__(self):
|
|
40
|
+
return (f"<lessPy ANOVA: {self.formula} "
|
|
41
|
+
f"[{self.design}], n={self.n_keep}>")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _anova_formula(my_formula):
|
|
45
|
+
""" "Y ~ X" -> (Y, [X], "oneway"); "Y ~ X1 * X2" ->
|
|
46
|
+
(Y, [X1,X2], "two-between"); "Y ~ X + Block" ->
|
|
47
|
+
(Y, [X,Block], "blocks"). R analog: the design detection of
|
|
48
|
+
ANOVA.R."""
|
|
49
|
+
if not isinstance(my_formula, str):
|
|
50
|
+
raise TypeError(
|
|
51
|
+
"the model is a formula string, such as "
|
|
52
|
+
'"Score ~ Group"')
|
|
53
|
+
if my_formula.count("~") != 1:
|
|
54
|
+
raise ValueError('the formula has one "~": "Y ~ X"')
|
|
55
|
+
lhs, rhs = (s.strip() for s in my_formula.split("~"))
|
|
56
|
+
if not lhs:
|
|
57
|
+
raise ValueError("the formula names the response "
|
|
58
|
+
'before the "~"')
|
|
59
|
+
if "*" in rhs:
|
|
60
|
+
facs = [t.strip() for t in rhs.split("*")]
|
|
61
|
+
design = "two-between"
|
|
62
|
+
elif "+" in rhs:
|
|
63
|
+
facs = [t.strip() for t in rhs.split("+")]
|
|
64
|
+
design = "blocks"
|
|
65
|
+
else:
|
|
66
|
+
facs = [rhs]
|
|
67
|
+
design = "oneway"
|
|
68
|
+
if len(facs) not in (1, 2) or any(not f for f in facs):
|
|
69
|
+
raise ValueError(
|
|
70
|
+
"ANOVA analyzes one or two factors:\n"
|
|
71
|
+
" one-way between groups: Y ~ X\n"
|
|
72
|
+
" one-way randomized blocks: Y ~ X + Blocks\n"
|
|
73
|
+
" two-way between groups: Y ~ X1 * X2")
|
|
74
|
+
return lhs, facs, design
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _levels(s):
|
|
78
|
+
"""Factor levels in order: declared for a Categorical, else
|
|
79
|
+
sorted, as an R factor."""
|
|
80
|
+
from .utils import category_order
|
|
81
|
+
return [str(lv) for lv in category_order(
|
|
82
|
+
s if isinstance(s.dtype, pd.CategoricalDtype)
|
|
83
|
+
else s.astype(str))]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _tukey(levels, means, ns, msw, df_w):
|
|
87
|
+
"""Tukey HSD pairwise comparisons, level_j - level_i for
|
|
88
|
+
i<j in level order, the diff, its 95% family-wise interval,
|
|
89
|
+
and adjusted p from the studentized range. means and ns are
|
|
90
|
+
per level. Matches R's TukeyHSD ordering and signs.
|
|
91
|
+
R analog: TukeyHSD()"""
|
|
92
|
+
from itertools import combinations
|
|
93
|
+
from scipy.stats import studentized_range
|
|
94
|
+
k = len(levels)
|
|
95
|
+
qc = float(studentized_range.ppf(0.95, k, df_w))
|
|
96
|
+
rows, idx = [], []
|
|
97
|
+
for a, b in combinations(levels, 2): # a before b
|
|
98
|
+
diff = means[b] - means[a]
|
|
99
|
+
se = np.sqrt(msw / 2 * (1 / ns[a] + 1 / ns[b]))
|
|
100
|
+
p = float(studentized_range.sf(abs(diff) / se, k, df_w))
|
|
101
|
+
rows.append([diff, diff - qc * se, diff + qc * se, p])
|
|
102
|
+
idx.append(f"{b}-{a}")
|
|
103
|
+
return pd.DataFrame(rows, index=idx,
|
|
104
|
+
columns=["diff", "lwr", "upr", "p adj"])
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _raw_means_ns(yv, gv, levels):
|
|
108
|
+
"""Raw group means and counts, keyed by level."""
|
|
109
|
+
means = {lv: float(yv[gv == lv].mean()) for lv in levels}
|
|
110
|
+
ns = {lv: int((gv == lv).sum()) for lv in levels}
|
|
111
|
+
return means, ns
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _seq_marginal_means(dfo, y_name, fcols, i, gv, levels):
|
|
115
|
+
"""The model.tables marginal means for factor i in the
|
|
116
|
+
sequential (Type I) fit: the first factor is unadjusted
|
|
117
|
+
(raw means), each later factor adjusted for the earlier
|
|
118
|
+
ones. R analog: model.tables(aov, "means"), used by
|
|
119
|
+
TukeyHSD.aov."""
|
|
120
|
+
import statsmodels.formula.api as smf
|
|
121
|
+
g = float(dfo[y_name].mean())
|
|
122
|
+
|
|
123
|
+
def fitted(k):
|
|
124
|
+
if k < 0:
|
|
125
|
+
return np.full(len(dfo), g)
|
|
126
|
+
terms = " + ".join(f"C({fcols[j]})" for j in range(k + 1))
|
|
127
|
+
return smf.ols(f"{y_name} ~ {terms}",
|
|
128
|
+
data=dfo).fit().fittedvalues.to_numpy()
|
|
129
|
+
|
|
130
|
+
proj = fitted(i) - fitted(i - 1)
|
|
131
|
+
return {lv: g + float(proj[gv == lv].mean()) for lv in levels}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def ANOVA(my_formula, data=None, filter=None, brief=False,
|
|
135
|
+
digits_d=None, res_rows=None, res_sort="zresid",
|
|
136
|
+
Rmd=None, Rmd_data=None, Rmd_format="html",
|
|
137
|
+
Rmd_browser=True, jitter_x=0.4, graphics=True):
|
|
138
|
+
"""Analysis of variance of a formula string — one-way
|
|
139
|
+
(Y ~ X), two-way between groups (Y ~ X1 * X2), or randomized
|
|
140
|
+
blocks (Y ~ X + Block). Reports descriptive statistics, the
|
|
141
|
+
ANOVA table, effect sizes, and Tukey comparisons, with the
|
|
142
|
+
plotly graphics. Always prints, as in R; returns an
|
|
143
|
+
ANOVAResults object with the figures in .plots."""
|
|
144
|
+
import statsmodels.formula.api as smf
|
|
145
|
+
from statsmodels.stats.anova import anova_lm
|
|
146
|
+
|
|
147
|
+
if data is None:
|
|
148
|
+
raise ValueError(
|
|
149
|
+
"data= is required: a pandas DataFrame containing "
|
|
150
|
+
"the model's variables")
|
|
151
|
+
if res_sort not in ("zresid", "fitted", "off"):
|
|
152
|
+
raise ValueError('res_sort: "zresid", "fitted", "off"')
|
|
153
|
+
if Rmd is not None:
|
|
154
|
+
if Rmd_format not in ("html", "pdf", "docx", "word",
|
|
155
|
+
"none"):
|
|
156
|
+
raise ValueError('Rmd_format: "html", "pdf", '
|
|
157
|
+
'"docx", or "none"')
|
|
158
|
+
if brief:
|
|
159
|
+
raise ValueError(
|
|
160
|
+
"a Quarto report needs the full analysis, so "
|
|
161
|
+
"Rmd= is not available with brief=True")
|
|
162
|
+
if filter is not None:
|
|
163
|
+
data = data.query(filter)
|
|
164
|
+
|
|
165
|
+
y_name, facs, design = _anova_formula(my_formula)
|
|
166
|
+
formula = (f"{y_name} ~ " + (" * ".join(facs)
|
|
167
|
+
if design == "two-between"
|
|
168
|
+
else " + ".join(facs)))
|
|
169
|
+
|
|
170
|
+
y_ser = get_column(data, y_name, "response")
|
|
171
|
+
if not pd.api.types.is_numeric_dtype(y_ser):
|
|
172
|
+
raise TypeError(f"the response '{y_name}' is numeric")
|
|
173
|
+
fac_sers = [get_column(data, f, "factor") for f in facs]
|
|
174
|
+
|
|
175
|
+
used = pd.concat([y_ser] + fac_sers, axis=1)
|
|
176
|
+
keep = ~used.isna().any(axis=1)
|
|
177
|
+
n_obs = len(data)
|
|
178
|
+
n_keep = int(keep.sum())
|
|
179
|
+
yv = y_ser[keep].to_numpy(dtype=float)
|
|
180
|
+
fvals = [s[keep].astype(str).to_numpy() for s in fac_sers]
|
|
181
|
+
flevs = [_levels(s[keep]) for s in fac_sers]
|
|
182
|
+
row_labels = y_ser[keep].index.astype(str)
|
|
183
|
+
|
|
184
|
+
if digits_d is None:
|
|
185
|
+
digits_d = _getdigits(yv, 2)
|
|
186
|
+
d = digits_d
|
|
187
|
+
|
|
188
|
+
# statsmodels model with categorical factors in level order
|
|
189
|
+
df = pd.DataFrame({y_name: yv})
|
|
190
|
+
fcols = []
|
|
191
|
+
for f, v, lv in zip(facs, fvals, flevs):
|
|
192
|
+
col = f"_f{len(fcols)}"
|
|
193
|
+
df[col] = pd.Categorical(v, categories=lv)
|
|
194
|
+
fcols.append(col)
|
|
195
|
+
rhs = (f"C({fcols[0]})" if design == "oneway"
|
|
196
|
+
else f"C({fcols[0]}) * C({fcols[1]})"
|
|
197
|
+
if design == "two-between"
|
|
198
|
+
else f"C({fcols[0]}) + C({fcols[1]})")
|
|
199
|
+
fit = smf.ols(f"{y_name} ~ {rhs}", data=df).fit()
|
|
200
|
+
|
|
201
|
+
lines = [f"\n BACKGROUND", ""]
|
|
202
|
+
lines += _background(y_name, facs, flevs, n_obs, n_keep,
|
|
203
|
+
design)
|
|
204
|
+
|
|
205
|
+
if design == "oneway":
|
|
206
|
+
out = _oneway(fit, yv, fvals[0], y_name, facs[0],
|
|
207
|
+
flevs[0], n_keep, d, brief, graphics,
|
|
208
|
+
jitter_x, lines, formula, n_obs)
|
|
209
|
+
else:
|
|
210
|
+
out = _twoway(fit, yv, fvals, y_name, facs, flevs,
|
|
211
|
+
n_keep, d, brief, graphics, design, lines,
|
|
212
|
+
formula, n_obs, df, fcols)
|
|
213
|
+
|
|
214
|
+
# residuals listing (shared), unless brief or res_rows=0
|
|
215
|
+
out.residuals = None
|
|
216
|
+
if not brief:
|
|
217
|
+
res_tbl = _residuals_section(
|
|
218
|
+
fit, facs, fvals, y_name, yv, row_labels, n_keep, d,
|
|
219
|
+
res_rows, res_sort, lines)
|
|
220
|
+
out.residuals = res_tbl
|
|
221
|
+
|
|
222
|
+
print("\n".join(lines))
|
|
223
|
+
|
|
224
|
+
if Rmd is not None:
|
|
225
|
+
from .anova_rmd import anova_rmd
|
|
226
|
+
anova_rmd(out, y_name, facs, formula, design,
|
|
227
|
+
list(data.columns), Rmd, Rmd_data, Rmd_format,
|
|
228
|
+
Rmd_browser)
|
|
229
|
+
return out
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _residuals_section(fit, facs, fvals, y_name, yv, row_labels,
|
|
233
|
+
n_keep, d, res_rows, res_sort, lines):
|
|
234
|
+
"""Fitted values, residuals, and internally standardized
|
|
235
|
+
residuals (rstandard), sorted by |z-resid| (or fitted),
|
|
236
|
+
the first res_rows rows. ~ ANOVA.R residuals block"""
|
|
237
|
+
if res_rows is None:
|
|
238
|
+
res_rows = n_keep if n_keep < 20 else 20
|
|
239
|
+
if res_rows == "all":
|
|
240
|
+
res_rows = n_keep
|
|
241
|
+
res_rows = min(int(res_rows), n_keep)
|
|
242
|
+
if res_rows == 0:
|
|
243
|
+
return None
|
|
244
|
+
|
|
245
|
+
zres = fit.get_influence().resid_studentized_internal
|
|
246
|
+
tbl = pd.DataFrame(
|
|
247
|
+
{f: v for f, v in zip(facs, fvals)}, index=row_labels)
|
|
248
|
+
tbl[y_name] = yv
|
|
249
|
+
tbl["fitted"] = np.asarray(fit.fittedvalues)
|
|
250
|
+
tbl["residual"] = np.asarray(fit.resid)
|
|
251
|
+
tbl["z-resid"] = zres
|
|
252
|
+
if res_sort == "zresid":
|
|
253
|
+
tbl = tbl.reindex(
|
|
254
|
+
tbl["z-resid"].abs().sort_values(
|
|
255
|
+
ascending=False).index)
|
|
256
|
+
elif res_sort == "fitted":
|
|
257
|
+
tbl = tbl.reindex(
|
|
258
|
+
tbl["fitted"].abs().sort_values(
|
|
259
|
+
ascending=False).index)
|
|
260
|
+
|
|
261
|
+
lines += ["", "", " RESIDUALS", "",
|
|
262
|
+
"Fitted Values, Residuals, Standardized Residuals"]
|
|
263
|
+
if res_sort == "zresid":
|
|
264
|
+
lines.append(" [sorted by Standardized Residuals, "
|
|
265
|
+
"ignoring + or - sign]")
|
|
266
|
+
elif res_sort == "fitted":
|
|
267
|
+
lines.append(" [sorted by Fitted Value, ignoring "
|
|
268
|
+
"+ or - sign]")
|
|
269
|
+
more = ("cases (rows) of data, or res_rows=\"all\"]"
|
|
270
|
+
if res_rows < n_keep else "]")
|
|
271
|
+
lines.append(f" [res_rows = {res_rows}, out of "
|
|
272
|
+
f"{n_keep} {more}")
|
|
273
|
+
dd = d - 1 if d > 2 else d
|
|
274
|
+
show = tbl.head(res_rows).copy()
|
|
275
|
+
for c in (y_name, "fitted", "residual", "z-resid"):
|
|
276
|
+
show[c] = [fmt(v, dd) for v in show[c]]
|
|
277
|
+
lines += _prntbl(show, dd).split("\n")
|
|
278
|
+
return tbl
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _background(y_name, facs, flevs, n_obs, n_keep, design):
|
|
282
|
+
out = [f"Response Variable: {y_name}", ""]
|
|
283
|
+
for i, (f, lv) in enumerate(zip(facs, flevs), start=1):
|
|
284
|
+
label = ("Factor Variable" if len(facs) == 1
|
|
285
|
+
else f"Factor Variable {i}"
|
|
286
|
+
if design != "blocks"
|
|
287
|
+
else ("Factor of Interest" if i == 1
|
|
288
|
+
else "Blocking Factor"))
|
|
289
|
+
out.append(f"{label}: {f}")
|
|
290
|
+
out.append(f" Levels: {', '.join(lv)}")
|
|
291
|
+
out += ["",
|
|
292
|
+
f"Number of cases (rows) of data: {n_obs}",
|
|
293
|
+
f"Number of cases retained for analysis: {n_keep}"]
|
|
294
|
+
return out
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _desc_oneway(yv, gv, levels, d):
|
|
298
|
+
"""Per-group n, mean, sd, min, max, and the grand mean."""
|
|
299
|
+
rows = []
|
|
300
|
+
for lv in levels:
|
|
301
|
+
v = yv[gv == lv]
|
|
302
|
+
rows.append([int(v.size), v.mean(), v.std(ddof=1),
|
|
303
|
+
v.min(), v.max()])
|
|
304
|
+
tbl = pd.DataFrame(rows, index=levels,
|
|
305
|
+
columns=["n", "mean", "sd", "min", "max"])
|
|
306
|
+
return tbl, float(yv.mean())
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _anova_table(fit, anova_lm, typ, y_name):
|
|
310
|
+
"""The ANOVA table (df, Sum Sq, Mean Sq, F-value, p-value)
|
|
311
|
+
with the factor rows and Residuals, from statsmodels."""
|
|
312
|
+
t = anova_lm(fit, typ=typ)
|
|
313
|
+
t = t.rename(columns={"sum_sq": "Sum Sq", "df": "df",
|
|
314
|
+
"mean_sq": "Mean Sq", "F": "F-value",
|
|
315
|
+
"PR(>F)": "p-value"})
|
|
316
|
+
if "Mean Sq" not in t.columns:
|
|
317
|
+
t["Mean Sq"] = t["Sum Sq"] / t["df"]
|
|
318
|
+
return t[["df", "Sum Sq", "Mean Sq", "F-value", "p-value"]]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _fmt_anova(tbl, labels, d):
|
|
322
|
+
"""Format the ANOVA table as lessR does: factor rows with
|
|
323
|
+
F and p, the Residuals row without."""
|
|
324
|
+
out = []
|
|
325
|
+
w1 = max(len("Residuals"), max(len(x) for x in labels))
|
|
326
|
+
hdr = (" " * w1 + f"{'df':>6}{'Sum Sq':>12}{'Mean Sq':>11}"
|
|
327
|
+
f"{'F-value':>10}{'p-value':>10}")
|
|
328
|
+
out.append(hdr)
|
|
329
|
+
for lbl, (_, r) in zip(labels, tbl.iterrows()):
|
|
330
|
+
line = (f"{lbl:<{w1}}{int(r['df']):>6}"
|
|
331
|
+
f"{fmt(r['Sum Sq'], d):>12}"
|
|
332
|
+
f"{fmt(r['Mean Sq'], d):>11}")
|
|
333
|
+
if lbl != "Residuals":
|
|
334
|
+
line += (f"{fmt(r['F-value'], d):>10}"
|
|
335
|
+
f"{fmt(r['p-value'], 4):>10}")
|
|
336
|
+
out.append(line)
|
|
337
|
+
return out
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _oneway(fit, yv, gv, y_name, x_name, levels, n_keep, d,
|
|
341
|
+
brief, graphics, jitter_x, lines, formula, n_obs):
|
|
342
|
+
from statsmodels.stats.anova import anova_lm
|
|
343
|
+
p = len(levels)
|
|
344
|
+
desc, grand = _desc_oneway(yv, gv, levels, d)
|
|
345
|
+
|
|
346
|
+
lines += ["", " DESCRIPTIVE STATISTICS", ""]
|
|
347
|
+
lines += _prntbl(desc, d).split("\n")
|
|
348
|
+
lines += ["", f"Grand Mean: {fmt(grand, d + 1)}"]
|
|
349
|
+
|
|
350
|
+
tbl = _anova_table(fit, anova_lm, 1, y_name)
|
|
351
|
+
tbl.index = [x_name, "Residuals"]
|
|
352
|
+
lines += ["", "", " ANOVA", "",
|
|
353
|
+
f"-- Summary Table for {y_name}", ""]
|
|
354
|
+
lines += _fmt_anova(tbl, [x_name, "Residuals"], d)
|
|
355
|
+
|
|
356
|
+
ssb = float(tbl.loc[x_name, "Sum Sq"])
|
|
357
|
+
ssw = float(tbl.loc["Residuals", "Sum Sq"])
|
|
358
|
+
msw = float(tbl.loc["Residuals", "Mean Sq"])
|
|
359
|
+
df_w = int(tbl.loc["Residuals", "df"])
|
|
360
|
+
sst = ssb + ssw
|
|
361
|
+
rsq = ssb / sst
|
|
362
|
+
rsq_adj = 1 - ((n_keep - 1) / (n_keep - p)) * (1 - rsq)
|
|
363
|
+
omsq = (ssb - (p - 1) * msw) / (sst + msw)
|
|
364
|
+
lines += ["", "",
|
|
365
|
+
f"-- Association and Effect Size for {y_name}", "",
|
|
366
|
+
f"R Squared: {fmt(rsq, 3)}",
|
|
367
|
+
f"R Sq Adjusted: {fmt(rsq_adj, 3)}",
|
|
368
|
+
f"Omega Squared: {fmt(omsq, 3)}"]
|
|
369
|
+
cohen_f = np.nan
|
|
370
|
+
if omsq > 0:
|
|
371
|
+
cohen_f = np.sqrt(omsq / (1 - omsq))
|
|
372
|
+
lines += ["", f"Cohen's f: {fmt(cohen_f, 3)}"]
|
|
373
|
+
|
|
374
|
+
tukey = None
|
|
375
|
+
lines += ["", "", " TUKEY MULTIPLE COMPARISONS OF MEANS"]
|
|
376
|
+
if not brief:
|
|
377
|
+
means, ns = _raw_means_ns(yv, gv, levels)
|
|
378
|
+
tukey = _tukey(levels, means, ns, msw, df_w)
|
|
379
|
+
lines += ["", "Family-wise Confidence Level: 0.95"]
|
|
380
|
+
lines += _prntbl(tukey, d).split("\n")
|
|
381
|
+
|
|
382
|
+
plots = {}
|
|
383
|
+
if graphics:
|
|
384
|
+
plots["means"] = _anova_means_plot(
|
|
385
|
+
yv, gv, y_name, x_name, levels, jitter_x, d)
|
|
386
|
+
|
|
387
|
+
return ANOVAResults(
|
|
388
|
+
formula=formula, design="oneway", n_obs=n_obs,
|
|
389
|
+
n_keep=n_keep, digits_d=d, response=y_name,
|
|
390
|
+
factors=[x_name], descriptive=desc, grand_mean=grand,
|
|
391
|
+
anova=tbl,
|
|
392
|
+
effects={"R_squared": rsq, "R_sq_adjusted": rsq_adj,
|
|
393
|
+
"omega_squared": omsq, "cohen_f": cohen_f},
|
|
394
|
+
tukey=tukey, plots=plots)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _anova_means_plot(yv, gv, y_name, x_name, levels, jitter_x,
|
|
398
|
+
d):
|
|
399
|
+
"""Scatterplot of the response by factor level, jittered
|
|
400
|
+
points with each cell mean marked and a level line.
|
|
401
|
+
~ .ANOVAz1 means plot"""
|
|
402
|
+
style = plotly_style()
|
|
403
|
+
pos = {lv: i for i, lv in enumerate(levels)}
|
|
404
|
+
rng = np.random.default_rng(0)
|
|
405
|
+
xn = np.array([pos[g] for g in gv], dtype=float)
|
|
406
|
+
xn = xn + rng.uniform(-jitter_x / 2, jitter_x / 2, len(xn))
|
|
407
|
+
pt = get_option("pt_color", "#324E5C")
|
|
408
|
+
fig = go.Figure()
|
|
409
|
+
fig.add_trace(go.Scatter(
|
|
410
|
+
x=xn, y=yv, mode="markers",
|
|
411
|
+
marker=dict(symbol="circle", size=6,
|
|
412
|
+
color=make_trans(pt, 0.85), opacity=1,
|
|
413
|
+
line=dict(color=to_hex(pt), width=1)),
|
|
414
|
+
hoverinfo="y", showlegend=False))
|
|
415
|
+
means = [yv[gv == lv].mean() for lv in levels]
|
|
416
|
+
for i, mval in enumerate(means):
|
|
417
|
+
fig.add_shape(type="line", xref="x", yref="y",
|
|
418
|
+
x0=-0.5, x1=len(levels) - 0.5,
|
|
419
|
+
y0=mval, y1=mval, layer="below",
|
|
420
|
+
line=dict(color=to_hex("gray70"), width=1))
|
|
421
|
+
fig.add_trace(go.Scatter(
|
|
422
|
+
x=list(range(len(levels))), y=means, mode="markers",
|
|
423
|
+
marker=dict(symbol="diamond", size=13,
|
|
424
|
+
color=to_hex(get_option("fit_color",
|
|
425
|
+
"#5C4032"))),
|
|
426
|
+
hoverinfo="x+y", showlegend=False))
|
|
427
|
+
axT2 = pretty(float(yv.min()), float(yv.max()))
|
|
428
|
+
ax_x = axis_cat(x_name)
|
|
429
|
+
ax_x.update(tickmode="array",
|
|
430
|
+
tickvals=list(range(len(levels))),
|
|
431
|
+
ticktext=levels, range=[-0.5, len(levels) - 0.5])
|
|
432
|
+
ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
|
|
433
|
+
ax_y.update(showgrid=True,
|
|
434
|
+
gridcolor=to_hex(style["grid_col"]),
|
|
435
|
+
gridwidth=1, griddash="dot")
|
|
436
|
+
fig.update_layout(
|
|
437
|
+
xaxis=ax_x, yaxis=ax_y, shapes=fig.layout.shapes
|
|
438
|
+
+ tuple(plot_border()), template=None,
|
|
439
|
+
plot_bgcolor=to_hex(style["panel_fill"]),
|
|
440
|
+
paper_bgcolor=to_hex(style["window_fill"]),
|
|
441
|
+
title=dict(text="Scatterplot with Cell Means",
|
|
442
|
+
x=0.5, xanchor="center",
|
|
443
|
+
font=dict(size=round(
|
|
444
|
+
16 * get_option("main_size", 1)))))
|
|
445
|
+
return fig
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _twoway(fit, yv, fvals, y_name, facs, flevs, n_keep, d,
|
|
449
|
+
brief, graphics, design, lines, formula, n_obs, df,
|
|
450
|
+
fcols):
|
|
451
|
+
"""Two-way between groups (Y ~ X1 * X2) or randomized blocks
|
|
452
|
+
(Y ~ X1 + X2). ~ .ANOVAz2"""
|
|
453
|
+
from statsmodels.stats.anova import anova_lm
|
|
454
|
+
bet = design == "two-between"
|
|
455
|
+
x1v, x2v = fvals[0], fvals[1]
|
|
456
|
+
l1, l2 = flevs[0], flevs[1]
|
|
457
|
+
f1, f2 = facs[0], facs[1]
|
|
458
|
+
p, q = len(l1), len(l2)
|
|
459
|
+
dp = pd.DataFrame({"y": yv, "a": x1v, "b": x2v})
|
|
460
|
+
cell_n = dp.pivot_table(index="a", columns="b", values="y",
|
|
461
|
+
aggfunc="size").reindex(
|
|
462
|
+
index=l1, columns=l2)
|
|
463
|
+
balanced = bool(cell_n.stack().nunique() == 1)
|
|
464
|
+
|
|
465
|
+
if bet:
|
|
466
|
+
lines += ["", "Two-way Between Groups ANOVA"]
|
|
467
|
+
else:
|
|
468
|
+
lines += ["", "Randomized Blocks ANOVA",
|
|
469
|
+
f" Factor of Interest: {f1}",
|
|
470
|
+
f" Blocking Factor: {f2}", "",
|
|
471
|
+
f"Note: For the F statistic for {f1} to be "
|
|
472
|
+
"distributed as F, the",
|
|
473
|
+
f" population covariances of {y_name} must "
|
|
474
|
+
"be spherical."]
|
|
475
|
+
|
|
476
|
+
lines += ["", " DESCRIPTIVE STATISTICS", ""]
|
|
477
|
+
if bet and not brief:
|
|
478
|
+
lines += ["-- Cell Sample Sizes", "",
|
|
479
|
+
("Equal cell sizes, so balanced design"
|
|
480
|
+
if balanced else
|
|
481
|
+
"Unequal cell sizes, so unbalanced design")]
|
|
482
|
+
if not balanced:
|
|
483
|
+
lines.append("ANOVA based on Type II Sums of Squares")
|
|
484
|
+
lines += [""]
|
|
485
|
+
cn = cell_n.T.astype(int) # factor2 rows, factor1 cols
|
|
486
|
+
lines += _prntbl(cn, 0, int_cols=list(cn.columns)
|
|
487
|
+
).split("\n")
|
|
488
|
+
cell_m = dp.pivot_table(index="b", columns="a",
|
|
489
|
+
values="y", aggfunc="mean"
|
|
490
|
+
).reindex(index=l2, columns=l1)
|
|
491
|
+
lines += ["", "-- Cell Means", ""]
|
|
492
|
+
lines += _prntbl(cell_m, d).split("\n")
|
|
493
|
+
|
|
494
|
+
# marginal means and grand mean
|
|
495
|
+
m1 = dp.groupby("a")["y"].mean().reindex(l1)
|
|
496
|
+
m2 = dp.groupby("b")["y"].mean().reindex(l2)
|
|
497
|
+
grand = float(yv.mean())
|
|
498
|
+
lines += ["", "-- Marginal Means", "", f1]
|
|
499
|
+
lines += _prntbl(pd.DataFrame([m1.to_numpy()], columns=l1,
|
|
500
|
+
index=[""]), d).split("\n")
|
|
501
|
+
lines += ["", f2]
|
|
502
|
+
lines += _prntbl(pd.DataFrame([m2.to_numpy()], columns=l2,
|
|
503
|
+
index=[""]), d).split("\n")
|
|
504
|
+
lines += ["", f"-- Grand Mean: {fmt(grand, d + 1)}"]
|
|
505
|
+
if bet and not brief:
|
|
506
|
+
cell_s = dp.pivot_table(index="b", columns="a",
|
|
507
|
+
values="y", aggfunc="std"
|
|
508
|
+
).reindex(index=l2, columns=l1)
|
|
509
|
+
lines += ["", "-- Cell Standard Deviations", ""]
|
|
510
|
+
lines += _prntbl(cell_s, d).split("\n")
|
|
511
|
+
|
|
512
|
+
# ANOVA table
|
|
513
|
+
typ = 2 if (bet and not balanced) else 1
|
|
514
|
+
t = anova_lm(fit, typ=typ)
|
|
515
|
+
t = t.rename(columns={"sum_sq": "Sum Sq", "df": "df",
|
|
516
|
+
"PR(>F)": "p-value", "F": "F-value"})
|
|
517
|
+
t["Mean Sq"] = t["Sum Sq"] / t["df"]
|
|
518
|
+
inter = f"{f1}:{f2}"
|
|
519
|
+
ren = {f"C({fcols[0]})": f1, f"C({fcols[1]})": f2,
|
|
520
|
+
f"C({fcols[0]}):C({fcols[1]})": inter,
|
|
521
|
+
"Residual": "Residuals"}
|
|
522
|
+
t.index = [ren.get(i, i) for i in t.index]
|
|
523
|
+
order = ([f1, f2, inter, "Residuals"] if bet
|
|
524
|
+
else [f1, f2, "Residuals"])
|
|
525
|
+
t = t.reindex(order)[["df", "Sum Sq", "Mean Sq", "F-value",
|
|
526
|
+
"p-value"]]
|
|
527
|
+
lines += ["", "", " ANOVA", "", "-- Summary Table"
|
|
528
|
+
+ (" from Type II Sums of Squares"
|
|
529
|
+
if typ == 2 else ""), ""]
|
|
530
|
+
lines += _fmt_anova(t, order, d)
|
|
531
|
+
|
|
532
|
+
msw = float(t.loc["Residuals", "Mean Sq"])
|
|
533
|
+
df_w = int(t.loc["Residuals", "df"])
|
|
534
|
+
|
|
535
|
+
# effect sizes use the Type I (sequential) F-values, as R's
|
|
536
|
+
# summary(aov), even when the displayed table is Type II
|
|
537
|
+
t1 = anova_lm(fit, typ=1).rename(
|
|
538
|
+
columns={"F": "F-value"})
|
|
539
|
+
t1.index = [ren.get(i, i) for i in t1.index]
|
|
540
|
+
fa = float(t1.loc[f1, "F-value"])
|
|
541
|
+
fb = float(t1.loc[f2, "F-value"])
|
|
542
|
+
lines += ["", "", "-- Association and Effect Size", ""]
|
|
543
|
+
eff = {}
|
|
544
|
+
if bet:
|
|
545
|
+
fab = float(t1.loc[inter, "F-value"])
|
|
546
|
+
nh = round(1 / np.mean(1 / cell_n.to_numpy().ravel()))
|
|
547
|
+
oa = ((p - 1) * (fa - 1)) / ((p - 1) * (fa - 1) + nh * p * q)
|
|
548
|
+
ob = ((q - 1) * (fb - 1)) / ((q - 1) * (fb - 1) + nh * p * q)
|
|
549
|
+
oab = (((p - 1) * (q - 1) * (fab - 1))
|
|
550
|
+
/ ((p - 1) * (q - 1) * (fab - 1) + nh * p * q))
|
|
551
|
+
eff = {"omega_sq_" + f1: oa, "omega_sq_" + f2: ob,
|
|
552
|
+
"omega_sq_interaction": oab}
|
|
553
|
+
lines += [f"Partial Omega Squared for {f1}: {fmt(oa, 3)}",
|
|
554
|
+
f"Partial Omega Squared for {f2}: {fmt(ob, 3)}",
|
|
555
|
+
f"Partial Omega Squared for {f1} & {f2}: "
|
|
556
|
+
f"{fmt(oab, 3)}", ""]
|
|
557
|
+
for nm, o in ((f1, oa), (f2, ob), (f"{f1} & {f2}", oab)):
|
|
558
|
+
if o > 0:
|
|
559
|
+
lines.append(f"Cohen's f for {nm}: "
|
|
560
|
+
f"{fmt(np.sqrt(o / (1 - o)), 3)}")
|
|
561
|
+
else:
|
|
562
|
+
oa = ((p - 1) * (fa - 1)) / ((p - 1) * (fa - 1) + q * p)
|
|
563
|
+
intra = (fb - 1) / ((p - 1) + fb)
|
|
564
|
+
eff = {"omega_sq_" + f1: oa, "intraclass_" + f2: intra}
|
|
565
|
+
lines += [f"Partial Omega Squared for {f1}: {fmt(oa, 3)}",
|
|
566
|
+
f"Partial Intraclass Correlation for {f2}: "
|
|
567
|
+
f"{fmt(intra, 3)}", ""]
|
|
568
|
+
if oa > 0:
|
|
569
|
+
lines.append(f"Cohen's f for {f1}: "
|
|
570
|
+
f"{fmt(np.sqrt(oa / (1 - oa)), 3)}")
|
|
571
|
+
if intra > 0:
|
|
572
|
+
lines.append(f"Cohen's f for {f2}: "
|
|
573
|
+
f"{fmt(np.sqrt(intra / (1 - intra)), 3)}")
|
|
574
|
+
|
|
575
|
+
# Tukey
|
|
576
|
+
tukey = None
|
|
577
|
+
lines += ["", "", " TUKEY MULTIPLE COMPARISONS OF MEANS"]
|
|
578
|
+
if not brief:
|
|
579
|
+
tukey = {}
|
|
580
|
+
lines += ["", "Family-wise Confidence Level: 0.95",
|
|
581
|
+
"", f"Factor: {f1}"]
|
|
582
|
+
m1m = _seq_marginal_means(df, y_name, fcols, 0, x1v, l1)
|
|
583
|
+
_, n1 = _raw_means_ns(yv, x1v, l1)
|
|
584
|
+
tukey[f1] = _tukey(l1, m1m, n1, msw, df_w)
|
|
585
|
+
lines += _prntbl(tukey[f1], d).split("\n")
|
|
586
|
+
if bet: # blocks: factor of interest only
|
|
587
|
+
lines += ["", f"Factor: {f2}"]
|
|
588
|
+
m2m = _seq_marginal_means(df, y_name, fcols, 1, x2v,
|
|
589
|
+
l2)
|
|
590
|
+
_, n2 = _raw_means_ns(yv, x2v, l2)
|
|
591
|
+
tukey[f2] = _tukey(l2, m2m, n2, msw, df_w)
|
|
592
|
+
lines += _prntbl(tukey[f2], d).split("\n")
|
|
593
|
+
cl = np.array([f"{a}:{b}" for a, b in zip(x1v, x2v)])
|
|
594
|
+
clv = [f"{a}:{b}" for b in l2 for a in l1]
|
|
595
|
+
clv = [c for c in clv if c in set(cl)]
|
|
596
|
+
cmn, cnn = _raw_means_ns(yv, cl, clv)
|
|
597
|
+
lines += ["", "Cell Means"]
|
|
598
|
+
tukey["cells"] = _tukey(clv, cmn, cnn, msw, df_w)
|
|
599
|
+
lines += _prntbl(tukey["cells"], d).split("\n")
|
|
600
|
+
|
|
601
|
+
plots = {}
|
|
602
|
+
if graphics:
|
|
603
|
+
if bet:
|
|
604
|
+
plots["interaction"] = _interaction_plot(
|
|
605
|
+
dp, y_name, f1, f2, l1, l2, d)
|
|
606
|
+
else:
|
|
607
|
+
plots["data"] = _blocks_plot(
|
|
608
|
+
x1v, yv, x2v, y_name, f1, f2, l1, l2, d,
|
|
609
|
+
"Data Values")
|
|
610
|
+
plots["fitted"] = _blocks_plot(
|
|
611
|
+
x1v, np.asarray(fit.fittedvalues), x2v,
|
|
612
|
+
"Fitted", f1, f2, l1, l2, d, "Fitted Values")
|
|
613
|
+
|
|
614
|
+
return ANOVAResults(
|
|
615
|
+
formula=formula, design=design, n_obs=n_obs,
|
|
616
|
+
n_keep=n_keep, digits_d=d, response=y_name,
|
|
617
|
+
factors=[f1, f2], cell_n=cell_n,
|
|
618
|
+
marginal_means={f1: m1, f2: m2}, grand_mean=grand,
|
|
619
|
+
anova=t, effects=eff, tukey=tukey, plots=plots)
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def _cat_line_fig(title, x_name, y_name, levels_x, series, d,
|
|
623
|
+
legend_title):
|
|
624
|
+
"""A categorical-x line-with-markers figure: one colored
|
|
625
|
+
line per by-group across the x levels. Shared by the
|
|
626
|
+
interaction and blocks plots."""
|
|
627
|
+
style = plotly_style()
|
|
628
|
+
fig = go.Figure()
|
|
629
|
+
ys_all = []
|
|
630
|
+
for gi, (gname, yv_line) in enumerate(series):
|
|
631
|
+
col = to_hex(BASE_COLORS[gi % len(BASE_COLORS)])
|
|
632
|
+
ys_all += [v for v in yv_line if v == v]
|
|
633
|
+
fig.add_trace(go.Scatter(
|
|
634
|
+
x=list(range(len(levels_x))), y=yv_line,
|
|
635
|
+
mode="lines+markers", name=str(gname),
|
|
636
|
+
line=dict(color=col, width=2),
|
|
637
|
+
marker=dict(size=9, color=col),
|
|
638
|
+
connectgaps=True, hoverinfo="x+y+name"))
|
|
639
|
+
axT2 = pretty(min(ys_all), max(ys_all))
|
|
640
|
+
ax_x = axis_cat(x_name)
|
|
641
|
+
ax_x.update(tickmode="array",
|
|
642
|
+
tickvals=list(range(len(levels_x))),
|
|
643
|
+
ticktext=levels_x,
|
|
644
|
+
range=[-0.4, len(levels_x) - 0.6])
|
|
645
|
+
ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
|
|
646
|
+
ax_y.update(showgrid=True,
|
|
647
|
+
gridcolor=to_hex(style["grid_col"]),
|
|
648
|
+
gridwidth=1, griddash="dot")
|
|
649
|
+
fig.update_layout(
|
|
650
|
+
xaxis=ax_x, yaxis=ax_y, shapes=plot_border(),
|
|
651
|
+
template=None, legend=dict(title=dict(text=legend_title)),
|
|
652
|
+
plot_bgcolor=to_hex(style["panel_fill"]),
|
|
653
|
+
paper_bgcolor=to_hex(style["window_fill"]),
|
|
654
|
+
title=dict(text=title, x=0.5, xanchor="center",
|
|
655
|
+
font=dict(size=round(
|
|
656
|
+
16 * get_option("main_size", 1)))))
|
|
657
|
+
return fig
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
def _interaction_plot(dp, y_name, f1, f2, l1, l2, d):
|
|
661
|
+
"""Cell means of the response across factor 1, one line per
|
|
662
|
+
factor 2 level. ~ .ANOVAz2 interaction plot"""
|
|
663
|
+
cm = dp.pivot_table(index="a", columns="b", values="y",
|
|
664
|
+
aggfunc="mean").reindex(index=l1,
|
|
665
|
+
columns=l2)
|
|
666
|
+
series = [(b, [cm.loc[a, b] for a in l1]) for b in l2]
|
|
667
|
+
return _cat_line_fig(f"Cell Means of {y_name}", f1, y_name,
|
|
668
|
+
l1, series, d, f2)
|
|
669
|
+
|
|
670
|
+
|
|
671
|
+
def _blocks_plot(x1v, yvals, x2v, y_name, f1, f2, l1, l2, d,
|
|
672
|
+
title):
|
|
673
|
+
"""Values across the factor of interest, one line per block.
|
|
674
|
+
~ .ANOVAz2 data / fitted plots"""
|
|
675
|
+
dp = pd.DataFrame({"y": yvals, "a": x1v, "b": x2v})
|
|
676
|
+
series = []
|
|
677
|
+
for b in l2:
|
|
678
|
+
sub = dp[dp["b"] == b].set_index("a")["y"]
|
|
679
|
+
series.append((b, [sub.get(a, np.nan) for a in l1]))
|
|
680
|
+
return _cat_line_fig(title, f1, y_name, l1, series, d, f2)
|