lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/ANOVA.py ADDED
@@ -0,0 +1,680 @@
1
+ # ANOVA.py — analog of ANOVA.R (one- and two-factor designs).
2
+ #
3
+ # ANOVA(): analysis of variance with the lessR output pipeline —
4
+ # background, descriptive statistics per cell, the ANOVA summary
5
+ # table, association and effect sizes, and Tukey multiple
6
+ # comparisons, plus the plotly graphics. Three designs, from the
7
+ # formula:
8
+ # Y ~ X one-way between groups
9
+ # Y ~ X1 * X2 two-way between groups (crossed)
10
+ # Y ~ X + Block one-way randomized blocks (blocking second)
11
+ # For more complex designs use statsmodels directly.
12
+ #
13
+ # The model is a formula string; the factors are treated as
14
+ # categorical. Rmd= writes a Quarto (.qmd) report (anova_rmd.py,
15
+ # ~ av.Rmd.R). Numerics through statsmodels (ols + anova_lm) and
16
+ # scipy (the studentized range for Tukey), imported lazily. As
17
+ # elsewhere, the pipeline is ported, not the lines. Returns an
18
+ # ANOVAResults object; figures in .plots are not auto-shown.
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+ import plotly.graph_objects as go
23
+
24
+ from .plotly_utils import (
25
+ BASE_COLORS, axis_cat, axis_format, axis_num, make_trans,
26
+ plot_border, plotly_style, to_hex, x_grid)
27
+ from .Regression import _getdigits, _prntbl
28
+ from .utils import fmt, get_column, get_option, pretty
29
+
30
+
31
+ class ANOVAResults:
32
+ """Numeric results and figures of ANOVA(): descriptive
33
+ statistics, the ANOVA table, effect sizes, Tukey comparisons
34
+ (DataFrames/dicts), and the plotly figures in .plots."""
35
+
36
+ def __init__(self, **kw):
37
+ self.__dict__.update(kw)
38
+
39
+ def __repr__(self):
40
+ return (f"<lessPy ANOVA: {self.formula} "
41
+ f"[{self.design}], n={self.n_keep}>")
42
+
43
+
44
+ def _anova_formula(my_formula):
45
+ """ "Y ~ X" -> (Y, [X], "oneway"); "Y ~ X1 * X2" ->
46
+ (Y, [X1,X2], "two-between"); "Y ~ X + Block" ->
47
+ (Y, [X,Block], "blocks"). R analog: the design detection of
48
+ ANOVA.R."""
49
+ if not isinstance(my_formula, str):
50
+ raise TypeError(
51
+ "the model is a formula string, such as "
52
+ '"Score ~ Group"')
53
+ if my_formula.count("~") != 1:
54
+ raise ValueError('the formula has one "~": "Y ~ X"')
55
+ lhs, rhs = (s.strip() for s in my_formula.split("~"))
56
+ if not lhs:
57
+ raise ValueError("the formula names the response "
58
+ 'before the "~"')
59
+ if "*" in rhs:
60
+ facs = [t.strip() for t in rhs.split("*")]
61
+ design = "two-between"
62
+ elif "+" in rhs:
63
+ facs = [t.strip() for t in rhs.split("+")]
64
+ design = "blocks"
65
+ else:
66
+ facs = [rhs]
67
+ design = "oneway"
68
+ if len(facs) not in (1, 2) or any(not f for f in facs):
69
+ raise ValueError(
70
+ "ANOVA analyzes one or two factors:\n"
71
+ " one-way between groups: Y ~ X\n"
72
+ " one-way randomized blocks: Y ~ X + Blocks\n"
73
+ " two-way between groups: Y ~ X1 * X2")
74
+ return lhs, facs, design
75
+
76
+
77
+ def _levels(s):
78
+ """Factor levels in order: declared for a Categorical, else
79
+ sorted, as an R factor."""
80
+ from .utils import category_order
81
+ return [str(lv) for lv in category_order(
82
+ s if isinstance(s.dtype, pd.CategoricalDtype)
83
+ else s.astype(str))]
84
+
85
+
86
+ def _tukey(levels, means, ns, msw, df_w):
87
+ """Tukey HSD pairwise comparisons, level_j - level_i for
88
+ i<j in level order, the diff, its 95% family-wise interval,
89
+ and adjusted p from the studentized range. means and ns are
90
+ per level. Matches R's TukeyHSD ordering and signs.
91
+ R analog: TukeyHSD()"""
92
+ from itertools import combinations
93
+ from scipy.stats import studentized_range
94
+ k = len(levels)
95
+ qc = float(studentized_range.ppf(0.95, k, df_w))
96
+ rows, idx = [], []
97
+ for a, b in combinations(levels, 2): # a before b
98
+ diff = means[b] - means[a]
99
+ se = np.sqrt(msw / 2 * (1 / ns[a] + 1 / ns[b]))
100
+ p = float(studentized_range.sf(abs(diff) / se, k, df_w))
101
+ rows.append([diff, diff - qc * se, diff + qc * se, p])
102
+ idx.append(f"{b}-{a}")
103
+ return pd.DataFrame(rows, index=idx,
104
+ columns=["diff", "lwr", "upr", "p adj"])
105
+
106
+
107
+ def _raw_means_ns(yv, gv, levels):
108
+ """Raw group means and counts, keyed by level."""
109
+ means = {lv: float(yv[gv == lv].mean()) for lv in levels}
110
+ ns = {lv: int((gv == lv).sum()) for lv in levels}
111
+ return means, ns
112
+
113
+
114
+ def _seq_marginal_means(dfo, y_name, fcols, i, gv, levels):
115
+ """The model.tables marginal means for factor i in the
116
+ sequential (Type I) fit: the first factor is unadjusted
117
+ (raw means), each later factor adjusted for the earlier
118
+ ones. R analog: model.tables(aov, "means"), used by
119
+ TukeyHSD.aov."""
120
+ import statsmodels.formula.api as smf
121
+ g = float(dfo[y_name].mean())
122
+
123
+ def fitted(k):
124
+ if k < 0:
125
+ return np.full(len(dfo), g)
126
+ terms = " + ".join(f"C({fcols[j]})" for j in range(k + 1))
127
+ return smf.ols(f"{y_name} ~ {terms}",
128
+ data=dfo).fit().fittedvalues.to_numpy()
129
+
130
+ proj = fitted(i) - fitted(i - 1)
131
+ return {lv: g + float(proj[gv == lv].mean()) for lv in levels}
132
+
133
+
134
+ def ANOVA(my_formula, data=None, filter=None, brief=False,
135
+ digits_d=None, res_rows=None, res_sort="zresid",
136
+ Rmd=None, Rmd_data=None, Rmd_format="html",
137
+ Rmd_browser=True, jitter_x=0.4, graphics=True):
138
+ """Analysis of variance of a formula string — one-way
139
+ (Y ~ X), two-way between groups (Y ~ X1 * X2), or randomized
140
+ blocks (Y ~ X + Block). Reports descriptive statistics, the
141
+ ANOVA table, effect sizes, and Tukey comparisons, with the
142
+ plotly graphics. Always prints, as in R; returns an
143
+ ANOVAResults object with the figures in .plots."""
144
+ import statsmodels.formula.api as smf
145
+ from statsmodels.stats.anova import anova_lm
146
+
147
+ if data is None:
148
+ raise ValueError(
149
+ "data= is required: a pandas DataFrame containing "
150
+ "the model's variables")
151
+ if res_sort not in ("zresid", "fitted", "off"):
152
+ raise ValueError('res_sort: "zresid", "fitted", "off"')
153
+ if Rmd is not None:
154
+ if Rmd_format not in ("html", "pdf", "docx", "word",
155
+ "none"):
156
+ raise ValueError('Rmd_format: "html", "pdf", '
157
+ '"docx", or "none"')
158
+ if brief:
159
+ raise ValueError(
160
+ "a Quarto report needs the full analysis, so "
161
+ "Rmd= is not available with brief=True")
162
+ if filter is not None:
163
+ data = data.query(filter)
164
+
165
+ y_name, facs, design = _anova_formula(my_formula)
166
+ formula = (f"{y_name} ~ " + (" * ".join(facs)
167
+ if design == "two-between"
168
+ else " + ".join(facs)))
169
+
170
+ y_ser = get_column(data, y_name, "response")
171
+ if not pd.api.types.is_numeric_dtype(y_ser):
172
+ raise TypeError(f"the response '{y_name}' is numeric")
173
+ fac_sers = [get_column(data, f, "factor") for f in facs]
174
+
175
+ used = pd.concat([y_ser] + fac_sers, axis=1)
176
+ keep = ~used.isna().any(axis=1)
177
+ n_obs = len(data)
178
+ n_keep = int(keep.sum())
179
+ yv = y_ser[keep].to_numpy(dtype=float)
180
+ fvals = [s[keep].astype(str).to_numpy() for s in fac_sers]
181
+ flevs = [_levels(s[keep]) for s in fac_sers]
182
+ row_labels = y_ser[keep].index.astype(str)
183
+
184
+ if digits_d is None:
185
+ digits_d = _getdigits(yv, 2)
186
+ d = digits_d
187
+
188
+ # statsmodels model with categorical factors in level order
189
+ df = pd.DataFrame({y_name: yv})
190
+ fcols = []
191
+ for f, v, lv in zip(facs, fvals, flevs):
192
+ col = f"_f{len(fcols)}"
193
+ df[col] = pd.Categorical(v, categories=lv)
194
+ fcols.append(col)
195
+ rhs = (f"C({fcols[0]})" if design == "oneway"
196
+ else f"C({fcols[0]}) * C({fcols[1]})"
197
+ if design == "two-between"
198
+ else f"C({fcols[0]}) + C({fcols[1]})")
199
+ fit = smf.ols(f"{y_name} ~ {rhs}", data=df).fit()
200
+
201
+ lines = [f"\n BACKGROUND", ""]
202
+ lines += _background(y_name, facs, flevs, n_obs, n_keep,
203
+ design)
204
+
205
+ if design == "oneway":
206
+ out = _oneway(fit, yv, fvals[0], y_name, facs[0],
207
+ flevs[0], n_keep, d, brief, graphics,
208
+ jitter_x, lines, formula, n_obs)
209
+ else:
210
+ out = _twoway(fit, yv, fvals, y_name, facs, flevs,
211
+ n_keep, d, brief, graphics, design, lines,
212
+ formula, n_obs, df, fcols)
213
+
214
+ # residuals listing (shared), unless brief or res_rows=0
215
+ out.residuals = None
216
+ if not brief:
217
+ res_tbl = _residuals_section(
218
+ fit, facs, fvals, y_name, yv, row_labels, n_keep, d,
219
+ res_rows, res_sort, lines)
220
+ out.residuals = res_tbl
221
+
222
+ print("\n".join(lines))
223
+
224
+ if Rmd is not None:
225
+ from .anova_rmd import anova_rmd
226
+ anova_rmd(out, y_name, facs, formula, design,
227
+ list(data.columns), Rmd, Rmd_data, Rmd_format,
228
+ Rmd_browser)
229
+ return out
230
+
231
+
232
+ def _residuals_section(fit, facs, fvals, y_name, yv, row_labels,
233
+ n_keep, d, res_rows, res_sort, lines):
234
+ """Fitted values, residuals, and internally standardized
235
+ residuals (rstandard), sorted by |z-resid| (or fitted),
236
+ the first res_rows rows. ~ ANOVA.R residuals block"""
237
+ if res_rows is None:
238
+ res_rows = n_keep if n_keep < 20 else 20
239
+ if res_rows == "all":
240
+ res_rows = n_keep
241
+ res_rows = min(int(res_rows), n_keep)
242
+ if res_rows == 0:
243
+ return None
244
+
245
+ zres = fit.get_influence().resid_studentized_internal
246
+ tbl = pd.DataFrame(
247
+ {f: v for f, v in zip(facs, fvals)}, index=row_labels)
248
+ tbl[y_name] = yv
249
+ tbl["fitted"] = np.asarray(fit.fittedvalues)
250
+ tbl["residual"] = np.asarray(fit.resid)
251
+ tbl["z-resid"] = zres
252
+ if res_sort == "zresid":
253
+ tbl = tbl.reindex(
254
+ tbl["z-resid"].abs().sort_values(
255
+ ascending=False).index)
256
+ elif res_sort == "fitted":
257
+ tbl = tbl.reindex(
258
+ tbl["fitted"].abs().sort_values(
259
+ ascending=False).index)
260
+
261
+ lines += ["", "", " RESIDUALS", "",
262
+ "Fitted Values, Residuals, Standardized Residuals"]
263
+ if res_sort == "zresid":
264
+ lines.append(" [sorted by Standardized Residuals, "
265
+ "ignoring + or - sign]")
266
+ elif res_sort == "fitted":
267
+ lines.append(" [sorted by Fitted Value, ignoring "
268
+ "+ or - sign]")
269
+ more = ("cases (rows) of data, or res_rows=\"all\"]"
270
+ if res_rows < n_keep else "]")
271
+ lines.append(f" [res_rows = {res_rows}, out of "
272
+ f"{n_keep} {more}")
273
+ dd = d - 1 if d > 2 else d
274
+ show = tbl.head(res_rows).copy()
275
+ for c in (y_name, "fitted", "residual", "z-resid"):
276
+ show[c] = [fmt(v, dd) for v in show[c]]
277
+ lines += _prntbl(show, dd).split("\n")
278
+ return tbl
279
+
280
+
281
+ def _background(y_name, facs, flevs, n_obs, n_keep, design):
282
+ out = [f"Response Variable: {y_name}", ""]
283
+ for i, (f, lv) in enumerate(zip(facs, flevs), start=1):
284
+ label = ("Factor Variable" if len(facs) == 1
285
+ else f"Factor Variable {i}"
286
+ if design != "blocks"
287
+ else ("Factor of Interest" if i == 1
288
+ else "Blocking Factor"))
289
+ out.append(f"{label}: {f}")
290
+ out.append(f" Levels: {', '.join(lv)}")
291
+ out += ["",
292
+ f"Number of cases (rows) of data: {n_obs}",
293
+ f"Number of cases retained for analysis: {n_keep}"]
294
+ return out
295
+
296
+
297
+ def _desc_oneway(yv, gv, levels, d):
298
+ """Per-group n, mean, sd, min, max, and the grand mean."""
299
+ rows = []
300
+ for lv in levels:
301
+ v = yv[gv == lv]
302
+ rows.append([int(v.size), v.mean(), v.std(ddof=1),
303
+ v.min(), v.max()])
304
+ tbl = pd.DataFrame(rows, index=levels,
305
+ columns=["n", "mean", "sd", "min", "max"])
306
+ return tbl, float(yv.mean())
307
+
308
+
309
+ def _anova_table(fit, anova_lm, typ, y_name):
310
+ """The ANOVA table (df, Sum Sq, Mean Sq, F-value, p-value)
311
+ with the factor rows and Residuals, from statsmodels."""
312
+ t = anova_lm(fit, typ=typ)
313
+ t = t.rename(columns={"sum_sq": "Sum Sq", "df": "df",
314
+ "mean_sq": "Mean Sq", "F": "F-value",
315
+ "PR(>F)": "p-value"})
316
+ if "Mean Sq" not in t.columns:
317
+ t["Mean Sq"] = t["Sum Sq"] / t["df"]
318
+ return t[["df", "Sum Sq", "Mean Sq", "F-value", "p-value"]]
319
+
320
+
321
+ def _fmt_anova(tbl, labels, d):
322
+ """Format the ANOVA table as lessR does: factor rows with
323
+ F and p, the Residuals row without."""
324
+ out = []
325
+ w1 = max(len("Residuals"), max(len(x) for x in labels))
326
+ hdr = (" " * w1 + f"{'df':>6}{'Sum Sq':>12}{'Mean Sq':>11}"
327
+ f"{'F-value':>10}{'p-value':>10}")
328
+ out.append(hdr)
329
+ for lbl, (_, r) in zip(labels, tbl.iterrows()):
330
+ line = (f"{lbl:<{w1}}{int(r['df']):>6}"
331
+ f"{fmt(r['Sum Sq'], d):>12}"
332
+ f"{fmt(r['Mean Sq'], d):>11}")
333
+ if lbl != "Residuals":
334
+ line += (f"{fmt(r['F-value'], d):>10}"
335
+ f"{fmt(r['p-value'], 4):>10}")
336
+ out.append(line)
337
+ return out
338
+
339
+
340
+ def _oneway(fit, yv, gv, y_name, x_name, levels, n_keep, d,
341
+ brief, graphics, jitter_x, lines, formula, n_obs):
342
+ from statsmodels.stats.anova import anova_lm
343
+ p = len(levels)
344
+ desc, grand = _desc_oneway(yv, gv, levels, d)
345
+
346
+ lines += ["", " DESCRIPTIVE STATISTICS", ""]
347
+ lines += _prntbl(desc, d).split("\n")
348
+ lines += ["", f"Grand Mean: {fmt(grand, d + 1)}"]
349
+
350
+ tbl = _anova_table(fit, anova_lm, 1, y_name)
351
+ tbl.index = [x_name, "Residuals"]
352
+ lines += ["", "", " ANOVA", "",
353
+ f"-- Summary Table for {y_name}", ""]
354
+ lines += _fmt_anova(tbl, [x_name, "Residuals"], d)
355
+
356
+ ssb = float(tbl.loc[x_name, "Sum Sq"])
357
+ ssw = float(tbl.loc["Residuals", "Sum Sq"])
358
+ msw = float(tbl.loc["Residuals", "Mean Sq"])
359
+ df_w = int(tbl.loc["Residuals", "df"])
360
+ sst = ssb + ssw
361
+ rsq = ssb / sst
362
+ rsq_adj = 1 - ((n_keep - 1) / (n_keep - p)) * (1 - rsq)
363
+ omsq = (ssb - (p - 1) * msw) / (sst + msw)
364
+ lines += ["", "",
365
+ f"-- Association and Effect Size for {y_name}", "",
366
+ f"R Squared: {fmt(rsq, 3)}",
367
+ f"R Sq Adjusted: {fmt(rsq_adj, 3)}",
368
+ f"Omega Squared: {fmt(omsq, 3)}"]
369
+ cohen_f = np.nan
370
+ if omsq > 0:
371
+ cohen_f = np.sqrt(omsq / (1 - omsq))
372
+ lines += ["", f"Cohen's f: {fmt(cohen_f, 3)}"]
373
+
374
+ tukey = None
375
+ lines += ["", "", " TUKEY MULTIPLE COMPARISONS OF MEANS"]
376
+ if not brief:
377
+ means, ns = _raw_means_ns(yv, gv, levels)
378
+ tukey = _tukey(levels, means, ns, msw, df_w)
379
+ lines += ["", "Family-wise Confidence Level: 0.95"]
380
+ lines += _prntbl(tukey, d).split("\n")
381
+
382
+ plots = {}
383
+ if graphics:
384
+ plots["means"] = _anova_means_plot(
385
+ yv, gv, y_name, x_name, levels, jitter_x, d)
386
+
387
+ return ANOVAResults(
388
+ formula=formula, design="oneway", n_obs=n_obs,
389
+ n_keep=n_keep, digits_d=d, response=y_name,
390
+ factors=[x_name], descriptive=desc, grand_mean=grand,
391
+ anova=tbl,
392
+ effects={"R_squared": rsq, "R_sq_adjusted": rsq_adj,
393
+ "omega_squared": omsq, "cohen_f": cohen_f},
394
+ tukey=tukey, plots=plots)
395
+
396
+
397
+ def _anova_means_plot(yv, gv, y_name, x_name, levels, jitter_x,
398
+ d):
399
+ """Scatterplot of the response by factor level, jittered
400
+ points with each cell mean marked and a level line.
401
+ ~ .ANOVAz1 means plot"""
402
+ style = plotly_style()
403
+ pos = {lv: i for i, lv in enumerate(levels)}
404
+ rng = np.random.default_rng(0)
405
+ xn = np.array([pos[g] for g in gv], dtype=float)
406
+ xn = xn + rng.uniform(-jitter_x / 2, jitter_x / 2, len(xn))
407
+ pt = get_option("pt_color", "#324E5C")
408
+ fig = go.Figure()
409
+ fig.add_trace(go.Scatter(
410
+ x=xn, y=yv, mode="markers",
411
+ marker=dict(symbol="circle", size=6,
412
+ color=make_trans(pt, 0.85), opacity=1,
413
+ line=dict(color=to_hex(pt), width=1)),
414
+ hoverinfo="y", showlegend=False))
415
+ means = [yv[gv == lv].mean() for lv in levels]
416
+ for i, mval in enumerate(means):
417
+ fig.add_shape(type="line", xref="x", yref="y",
418
+ x0=-0.5, x1=len(levels) - 0.5,
419
+ y0=mval, y1=mval, layer="below",
420
+ line=dict(color=to_hex("gray70"), width=1))
421
+ fig.add_trace(go.Scatter(
422
+ x=list(range(len(levels))), y=means, mode="markers",
423
+ marker=dict(symbol="diamond", size=13,
424
+ color=to_hex(get_option("fit_color",
425
+ "#5C4032"))),
426
+ hoverinfo="x+y", showlegend=False))
427
+ axT2 = pretty(float(yv.min()), float(yv.max()))
428
+ ax_x = axis_cat(x_name)
429
+ ax_x.update(tickmode="array",
430
+ tickvals=list(range(len(levels))),
431
+ ticktext=levels, range=[-0.5, len(levels) - 0.5])
432
+ ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
433
+ ax_y.update(showgrid=True,
434
+ gridcolor=to_hex(style["grid_col"]),
435
+ gridwidth=1, griddash="dot")
436
+ fig.update_layout(
437
+ xaxis=ax_x, yaxis=ax_y, shapes=fig.layout.shapes
438
+ + tuple(plot_border()), template=None,
439
+ plot_bgcolor=to_hex(style["panel_fill"]),
440
+ paper_bgcolor=to_hex(style["window_fill"]),
441
+ title=dict(text="Scatterplot with Cell Means",
442
+ x=0.5, xanchor="center",
443
+ font=dict(size=round(
444
+ 16 * get_option("main_size", 1)))))
445
+ return fig
446
+
447
+
448
+ def _twoway(fit, yv, fvals, y_name, facs, flevs, n_keep, d,
449
+ brief, graphics, design, lines, formula, n_obs, df,
450
+ fcols):
451
+ """Two-way between groups (Y ~ X1 * X2) or randomized blocks
452
+ (Y ~ X1 + X2). ~ .ANOVAz2"""
453
+ from statsmodels.stats.anova import anova_lm
454
+ bet = design == "two-between"
455
+ x1v, x2v = fvals[0], fvals[1]
456
+ l1, l2 = flevs[0], flevs[1]
457
+ f1, f2 = facs[0], facs[1]
458
+ p, q = len(l1), len(l2)
459
+ dp = pd.DataFrame({"y": yv, "a": x1v, "b": x2v})
460
+ cell_n = dp.pivot_table(index="a", columns="b", values="y",
461
+ aggfunc="size").reindex(
462
+ index=l1, columns=l2)
463
+ balanced = bool(cell_n.stack().nunique() == 1)
464
+
465
+ if bet:
466
+ lines += ["", "Two-way Between Groups ANOVA"]
467
+ else:
468
+ lines += ["", "Randomized Blocks ANOVA",
469
+ f" Factor of Interest: {f1}",
470
+ f" Blocking Factor: {f2}", "",
471
+ f"Note: For the F statistic for {f1} to be "
472
+ "distributed as F, the",
473
+ f" population covariances of {y_name} must "
474
+ "be spherical."]
475
+
476
+ lines += ["", " DESCRIPTIVE STATISTICS", ""]
477
+ if bet and not brief:
478
+ lines += ["-- Cell Sample Sizes", "",
479
+ ("Equal cell sizes, so balanced design"
480
+ if balanced else
481
+ "Unequal cell sizes, so unbalanced design")]
482
+ if not balanced:
483
+ lines.append("ANOVA based on Type II Sums of Squares")
484
+ lines += [""]
485
+ cn = cell_n.T.astype(int) # factor2 rows, factor1 cols
486
+ lines += _prntbl(cn, 0, int_cols=list(cn.columns)
487
+ ).split("\n")
488
+ cell_m = dp.pivot_table(index="b", columns="a",
489
+ values="y", aggfunc="mean"
490
+ ).reindex(index=l2, columns=l1)
491
+ lines += ["", "-- Cell Means", ""]
492
+ lines += _prntbl(cell_m, d).split("\n")
493
+
494
+ # marginal means and grand mean
495
+ m1 = dp.groupby("a")["y"].mean().reindex(l1)
496
+ m2 = dp.groupby("b")["y"].mean().reindex(l2)
497
+ grand = float(yv.mean())
498
+ lines += ["", "-- Marginal Means", "", f1]
499
+ lines += _prntbl(pd.DataFrame([m1.to_numpy()], columns=l1,
500
+ index=[""]), d).split("\n")
501
+ lines += ["", f2]
502
+ lines += _prntbl(pd.DataFrame([m2.to_numpy()], columns=l2,
503
+ index=[""]), d).split("\n")
504
+ lines += ["", f"-- Grand Mean: {fmt(grand, d + 1)}"]
505
+ if bet and not brief:
506
+ cell_s = dp.pivot_table(index="b", columns="a",
507
+ values="y", aggfunc="std"
508
+ ).reindex(index=l2, columns=l1)
509
+ lines += ["", "-- Cell Standard Deviations", ""]
510
+ lines += _prntbl(cell_s, d).split("\n")
511
+
512
+ # ANOVA table
513
+ typ = 2 if (bet and not balanced) else 1
514
+ t = anova_lm(fit, typ=typ)
515
+ t = t.rename(columns={"sum_sq": "Sum Sq", "df": "df",
516
+ "PR(>F)": "p-value", "F": "F-value"})
517
+ t["Mean Sq"] = t["Sum Sq"] / t["df"]
518
+ inter = f"{f1}:{f2}"
519
+ ren = {f"C({fcols[0]})": f1, f"C({fcols[1]})": f2,
520
+ f"C({fcols[0]}):C({fcols[1]})": inter,
521
+ "Residual": "Residuals"}
522
+ t.index = [ren.get(i, i) for i in t.index]
523
+ order = ([f1, f2, inter, "Residuals"] if bet
524
+ else [f1, f2, "Residuals"])
525
+ t = t.reindex(order)[["df", "Sum Sq", "Mean Sq", "F-value",
526
+ "p-value"]]
527
+ lines += ["", "", " ANOVA", "", "-- Summary Table"
528
+ + (" from Type II Sums of Squares"
529
+ if typ == 2 else ""), ""]
530
+ lines += _fmt_anova(t, order, d)
531
+
532
+ msw = float(t.loc["Residuals", "Mean Sq"])
533
+ df_w = int(t.loc["Residuals", "df"])
534
+
535
+ # effect sizes use the Type I (sequential) F-values, as R's
536
+ # summary(aov), even when the displayed table is Type II
537
+ t1 = anova_lm(fit, typ=1).rename(
538
+ columns={"F": "F-value"})
539
+ t1.index = [ren.get(i, i) for i in t1.index]
540
+ fa = float(t1.loc[f1, "F-value"])
541
+ fb = float(t1.loc[f2, "F-value"])
542
+ lines += ["", "", "-- Association and Effect Size", ""]
543
+ eff = {}
544
+ if bet:
545
+ fab = float(t1.loc[inter, "F-value"])
546
+ nh = round(1 / np.mean(1 / cell_n.to_numpy().ravel()))
547
+ oa = ((p - 1) * (fa - 1)) / ((p - 1) * (fa - 1) + nh * p * q)
548
+ ob = ((q - 1) * (fb - 1)) / ((q - 1) * (fb - 1) + nh * p * q)
549
+ oab = (((p - 1) * (q - 1) * (fab - 1))
550
+ / ((p - 1) * (q - 1) * (fab - 1) + nh * p * q))
551
+ eff = {"omega_sq_" + f1: oa, "omega_sq_" + f2: ob,
552
+ "omega_sq_interaction": oab}
553
+ lines += [f"Partial Omega Squared for {f1}: {fmt(oa, 3)}",
554
+ f"Partial Omega Squared for {f2}: {fmt(ob, 3)}",
555
+ f"Partial Omega Squared for {f1} & {f2}: "
556
+ f"{fmt(oab, 3)}", ""]
557
+ for nm, o in ((f1, oa), (f2, ob), (f"{f1} & {f2}", oab)):
558
+ if o > 0:
559
+ lines.append(f"Cohen's f for {nm}: "
560
+ f"{fmt(np.sqrt(o / (1 - o)), 3)}")
561
+ else:
562
+ oa = ((p - 1) * (fa - 1)) / ((p - 1) * (fa - 1) + q * p)
563
+ intra = (fb - 1) / ((p - 1) + fb)
564
+ eff = {"omega_sq_" + f1: oa, "intraclass_" + f2: intra}
565
+ lines += [f"Partial Omega Squared for {f1}: {fmt(oa, 3)}",
566
+ f"Partial Intraclass Correlation for {f2}: "
567
+ f"{fmt(intra, 3)}", ""]
568
+ if oa > 0:
569
+ lines.append(f"Cohen's f for {f1}: "
570
+ f"{fmt(np.sqrt(oa / (1 - oa)), 3)}")
571
+ if intra > 0:
572
+ lines.append(f"Cohen's f for {f2}: "
573
+ f"{fmt(np.sqrt(intra / (1 - intra)), 3)}")
574
+
575
+ # Tukey
576
+ tukey = None
577
+ lines += ["", "", " TUKEY MULTIPLE COMPARISONS OF MEANS"]
578
+ if not brief:
579
+ tukey = {}
580
+ lines += ["", "Family-wise Confidence Level: 0.95",
581
+ "", f"Factor: {f1}"]
582
+ m1m = _seq_marginal_means(df, y_name, fcols, 0, x1v, l1)
583
+ _, n1 = _raw_means_ns(yv, x1v, l1)
584
+ tukey[f1] = _tukey(l1, m1m, n1, msw, df_w)
585
+ lines += _prntbl(tukey[f1], d).split("\n")
586
+ if bet: # blocks: factor of interest only
587
+ lines += ["", f"Factor: {f2}"]
588
+ m2m = _seq_marginal_means(df, y_name, fcols, 1, x2v,
589
+ l2)
590
+ _, n2 = _raw_means_ns(yv, x2v, l2)
591
+ tukey[f2] = _tukey(l2, m2m, n2, msw, df_w)
592
+ lines += _prntbl(tukey[f2], d).split("\n")
593
+ cl = np.array([f"{a}:{b}" for a, b in zip(x1v, x2v)])
594
+ clv = [f"{a}:{b}" for b in l2 for a in l1]
595
+ clv = [c for c in clv if c in set(cl)]
596
+ cmn, cnn = _raw_means_ns(yv, cl, clv)
597
+ lines += ["", "Cell Means"]
598
+ tukey["cells"] = _tukey(clv, cmn, cnn, msw, df_w)
599
+ lines += _prntbl(tukey["cells"], d).split("\n")
600
+
601
+ plots = {}
602
+ if graphics:
603
+ if bet:
604
+ plots["interaction"] = _interaction_plot(
605
+ dp, y_name, f1, f2, l1, l2, d)
606
+ else:
607
+ plots["data"] = _blocks_plot(
608
+ x1v, yv, x2v, y_name, f1, f2, l1, l2, d,
609
+ "Data Values")
610
+ plots["fitted"] = _blocks_plot(
611
+ x1v, np.asarray(fit.fittedvalues), x2v,
612
+ "Fitted", f1, f2, l1, l2, d, "Fitted Values")
613
+
614
+ return ANOVAResults(
615
+ formula=formula, design=design, n_obs=n_obs,
616
+ n_keep=n_keep, digits_d=d, response=y_name,
617
+ factors=[f1, f2], cell_n=cell_n,
618
+ marginal_means={f1: m1, f2: m2}, grand_mean=grand,
619
+ anova=t, effects=eff, tukey=tukey, plots=plots)
620
+
621
+
622
+ def _cat_line_fig(title, x_name, y_name, levels_x, series, d,
623
+ legend_title):
624
+ """A categorical-x line-with-markers figure: one colored
625
+ line per by-group across the x levels. Shared by the
626
+ interaction and blocks plots."""
627
+ style = plotly_style()
628
+ fig = go.Figure()
629
+ ys_all = []
630
+ for gi, (gname, yv_line) in enumerate(series):
631
+ col = to_hex(BASE_COLORS[gi % len(BASE_COLORS)])
632
+ ys_all += [v for v in yv_line if v == v]
633
+ fig.add_trace(go.Scatter(
634
+ x=list(range(len(levels_x))), y=yv_line,
635
+ mode="lines+markers", name=str(gname),
636
+ line=dict(color=col, width=2),
637
+ marker=dict(size=9, color=col),
638
+ connectgaps=True, hoverinfo="x+y+name"))
639
+ axT2 = pretty(min(ys_all), max(ys_all))
640
+ ax_x = axis_cat(x_name)
641
+ ax_x.update(tickmode="array",
642
+ tickvals=list(range(len(levels_x))),
643
+ ticktext=levels_x,
644
+ range=[-0.4, len(levels_x) - 0.6])
645
+ ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
646
+ ax_y.update(showgrid=True,
647
+ gridcolor=to_hex(style["grid_col"]),
648
+ gridwidth=1, griddash="dot")
649
+ fig.update_layout(
650
+ xaxis=ax_x, yaxis=ax_y, shapes=plot_border(),
651
+ template=None, legend=dict(title=dict(text=legend_title)),
652
+ plot_bgcolor=to_hex(style["panel_fill"]),
653
+ paper_bgcolor=to_hex(style["window_fill"]),
654
+ title=dict(text=title, x=0.5, xanchor="center",
655
+ font=dict(size=round(
656
+ 16 * get_option("main_size", 1)))))
657
+ return fig
658
+
659
+
660
+ def _interaction_plot(dp, y_name, f1, f2, l1, l2, d):
661
+ """Cell means of the response across factor 1, one line per
662
+ factor 2 level. ~ .ANOVAz2 interaction plot"""
663
+ cm = dp.pivot_table(index="a", columns="b", values="y",
664
+ aggfunc="mean").reindex(index=l1,
665
+ columns=l2)
666
+ series = [(b, [cm.loc[a, b] for a in l1]) for b in l2]
667
+ return _cat_line_fig(f"Cell Means of {y_name}", f1, y_name,
668
+ l1, series, d, f2)
669
+
670
+
671
+ def _blocks_plot(x1v, yvals, x2v, y_name, f1, f2, l1, l2, d,
672
+ title):
673
+ """Values across the factor of interest, one line per block.
674
+ ~ .ANOVAz2 data / fitted plots"""
675
+ dp = pd.DataFrame({"y": yvals, "a": x1v, "b": x2v})
676
+ series = []
677
+ for b in l2:
678
+ sub = dp[dp["b"] == b].set_index("a")["y"]
679
+ series.append((b, [sub.get(a, np.nan) for a in l1]))
680
+ return _cat_line_fig(title, f1, y_name, l1, series, d, f2)