lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/ttest.py ADDED
@@ -0,0 +1,641 @@
1
+ # ttest.py — analog of ttest.R (tt.2group, tt.1group,
2
+ # tt.formula).
3
+ #
4
+ # ttest(): the t-test of a mean or a mean difference, with the
5
+ # lessR output pipeline — Describe, Assumptions (normality and,
6
+ # for two groups, homogeneity of variance), Infer (the equal-
7
+ # variance test and, for two groups, the Welch test), Effect
8
+ # Size (Cohen's d), Practical Importance, and Needed Sample Size,
9
+ # plus a plotly density plot. Modes, from the arguments:
10
+ # ttest("Y ~ Group", data=d) two independent groups
11
+ # ttest("V1", "V2", data=d) two independent groups
12
+ # ttest("V1", "V2", data=d, paired=True) paired
13
+ # ttest("V", data=d, mu=100) one group vs mu
14
+ # ttest("V", data=d) one group, CI only
15
+ # ttest(n1=, m1=, s1=, n2=, m2=, s2=) two groups from stats
16
+ # ttest(n=, m=, s=, mu=) one group from stats
17
+ #
18
+ # The larger-mean group is always reported first, as R. Numerics
19
+ # through scipy.stats (imported lazily); the pipeline is ported,
20
+ # not the lines. Returns a ttestResults object; the figure is in
21
+ # .plots and is not auto-shown.
22
+
23
+ import numpy as np
24
+ import pandas as pd
25
+
26
+ from .Regression import _getdigits
27
+ from .utils import fmt, get_column, get_option
28
+
29
+
30
+ class ttestResults:
31
+ """Numeric results of ttest(): the per-group summary, the
32
+ equal-variance and (two-group) Welch inference, the effect
33
+ size, and the plotly figure in .plots."""
34
+
35
+ def __init__(self, **kw):
36
+ self.__dict__.update(kw)
37
+
38
+ def __repr__(self):
39
+ return f"<lessPy ttest: {self.kind}, n={self.n_total}>"
40
+
41
+
42
+ def _F(x, d):
43
+ return fmt(x, d)
44
+
45
+
46
+ def ttest(x=None, y=None, data=None, filter=None, paired=False,
47
+ n=None, m=None, s=None, mu=None,
48
+ n1=None, n2=None, m1=None, m2=None, s1=None, s2=None,
49
+ Ynm="Y", Xnm="X", X1nm="Group1", X2nm="Group2",
50
+ brief=False, digits_d=None, conf_level=0.95,
51
+ alternative="two_sided", mmd=None, msmd=None,
52
+ Edesired=None, graph=True):
53
+ """t-test of a mean (one group, vs mu) or a mean difference
54
+ (two independent groups or paired). Always prints, as in R;
55
+ returns a ttestResults object with the density figure in
56
+ .plots."""
57
+ if alternative not in ("two_sided", "less", "greater"):
58
+ raise ValueError(
59
+ 'alternative: "two_sided", "less", "greater"')
60
+ if data is not None and filter is not None:
61
+ data = data.query(filter)
62
+
63
+ from_stats = n is not None or n1 is not None
64
+ two_group = n1 is not None or (not from_stats and y is not None
65
+ and not paired)
66
+
67
+ if from_stats:
68
+ if n1 is not None:
69
+ return _two_group_stats(
70
+ n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm, X2nm,
71
+ brief, digits_d, conf_level, alternative, mmd,
72
+ msmd, Edesired)
73
+ return _one_group(
74
+ None, n, m, s, mu, Ynm, brief, conf_level,
75
+ alternative, digits_d, mmd, msmd, Edesired, False,
76
+ graph=False)
77
+
78
+ # from data: resolve x (and y) as column names or arrays
79
+ if isinstance(x, str) and "~" in x:
80
+ yv, xg, Ynm, Xnm, X1nm, X2nm = _formula(x, data)
81
+ return _two_group_data(
82
+ yv, xg, Ynm, Xnm, X1nm, X2nm, brief, digits_d,
83
+ conf_level, alternative, mmd, msmd, Edesired, graph)
84
+
85
+ xv = _resolve(x, data, "x")
86
+ if y is None:
87
+ Ynm = x if isinstance(x, str) else Ynm
88
+ return _one_group(
89
+ xv, None, None, None, mu, Ynm, brief, conf_level,
90
+ alternative, digits_d, mmd, msmd, Edesired, False,
91
+ graph=graph)
92
+
93
+ yv = _resolve(y, data, "y")
94
+ if paired:
95
+ if len(xv) != len(yv):
96
+ raise ValueError("paired samples must be equal length")
97
+ keep = ~(np.isnan(xv) | np.isnan(yv))
98
+ diff = yv[keep] - xv[keep] # y - x, as R
99
+ return _one_group(
100
+ diff, None, None, None, 0.0, "Difference", brief,
101
+ conf_level, alternative, digits_d, mmd, msmd,
102
+ Edesired, True, graph=graph)
103
+
104
+ X1nm = x if isinstance(x, str) else X1nm
105
+ X2nm = y if isinstance(y, str) else X2nm
106
+ Ynm = "Y"
107
+ return _two_group_data(
108
+ {X1nm: xv, X2nm: yv}, None, Ynm, Xnm, X1nm, X2nm, brief,
109
+ digits_d, conf_level, alternative, mmd, msmd, Edesired,
110
+ graph)
111
+
112
+
113
+ def _resolve(v, data, arg):
114
+ if isinstance(v, str):
115
+ if data is None:
116
+ raise ValueError(f"{arg}='{v}' is a column name, so "
117
+ "data= is required")
118
+ return get_column(data, v, arg).to_numpy(dtype=float)
119
+ return np.asarray(v, dtype=float)
120
+
121
+
122
+ def _formula(f, data):
123
+ if data is None:
124
+ raise ValueError("a formula needs data=")
125
+ lhs, rhs = (t.strip() for t in f.split("~"))
126
+ y_ser = get_column(data, lhs, "response")
127
+ g_ser = get_column(data, rhs, "grouping")
128
+ if not pd.api.types.is_numeric_dtype(y_ser):
129
+ raise TypeError(f"the response '{lhs}' must be numeric")
130
+ levels = list(pd.unique(g_ser.dropna().astype(str)))
131
+ if len(levels) != 2:
132
+ raise ValueError(
133
+ f"the grouping variable '{rhs}' must have exactly "
134
+ f"two values; found {len(levels)}. Use ANOVA for "
135
+ "more than two groups.")
136
+ levels = sorted(levels)
137
+ gs = g_ser.astype(str)
138
+ groups = {lv: y_ser[gs == lv].to_numpy(dtype=float)
139
+ for lv in levels}
140
+ return groups, None, lhs, rhs, levels[0], levels[1]
141
+
142
+
143
+ # ------------------------------------------------------------
144
+ # formatting helpers for the group summary
145
+ # ------------------------------------------------------------
146
+
147
+ def _alt_word(alt):
148
+ return {"two_sided": "two.sided", "less": "less",
149
+ "greater": "greater"}[alt]
150
+
151
+
152
+ def _tcrit(conf_level, df, alt):
153
+ from scipy.stats import t as tdist
154
+ if alt == "two_sided":
155
+ return float(tdist.ppf(1 - (1 - conf_level) / 2, df))
156
+ return float(tdist.ppf(conf_level, df))
157
+
158
+
159
+ def _pval(tvalue, df, alt):
160
+ from scipy.stats import t as tdist
161
+ if alt == "two_sided":
162
+ return float(2 * tdist.sf(abs(tvalue), df))
163
+ if alt == "less":
164
+ return float(tdist.cdf(tvalue, df))
165
+ return float(tdist.sf(tvalue, df))
166
+
167
+
168
+ # ------------------------------------------------------------
169
+ # two independent groups
170
+ # ------------------------------------------------------------
171
+
172
+ def _two_group_data(groups, _g, Ynm, Xnm, X1nm, X2nm, brief,
173
+ digits_d, conf_level, alternative, mmd, msmd,
174
+ Edesired, graph):
175
+ YA0 = groups[X1nm]
176
+ YB0 = groups[X2nm]
177
+ n1m = int(np.isnan(YA0).sum())
178
+ n2m = int(np.isnan(YB0).sum())
179
+ YA = YA0[~np.isnan(YA0)]
180
+ YB = YB0[~np.isnan(YB0)]
181
+ if len(YA) < 2 or len(YB) < 2:
182
+ raise ValueError("need at least two cases per sample")
183
+ # the larger-mean group is reported first, as R
184
+ if YA.mean() < YB.mean():
185
+ YA, YB = YB, YA
186
+ X1nm, X2nm = X2nm, X1nm
187
+ n1m, n2m = n2m, n1m
188
+ d = digits_d if digits_d is not None else _getdigits(
189
+ np.concatenate([YA, YB]), 3)
190
+ return _two_group(
191
+ len(YA), YA.mean(), YA.std(ddof=1),
192
+ len(YB), YB.mean(), YB.std(ddof=1),
193
+ Ynm, Xnm, X1nm, X2nm, brief, d, conf_level, alternative,
194
+ mmd, msmd, Edesired, YA, YB, n1m, n2m, graph)
195
+
196
+
197
+ def _two_group_stats(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm,
198
+ X2nm, brief, digits_d, conf_level,
199
+ alternative, mmd, msmd, Edesired):
200
+ if m1 < m2: # larger mean first
201
+ n1, n2 = n2, n1
202
+ m1, m2 = m2, m1
203
+ s1, s2 = s2, s1
204
+ X1nm, X2nm = X2nm, X1nm
205
+ d = digits_d if digits_d is not None else 3
206
+ return _two_group(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm,
207
+ X2nm, brief, d, conf_level, alternative,
208
+ mmd, msmd, Edesired, None, None, 0, 0,
209
+ False)
210
+
211
+
212
+ def _two_group(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm, X2nm,
213
+ brief, d, conf_level, alternative, mmd, msmd,
214
+ Edesired, YA, YB, n1m, n2m, graph):
215
+ from scipy.stats import f as fdist
216
+ from_data = YA is not None
217
+ v1, v2 = s1 ** 2, s2 ** 2
218
+ sd_d = d if from_data else d - 1
219
+ clpct = f"{round(conf_level * 100, 2):g}%"
220
+ alt = alternative
221
+
222
+ L = ["", f"Compare {Ynm} across {Xnm} with levels "
223
+ f"{X1nm} and {X2nm}",
224
+ f"Grouping Variable: {Xnm}",
225
+ f"Response Variable: {Ynm}", ""]
226
+
227
+ L.append("------ Describe ------" if not brief
228
+ else " --- Describe ---")
229
+ L.append("")
230
+ miss1 = f"n.miss = {n1m}, " if from_data else ""
231
+ miss2 = f"n.miss = {n2m}, " if from_data else ""
232
+ L.append(f"{Ynm} for {Xnm} {X1nm}: {miss1}n = {n1}, "
233
+ f" mean = {_F(m1, sd_d)}, sd = {_F(s1, sd_d)}")
234
+ L.append(f"{Ynm} for {Xnm} {X2nm}: {miss2}n = {n2}, "
235
+ f" mean = {_F(m2, sd_d)}, sd = {_F(s2, sd_d)}")
236
+ L += ["", f"Mean Difference of {Ynm}: {_F(m1 - m2, d)}"]
237
+
238
+ df1, df2 = n1 - 1, n2 - 1
239
+ swsq = (df1 * v1 + df2 * v2) / (df1 + df2)
240
+ sw = np.sqrt(swsq)
241
+ smd = (m1 - m2) / sw
242
+ L += ["", f"Weighted Average Standard Deviation: {_F(sw, d)}"]
243
+ if brief:
244
+ L.append(f"Standardized Mean Difference of {Ynm}: "
245
+ f"{_F(smd, d)}")
246
+
247
+ if not brief:
248
+ L += _assumptions(YA, YB, X1nm, X2nm, Ynm, n1, n2, v1,
249
+ v2, df1, df2, from_data, d)
250
+
251
+ # ----- Infer: equal variances (pooled) -----
252
+ L.append("")
253
+ L.append("------ Infer ------" if not brief
254
+ else " --- Infer ---")
255
+ L.append("")
256
+ if not brief:
257
+ L.append("--- Assume equal population variances of "
258
+ f"{Ynm} for each {Xnm}")
259
+ L.append("")
260
+ sterr = sw * np.sqrt(1 / n1 + 1 / n2)
261
+ df = df1 + df2
262
+ tcut = _tcrit(conf_level, df, alt)
263
+ tvalue = (m1 - m2) / sterr
264
+ pvalue = _pval(tvalue, df, alt)
265
+ E = tcut * sterr
266
+ lb, ub = (m1 - m2) - E, (m1 - m2) + E
267
+ if alt != "two_sided":
268
+ L.append("Alternative hypothesis: Population mean "
269
+ f"difference is {_alt_word(alt)} than 0")
270
+ L += [f"t-cutoff for {clpct} range of variation: "
271
+ f"tcut = {_F(tcut, 3)}",
272
+ f"Standard Error of Mean Difference: SE = {_F(sterr, d)}",
273
+ "",
274
+ f"Hypothesis Test of 0 Mean Diff: t-value = "
275
+ f"{_F(tvalue, 3)}, df = {df}, p-value = "
276
+ f"{_F(pvalue, 3)}", "",
277
+ f"Margin of Error for {clpct} Confidence Level: "
278
+ f"{_F(E, d)}",
279
+ f"{clpct} Confidence Interval for Mean Difference: "
280
+ f"{_F(lb, d)} to {_F(ub, d)}"]
281
+
282
+ welch = None
283
+ if not brief:
284
+ welch = _welch(m1, m2, v1, v2, n1, n2, conf_level, alt,
285
+ YA, YB)
286
+ L += ["", "--- Do not assume equal population variances "
287
+ f"of {Ynm} for each {Xnm}", "",
288
+ f"t-cutoff: tcut = {_F(welch['tcut'], 3)}",
289
+ f"Standard Error of Mean Difference: SE = "
290
+ f"{_F(welch['sterr'], d)}", "",
291
+ f"Hypothesis Test of 0 Mean Diff: t = "
292
+ f"{_F(welch['t'], 3)}, df = {_F(welch['df'], 3)}, "
293
+ f"p-value = {_F(welch['p'], 3)}", "",
294
+ f"Margin of Error for {clpct} Confidence Level: "
295
+ f"{_F(welch['E'], d)}",
296
+ f"{clpct} Confidence Interval for Mean Difference: "
297
+ f"{_F(welch['lb'], d)} to {_F(welch['ub'], d)}"]
298
+
299
+ L += ["", "------ Effect Size ------", "",
300
+ "--- Assume equal population variances of "
301
+ f"{Ynm} for each {Xnm}", "",
302
+ f"Standardized Mean Difference of {Ynm}, "
303
+ f"Cohen's d: {_F(smd, d)}"]
304
+ L += _practical(mmd, msmd, sw, m1 - m2, smd, lb, ub, d)
305
+ L += _needed_2(Edesired, conf_level, sw, E, n1, n2, d)
306
+
307
+ print("\n".join(L))
308
+ plots = {}
309
+ if graph and from_data:
310
+ plots["two_group"] = _two_group_plot(
311
+ YA, YB, Ynm, X1nm, X2nm, m1, m2, d)
312
+ return ttestResults(
313
+ kind="two-group", n_total=n1 + n2, digits_d=d,
314
+ group1={"name": X1nm, "n": n1, "mean": m1, "sd": s1},
315
+ group2={"name": X2nm, "n": n2, "mean": m2, "sd": s2},
316
+ mean_diff=m1 - m2, pooled_sd=sw, cohen_d=smd,
317
+ equal_var={"t": tvalue, "df": df, "p_value": pvalue,
318
+ "se": sterr, "lb": lb, "ub": ub},
319
+ welch=welch, plots=plots)
320
+
321
+
322
+ def _assumptions(YA, YB, X1nm, X2nm, Ynm, n1, n2, v1, v2, df1,
323
+ df2, from_data, d):
324
+ from scipy.stats import f as fdist, shapiro
325
+ L = ["", "", "------ Assumptions ------", "",
326
+ "Note: These hypothesis tests can perform poorly, and "
327
+ "the",
328
+ " t-test is typically robust to violations of "
329
+ "assumptions.",
330
+ " Use as heuristic guides instead of interpreting "
331
+ "literally.", ""]
332
+ if from_data:
333
+ L.append("Null hypothesis, for each group, is a normal "
334
+ f"distribution of {Ynm}.")
335
+ for nm, Y, ni in ((X1nm, YA, n1), (X2nm, YB, n2)):
336
+ if ni > 30:
337
+ L.append(f"Group {nm}: Sample mean assumed "
338
+ "normal because n > 30, so no test "
339
+ "needed.")
340
+ elif 2 < ni < 5000:
341
+ W, p = shapiro(Y)
342
+ L.append(f"Group {nm} Shapiro-Wilk normality "
343
+ f"test: W = {_F(W, 3)}, p-value = "
344
+ f"{_F(p, 3)}")
345
+ else:
346
+ L.append(f"Group {nm} Sample size out of range "
347
+ "for Shapiro-Wilk normality test.")
348
+ L.append("")
349
+ # variance ratio F test
350
+ if v1 >= v2:
351
+ vr, dfn, dfd = v1 / v2, df1, df2
352
+ vrs = f"{_F(v1, d)}/{_F(v2, d)}"
353
+ else:
354
+ vr, dfn, dfd = v2 / v1, df2, df1
355
+ vrs = f"{_F(v2, d)}/{_F(v1, d)}"
356
+ pv = float(fdist.cdf(vr, dfn, dfd))
357
+ pv = 2 * min(pv, 1 - pv)
358
+ L.append(f"Null hypothesis is equal variances of {Ynm}, "
359
+ "homogeneous.")
360
+ L.append(f"Variance Ratio test: F = {vrs} = {_F(vr, d)}, "
361
+ f" df = {dfn};{dfd}, p-value = {_F(pv, 3)}")
362
+ if from_data:
363
+ # Levene, Brown-Forsythe: pooled t on |Y - median|
364
+ a = np.abs(YA - np.median(YA))
365
+ b = np.abs(YB - np.median(YB))
366
+ t_bf, df_bf, p_bf = _pooled_t(a, b)
367
+ L.append(f"Levene's test, Brown-Forsythe: t = "
368
+ f"{_F(t_bf, 3)}, df = {df_bf}, p-value = "
369
+ f"{_F(p_bf, 3)}")
370
+ return L
371
+
372
+
373
+ def _pooled_t(a, b):
374
+ na, nb = len(a), len(b)
375
+ va, vb = a.var(ddof=1), b.var(ddof=1)
376
+ dfa, dfb = na - 1, nb - 1
377
+ sw = np.sqrt((dfa * va + dfb * vb) / (dfa + dfb))
378
+ se = sw * np.sqrt(1 / na + 1 / nb)
379
+ t = (a.mean() - b.mean()) / se
380
+ df = dfa + dfb
381
+ return t, df, _pval(t, df, "two_sided")
382
+
383
+
384
+ def _welch(m1, m2, v1, v2, n1, n2, conf_level, alt, YA, YB):
385
+ k1, k2 = v1 / n1, v2 / n2
386
+ df = (k1 + k2) ** 2 / (k1 ** 2 / (n1 - 1) + k2 ** 2 / (n2 - 1))
387
+ sterr = np.sqrt(k1 + k2)
388
+ tcut = _tcrit(conf_level, df, alt)
389
+ t = (m1 - m2) / sterr
390
+ p = _pval(t, df, alt)
391
+ E = tcut * sterr
392
+ return {"t": t, "df": df, "p": p, "se": sterr, "tcut": tcut,
393
+ "E": E, "lb": (m1 - m2) - E, "ub": (m1 - m2) + E,
394
+ "sterr": sterr}
395
+
396
+
397
+ def _practical(mmd, msmd, sw, mdiff, smd, lb, ub, d):
398
+ L = ["", "", "------ Practical Importance ------", "",
399
+ "Minimum Mean Difference of practical importance: mmd"]
400
+ if mmd is not None or msmd is not None:
401
+ if mmd is not None:
402
+ msmd = mmd / sw
403
+ else:
404
+ mmd = msmd * sw
405
+ L += [f"Compare mmd = {_F(mmd, d)} to the obtained value "
406
+ f"of md = {_F(mdiff, d)}",
407
+ f"Compare mmd to the confidence interval for md: "
408
+ f"{_F(lb, d)} to {_F(ub, d)}", "",
409
+ "Minimum Standardized Mean Difference of practical "
410
+ "importance: msmd",
411
+ f"Compare msmd = {_F(msmd, d)} to the obtained "
412
+ f"value of smd = {_F(smd, d)}"]
413
+ else:
414
+ L += ["Minimum Standardized Mean Difference of "
415
+ "practical importance: msmd",
416
+ "Neither value specified, so no analysis"]
417
+ return L
418
+
419
+
420
+ def _needed_2(Edesired, conf_level, sw, E, n1, n2, d):
421
+ if Edesired is None:
422
+ return []
423
+ from scipy.stats import norm
424
+ zcut = norm.ppf((1 - conf_level) / 2)
425
+ ns = 2 * ((zcut * sw) / Edesired) ** 2
426
+ needed = int(np.ceil(1.099 * ns + 4.863))
427
+ L = ["", "", "------ Needed Sample Size ------", ""]
428
+ if Edesired > E:
429
+ L.append(f"Note: Desired margin of error, {_F(Edesired, d)}"
430
+ f" is worse than what was obtained, {_F(E, d)}")
431
+ L.append("")
432
+ L += [f"Desired Margin of Error: {_F(Edesired, d)}", "",
433
+ "For the following sample size there is a 0.9 "
434
+ "probability of obtaining",
435
+ f"the desired margin of error for the resulting "
436
+ "confidence interval.",
437
+ f"Needed sample size per group: {needed}", "",
438
+ f"Additional data values needed Group 1: {needed - n1}",
439
+ f"Additional data values needed Group 2: {needed - n2}"]
440
+ return L
441
+
442
+
443
+ # ------------------------------------------------------------
444
+ # one group (and paired, on the differences)
445
+ # ------------------------------------------------------------
446
+
447
+ def _one_group(Y, n, m, s, mu, Ynm, brief, conf_level,
448
+ alternative, digits_d, mmd, msmd, Edesired,
449
+ paired, graph):
450
+ from scipy.stats import shapiro
451
+ from_data = Y is not None
452
+ if from_data:
453
+ Y0 = np.asarray(Y, dtype=float)
454
+ n_miss = int(np.isnan(Y0).sum())
455
+ Y = Y0[~np.isnan(Y0)]
456
+ n, m, s = len(Y), Y.mean(), Y.std(ddof=1)
457
+ d = digits_d if digits_d is not None else _getdigits(Y, 3)
458
+ else:
459
+ n_miss = 0
460
+ d = digits_d if digits_d is not None else 3
461
+ sd_d = d if from_data else d - 1
462
+ clpct = f"{round(conf_level * 100, 2):g}%"
463
+ alt = alternative
464
+
465
+ L = ["", f"------ Describe ------" if not brief
466
+ else " --- Describe ---", ""]
467
+ lead = f"{Ynm}: " if Ynm not in ("Y",) else ""
468
+ miss = f"n.miss = {n_miss}, " if from_data else ""
469
+ L.append(f"{lead}{miss}n = {n}, mean = {_F(m, sd_d)}, "
470
+ f" sd = {_F(s, sd_d)}")
471
+
472
+ if not brief and from_data:
473
+ L += ["", "", "------ Normality Assumption ------", ""]
474
+ if n > 30:
475
+ L.append("Sample mean assumed normal because n > 30, "
476
+ "so no test needed.")
477
+ elif 2 < n < 5000:
478
+ W, p = shapiro(Y)
479
+ L += [f"Null hypothesis is a normal distribution of "
480
+ f"{Ynm}.",
481
+ f"Shapiro-Wilk normality test: W = {_F(W, 3)}, "
482
+ f" p-value = {_F(p, 3)}"]
483
+ else:
484
+ L.append("Sample size out of range for Shapiro-Wilk "
485
+ "normality test.")
486
+
487
+ L += ["", "", "------ Infer ------" if not brief
488
+ else " --- Infer ---", ""]
489
+ df = n - 1
490
+ sterr = s * np.sqrt(1 / n)
491
+ tcut = _tcrit(conf_level, df, alt)
492
+ E = tcut * sterr
493
+ lb, ub = m - E, m + E
494
+ tvalue = pvalue = None
495
+ if mu is not None:
496
+ tvalue = (m - mu) / sterr
497
+ pvalue = _pval(tvalue, df, alt)
498
+ L += [f"t-cutoff for {clpct} range of variation: "
499
+ f"tcut = {_F(tcut, 3)}",
500
+ f"Standard Error of Mean: SE = {_F(sterr, d)}", ""]
501
+ if mu is not None:
502
+ if alt != "two_sided":
503
+ L.append("Alternative hypothesis: Population mean is "
504
+ f"{_alt_word(alt)} than {mu}")
505
+ L += [f"Hypothesized Value H0: mu = {mu}",
506
+ f"Hypothesis Test of Mean: t-value = "
507
+ f"{_F(tvalue, 3)}, df = {df}, p-value = "
508
+ f"{_F(pvalue, 3)}", ""]
509
+ L += [f"Margin of Error for {clpct} Confidence Level: "
510
+ f"{_F(E, d)}",
511
+ f"{clpct} Confidence Interval for Mean: {_F(lb, d)} "
512
+ f"to {_F(ub, d)}"]
513
+
514
+ cohen = None
515
+ if mu is not None:
516
+ mdiff = m - mu
517
+ cohen = abs(mdiff / s)
518
+ L += ["", "", "------ Effect Size ------", "",
519
+ f"Distance of sample mean from hypothesized: "
520
+ f"{_F(mdiff, d)}",
521
+ f"Standardized Distance, Cohen's d: "
522
+ f"{_F(cohen, d)}"]
523
+
524
+ if Edesired is not None and from_data:
525
+ L += _needed_1(Edesired, conf_level, s, E, n, d)
526
+
527
+ print("\n".join(L))
528
+ plots = {}
529
+ if graph and from_data:
530
+ plots["one_group"] = _one_group_plot(
531
+ Y, Ynm, m, lb, ub, d, paired)
532
+ return ttestResults(
533
+ kind="paired" if paired else "one-group", n_total=n,
534
+ digits_d=d, n=n, mean=m, sd=s, mu=mu,
535
+ infer={"t": tvalue, "df": df, "p_value": pvalue,
536
+ "se": sterr, "lb": lb, "ub": ub},
537
+ cohen_d=cohen, plots=plots)
538
+
539
+
540
+ def _needed_1(Edesired, conf_level, s, E, n, d):
541
+ from scipy.stats import norm
542
+ zcut = norm.ppf((1 - conf_level) / 2)
543
+ ns = ((zcut * s) / Edesired) ** 2
544
+ needed = int(np.ceil(1.132 * ns + 7.368))
545
+ L = ["", "", "------ Needed Sample Size ------", ""]
546
+ if Edesired > E:
547
+ L += [f"Note: Desired margin of error, {_F(Edesired, d)}"
548
+ f" is worse than what was obtained, {_F(E, d)}", ""]
549
+ L += [f"Desired Margin of Error: {_F(Edesired, d)}", "",
550
+ "For the following sample size there is a 0.9 "
551
+ "probability of obtaining",
552
+ "the desired margin of error for the confidence "
553
+ "interval.",
554
+ f"Needed sample size: {needed}", "",
555
+ f"Additional data values needed: {needed - n}"]
556
+ return L
557
+
558
+
559
+ # ------------------------------------------------------------
560
+ # density plots
561
+ # ------------------------------------------------------------
562
+
563
+ def _kde(v):
564
+ from scipy.stats import gaussian_kde
565
+ k = gaussian_kde(v)
566
+ lo, hi = v.min(), v.max()
567
+ pad = 0.15 * (hi - lo if hi > lo else 1)
568
+ xs = np.linspace(lo - pad, hi + pad, 200)
569
+ return xs, k(xs)
570
+
571
+
572
+ def _two_group_plot(YA, YB, Ynm, X1nm, X2nm, m1, m2, d):
573
+ import plotly.graph_objects as go
574
+ from .plotly_utils import (
575
+ BASE_COLORS, axis_format, axis_num, make_trans,
576
+ plot_border, plotly_style, to_hex, x_grid)
577
+ from .utils import pretty
578
+ style = plotly_style()
579
+ fig = go.Figure()
580
+ allx, ally = [], []
581
+ for (nm, v, col) in ((X1nm, YA, BASE_COLORS[1]),
582
+ (X2nm, YB, BASE_COLORS[0])):
583
+ xs, ys = _kde(v)
584
+ allx += [xs.min(), xs.max()]
585
+ ally.append(ys.max())
586
+ c = to_hex(col)
587
+ fig.add_trace(go.Scatter(
588
+ x=xs, y=ys, mode="lines", name=str(nm),
589
+ line=dict(color=c, width=2), fill="tozeroy",
590
+ fillcolor=make_trans(col, 0.8), hoverinfo="x+name"))
591
+ axT1 = pretty(min(allx), max(allx))
592
+ ax_x = axis_num(Ynm, axT1, axis_format(axT1, d))
593
+ ax_y = axis_num("Density", pretty(0, max(ally)),
594
+ axis_format(pretty(0, max(ally)), d))
595
+ fig.update_layout(
596
+ xaxis=ax_x, yaxis=ax_y,
597
+ shapes=x_grid(axT1) + plot_border(), template=None,
598
+ plot_bgcolor=to_hex(style["panel_fill"]),
599
+ paper_bgcolor=to_hex(style["window_fill"]),
600
+ legend=dict(title=dict(text=str(X1nm) + " / " + str(X2nm))),
601
+ title=dict(text="Two-Group Density Plot", x=0.5,
602
+ xanchor="center",
603
+ font=dict(size=round(
604
+ 16 * get_option("main_size", 1)))))
605
+ return fig
606
+
607
+
608
+ def _one_group_plot(Y, Ynm, m, lb, ub, d, paired):
609
+ import plotly.graph_objects as go
610
+ from .plotly_utils import (
611
+ BASE_COLORS, axis_format, axis_num, make_trans,
612
+ plot_border, plotly_style, to_hex, x_grid)
613
+ from .utils import pretty
614
+ style = plotly_style()
615
+ xs, ys = _kde(Y)
616
+ c = BASE_COLORS[0]
617
+ fig = go.Figure()
618
+ fig.add_trace(go.Scatter(
619
+ x=xs, y=ys, mode="lines", line=dict(color=to_hex(c),
620
+ width=2),
621
+ fill="tozeroy", fillcolor=make_trans(c, 0.85),
622
+ hoverinfo="x", showlegend=False))
623
+ ym = float(np.interp(m, xs, ys))
624
+ fig.add_shape(type="line", x0=m, x1=m, y0=0, y1=ym,
625
+ line=dict(color=to_hex("gray50"), width=1))
626
+ axT1 = pretty(float(xs.min()), float(xs.max()))
627
+ xlab = ("Differences of Matched Pairs" if paired else Ynm)
628
+ ax_x = axis_num(xlab, axT1, axis_format(axT1, d))
629
+ ax_y = axis_num("Density", pretty(0, float(ys.max())),
630
+ axis_format(pretty(0, float(ys.max())), d))
631
+ fig.update_layout(
632
+ xaxis=ax_x, yaxis=ax_y,
633
+ shapes=x_grid(axT1) + plot_border(), template=None,
634
+ plot_bgcolor=to_hex(style["panel_fill"]),
635
+ paper_bgcolor=to_hex(style["window_fill"]),
636
+ title=dict(
637
+ text="One-Group Density Plot", x=0.5,
638
+ xanchor="center",
639
+ font=dict(size=round(16 * get_option("main_size",
640
+ 1)))))
641
+ return fig