lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/ttest.py
ADDED
|
@@ -0,0 +1,641 @@
|
|
|
1
|
+
# ttest.py — analog of ttest.R (tt.2group, tt.1group,
|
|
2
|
+
# tt.formula).
|
|
3
|
+
#
|
|
4
|
+
# ttest(): the t-test of a mean or a mean difference, with the
|
|
5
|
+
# lessR output pipeline — Describe, Assumptions (normality and,
|
|
6
|
+
# for two groups, homogeneity of variance), Infer (the equal-
|
|
7
|
+
# variance test and, for two groups, the Welch test), Effect
|
|
8
|
+
# Size (Cohen's d), Practical Importance, and Needed Sample Size,
|
|
9
|
+
# plus a plotly density plot. Modes, from the arguments:
|
|
10
|
+
# ttest("Y ~ Group", data=d) two independent groups
|
|
11
|
+
# ttest("V1", "V2", data=d) two independent groups
|
|
12
|
+
# ttest("V1", "V2", data=d, paired=True) paired
|
|
13
|
+
# ttest("V", data=d, mu=100) one group vs mu
|
|
14
|
+
# ttest("V", data=d) one group, CI only
|
|
15
|
+
# ttest(n1=, m1=, s1=, n2=, m2=, s2=) two groups from stats
|
|
16
|
+
# ttest(n=, m=, s=, mu=) one group from stats
|
|
17
|
+
#
|
|
18
|
+
# The larger-mean group is always reported first, as R. Numerics
|
|
19
|
+
# through scipy.stats (imported lazily); the pipeline is ported,
|
|
20
|
+
# not the lines. Returns a ttestResults object; the figure is in
|
|
21
|
+
# .plots and is not auto-shown.
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
import pandas as pd
|
|
25
|
+
|
|
26
|
+
from .Regression import _getdigits
|
|
27
|
+
from .utils import fmt, get_column, get_option
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ttestResults:
|
|
31
|
+
"""Numeric results of ttest(): the per-group summary, the
|
|
32
|
+
equal-variance and (two-group) Welch inference, the effect
|
|
33
|
+
size, and the plotly figure in .plots."""
|
|
34
|
+
|
|
35
|
+
def __init__(self, **kw):
|
|
36
|
+
self.__dict__.update(kw)
|
|
37
|
+
|
|
38
|
+
def __repr__(self):
|
|
39
|
+
return f"<lessPy ttest: {self.kind}, n={self.n_total}>"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _F(x, d):
|
|
43
|
+
return fmt(x, d)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def ttest(x=None, y=None, data=None, filter=None, paired=False,
|
|
47
|
+
n=None, m=None, s=None, mu=None,
|
|
48
|
+
n1=None, n2=None, m1=None, m2=None, s1=None, s2=None,
|
|
49
|
+
Ynm="Y", Xnm="X", X1nm="Group1", X2nm="Group2",
|
|
50
|
+
brief=False, digits_d=None, conf_level=0.95,
|
|
51
|
+
alternative="two_sided", mmd=None, msmd=None,
|
|
52
|
+
Edesired=None, graph=True):
|
|
53
|
+
"""t-test of a mean (one group, vs mu) or a mean difference
|
|
54
|
+
(two independent groups or paired). Always prints, as in R;
|
|
55
|
+
returns a ttestResults object with the density figure in
|
|
56
|
+
.plots."""
|
|
57
|
+
if alternative not in ("two_sided", "less", "greater"):
|
|
58
|
+
raise ValueError(
|
|
59
|
+
'alternative: "two_sided", "less", "greater"')
|
|
60
|
+
if data is not None and filter is not None:
|
|
61
|
+
data = data.query(filter)
|
|
62
|
+
|
|
63
|
+
from_stats = n is not None or n1 is not None
|
|
64
|
+
two_group = n1 is not None or (not from_stats and y is not None
|
|
65
|
+
and not paired)
|
|
66
|
+
|
|
67
|
+
if from_stats:
|
|
68
|
+
if n1 is not None:
|
|
69
|
+
return _two_group_stats(
|
|
70
|
+
n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm, X2nm,
|
|
71
|
+
brief, digits_d, conf_level, alternative, mmd,
|
|
72
|
+
msmd, Edesired)
|
|
73
|
+
return _one_group(
|
|
74
|
+
None, n, m, s, mu, Ynm, brief, conf_level,
|
|
75
|
+
alternative, digits_d, mmd, msmd, Edesired, False,
|
|
76
|
+
graph=False)
|
|
77
|
+
|
|
78
|
+
# from data: resolve x (and y) as column names or arrays
|
|
79
|
+
if isinstance(x, str) and "~" in x:
|
|
80
|
+
yv, xg, Ynm, Xnm, X1nm, X2nm = _formula(x, data)
|
|
81
|
+
return _two_group_data(
|
|
82
|
+
yv, xg, Ynm, Xnm, X1nm, X2nm, brief, digits_d,
|
|
83
|
+
conf_level, alternative, mmd, msmd, Edesired, graph)
|
|
84
|
+
|
|
85
|
+
xv = _resolve(x, data, "x")
|
|
86
|
+
if y is None:
|
|
87
|
+
Ynm = x if isinstance(x, str) else Ynm
|
|
88
|
+
return _one_group(
|
|
89
|
+
xv, None, None, None, mu, Ynm, brief, conf_level,
|
|
90
|
+
alternative, digits_d, mmd, msmd, Edesired, False,
|
|
91
|
+
graph=graph)
|
|
92
|
+
|
|
93
|
+
yv = _resolve(y, data, "y")
|
|
94
|
+
if paired:
|
|
95
|
+
if len(xv) != len(yv):
|
|
96
|
+
raise ValueError("paired samples must be equal length")
|
|
97
|
+
keep = ~(np.isnan(xv) | np.isnan(yv))
|
|
98
|
+
diff = yv[keep] - xv[keep] # y - x, as R
|
|
99
|
+
return _one_group(
|
|
100
|
+
diff, None, None, None, 0.0, "Difference", brief,
|
|
101
|
+
conf_level, alternative, digits_d, mmd, msmd,
|
|
102
|
+
Edesired, True, graph=graph)
|
|
103
|
+
|
|
104
|
+
X1nm = x if isinstance(x, str) else X1nm
|
|
105
|
+
X2nm = y if isinstance(y, str) else X2nm
|
|
106
|
+
Ynm = "Y"
|
|
107
|
+
return _two_group_data(
|
|
108
|
+
{X1nm: xv, X2nm: yv}, None, Ynm, Xnm, X1nm, X2nm, brief,
|
|
109
|
+
digits_d, conf_level, alternative, mmd, msmd, Edesired,
|
|
110
|
+
graph)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _resolve(v, data, arg):
|
|
114
|
+
if isinstance(v, str):
|
|
115
|
+
if data is None:
|
|
116
|
+
raise ValueError(f"{arg}='{v}' is a column name, so "
|
|
117
|
+
"data= is required")
|
|
118
|
+
return get_column(data, v, arg).to_numpy(dtype=float)
|
|
119
|
+
return np.asarray(v, dtype=float)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _formula(f, data):
|
|
123
|
+
if data is None:
|
|
124
|
+
raise ValueError("a formula needs data=")
|
|
125
|
+
lhs, rhs = (t.strip() for t in f.split("~"))
|
|
126
|
+
y_ser = get_column(data, lhs, "response")
|
|
127
|
+
g_ser = get_column(data, rhs, "grouping")
|
|
128
|
+
if not pd.api.types.is_numeric_dtype(y_ser):
|
|
129
|
+
raise TypeError(f"the response '{lhs}' must be numeric")
|
|
130
|
+
levels = list(pd.unique(g_ser.dropna().astype(str)))
|
|
131
|
+
if len(levels) != 2:
|
|
132
|
+
raise ValueError(
|
|
133
|
+
f"the grouping variable '{rhs}' must have exactly "
|
|
134
|
+
f"two values; found {len(levels)}. Use ANOVA for "
|
|
135
|
+
"more than two groups.")
|
|
136
|
+
levels = sorted(levels)
|
|
137
|
+
gs = g_ser.astype(str)
|
|
138
|
+
groups = {lv: y_ser[gs == lv].to_numpy(dtype=float)
|
|
139
|
+
for lv in levels}
|
|
140
|
+
return groups, None, lhs, rhs, levels[0], levels[1]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# ------------------------------------------------------------
|
|
144
|
+
# formatting helpers for the group summary
|
|
145
|
+
# ------------------------------------------------------------
|
|
146
|
+
|
|
147
|
+
def _alt_word(alt):
|
|
148
|
+
return {"two_sided": "two.sided", "less": "less",
|
|
149
|
+
"greater": "greater"}[alt]
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _tcrit(conf_level, df, alt):
|
|
153
|
+
from scipy.stats import t as tdist
|
|
154
|
+
if alt == "two_sided":
|
|
155
|
+
return float(tdist.ppf(1 - (1 - conf_level) / 2, df))
|
|
156
|
+
return float(tdist.ppf(conf_level, df))
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _pval(tvalue, df, alt):
|
|
160
|
+
from scipy.stats import t as tdist
|
|
161
|
+
if alt == "two_sided":
|
|
162
|
+
return float(2 * tdist.sf(abs(tvalue), df))
|
|
163
|
+
if alt == "less":
|
|
164
|
+
return float(tdist.cdf(tvalue, df))
|
|
165
|
+
return float(tdist.sf(tvalue, df))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# ------------------------------------------------------------
|
|
169
|
+
# two independent groups
|
|
170
|
+
# ------------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
def _two_group_data(groups, _g, Ynm, Xnm, X1nm, X2nm, brief,
|
|
173
|
+
digits_d, conf_level, alternative, mmd, msmd,
|
|
174
|
+
Edesired, graph):
|
|
175
|
+
YA0 = groups[X1nm]
|
|
176
|
+
YB0 = groups[X2nm]
|
|
177
|
+
n1m = int(np.isnan(YA0).sum())
|
|
178
|
+
n2m = int(np.isnan(YB0).sum())
|
|
179
|
+
YA = YA0[~np.isnan(YA0)]
|
|
180
|
+
YB = YB0[~np.isnan(YB0)]
|
|
181
|
+
if len(YA) < 2 or len(YB) < 2:
|
|
182
|
+
raise ValueError("need at least two cases per sample")
|
|
183
|
+
# the larger-mean group is reported first, as R
|
|
184
|
+
if YA.mean() < YB.mean():
|
|
185
|
+
YA, YB = YB, YA
|
|
186
|
+
X1nm, X2nm = X2nm, X1nm
|
|
187
|
+
n1m, n2m = n2m, n1m
|
|
188
|
+
d = digits_d if digits_d is not None else _getdigits(
|
|
189
|
+
np.concatenate([YA, YB]), 3)
|
|
190
|
+
return _two_group(
|
|
191
|
+
len(YA), YA.mean(), YA.std(ddof=1),
|
|
192
|
+
len(YB), YB.mean(), YB.std(ddof=1),
|
|
193
|
+
Ynm, Xnm, X1nm, X2nm, brief, d, conf_level, alternative,
|
|
194
|
+
mmd, msmd, Edesired, YA, YB, n1m, n2m, graph)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _two_group_stats(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm,
|
|
198
|
+
X2nm, brief, digits_d, conf_level,
|
|
199
|
+
alternative, mmd, msmd, Edesired):
|
|
200
|
+
if m1 < m2: # larger mean first
|
|
201
|
+
n1, n2 = n2, n1
|
|
202
|
+
m1, m2 = m2, m1
|
|
203
|
+
s1, s2 = s2, s1
|
|
204
|
+
X1nm, X2nm = X2nm, X1nm
|
|
205
|
+
d = digits_d if digits_d is not None else 3
|
|
206
|
+
return _two_group(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm,
|
|
207
|
+
X2nm, brief, d, conf_level, alternative,
|
|
208
|
+
mmd, msmd, Edesired, None, None, 0, 0,
|
|
209
|
+
False)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _two_group(n1, m1, s1, n2, m2, s2, Ynm, Xnm, X1nm, X2nm,
|
|
213
|
+
brief, d, conf_level, alternative, mmd, msmd,
|
|
214
|
+
Edesired, YA, YB, n1m, n2m, graph):
|
|
215
|
+
from scipy.stats import f as fdist
|
|
216
|
+
from_data = YA is not None
|
|
217
|
+
v1, v2 = s1 ** 2, s2 ** 2
|
|
218
|
+
sd_d = d if from_data else d - 1
|
|
219
|
+
clpct = f"{round(conf_level * 100, 2):g}%"
|
|
220
|
+
alt = alternative
|
|
221
|
+
|
|
222
|
+
L = ["", f"Compare {Ynm} across {Xnm} with levels "
|
|
223
|
+
f"{X1nm} and {X2nm}",
|
|
224
|
+
f"Grouping Variable: {Xnm}",
|
|
225
|
+
f"Response Variable: {Ynm}", ""]
|
|
226
|
+
|
|
227
|
+
L.append("------ Describe ------" if not brief
|
|
228
|
+
else " --- Describe ---")
|
|
229
|
+
L.append("")
|
|
230
|
+
miss1 = f"n.miss = {n1m}, " if from_data else ""
|
|
231
|
+
miss2 = f"n.miss = {n2m}, " if from_data else ""
|
|
232
|
+
L.append(f"{Ynm} for {Xnm} {X1nm}: {miss1}n = {n1}, "
|
|
233
|
+
f" mean = {_F(m1, sd_d)}, sd = {_F(s1, sd_d)}")
|
|
234
|
+
L.append(f"{Ynm} for {Xnm} {X2nm}: {miss2}n = {n2}, "
|
|
235
|
+
f" mean = {_F(m2, sd_d)}, sd = {_F(s2, sd_d)}")
|
|
236
|
+
L += ["", f"Mean Difference of {Ynm}: {_F(m1 - m2, d)}"]
|
|
237
|
+
|
|
238
|
+
df1, df2 = n1 - 1, n2 - 1
|
|
239
|
+
swsq = (df1 * v1 + df2 * v2) / (df1 + df2)
|
|
240
|
+
sw = np.sqrt(swsq)
|
|
241
|
+
smd = (m1 - m2) / sw
|
|
242
|
+
L += ["", f"Weighted Average Standard Deviation: {_F(sw, d)}"]
|
|
243
|
+
if brief:
|
|
244
|
+
L.append(f"Standardized Mean Difference of {Ynm}: "
|
|
245
|
+
f"{_F(smd, d)}")
|
|
246
|
+
|
|
247
|
+
if not brief:
|
|
248
|
+
L += _assumptions(YA, YB, X1nm, X2nm, Ynm, n1, n2, v1,
|
|
249
|
+
v2, df1, df2, from_data, d)
|
|
250
|
+
|
|
251
|
+
# ----- Infer: equal variances (pooled) -----
|
|
252
|
+
L.append("")
|
|
253
|
+
L.append("------ Infer ------" if not brief
|
|
254
|
+
else " --- Infer ---")
|
|
255
|
+
L.append("")
|
|
256
|
+
if not brief:
|
|
257
|
+
L.append("--- Assume equal population variances of "
|
|
258
|
+
f"{Ynm} for each {Xnm}")
|
|
259
|
+
L.append("")
|
|
260
|
+
sterr = sw * np.sqrt(1 / n1 + 1 / n2)
|
|
261
|
+
df = df1 + df2
|
|
262
|
+
tcut = _tcrit(conf_level, df, alt)
|
|
263
|
+
tvalue = (m1 - m2) / sterr
|
|
264
|
+
pvalue = _pval(tvalue, df, alt)
|
|
265
|
+
E = tcut * sterr
|
|
266
|
+
lb, ub = (m1 - m2) - E, (m1 - m2) + E
|
|
267
|
+
if alt != "two_sided":
|
|
268
|
+
L.append("Alternative hypothesis: Population mean "
|
|
269
|
+
f"difference is {_alt_word(alt)} than 0")
|
|
270
|
+
L += [f"t-cutoff for {clpct} range of variation: "
|
|
271
|
+
f"tcut = {_F(tcut, 3)}",
|
|
272
|
+
f"Standard Error of Mean Difference: SE = {_F(sterr, d)}",
|
|
273
|
+
"",
|
|
274
|
+
f"Hypothesis Test of 0 Mean Diff: t-value = "
|
|
275
|
+
f"{_F(tvalue, 3)}, df = {df}, p-value = "
|
|
276
|
+
f"{_F(pvalue, 3)}", "",
|
|
277
|
+
f"Margin of Error for {clpct} Confidence Level: "
|
|
278
|
+
f"{_F(E, d)}",
|
|
279
|
+
f"{clpct} Confidence Interval for Mean Difference: "
|
|
280
|
+
f"{_F(lb, d)} to {_F(ub, d)}"]
|
|
281
|
+
|
|
282
|
+
welch = None
|
|
283
|
+
if not brief:
|
|
284
|
+
welch = _welch(m1, m2, v1, v2, n1, n2, conf_level, alt,
|
|
285
|
+
YA, YB)
|
|
286
|
+
L += ["", "--- Do not assume equal population variances "
|
|
287
|
+
f"of {Ynm} for each {Xnm}", "",
|
|
288
|
+
f"t-cutoff: tcut = {_F(welch['tcut'], 3)}",
|
|
289
|
+
f"Standard Error of Mean Difference: SE = "
|
|
290
|
+
f"{_F(welch['sterr'], d)}", "",
|
|
291
|
+
f"Hypothesis Test of 0 Mean Diff: t = "
|
|
292
|
+
f"{_F(welch['t'], 3)}, df = {_F(welch['df'], 3)}, "
|
|
293
|
+
f"p-value = {_F(welch['p'], 3)}", "",
|
|
294
|
+
f"Margin of Error for {clpct} Confidence Level: "
|
|
295
|
+
f"{_F(welch['E'], d)}",
|
|
296
|
+
f"{clpct} Confidence Interval for Mean Difference: "
|
|
297
|
+
f"{_F(welch['lb'], d)} to {_F(welch['ub'], d)}"]
|
|
298
|
+
|
|
299
|
+
L += ["", "------ Effect Size ------", "",
|
|
300
|
+
"--- Assume equal population variances of "
|
|
301
|
+
f"{Ynm} for each {Xnm}", "",
|
|
302
|
+
f"Standardized Mean Difference of {Ynm}, "
|
|
303
|
+
f"Cohen's d: {_F(smd, d)}"]
|
|
304
|
+
L += _practical(mmd, msmd, sw, m1 - m2, smd, lb, ub, d)
|
|
305
|
+
L += _needed_2(Edesired, conf_level, sw, E, n1, n2, d)
|
|
306
|
+
|
|
307
|
+
print("\n".join(L))
|
|
308
|
+
plots = {}
|
|
309
|
+
if graph and from_data:
|
|
310
|
+
plots["two_group"] = _two_group_plot(
|
|
311
|
+
YA, YB, Ynm, X1nm, X2nm, m1, m2, d)
|
|
312
|
+
return ttestResults(
|
|
313
|
+
kind="two-group", n_total=n1 + n2, digits_d=d,
|
|
314
|
+
group1={"name": X1nm, "n": n1, "mean": m1, "sd": s1},
|
|
315
|
+
group2={"name": X2nm, "n": n2, "mean": m2, "sd": s2},
|
|
316
|
+
mean_diff=m1 - m2, pooled_sd=sw, cohen_d=smd,
|
|
317
|
+
equal_var={"t": tvalue, "df": df, "p_value": pvalue,
|
|
318
|
+
"se": sterr, "lb": lb, "ub": ub},
|
|
319
|
+
welch=welch, plots=plots)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _assumptions(YA, YB, X1nm, X2nm, Ynm, n1, n2, v1, v2, df1,
|
|
323
|
+
df2, from_data, d):
|
|
324
|
+
from scipy.stats import f as fdist, shapiro
|
|
325
|
+
L = ["", "", "------ Assumptions ------", "",
|
|
326
|
+
"Note: These hypothesis tests can perform poorly, and "
|
|
327
|
+
"the",
|
|
328
|
+
" t-test is typically robust to violations of "
|
|
329
|
+
"assumptions.",
|
|
330
|
+
" Use as heuristic guides instead of interpreting "
|
|
331
|
+
"literally.", ""]
|
|
332
|
+
if from_data:
|
|
333
|
+
L.append("Null hypothesis, for each group, is a normal "
|
|
334
|
+
f"distribution of {Ynm}.")
|
|
335
|
+
for nm, Y, ni in ((X1nm, YA, n1), (X2nm, YB, n2)):
|
|
336
|
+
if ni > 30:
|
|
337
|
+
L.append(f"Group {nm}: Sample mean assumed "
|
|
338
|
+
"normal because n > 30, so no test "
|
|
339
|
+
"needed.")
|
|
340
|
+
elif 2 < ni < 5000:
|
|
341
|
+
W, p = shapiro(Y)
|
|
342
|
+
L.append(f"Group {nm} Shapiro-Wilk normality "
|
|
343
|
+
f"test: W = {_F(W, 3)}, p-value = "
|
|
344
|
+
f"{_F(p, 3)}")
|
|
345
|
+
else:
|
|
346
|
+
L.append(f"Group {nm} Sample size out of range "
|
|
347
|
+
"for Shapiro-Wilk normality test.")
|
|
348
|
+
L.append("")
|
|
349
|
+
# variance ratio F test
|
|
350
|
+
if v1 >= v2:
|
|
351
|
+
vr, dfn, dfd = v1 / v2, df1, df2
|
|
352
|
+
vrs = f"{_F(v1, d)}/{_F(v2, d)}"
|
|
353
|
+
else:
|
|
354
|
+
vr, dfn, dfd = v2 / v1, df2, df1
|
|
355
|
+
vrs = f"{_F(v2, d)}/{_F(v1, d)}"
|
|
356
|
+
pv = float(fdist.cdf(vr, dfn, dfd))
|
|
357
|
+
pv = 2 * min(pv, 1 - pv)
|
|
358
|
+
L.append(f"Null hypothesis is equal variances of {Ynm}, "
|
|
359
|
+
"homogeneous.")
|
|
360
|
+
L.append(f"Variance Ratio test: F = {vrs} = {_F(vr, d)}, "
|
|
361
|
+
f" df = {dfn};{dfd}, p-value = {_F(pv, 3)}")
|
|
362
|
+
if from_data:
|
|
363
|
+
# Levene, Brown-Forsythe: pooled t on |Y - median|
|
|
364
|
+
a = np.abs(YA - np.median(YA))
|
|
365
|
+
b = np.abs(YB - np.median(YB))
|
|
366
|
+
t_bf, df_bf, p_bf = _pooled_t(a, b)
|
|
367
|
+
L.append(f"Levene's test, Brown-Forsythe: t = "
|
|
368
|
+
f"{_F(t_bf, 3)}, df = {df_bf}, p-value = "
|
|
369
|
+
f"{_F(p_bf, 3)}")
|
|
370
|
+
return L
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _pooled_t(a, b):
|
|
374
|
+
na, nb = len(a), len(b)
|
|
375
|
+
va, vb = a.var(ddof=1), b.var(ddof=1)
|
|
376
|
+
dfa, dfb = na - 1, nb - 1
|
|
377
|
+
sw = np.sqrt((dfa * va + dfb * vb) / (dfa + dfb))
|
|
378
|
+
se = sw * np.sqrt(1 / na + 1 / nb)
|
|
379
|
+
t = (a.mean() - b.mean()) / se
|
|
380
|
+
df = dfa + dfb
|
|
381
|
+
return t, df, _pval(t, df, "two_sided")
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _welch(m1, m2, v1, v2, n1, n2, conf_level, alt, YA, YB):
|
|
385
|
+
k1, k2 = v1 / n1, v2 / n2
|
|
386
|
+
df = (k1 + k2) ** 2 / (k1 ** 2 / (n1 - 1) + k2 ** 2 / (n2 - 1))
|
|
387
|
+
sterr = np.sqrt(k1 + k2)
|
|
388
|
+
tcut = _tcrit(conf_level, df, alt)
|
|
389
|
+
t = (m1 - m2) / sterr
|
|
390
|
+
p = _pval(t, df, alt)
|
|
391
|
+
E = tcut * sterr
|
|
392
|
+
return {"t": t, "df": df, "p": p, "se": sterr, "tcut": tcut,
|
|
393
|
+
"E": E, "lb": (m1 - m2) - E, "ub": (m1 - m2) + E,
|
|
394
|
+
"sterr": sterr}
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _practical(mmd, msmd, sw, mdiff, smd, lb, ub, d):
|
|
398
|
+
L = ["", "", "------ Practical Importance ------", "",
|
|
399
|
+
"Minimum Mean Difference of practical importance: mmd"]
|
|
400
|
+
if mmd is not None or msmd is not None:
|
|
401
|
+
if mmd is not None:
|
|
402
|
+
msmd = mmd / sw
|
|
403
|
+
else:
|
|
404
|
+
mmd = msmd * sw
|
|
405
|
+
L += [f"Compare mmd = {_F(mmd, d)} to the obtained value "
|
|
406
|
+
f"of md = {_F(mdiff, d)}",
|
|
407
|
+
f"Compare mmd to the confidence interval for md: "
|
|
408
|
+
f"{_F(lb, d)} to {_F(ub, d)}", "",
|
|
409
|
+
"Minimum Standardized Mean Difference of practical "
|
|
410
|
+
"importance: msmd",
|
|
411
|
+
f"Compare msmd = {_F(msmd, d)} to the obtained "
|
|
412
|
+
f"value of smd = {_F(smd, d)}"]
|
|
413
|
+
else:
|
|
414
|
+
L += ["Minimum Standardized Mean Difference of "
|
|
415
|
+
"practical importance: msmd",
|
|
416
|
+
"Neither value specified, so no analysis"]
|
|
417
|
+
return L
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _needed_2(Edesired, conf_level, sw, E, n1, n2, d):
|
|
421
|
+
if Edesired is None:
|
|
422
|
+
return []
|
|
423
|
+
from scipy.stats import norm
|
|
424
|
+
zcut = norm.ppf((1 - conf_level) / 2)
|
|
425
|
+
ns = 2 * ((zcut * sw) / Edesired) ** 2
|
|
426
|
+
needed = int(np.ceil(1.099 * ns + 4.863))
|
|
427
|
+
L = ["", "", "------ Needed Sample Size ------", ""]
|
|
428
|
+
if Edesired > E:
|
|
429
|
+
L.append(f"Note: Desired margin of error, {_F(Edesired, d)}"
|
|
430
|
+
f" is worse than what was obtained, {_F(E, d)}")
|
|
431
|
+
L.append("")
|
|
432
|
+
L += [f"Desired Margin of Error: {_F(Edesired, d)}", "",
|
|
433
|
+
"For the following sample size there is a 0.9 "
|
|
434
|
+
"probability of obtaining",
|
|
435
|
+
f"the desired margin of error for the resulting "
|
|
436
|
+
"confidence interval.",
|
|
437
|
+
f"Needed sample size per group: {needed}", "",
|
|
438
|
+
f"Additional data values needed Group 1: {needed - n1}",
|
|
439
|
+
f"Additional data values needed Group 2: {needed - n2}"]
|
|
440
|
+
return L
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
# ------------------------------------------------------------
|
|
444
|
+
# one group (and paired, on the differences)
|
|
445
|
+
# ------------------------------------------------------------
|
|
446
|
+
|
|
447
|
+
def _one_group(Y, n, m, s, mu, Ynm, brief, conf_level,
|
|
448
|
+
alternative, digits_d, mmd, msmd, Edesired,
|
|
449
|
+
paired, graph):
|
|
450
|
+
from scipy.stats import shapiro
|
|
451
|
+
from_data = Y is not None
|
|
452
|
+
if from_data:
|
|
453
|
+
Y0 = np.asarray(Y, dtype=float)
|
|
454
|
+
n_miss = int(np.isnan(Y0).sum())
|
|
455
|
+
Y = Y0[~np.isnan(Y0)]
|
|
456
|
+
n, m, s = len(Y), Y.mean(), Y.std(ddof=1)
|
|
457
|
+
d = digits_d if digits_d is not None else _getdigits(Y, 3)
|
|
458
|
+
else:
|
|
459
|
+
n_miss = 0
|
|
460
|
+
d = digits_d if digits_d is not None else 3
|
|
461
|
+
sd_d = d if from_data else d - 1
|
|
462
|
+
clpct = f"{round(conf_level * 100, 2):g}%"
|
|
463
|
+
alt = alternative
|
|
464
|
+
|
|
465
|
+
L = ["", f"------ Describe ------" if not brief
|
|
466
|
+
else " --- Describe ---", ""]
|
|
467
|
+
lead = f"{Ynm}: " if Ynm not in ("Y",) else ""
|
|
468
|
+
miss = f"n.miss = {n_miss}, " if from_data else ""
|
|
469
|
+
L.append(f"{lead}{miss}n = {n}, mean = {_F(m, sd_d)}, "
|
|
470
|
+
f" sd = {_F(s, sd_d)}")
|
|
471
|
+
|
|
472
|
+
if not brief and from_data:
|
|
473
|
+
L += ["", "", "------ Normality Assumption ------", ""]
|
|
474
|
+
if n > 30:
|
|
475
|
+
L.append("Sample mean assumed normal because n > 30, "
|
|
476
|
+
"so no test needed.")
|
|
477
|
+
elif 2 < n < 5000:
|
|
478
|
+
W, p = shapiro(Y)
|
|
479
|
+
L += [f"Null hypothesis is a normal distribution of "
|
|
480
|
+
f"{Ynm}.",
|
|
481
|
+
f"Shapiro-Wilk normality test: W = {_F(W, 3)}, "
|
|
482
|
+
f" p-value = {_F(p, 3)}"]
|
|
483
|
+
else:
|
|
484
|
+
L.append("Sample size out of range for Shapiro-Wilk "
|
|
485
|
+
"normality test.")
|
|
486
|
+
|
|
487
|
+
L += ["", "", "------ Infer ------" if not brief
|
|
488
|
+
else " --- Infer ---", ""]
|
|
489
|
+
df = n - 1
|
|
490
|
+
sterr = s * np.sqrt(1 / n)
|
|
491
|
+
tcut = _tcrit(conf_level, df, alt)
|
|
492
|
+
E = tcut * sterr
|
|
493
|
+
lb, ub = m - E, m + E
|
|
494
|
+
tvalue = pvalue = None
|
|
495
|
+
if mu is not None:
|
|
496
|
+
tvalue = (m - mu) / sterr
|
|
497
|
+
pvalue = _pval(tvalue, df, alt)
|
|
498
|
+
L += [f"t-cutoff for {clpct} range of variation: "
|
|
499
|
+
f"tcut = {_F(tcut, 3)}",
|
|
500
|
+
f"Standard Error of Mean: SE = {_F(sterr, d)}", ""]
|
|
501
|
+
if mu is not None:
|
|
502
|
+
if alt != "two_sided":
|
|
503
|
+
L.append("Alternative hypothesis: Population mean is "
|
|
504
|
+
f"{_alt_word(alt)} than {mu}")
|
|
505
|
+
L += [f"Hypothesized Value H0: mu = {mu}",
|
|
506
|
+
f"Hypothesis Test of Mean: t-value = "
|
|
507
|
+
f"{_F(tvalue, 3)}, df = {df}, p-value = "
|
|
508
|
+
f"{_F(pvalue, 3)}", ""]
|
|
509
|
+
L += [f"Margin of Error for {clpct} Confidence Level: "
|
|
510
|
+
f"{_F(E, d)}",
|
|
511
|
+
f"{clpct} Confidence Interval for Mean: {_F(lb, d)} "
|
|
512
|
+
f"to {_F(ub, d)}"]
|
|
513
|
+
|
|
514
|
+
cohen = None
|
|
515
|
+
if mu is not None:
|
|
516
|
+
mdiff = m - mu
|
|
517
|
+
cohen = abs(mdiff / s)
|
|
518
|
+
L += ["", "", "------ Effect Size ------", "",
|
|
519
|
+
f"Distance of sample mean from hypothesized: "
|
|
520
|
+
f"{_F(mdiff, d)}",
|
|
521
|
+
f"Standardized Distance, Cohen's d: "
|
|
522
|
+
f"{_F(cohen, d)}"]
|
|
523
|
+
|
|
524
|
+
if Edesired is not None and from_data:
|
|
525
|
+
L += _needed_1(Edesired, conf_level, s, E, n, d)
|
|
526
|
+
|
|
527
|
+
print("\n".join(L))
|
|
528
|
+
plots = {}
|
|
529
|
+
if graph and from_data:
|
|
530
|
+
plots["one_group"] = _one_group_plot(
|
|
531
|
+
Y, Ynm, m, lb, ub, d, paired)
|
|
532
|
+
return ttestResults(
|
|
533
|
+
kind="paired" if paired else "one-group", n_total=n,
|
|
534
|
+
digits_d=d, n=n, mean=m, sd=s, mu=mu,
|
|
535
|
+
infer={"t": tvalue, "df": df, "p_value": pvalue,
|
|
536
|
+
"se": sterr, "lb": lb, "ub": ub},
|
|
537
|
+
cohen_d=cohen, plots=plots)
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _needed_1(Edesired, conf_level, s, E, n, d):
|
|
541
|
+
from scipy.stats import norm
|
|
542
|
+
zcut = norm.ppf((1 - conf_level) / 2)
|
|
543
|
+
ns = ((zcut * s) / Edesired) ** 2
|
|
544
|
+
needed = int(np.ceil(1.132 * ns + 7.368))
|
|
545
|
+
L = ["", "", "------ Needed Sample Size ------", ""]
|
|
546
|
+
if Edesired > E:
|
|
547
|
+
L += [f"Note: Desired margin of error, {_F(Edesired, d)}"
|
|
548
|
+
f" is worse than what was obtained, {_F(E, d)}", ""]
|
|
549
|
+
L += [f"Desired Margin of Error: {_F(Edesired, d)}", "",
|
|
550
|
+
"For the following sample size there is a 0.9 "
|
|
551
|
+
"probability of obtaining",
|
|
552
|
+
"the desired margin of error for the confidence "
|
|
553
|
+
"interval.",
|
|
554
|
+
f"Needed sample size: {needed}", "",
|
|
555
|
+
f"Additional data values needed: {needed - n}"]
|
|
556
|
+
return L
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
# ------------------------------------------------------------
|
|
560
|
+
# density plots
|
|
561
|
+
# ------------------------------------------------------------
|
|
562
|
+
|
|
563
|
+
def _kde(v):
|
|
564
|
+
from scipy.stats import gaussian_kde
|
|
565
|
+
k = gaussian_kde(v)
|
|
566
|
+
lo, hi = v.min(), v.max()
|
|
567
|
+
pad = 0.15 * (hi - lo if hi > lo else 1)
|
|
568
|
+
xs = np.linspace(lo - pad, hi + pad, 200)
|
|
569
|
+
return xs, k(xs)
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _two_group_plot(YA, YB, Ynm, X1nm, X2nm, m1, m2, d):
|
|
573
|
+
import plotly.graph_objects as go
|
|
574
|
+
from .plotly_utils import (
|
|
575
|
+
BASE_COLORS, axis_format, axis_num, make_trans,
|
|
576
|
+
plot_border, plotly_style, to_hex, x_grid)
|
|
577
|
+
from .utils import pretty
|
|
578
|
+
style = plotly_style()
|
|
579
|
+
fig = go.Figure()
|
|
580
|
+
allx, ally = [], []
|
|
581
|
+
for (nm, v, col) in ((X1nm, YA, BASE_COLORS[1]),
|
|
582
|
+
(X2nm, YB, BASE_COLORS[0])):
|
|
583
|
+
xs, ys = _kde(v)
|
|
584
|
+
allx += [xs.min(), xs.max()]
|
|
585
|
+
ally.append(ys.max())
|
|
586
|
+
c = to_hex(col)
|
|
587
|
+
fig.add_trace(go.Scatter(
|
|
588
|
+
x=xs, y=ys, mode="lines", name=str(nm),
|
|
589
|
+
line=dict(color=c, width=2), fill="tozeroy",
|
|
590
|
+
fillcolor=make_trans(col, 0.8), hoverinfo="x+name"))
|
|
591
|
+
axT1 = pretty(min(allx), max(allx))
|
|
592
|
+
ax_x = axis_num(Ynm, axT1, axis_format(axT1, d))
|
|
593
|
+
ax_y = axis_num("Density", pretty(0, max(ally)),
|
|
594
|
+
axis_format(pretty(0, max(ally)), d))
|
|
595
|
+
fig.update_layout(
|
|
596
|
+
xaxis=ax_x, yaxis=ax_y,
|
|
597
|
+
shapes=x_grid(axT1) + plot_border(), template=None,
|
|
598
|
+
plot_bgcolor=to_hex(style["panel_fill"]),
|
|
599
|
+
paper_bgcolor=to_hex(style["window_fill"]),
|
|
600
|
+
legend=dict(title=dict(text=str(X1nm) + " / " + str(X2nm))),
|
|
601
|
+
title=dict(text="Two-Group Density Plot", x=0.5,
|
|
602
|
+
xanchor="center",
|
|
603
|
+
font=dict(size=round(
|
|
604
|
+
16 * get_option("main_size", 1)))))
|
|
605
|
+
return fig
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _one_group_plot(Y, Ynm, m, lb, ub, d, paired):
|
|
609
|
+
import plotly.graph_objects as go
|
|
610
|
+
from .plotly_utils import (
|
|
611
|
+
BASE_COLORS, axis_format, axis_num, make_trans,
|
|
612
|
+
plot_border, plotly_style, to_hex, x_grid)
|
|
613
|
+
from .utils import pretty
|
|
614
|
+
style = plotly_style()
|
|
615
|
+
xs, ys = _kde(Y)
|
|
616
|
+
c = BASE_COLORS[0]
|
|
617
|
+
fig = go.Figure()
|
|
618
|
+
fig.add_trace(go.Scatter(
|
|
619
|
+
x=xs, y=ys, mode="lines", line=dict(color=to_hex(c),
|
|
620
|
+
width=2),
|
|
621
|
+
fill="tozeroy", fillcolor=make_trans(c, 0.85),
|
|
622
|
+
hoverinfo="x", showlegend=False))
|
|
623
|
+
ym = float(np.interp(m, xs, ys))
|
|
624
|
+
fig.add_shape(type="line", x0=m, x1=m, y0=0, y1=ym,
|
|
625
|
+
line=dict(color=to_hex("gray50"), width=1))
|
|
626
|
+
axT1 = pretty(float(xs.min()), float(xs.max()))
|
|
627
|
+
xlab = ("Differences of Matched Pairs" if paired else Ynm)
|
|
628
|
+
ax_x = axis_num(xlab, axT1, axis_format(axT1, d))
|
|
629
|
+
ax_y = axis_num("Density", pretty(0, float(ys.max())),
|
|
630
|
+
axis_format(pretty(0, float(ys.max())), d))
|
|
631
|
+
fig.update_layout(
|
|
632
|
+
xaxis=ax_x, yaxis=ax_y,
|
|
633
|
+
shapes=x_grid(axT1) + plot_border(), template=None,
|
|
634
|
+
plot_bgcolor=to_hex(style["panel_fill"]),
|
|
635
|
+
paper_bgcolor=to_hex(style["window_fill"]),
|
|
636
|
+
title=dict(
|
|
637
|
+
text="One-Group Density Plot", x=0.5,
|
|
638
|
+
xanchor="center",
|
|
639
|
+
font=dict(size=round(16 * get_option("main_size",
|
|
640
|
+
1)))))
|
|
641
|
+
return fig
|