lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Prop_test.py
ADDED
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
# Prop_test.py — analog of Prop_test.R.
|
|
2
|
+
#
|
|
3
|
+
# Prop_test(): four proportion analyses, chosen from the
|
|
4
|
+
# arguments —
|
|
5
|
+
# one proportion variable + success (data) | n_succ + n_tot
|
|
6
|
+
# exact binomial test vs pi (default 0.5)
|
|
7
|
+
# many proportions variable + success + by | n_succ,n_tot lists
|
|
8
|
+
# chi-square test of equal proportions
|
|
9
|
+
# goodness-of-fit variable (data) | n_tot vector
|
|
10
|
+
# chi-square vs equal proportions
|
|
11
|
+
# independence variable + by (data) | n_table
|
|
12
|
+
# chi-square cross-tab test, with Cramer's V
|
|
13
|
+
# Numerics through scipy.stats (binomtest, chi2_contingency,
|
|
14
|
+
# chisquare). Prints as R; returns a PropResults object.
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
from .utils import fmt, get_column
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class PropResults:
|
|
23
|
+
"""Results of Prop_test(): the analysis kind and its
|
|
24
|
+
statistics (proportion/p-value/CI, or chi-square/df/p and
|
|
25
|
+
the tables)."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, **kw):
|
|
28
|
+
self.__dict__.update(kw)
|
|
29
|
+
|
|
30
|
+
def __repr__(self):
|
|
31
|
+
return f"<lessPy Prop_test: {self.kind}>"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def Prop_test(variable=None, success=None, by=None, data=None,
|
|
35
|
+
n_succ=None, n_fail=None, n_tot=None, n_table=None,
|
|
36
|
+
Yates=False, pi=None, digits_d=3,
|
|
37
|
+
alternative="two_sided"):
|
|
38
|
+
"""Test of one or more proportions, goodness-of-fit, or
|
|
39
|
+
cross-tab independence, from data columns or from summary
|
|
40
|
+
counts. R analog: Prop_test()"""
|
|
41
|
+
if alternative not in ("two_sided", "less", "greater"):
|
|
42
|
+
raise ValueError('alternative: "two_sided", "less", '
|
|
43
|
+
'"greater"')
|
|
44
|
+
do_data = (n_succ is None and n_tot is None
|
|
45
|
+
and n_table is None)
|
|
46
|
+
d = digits_d
|
|
47
|
+
|
|
48
|
+
if do_data:
|
|
49
|
+
if variable is None:
|
|
50
|
+
raise ValueError("variable= is required with data=")
|
|
51
|
+
v = get_column(data, variable, "variable")
|
|
52
|
+
g = get_column(data, by, "by") if by is not None else None
|
|
53
|
+
if success is not None and g is None:
|
|
54
|
+
return _one_prop_data(v, success, variable, pi, d,
|
|
55
|
+
alternative)
|
|
56
|
+
if success is not None and g is not None:
|
|
57
|
+
return _many_prop_data(v, g, success, variable, by,
|
|
58
|
+
Yates, d)
|
|
59
|
+
if success is None and g is None:
|
|
60
|
+
return _gof_data(v, variable, d)
|
|
61
|
+
return _crosstab_data(v, g, variable, by, Yates, d)
|
|
62
|
+
|
|
63
|
+
# summary counts
|
|
64
|
+
if n_fail is not None and n_tot is None:
|
|
65
|
+
n_tot = (np.add(n_succ, n_fail) if np.ndim(n_succ)
|
|
66
|
+
else n_succ + n_fail)
|
|
67
|
+
if n_table is not None:
|
|
68
|
+
return _crosstab_table(n_table, Yates, d)
|
|
69
|
+
if n_succ is None:
|
|
70
|
+
return _gof_counts(np.asarray(n_tot, dtype=float), d)
|
|
71
|
+
if np.ndim(n_succ) == 0:
|
|
72
|
+
return _one_prop(int(n_succ), int(n_tot), pi, d,
|
|
73
|
+
alternative)
|
|
74
|
+
return _many_prop(np.asarray(n_succ), np.asarray(n_tot), d)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ------------------------------------------------------------
|
|
78
|
+
# one proportion (exact binomial)
|
|
79
|
+
# ------------------------------------------------------------
|
|
80
|
+
|
|
81
|
+
def _one_prop_data(v, success, name, pi, d, alt):
|
|
82
|
+
s = v.astype(str)
|
|
83
|
+
n_na = int(v.isna().sum())
|
|
84
|
+
n_tot = int(s.notna().sum() if False else (~v.isna()).sum())
|
|
85
|
+
n_succ = int((s == str(success)).sum())
|
|
86
|
+
return _one_prop(n_succ, n_tot, pi, d, alt, name, success,
|
|
87
|
+
n_na)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _one_prop(n_succ, n_tot, pi, d, alt, name=None, success=None,
|
|
91
|
+
n_na=None):
|
|
92
|
+
from scipy.stats import binomtest
|
|
93
|
+
if pi is None:
|
|
94
|
+
pi = 0.5
|
|
95
|
+
sci_alt = {"two_sided": "two-sided", "less": "less",
|
|
96
|
+
"greater": "greater"}[alt]
|
|
97
|
+
r = binomtest(n_succ, n_tot, pi, alternative=sci_alt)
|
|
98
|
+
lo, hi = r.proportion_ci(0.95, method="exact")
|
|
99
|
+
prop = n_succ / n_tot
|
|
100
|
+
|
|
101
|
+
L = ["", "<<< Exact binomial test of a proportion", ""]
|
|
102
|
+
if name is not None:
|
|
103
|
+
L += [f"variable: {name}", f"success: {success}", ""]
|
|
104
|
+
L += ["------ Describe ------", ""]
|
|
105
|
+
if n_na is not None:
|
|
106
|
+
L.append(f"Number of missing values: {n_na}")
|
|
107
|
+
L += [f"Number of successes: {n_succ}",
|
|
108
|
+
f"Number of failures: {n_tot - n_succ}",
|
|
109
|
+
f"Number of trials: {n_tot}",
|
|
110
|
+
f"Sample proportion: {fmt(prop, d)}", "",
|
|
111
|
+
"------ Infer ------", ""]
|
|
112
|
+
if alt != "two_sided":
|
|
113
|
+
L.append(f"Alternative hypothesis: Population proportion "
|
|
114
|
+
f"is {sci_alt} than {pi}")
|
|
115
|
+
L += [f"Hypothesis test for null of {pi}, p-value: "
|
|
116
|
+
f"{fmt(r.pvalue, d)}",
|
|
117
|
+
f"95% Confidence interval: {fmt(lo, d)} to "
|
|
118
|
+
f"{fmt(hi, d)}"]
|
|
119
|
+
print("\n".join(L))
|
|
120
|
+
return PropResults(kind="one-proportion", proportion=prop,
|
|
121
|
+
n_succ=n_succ, n_tot=n_tot,
|
|
122
|
+
p_value=float(r.pvalue),
|
|
123
|
+
conf_int=(float(lo), float(hi)), pi=pi)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ------------------------------------------------------------
|
|
127
|
+
# many proportions (chi-square of equal proportions)
|
|
128
|
+
# ------------------------------------------------------------
|
|
129
|
+
|
|
130
|
+
def _many_prop_data(v, g, success, name, by, Yates, d):
|
|
131
|
+
outcome = np.where(v.astype(str) == str(success),
|
|
132
|
+
str(success), "fail")
|
|
133
|
+
tab = pd.crosstab(g.astype(str), outcome)
|
|
134
|
+
cols = [str(success), "fail"]
|
|
135
|
+
tab = tab.reindex(columns=[c for c in cols if c in
|
|
136
|
+
tab.columns], fill_value=0)
|
|
137
|
+
return _many_prop_table(tab.to_numpy(), list(tab.index), d,
|
|
138
|
+
name, by, success)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _many_prop(n_succ, n_tot, d):
|
|
142
|
+
tab = np.column_stack([n_succ, n_tot - n_succ])
|
|
143
|
+
return _many_prop_table(tab, [str(i + 1) for i in
|
|
144
|
+
range(len(n_succ))], d)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _many_prop_table(tab, groups, d, name=None, by=None,
|
|
148
|
+
success=None):
|
|
149
|
+
from scipy.stats import chi2_contingency
|
|
150
|
+
chi2, p, dof, _ = chi2_contingency(tab, correction=False)
|
|
151
|
+
props = tab[:, 0] / tab.sum(axis=1)
|
|
152
|
+
L = ["", "<<< Chi-square test of equal proportions", ""]
|
|
153
|
+
if name is not None:
|
|
154
|
+
L += [f"variable: {name}", f"success: {success}",
|
|
155
|
+
f"by: {by}", ""]
|
|
156
|
+
L += ["--- Description", ""]
|
|
157
|
+
desc = pd.DataFrame(
|
|
158
|
+
{g: [int(tab[i, 0]), int(tab[i].sum()),
|
|
159
|
+
round(props[i], d)]
|
|
160
|
+
for i, g in enumerate(groups)},
|
|
161
|
+
index=[f"n_{success}" if success else "n_success",
|
|
162
|
+
"n_total", "proportion"])
|
|
163
|
+
L.append(desc.to_string())
|
|
164
|
+
L += ["", "--- Inference", "",
|
|
165
|
+
f"Chi-square statistic: {fmt(chi2, d)}",
|
|
166
|
+
f"Degrees of freedom: {dof}",
|
|
167
|
+
"Hypothesis test of equal population proportions: "
|
|
168
|
+
f"p-value = {fmt(p, d)}"]
|
|
169
|
+
print("\n".join(L))
|
|
170
|
+
return PropResults(kind="many-proportions",
|
|
171
|
+
proportions=props, chi2=float(chi2),
|
|
172
|
+
df=int(dof), p_value=float(p))
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
# ------------------------------------------------------------
|
|
176
|
+
# goodness-of-fit
|
|
177
|
+
# ------------------------------------------------------------
|
|
178
|
+
|
|
179
|
+
def _gof_data(v, name, d):
|
|
180
|
+
tab = v.astype(str).value_counts().sort_index()
|
|
181
|
+
return _gof(tab.to_numpy(dtype=float), list(tab.index), d,
|
|
182
|
+
name)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _gof_counts(n_tot, d):
|
|
186
|
+
return _gof(n_tot, [str(i + 1) for i in range(len(n_tot))], d)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _gof(obs, names, d, name=None):
|
|
190
|
+
from scipy.stats import chisquare
|
|
191
|
+
n = obs.sum()
|
|
192
|
+
exp = np.full(len(obs), n / len(obs))
|
|
193
|
+
chi2, p = chisquare(obs, exp)
|
|
194
|
+
dof = len(obs) - 1
|
|
195
|
+
resid = (obs - exp) / np.sqrt(exp)
|
|
196
|
+
stdres = (obs - exp) / np.sqrt(exp * (1 - exp / n))
|
|
197
|
+
|
|
198
|
+
L = ["", "<<< Chi-squared test for given probabilities", ""]
|
|
199
|
+
if name is not None:
|
|
200
|
+
L.append(f"variable: {name}")
|
|
201
|
+
L += ["", "--- Description", ""]
|
|
202
|
+
tbl = pd.DataFrame(
|
|
203
|
+
{nm: [int(obs[i]), round(exp[i], d), round(resid[i], d),
|
|
204
|
+
round(stdres[i], d)] for i, nm in enumerate(names)},
|
|
205
|
+
index=["observed", "expected", "residual", "stdn res"])
|
|
206
|
+
L += [tbl.to_string(), "", "--- Inference", "",
|
|
207
|
+
f"Chi-square statistic: {fmt(chi2, d)}",
|
|
208
|
+
f"Degrees of freedom: {dof}",
|
|
209
|
+
"Hypothesis test of equal population proportions: "
|
|
210
|
+
f"p-value = {fmt(p, d)}"]
|
|
211
|
+
print("\n".join(L))
|
|
212
|
+
return PropResults(kind="goodness-of-fit", observed=obs,
|
|
213
|
+
expected=exp, residuals=resid,
|
|
214
|
+
stdres=stdres, chi2=float(chi2), df=dof,
|
|
215
|
+
p_value=float(p))
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
# ------------------------------------------------------------
|
|
219
|
+
# cross-tab independence
|
|
220
|
+
# ------------------------------------------------------------
|
|
221
|
+
|
|
222
|
+
def _crosstab_data(v, g, name, by, Yates, d):
|
|
223
|
+
tab = pd.crosstab(g.astype(str), v.astype(str))
|
|
224
|
+
return _crosstab(tab, Yates, d, name, by)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _crosstab_table(n_table, Yates, d):
|
|
228
|
+
if isinstance(n_table, (pd.DataFrame, np.ndarray)):
|
|
229
|
+
tab = pd.DataFrame(n_table)
|
|
230
|
+
else: # a file path
|
|
231
|
+
tab = pd.read_csv(n_table, header=None)
|
|
232
|
+
return _crosstab(tab, Yates, d, None, None)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _crosstab(tab, Yates, d, name, by):
|
|
236
|
+
from scipy.stats import chi2_contingency
|
|
237
|
+
O = tab.to_numpy(dtype=float)
|
|
238
|
+
chi2, p, dof, E = chi2_contingency(O, correction=Yates)
|
|
239
|
+
n = O.sum()
|
|
240
|
+
rs = O.sum(axis=1, keepdims=True)
|
|
241
|
+
cs = O.sum(axis=0, keepdims=True)
|
|
242
|
+
stdres = (O - E) / np.sqrt(E * (1 - rs / n) * (1 - cs / n))
|
|
243
|
+
V = np.sqrt(chi2 / (min(O.shape[0] - 1, O.shape[1] - 1) * n))
|
|
244
|
+
|
|
245
|
+
L = ["", "<<< Pearson's Chi-squared test", ""]
|
|
246
|
+
if name is not None:
|
|
247
|
+
L += [f"variable: {name}", f"by: {by}"]
|
|
248
|
+
L += ["", "--- Description", "", tab.to_string(),
|
|
249
|
+
"",
|
|
250
|
+
f"Cramer's V{' (phi)' if dof == 1 else ''}: "
|
|
251
|
+
f"{fmt(V, 3)}", ""]
|
|
252
|
+
cells = pd.DataFrame(
|
|
253
|
+
[[i + 1, j + 1, int(O[i, j]), round(E[i, j], d),
|
|
254
|
+
round(O[i, j] - E[i, j], d), round(stdres[i, j], d)]
|
|
255
|
+
for i in range(O.shape[0]) for j in range(O.shape[1])],
|
|
256
|
+
columns=["Row", "Col", "Observed", "Expected",
|
|
257
|
+
"Residual", "Stnd Res"])
|
|
258
|
+
L += [cells.to_string(index=False), "", "--- Inference", "",
|
|
259
|
+
f"Chi-square statistic: {fmt(chi2, d)}",
|
|
260
|
+
f"Degrees of freedom: {dof}",
|
|
261
|
+
f"Hypothesis test of independence: p-value = "
|
|
262
|
+
f"{fmt(p, d)}"]
|
|
263
|
+
print("\n".join(L))
|
|
264
|
+
return PropResults(kind="independence", observed=O,
|
|
265
|
+
expected=E, stdres=stdres,
|
|
266
|
+
cramers_v=float(V), chi2=float(chi2),
|
|
267
|
+
df=int(dof), p_value=float(p))
|