lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Prop_test.py ADDED
@@ -0,0 +1,267 @@
1
+ # Prop_test.py — analog of Prop_test.R.
2
+ #
3
+ # Prop_test(): four proportion analyses, chosen from the
4
+ # arguments —
5
+ # one proportion variable + success (data) | n_succ + n_tot
6
+ # exact binomial test vs pi (default 0.5)
7
+ # many proportions variable + success + by | n_succ,n_tot lists
8
+ # chi-square test of equal proportions
9
+ # goodness-of-fit variable (data) | n_tot vector
10
+ # chi-square vs equal proportions
11
+ # independence variable + by (data) | n_table
12
+ # chi-square cross-tab test, with Cramer's V
13
+ # Numerics through scipy.stats (binomtest, chi2_contingency,
14
+ # chisquare). Prints as R; returns a PropResults object.
15
+
16
+ import numpy as np
17
+ import pandas as pd
18
+
19
+ from .utils import fmt, get_column
20
+
21
+
22
+ class PropResults:
23
+ """Results of Prop_test(): the analysis kind and its
24
+ statistics (proportion/p-value/CI, or chi-square/df/p and
25
+ the tables)."""
26
+
27
+ def __init__(self, **kw):
28
+ self.__dict__.update(kw)
29
+
30
+ def __repr__(self):
31
+ return f"<lessPy Prop_test: {self.kind}>"
32
+
33
+
34
+ def Prop_test(variable=None, success=None, by=None, data=None,
35
+ n_succ=None, n_fail=None, n_tot=None, n_table=None,
36
+ Yates=False, pi=None, digits_d=3,
37
+ alternative="two_sided"):
38
+ """Test of one or more proportions, goodness-of-fit, or
39
+ cross-tab independence, from data columns or from summary
40
+ counts. R analog: Prop_test()"""
41
+ if alternative not in ("two_sided", "less", "greater"):
42
+ raise ValueError('alternative: "two_sided", "less", '
43
+ '"greater"')
44
+ do_data = (n_succ is None and n_tot is None
45
+ and n_table is None)
46
+ d = digits_d
47
+
48
+ if do_data:
49
+ if variable is None:
50
+ raise ValueError("variable= is required with data=")
51
+ v = get_column(data, variable, "variable")
52
+ g = get_column(data, by, "by") if by is not None else None
53
+ if success is not None and g is None:
54
+ return _one_prop_data(v, success, variable, pi, d,
55
+ alternative)
56
+ if success is not None and g is not None:
57
+ return _many_prop_data(v, g, success, variable, by,
58
+ Yates, d)
59
+ if success is None and g is None:
60
+ return _gof_data(v, variable, d)
61
+ return _crosstab_data(v, g, variable, by, Yates, d)
62
+
63
+ # summary counts
64
+ if n_fail is not None and n_tot is None:
65
+ n_tot = (np.add(n_succ, n_fail) if np.ndim(n_succ)
66
+ else n_succ + n_fail)
67
+ if n_table is not None:
68
+ return _crosstab_table(n_table, Yates, d)
69
+ if n_succ is None:
70
+ return _gof_counts(np.asarray(n_tot, dtype=float), d)
71
+ if np.ndim(n_succ) == 0:
72
+ return _one_prop(int(n_succ), int(n_tot), pi, d,
73
+ alternative)
74
+ return _many_prop(np.asarray(n_succ), np.asarray(n_tot), d)
75
+
76
+
77
+ # ------------------------------------------------------------
78
+ # one proportion (exact binomial)
79
+ # ------------------------------------------------------------
80
+
81
+ def _one_prop_data(v, success, name, pi, d, alt):
82
+ s = v.astype(str)
83
+ n_na = int(v.isna().sum())
84
+ n_tot = int(s.notna().sum() if False else (~v.isna()).sum())
85
+ n_succ = int((s == str(success)).sum())
86
+ return _one_prop(n_succ, n_tot, pi, d, alt, name, success,
87
+ n_na)
88
+
89
+
90
+ def _one_prop(n_succ, n_tot, pi, d, alt, name=None, success=None,
91
+ n_na=None):
92
+ from scipy.stats import binomtest
93
+ if pi is None:
94
+ pi = 0.5
95
+ sci_alt = {"two_sided": "two-sided", "less": "less",
96
+ "greater": "greater"}[alt]
97
+ r = binomtest(n_succ, n_tot, pi, alternative=sci_alt)
98
+ lo, hi = r.proportion_ci(0.95, method="exact")
99
+ prop = n_succ / n_tot
100
+
101
+ L = ["", "<<< Exact binomial test of a proportion", ""]
102
+ if name is not None:
103
+ L += [f"variable: {name}", f"success: {success}", ""]
104
+ L += ["------ Describe ------", ""]
105
+ if n_na is not None:
106
+ L.append(f"Number of missing values: {n_na}")
107
+ L += [f"Number of successes: {n_succ}",
108
+ f"Number of failures: {n_tot - n_succ}",
109
+ f"Number of trials: {n_tot}",
110
+ f"Sample proportion: {fmt(prop, d)}", "",
111
+ "------ Infer ------", ""]
112
+ if alt != "two_sided":
113
+ L.append(f"Alternative hypothesis: Population proportion "
114
+ f"is {sci_alt} than {pi}")
115
+ L += [f"Hypothesis test for null of {pi}, p-value: "
116
+ f"{fmt(r.pvalue, d)}",
117
+ f"95% Confidence interval: {fmt(lo, d)} to "
118
+ f"{fmt(hi, d)}"]
119
+ print("\n".join(L))
120
+ return PropResults(kind="one-proportion", proportion=prop,
121
+ n_succ=n_succ, n_tot=n_tot,
122
+ p_value=float(r.pvalue),
123
+ conf_int=(float(lo), float(hi)), pi=pi)
124
+
125
+
126
+ # ------------------------------------------------------------
127
+ # many proportions (chi-square of equal proportions)
128
+ # ------------------------------------------------------------
129
+
130
+ def _many_prop_data(v, g, success, name, by, Yates, d):
131
+ outcome = np.where(v.astype(str) == str(success),
132
+ str(success), "fail")
133
+ tab = pd.crosstab(g.astype(str), outcome)
134
+ cols = [str(success), "fail"]
135
+ tab = tab.reindex(columns=[c for c in cols if c in
136
+ tab.columns], fill_value=0)
137
+ return _many_prop_table(tab.to_numpy(), list(tab.index), d,
138
+ name, by, success)
139
+
140
+
141
+ def _many_prop(n_succ, n_tot, d):
142
+ tab = np.column_stack([n_succ, n_tot - n_succ])
143
+ return _many_prop_table(tab, [str(i + 1) for i in
144
+ range(len(n_succ))], d)
145
+
146
+
147
+ def _many_prop_table(tab, groups, d, name=None, by=None,
148
+ success=None):
149
+ from scipy.stats import chi2_contingency
150
+ chi2, p, dof, _ = chi2_contingency(tab, correction=False)
151
+ props = tab[:, 0] / tab.sum(axis=1)
152
+ L = ["", "<<< Chi-square test of equal proportions", ""]
153
+ if name is not None:
154
+ L += [f"variable: {name}", f"success: {success}",
155
+ f"by: {by}", ""]
156
+ L += ["--- Description", ""]
157
+ desc = pd.DataFrame(
158
+ {g: [int(tab[i, 0]), int(tab[i].sum()),
159
+ round(props[i], d)]
160
+ for i, g in enumerate(groups)},
161
+ index=[f"n_{success}" if success else "n_success",
162
+ "n_total", "proportion"])
163
+ L.append(desc.to_string())
164
+ L += ["", "--- Inference", "",
165
+ f"Chi-square statistic: {fmt(chi2, d)}",
166
+ f"Degrees of freedom: {dof}",
167
+ "Hypothesis test of equal population proportions: "
168
+ f"p-value = {fmt(p, d)}"]
169
+ print("\n".join(L))
170
+ return PropResults(kind="many-proportions",
171
+ proportions=props, chi2=float(chi2),
172
+ df=int(dof), p_value=float(p))
173
+
174
+
175
+ # ------------------------------------------------------------
176
+ # goodness-of-fit
177
+ # ------------------------------------------------------------
178
+
179
+ def _gof_data(v, name, d):
180
+ tab = v.astype(str).value_counts().sort_index()
181
+ return _gof(tab.to_numpy(dtype=float), list(tab.index), d,
182
+ name)
183
+
184
+
185
+ def _gof_counts(n_tot, d):
186
+ return _gof(n_tot, [str(i + 1) for i in range(len(n_tot))], d)
187
+
188
+
189
+ def _gof(obs, names, d, name=None):
190
+ from scipy.stats import chisquare
191
+ n = obs.sum()
192
+ exp = np.full(len(obs), n / len(obs))
193
+ chi2, p = chisquare(obs, exp)
194
+ dof = len(obs) - 1
195
+ resid = (obs - exp) / np.sqrt(exp)
196
+ stdres = (obs - exp) / np.sqrt(exp * (1 - exp / n))
197
+
198
+ L = ["", "<<< Chi-squared test for given probabilities", ""]
199
+ if name is not None:
200
+ L.append(f"variable: {name}")
201
+ L += ["", "--- Description", ""]
202
+ tbl = pd.DataFrame(
203
+ {nm: [int(obs[i]), round(exp[i], d), round(resid[i], d),
204
+ round(stdres[i], d)] for i, nm in enumerate(names)},
205
+ index=["observed", "expected", "residual", "stdn res"])
206
+ L += [tbl.to_string(), "", "--- Inference", "",
207
+ f"Chi-square statistic: {fmt(chi2, d)}",
208
+ f"Degrees of freedom: {dof}",
209
+ "Hypothesis test of equal population proportions: "
210
+ f"p-value = {fmt(p, d)}"]
211
+ print("\n".join(L))
212
+ return PropResults(kind="goodness-of-fit", observed=obs,
213
+ expected=exp, residuals=resid,
214
+ stdres=stdres, chi2=float(chi2), df=dof,
215
+ p_value=float(p))
216
+
217
+
218
+ # ------------------------------------------------------------
219
+ # cross-tab independence
220
+ # ------------------------------------------------------------
221
+
222
+ def _crosstab_data(v, g, name, by, Yates, d):
223
+ tab = pd.crosstab(g.astype(str), v.astype(str))
224
+ return _crosstab(tab, Yates, d, name, by)
225
+
226
+
227
+ def _crosstab_table(n_table, Yates, d):
228
+ if isinstance(n_table, (pd.DataFrame, np.ndarray)):
229
+ tab = pd.DataFrame(n_table)
230
+ else: # a file path
231
+ tab = pd.read_csv(n_table, header=None)
232
+ return _crosstab(tab, Yates, d, None, None)
233
+
234
+
235
+ def _crosstab(tab, Yates, d, name, by):
236
+ from scipy.stats import chi2_contingency
237
+ O = tab.to_numpy(dtype=float)
238
+ chi2, p, dof, E = chi2_contingency(O, correction=Yates)
239
+ n = O.sum()
240
+ rs = O.sum(axis=1, keepdims=True)
241
+ cs = O.sum(axis=0, keepdims=True)
242
+ stdres = (O - E) / np.sqrt(E * (1 - rs / n) * (1 - cs / n))
243
+ V = np.sqrt(chi2 / (min(O.shape[0] - 1, O.shape[1] - 1) * n))
244
+
245
+ L = ["", "<<< Pearson's Chi-squared test", ""]
246
+ if name is not None:
247
+ L += [f"variable: {name}", f"by: {by}"]
248
+ L += ["", "--- Description", "", tab.to_string(),
249
+ "",
250
+ f"Cramer's V{' (phi)' if dof == 1 else ''}: "
251
+ f"{fmt(V, 3)}", ""]
252
+ cells = pd.DataFrame(
253
+ [[i + 1, j + 1, int(O[i, j]), round(E[i, j], d),
254
+ round(O[i, j] - E[i, j], d), round(stdres[i, j], d)]
255
+ for i in range(O.shape[0]) for j in range(O.shape[1])],
256
+ columns=["Row", "Col", "Observed", "Expected",
257
+ "Residual", "Stnd Res"])
258
+ L += [cells.to_string(index=False), "", "--- Inference", "",
259
+ f"Chi-square statistic: {fmt(chi2, d)}",
260
+ f"Degrees of freedom: {dof}",
261
+ f"Hypothesis test of independence: p-value = "
262
+ f"{fmt(p, d)}"]
263
+ print("\n".join(L))
264
+ return PropResults(kind="independence", observed=O,
265
+ expected=E, stdres=stdres,
266
+ cramers_v=float(V), chi2=float(chi2),
267
+ df=int(dof), p_value=float(p))