lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/stats_out.py ADDED
@@ -0,0 +1,189 @@
1
+ # stats_out.py — accompanying statistics: the console numbers
2
+ # that pair with each analytic view, a core differentiator of
3
+ # the framework. R analogs: SummaryStats.R and bx.stats.R for
4
+ # X(), the correlation and fit report of XY.R, and the
5
+ # frequency / chi-square output of the Chart() bar path.
6
+ #
7
+ # This increment prints (quiet= suppresses, default from the
8
+ # "quiet" option, as R getOption("quiet")). Returning a stats
9
+ # object is deferred: the functions return the plotly Figure,
10
+ # which notebooks display automatically; wrapping it would
11
+ # break that.
12
+
13
+ import numpy as np
14
+ import pandas as pd
15
+ from scipy import stats as sps
16
+
17
+ from .utils import fmt, get_option
18
+
19
+
20
+ def resolve_quiet(quiet):
21
+ return get_option("quiet", False) if quiet is None else quiet
22
+
23
+
24
+ def x_stats(xv, x_name, digits_d=2):
25
+ """n, missing, mean and sd, five-number summary, and IQR
26
+ outliers for the distribution X() displays."""
27
+ v = np.asarray(xv, float)
28
+ miss = int(np.isnan(v).sum())
29
+ v = v[np.isfinite(v)]
30
+ n = len(v)
31
+ d = digits_d
32
+ lines = [f"--- {x_name} ---",
33
+ f"n: {n} missing: {miss}"]
34
+ if n < 2:
35
+ return lines
36
+ q1, med, q3 = np.percentile(v, [25, 50, 75])
37
+ iqr = q3 - q1
38
+ out_v = np.sort(v[(v < q1 - 1.5 * iqr)
39
+ | (v > q3 + 1.5 * iqr)])
40
+ lines += [
41
+ f"mean: {fmt(v.mean(), d)} sd: {fmt(v.std(ddof=1), d)}",
42
+ f"min: {fmt(v.min(), d)} 1st Qu: {fmt(q1, d)} "
43
+ f"median: {fmt(med, d)} 3rd Qu: {fmt(q3, d)} "
44
+ f"max: {fmt(v.max(), d)}"]
45
+ if len(out_v):
46
+ vals = " ".join(fmt(o, d) for o in out_v[:12])
47
+ more = ("" if len(out_v) <= 12
48
+ else f" ... {len(out_v)} total")
49
+ lines.append(f"outliers (1.5 IQR): {vals}{more}")
50
+ else:
51
+ lines.append("outliers (1.5 IQR): none")
52
+ return lines
53
+
54
+
55
+ def facet_summary(x_vec, grp_vec, grp_label, digits_d=2):
56
+ """Summary table of x over the levels of one grouping
57
+ variable: n, Mean, Median, SD, IQR, Min, Max per level.
58
+ Printed with the faceted scatterplot, one table per by=
59
+ and facet variable. R analog: .vbs_summary_table()"""
60
+ s = pd.Series(np.asarray(x_vec, dtype=float))
61
+ g = pd.Series(np.asarray(grp_vec).astype(str))
62
+ rows = []
63
+ for lv, z in s.groupby(g, sort=True):
64
+ z = z.dropna()
65
+ rows.append({
66
+ grp_label: lv,
67
+ "n": len(z),
68
+ "Mean": fmt(z.mean(), digits_d),
69
+ "Median": fmt(z.median(), digits_d),
70
+ "SD": fmt(z.std(ddof=1), digits_d),
71
+ "IQR": fmt(z.quantile(.75) - z.quantile(.25),
72
+ digits_d),
73
+ "Min": fmt(z.min(), digits_d),
74
+ "Max": fmt(z.max(), digits_d),
75
+ })
76
+ return pd.DataFrame(rows).to_string(index=False)
77
+
78
+
79
+ def _cor_block(xg, yg, label, d):
80
+ n = len(xg)
81
+ if n < 4:
82
+ return [f"correlation{label}: n = {n}, too few points"]
83
+ r = float(np.corrcoef(xg, yg)[0, 1])
84
+ tval = r * np.sqrt((n - 2) / max(1 - r * r, 1e-12))
85
+ p = 2 * sps.t.sf(abs(tval), n - 2)
86
+ zlo = np.arctanh(r) - 1.959964 / np.sqrt(n - 3)
87
+ zhi = np.arctanh(r) + 1.959964 / np.sqrt(n - 3)
88
+ return [
89
+ f"correlation{label}: r = {fmt(r, 3)}",
90
+ f" n = {n}, t = {fmt(tval, 2)} (df = {n - 2}), "
91
+ f"p-value = {fmt(p, 4)}",
92
+ f" 95% CI: {fmt(np.tanh(zlo), 3)} to "
93
+ f"{fmt(np.tanh(zhi), 3)}"]
94
+
95
+
96
+ def xy_stats(groups, x_name, y_name, fit_stats=None,
97
+ digits_d=2):
98
+ """Correlation (t test, Fisher-z CI) for the relationship
99
+ XY() displays, overall or per by= group, plus fit-line MSE
100
+ when a fit line is drawn."""
101
+ lines = [f"--- {x_name} and {y_name} ---"]
102
+ for nm, xg, yg in groups:
103
+ label = "" if nm is None else f" ({nm})"
104
+ lines += _cor_block(np.asarray(xg, float),
105
+ np.asarray(yg, float), label,
106
+ digits_d)
107
+ for fs in (fit_stats or []):
108
+ nm, fit, ys, f = fs
109
+ label = "" if nm is None else f" ({nm})"
110
+ mse = float(np.mean((ys - f) ** 2))
111
+ tss = float(((ys - ys.mean()) ** 2).sum())
112
+ rsq = 1 - float(((ys - f) ** 2).sum()) / tss \
113
+ if tss > 0 else np.nan
114
+ lines.append(
115
+ f'fit="{fit}"{label}: MSE: {fmt(mse, digits_d + 1)}'
116
+ f" R-squared: {fmt(rsq, 3)}")
117
+ return lines
118
+
119
+
120
+ def md_outliers(xv, yv, ids, MD_cut, out_cut):
121
+ """Bivariate outliers by Mahalanobis distance from the
122
+ centroid: MD_cut an absolute threshold, out_cut a proportion
123
+ (< 1) or count (>= 1) of the most extreme points. Returns the
124
+ flagged indices and the sorted MD/ID table.
125
+ R analog: .plt.MD (plt.MD.R)"""
126
+ v = np.column_stack((xv, yv))
127
+ center = v.mean(axis=0)
128
+ cov = np.cov(v, rowvar=False, ddof=1)
129
+ diff = v - center
130
+ dst = np.einsum("ij,jk,ik->i", diff,
131
+ np.linalg.pinv(cov), diff)
132
+
133
+ if MD_cut > 0: # absolute threshold
134
+ out_idx = np.flatnonzero(dst >= MD_cut)
135
+ elif 0 < out_cut < 1: # a proportion
136
+ out_idx = np.flatnonzero(
137
+ dst > np.quantile(dst, 1 - out_cut))
138
+ else: # a count
139
+ cut = np.sort(dst)[::-1][:int(out_cut)].min()
140
+ out_idx = np.flatnonzero(dst >= cut)
141
+
142
+ ids = np.asarray(ids).astype(str)
143
+ ord_ = np.argsort(dst)[::-1]
144
+ n_lines = min(len(out_idx) + 3, len(dst))
145
+ w_id = max(len(s) for s in ids)
146
+ w_md = max(len(fmt(d, 2)) for d in dst)
147
+ lines = [">>> Outlier analysis with Mahalanobis Distance",
148
+ "",
149
+ f"{'MD':>{w_md}} {'ID':>{w_id + 1}}",
150
+ f"{'-----':>{w_md}} {'-----':>{w_id + 1}}"]
151
+ for i in range(n_lines):
152
+ if i == len(out_idx) and len(out_idx) > 0:
153
+ lines.append("")
154
+ j = ord_[i]
155
+ lines.append(f"{fmt(dst[j], 2):>{w_md}} "
156
+ f"{ids[j]:>{w_id + 1}}")
157
+ if n_lines < len(dst):
158
+ lines.append(f"{'...':>{w_md}} {'...':>{w_id + 1}}")
159
+ return out_idx, lines
160
+
161
+
162
+ def chart_stats(x_ser, by_ser, y_ser, stat_lbl, x_name,
163
+ by_name, y_name, digits_d=2):
164
+ """Frequencies with proportions for one categorical
165
+ variable; cross-tabulation with a chi-square test with by=;
166
+ the per-level summary when y= is charted."""
167
+ lines = [f"--- {x_name} ---"]
168
+ if y_ser is not None: # stat of y per level
169
+ g = y_ser.groupby(x_ser, observed=True)
170
+ agg = {"sd": "std"}.get(stat_lbl, stat_lbl)
171
+ tbl = pd.DataFrame({"n": g.size(),
172
+ stat_lbl: g.agg(agg)})
173
+ lines += tbl.to_string(
174
+ float_format=lambda v: fmt(v, digits_d)).split("\n")
175
+ elif by_ser is None: # one-way frequencies
176
+ cnt = x_ser.groupby(x_ser, observed=True).size()
177
+ tbl = pd.DataFrame({"n": cnt,
178
+ "prop": cnt / cnt.sum()})
179
+ tbl.loc["Total"] = [cnt.sum(), 1.0]
180
+ lines += tbl.to_string(
181
+ float_format=lambda v: fmt(v, digits_d)).split("\n")
182
+ else: # two-way + chi-square
183
+ ct = pd.crosstab(by_ser, x_ser)
184
+ lines += ct.to_string().split("\n")
185
+ chi2, p, dof, _ = sps.chi2_contingency(ct.to_numpy())
186
+ lines += ["",
187
+ f"chi-square({dof}) = {fmt(chi2, 2)}, "
188
+ f"p-value = {fmt(p, 4)}"]
189
+ return lines