lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/stats_out.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
# stats_out.py — accompanying statistics: the console numbers
|
|
2
|
+
# that pair with each analytic view, a core differentiator of
|
|
3
|
+
# the framework. R analogs: SummaryStats.R and bx.stats.R for
|
|
4
|
+
# X(), the correlation and fit report of XY.R, and the
|
|
5
|
+
# frequency / chi-square output of the Chart() bar path.
|
|
6
|
+
#
|
|
7
|
+
# This increment prints (quiet= suppresses, default from the
|
|
8
|
+
# "quiet" option, as R getOption("quiet")). Returning a stats
|
|
9
|
+
# object is deferred: the functions return the plotly Figure,
|
|
10
|
+
# which notebooks display automatically; wrapping it would
|
|
11
|
+
# break that.
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
from scipy import stats as sps
|
|
16
|
+
|
|
17
|
+
from .utils import fmt, get_option
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def resolve_quiet(quiet):
|
|
21
|
+
return get_option("quiet", False) if quiet is None else quiet
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def x_stats(xv, x_name, digits_d=2):
|
|
25
|
+
"""n, missing, mean and sd, five-number summary, and IQR
|
|
26
|
+
outliers for the distribution X() displays."""
|
|
27
|
+
v = np.asarray(xv, float)
|
|
28
|
+
miss = int(np.isnan(v).sum())
|
|
29
|
+
v = v[np.isfinite(v)]
|
|
30
|
+
n = len(v)
|
|
31
|
+
d = digits_d
|
|
32
|
+
lines = [f"--- {x_name} ---",
|
|
33
|
+
f"n: {n} missing: {miss}"]
|
|
34
|
+
if n < 2:
|
|
35
|
+
return lines
|
|
36
|
+
q1, med, q3 = np.percentile(v, [25, 50, 75])
|
|
37
|
+
iqr = q3 - q1
|
|
38
|
+
out_v = np.sort(v[(v < q1 - 1.5 * iqr)
|
|
39
|
+
| (v > q3 + 1.5 * iqr)])
|
|
40
|
+
lines += [
|
|
41
|
+
f"mean: {fmt(v.mean(), d)} sd: {fmt(v.std(ddof=1), d)}",
|
|
42
|
+
f"min: {fmt(v.min(), d)} 1st Qu: {fmt(q1, d)} "
|
|
43
|
+
f"median: {fmt(med, d)} 3rd Qu: {fmt(q3, d)} "
|
|
44
|
+
f"max: {fmt(v.max(), d)}"]
|
|
45
|
+
if len(out_v):
|
|
46
|
+
vals = " ".join(fmt(o, d) for o in out_v[:12])
|
|
47
|
+
more = ("" if len(out_v) <= 12
|
|
48
|
+
else f" ... {len(out_v)} total")
|
|
49
|
+
lines.append(f"outliers (1.5 IQR): {vals}{more}")
|
|
50
|
+
else:
|
|
51
|
+
lines.append("outliers (1.5 IQR): none")
|
|
52
|
+
return lines
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def facet_summary(x_vec, grp_vec, grp_label, digits_d=2):
|
|
56
|
+
"""Summary table of x over the levels of one grouping
|
|
57
|
+
variable: n, Mean, Median, SD, IQR, Min, Max per level.
|
|
58
|
+
Printed with the faceted scatterplot, one table per by=
|
|
59
|
+
and facet variable. R analog: .vbs_summary_table()"""
|
|
60
|
+
s = pd.Series(np.asarray(x_vec, dtype=float))
|
|
61
|
+
g = pd.Series(np.asarray(grp_vec).astype(str))
|
|
62
|
+
rows = []
|
|
63
|
+
for lv, z in s.groupby(g, sort=True):
|
|
64
|
+
z = z.dropna()
|
|
65
|
+
rows.append({
|
|
66
|
+
grp_label: lv,
|
|
67
|
+
"n": len(z),
|
|
68
|
+
"Mean": fmt(z.mean(), digits_d),
|
|
69
|
+
"Median": fmt(z.median(), digits_d),
|
|
70
|
+
"SD": fmt(z.std(ddof=1), digits_d),
|
|
71
|
+
"IQR": fmt(z.quantile(.75) - z.quantile(.25),
|
|
72
|
+
digits_d),
|
|
73
|
+
"Min": fmt(z.min(), digits_d),
|
|
74
|
+
"Max": fmt(z.max(), digits_d),
|
|
75
|
+
})
|
|
76
|
+
return pd.DataFrame(rows).to_string(index=False)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _cor_block(xg, yg, label, d):
|
|
80
|
+
n = len(xg)
|
|
81
|
+
if n < 4:
|
|
82
|
+
return [f"correlation{label}: n = {n}, too few points"]
|
|
83
|
+
r = float(np.corrcoef(xg, yg)[0, 1])
|
|
84
|
+
tval = r * np.sqrt((n - 2) / max(1 - r * r, 1e-12))
|
|
85
|
+
p = 2 * sps.t.sf(abs(tval), n - 2)
|
|
86
|
+
zlo = np.arctanh(r) - 1.959964 / np.sqrt(n - 3)
|
|
87
|
+
zhi = np.arctanh(r) + 1.959964 / np.sqrt(n - 3)
|
|
88
|
+
return [
|
|
89
|
+
f"correlation{label}: r = {fmt(r, 3)}",
|
|
90
|
+
f" n = {n}, t = {fmt(tval, 2)} (df = {n - 2}), "
|
|
91
|
+
f"p-value = {fmt(p, 4)}",
|
|
92
|
+
f" 95% CI: {fmt(np.tanh(zlo), 3)} to "
|
|
93
|
+
f"{fmt(np.tanh(zhi), 3)}"]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def xy_stats(groups, x_name, y_name, fit_stats=None,
|
|
97
|
+
digits_d=2):
|
|
98
|
+
"""Correlation (t test, Fisher-z CI) for the relationship
|
|
99
|
+
XY() displays, overall or per by= group, plus fit-line MSE
|
|
100
|
+
when a fit line is drawn."""
|
|
101
|
+
lines = [f"--- {x_name} and {y_name} ---"]
|
|
102
|
+
for nm, xg, yg in groups:
|
|
103
|
+
label = "" if nm is None else f" ({nm})"
|
|
104
|
+
lines += _cor_block(np.asarray(xg, float),
|
|
105
|
+
np.asarray(yg, float), label,
|
|
106
|
+
digits_d)
|
|
107
|
+
for fs in (fit_stats or []):
|
|
108
|
+
nm, fit, ys, f = fs
|
|
109
|
+
label = "" if nm is None else f" ({nm})"
|
|
110
|
+
mse = float(np.mean((ys - f) ** 2))
|
|
111
|
+
tss = float(((ys - ys.mean()) ** 2).sum())
|
|
112
|
+
rsq = 1 - float(((ys - f) ** 2).sum()) / tss \
|
|
113
|
+
if tss > 0 else np.nan
|
|
114
|
+
lines.append(
|
|
115
|
+
f'fit="{fit}"{label}: MSE: {fmt(mse, digits_d + 1)}'
|
|
116
|
+
f" R-squared: {fmt(rsq, 3)}")
|
|
117
|
+
return lines
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def md_outliers(xv, yv, ids, MD_cut, out_cut):
|
|
121
|
+
"""Bivariate outliers by Mahalanobis distance from the
|
|
122
|
+
centroid: MD_cut an absolute threshold, out_cut a proportion
|
|
123
|
+
(< 1) or count (>= 1) of the most extreme points. Returns the
|
|
124
|
+
flagged indices and the sorted MD/ID table.
|
|
125
|
+
R analog: .plt.MD (plt.MD.R)"""
|
|
126
|
+
v = np.column_stack((xv, yv))
|
|
127
|
+
center = v.mean(axis=0)
|
|
128
|
+
cov = np.cov(v, rowvar=False, ddof=1)
|
|
129
|
+
diff = v - center
|
|
130
|
+
dst = np.einsum("ij,jk,ik->i", diff,
|
|
131
|
+
np.linalg.pinv(cov), diff)
|
|
132
|
+
|
|
133
|
+
if MD_cut > 0: # absolute threshold
|
|
134
|
+
out_idx = np.flatnonzero(dst >= MD_cut)
|
|
135
|
+
elif 0 < out_cut < 1: # a proportion
|
|
136
|
+
out_idx = np.flatnonzero(
|
|
137
|
+
dst > np.quantile(dst, 1 - out_cut))
|
|
138
|
+
else: # a count
|
|
139
|
+
cut = np.sort(dst)[::-1][:int(out_cut)].min()
|
|
140
|
+
out_idx = np.flatnonzero(dst >= cut)
|
|
141
|
+
|
|
142
|
+
ids = np.asarray(ids).astype(str)
|
|
143
|
+
ord_ = np.argsort(dst)[::-1]
|
|
144
|
+
n_lines = min(len(out_idx) + 3, len(dst))
|
|
145
|
+
w_id = max(len(s) for s in ids)
|
|
146
|
+
w_md = max(len(fmt(d, 2)) for d in dst)
|
|
147
|
+
lines = [">>> Outlier analysis with Mahalanobis Distance",
|
|
148
|
+
"",
|
|
149
|
+
f"{'MD':>{w_md}} {'ID':>{w_id + 1}}",
|
|
150
|
+
f"{'-----':>{w_md}} {'-----':>{w_id + 1}}"]
|
|
151
|
+
for i in range(n_lines):
|
|
152
|
+
if i == len(out_idx) and len(out_idx) > 0:
|
|
153
|
+
lines.append("")
|
|
154
|
+
j = ord_[i]
|
|
155
|
+
lines.append(f"{fmt(dst[j], 2):>{w_md}} "
|
|
156
|
+
f"{ids[j]:>{w_id + 1}}")
|
|
157
|
+
if n_lines < len(dst):
|
|
158
|
+
lines.append(f"{'...':>{w_md}} {'...':>{w_id + 1}}")
|
|
159
|
+
return out_idx, lines
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def chart_stats(x_ser, by_ser, y_ser, stat_lbl, x_name,
|
|
163
|
+
by_name, y_name, digits_d=2):
|
|
164
|
+
"""Frequencies with proportions for one categorical
|
|
165
|
+
variable; cross-tabulation with a chi-square test with by=;
|
|
166
|
+
the per-level summary when y= is charted."""
|
|
167
|
+
lines = [f"--- {x_name} ---"]
|
|
168
|
+
if y_ser is not None: # stat of y per level
|
|
169
|
+
g = y_ser.groupby(x_ser, observed=True)
|
|
170
|
+
agg = {"sd": "std"}.get(stat_lbl, stat_lbl)
|
|
171
|
+
tbl = pd.DataFrame({"n": g.size(),
|
|
172
|
+
stat_lbl: g.agg(agg)})
|
|
173
|
+
lines += tbl.to_string(
|
|
174
|
+
float_format=lambda v: fmt(v, digits_d)).split("\n")
|
|
175
|
+
elif by_ser is None: # one-way frequencies
|
|
176
|
+
cnt = x_ser.groupby(x_ser, observed=True).size()
|
|
177
|
+
tbl = pd.DataFrame({"n": cnt,
|
|
178
|
+
"prop": cnt / cnt.sum()})
|
|
179
|
+
tbl.loc["Total"] = [cnt.sum(), 1.0]
|
|
180
|
+
lines += tbl.to_string(
|
|
181
|
+
float_format=lambda v: fmt(v, digits_d)).split("\n")
|
|
182
|
+
else: # two-way + chi-square
|
|
183
|
+
ct = pd.crosstab(by_ser, x_ser)
|
|
184
|
+
lines += ct.to_string().split("\n")
|
|
185
|
+
chi2, p, dof, _ = sps.chi2_contingency(ct.to_numpy())
|
|
186
|
+
lines += ["",
|
|
187
|
+
f"chi-square({dof}) = {fmt(chi2, 2)}, "
|
|
188
|
+
f"p-value = {fmt(p, 4)}"]
|
|
189
|
+
return lines
|