lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/utils.py
ADDED
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
# utils.py — analog of the general utilities in lessR zzz.R
|
|
2
|
+
#
|
|
3
|
+
# Includes the option (style) system that replaces R's getOption()
|
|
4
|
+
# calls, number formatting (.fmt), and an R-style pretty() for axis
|
|
5
|
+
# tick values, which lessR computes upstream of the render functions.
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
|
|
9
|
+
# style settings; R analog: options() set by lessR style()
|
|
10
|
+
_OPTIONS = {
|
|
11
|
+
"main_size": 1.0,
|
|
12
|
+
"lab_size": 1.0,
|
|
13
|
+
"axis_size": 0.9,
|
|
14
|
+
"axis_color": "black",
|
|
15
|
+
"lab_color": "black",
|
|
16
|
+
"grid_color": "gray90",
|
|
17
|
+
"grid_col": "gray85",
|
|
18
|
+
"grid_lwd": 0.5,
|
|
19
|
+
"grid_lty": None,
|
|
20
|
+
"axis_lwd": 1,
|
|
21
|
+
"panel_border": "#808080",
|
|
22
|
+
"panel_lwd": 1,
|
|
23
|
+
"panel_fill": "white",
|
|
24
|
+
"window_fill": "white",
|
|
25
|
+
"digits_d": 2,
|
|
26
|
+
"pt_color": "#324E5C", # rgb(50,78,92), zzz_on.R
|
|
27
|
+
"trans_pt_fill": 0.10,
|
|
28
|
+
"segment_color": "gray40",
|
|
29
|
+
"fit_color": "#5C4032", # rgb(92,64,50), zzz_on.R
|
|
30
|
+
"fit_lwd": 2,
|
|
31
|
+
"se_fill": "#1A1A1A19",
|
|
32
|
+
"ellipse_fill": "#92806F28",
|
|
33
|
+
"ellipse_color": "gray20",
|
|
34
|
+
"ellipse_lwd": 1,
|
|
35
|
+
"violin_fill": "#7485975A",
|
|
36
|
+
"violin_color": "gray15",
|
|
37
|
+
"box_fill": "#419BD2", # rgb(65,155,210), zzz_on.R
|
|
38
|
+
"box_color": "gray15",
|
|
39
|
+
"out_fill": "#8B1A1A", # firebrick4
|
|
40
|
+
"out_color": "#8B1A1A",
|
|
41
|
+
"out2_fill": "#EE2C2C", # firebrick2
|
|
42
|
+
"out2_color": "#EE2C2C",
|
|
43
|
+
"strip_fill": "#7F7F7F37",
|
|
44
|
+
"strip_color": "gray40",
|
|
45
|
+
"strip_text_color": "gray15",
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def get_option(name, default=None):
|
|
50
|
+
val = _OPTIONS.get(name, default)
|
|
51
|
+
return default if val is None else val
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def set_option(name, value):
|
|
55
|
+
_OPTIONS[name] = value
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def fmt(k, d=None):
|
|
59
|
+
"""Format one number with d decimal digits. R analog: .fmt()"""
|
|
60
|
+
if d is None:
|
|
61
|
+
d = get_option("digits_d", 2)
|
|
62
|
+
return f"{k:.{d}f}"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# aggregation functions for stat=; sd uses n-1 as in R
|
|
66
|
+
# R analog: .stat_fun()
|
|
67
|
+
STAT_FUN = {
|
|
68
|
+
"mean": lambda s: s.mean(),
|
|
69
|
+
"sum": lambda s: s.sum(),
|
|
70
|
+
"sd": lambda s: s.std(ddof=1),
|
|
71
|
+
"min": lambda s: s.min(),
|
|
72
|
+
"median": lambda s: s.median(),
|
|
73
|
+
"max": lambda s: s.max(),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
# display label for a stat= value; R analog: .stat_lbl()
|
|
77
|
+
STAT_LBL = {
|
|
78
|
+
"sum": "Sum",
|
|
79
|
+
"mean": "Mean",
|
|
80
|
+
"sd": "Standard Deviation",
|
|
81
|
+
"deviation": "Mean Deviation",
|
|
82
|
+
"min": "Minimum",
|
|
83
|
+
"median": "Median",
|
|
84
|
+
"max": "Maximum",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def category_order(s):
|
|
89
|
+
"""Category order: declared order for a pandas Categorical,
|
|
90
|
+
else sorted unique values -- the same convention as an R factor
|
|
91
|
+
built by table(). Shared by Chart() and XY()."""
|
|
92
|
+
import pandas as pd
|
|
93
|
+
if isinstance(s.dtype, pd.CategoricalDtype):
|
|
94
|
+
return [c for c in s.cat.categories if c in set(s.dropna())]
|
|
95
|
+
return sorted(s.dropna().unique().tolist())
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def get_column(data, name, arg):
|
|
99
|
+
"""Resolve a string column name against a DataFrame, with the
|
|
100
|
+
errors the string interface implies. Shared by Chart() and X()."""
|
|
101
|
+
if not isinstance(name, str):
|
|
102
|
+
raise TypeError(
|
|
103
|
+
f"{arg} must be a string naming a column of data, "
|
|
104
|
+
f"not {type(name).__name__}. (Python has no equivalent "
|
|
105
|
+
"of R's unquoted variable names.)")
|
|
106
|
+
if name not in data.columns:
|
|
107
|
+
raise KeyError(
|
|
108
|
+
f"{arg}='{name}' is not a column of data. "
|
|
109
|
+
f"Columns: {', '.join(map(str, data.columns))}")
|
|
110
|
+
return data[name]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def facet_values(data, values, arg):
|
|
114
|
+
"""Resolve a computed facet -- a pandas Series or numpy array
|
|
115
|
+
of values aligned with data, the Python analog of R's facet
|
|
116
|
+
expression. Length-checked against data, as R checks the
|
|
117
|
+
expression length. Returns an aligned Series."""
|
|
118
|
+
import numpy as np
|
|
119
|
+
import pandas as pd
|
|
120
|
+
if isinstance(values, pd.Series):
|
|
121
|
+
if len(values) != len(data):
|
|
122
|
+
raise ValueError(
|
|
123
|
+
f"data has {len(data)} rows, but the {arg} Series "
|
|
124
|
+
f"has {len(values)} values")
|
|
125
|
+
name = values.name if values.name else arg
|
|
126
|
+
return pd.Series(values.to_numpy(), index=data.index,
|
|
127
|
+
name=name)
|
|
128
|
+
if isinstance(values, np.ndarray):
|
|
129
|
+
if len(values) != len(data):
|
|
130
|
+
raise ValueError(
|
|
131
|
+
f"data has {len(data)} rows, but the {arg} array "
|
|
132
|
+
f"has {len(values)} values")
|
|
133
|
+
return pd.Series(values, index=data.index, name=arg)
|
|
134
|
+
raise TypeError(
|
|
135
|
+
f"{arg} must be a string naming a column of data, a "
|
|
136
|
+
f"list of such names, or a pandas Series / numpy array "
|
|
137
|
+
f"of computed values, not {type(values).__name__}")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def resolve_facet(data, facet, fun):
|
|
141
|
+
"""Resolve facet= into up to two aligned Series:
|
|
142
|
+
(facet1, facet1_name, facet2, facet2_name). Accepts a column
|
|
143
|
+
name, a list/tuple of names (facet1 = panel columns, facet2 =
|
|
144
|
+
panel rows; more than two uses the first two, with a message),
|
|
145
|
+
or a Series/array of computed values (facet_values). R analog:
|
|
146
|
+
the --- resolve facet --- block of X.R / XY.R."""
|
|
147
|
+
if facet is None:
|
|
148
|
+
return None, None, None, None
|
|
149
|
+
if isinstance(facet, str):
|
|
150
|
+
return get_column(data, facet, "facet"), facet, None, None
|
|
151
|
+
if isinstance(facet, (list, tuple)):
|
|
152
|
+
if not all(isinstance(f, str) for f in facet):
|
|
153
|
+
raise TypeError(
|
|
154
|
+
"a facet list gives column names; for computed "
|
|
155
|
+
"values pass a single pandas Series or numpy "
|
|
156
|
+
"array")
|
|
157
|
+
if len(facet) == 0:
|
|
158
|
+
raise ValueError(
|
|
159
|
+
"facet names a categorical variable for its "
|
|
160
|
+
"panels")
|
|
161
|
+
if len(facet) > 2:
|
|
162
|
+
print(f"facet has {len(facet)} variables. {fun}() "
|
|
163
|
+
"uses the first two: "
|
|
164
|
+
f"{facet[0]}, {facet[1]}.")
|
|
165
|
+
f1 = get_column(data, facet[0], "facet")
|
|
166
|
+
if len(facet) == 1:
|
|
167
|
+
return f1, facet[0], None, None
|
|
168
|
+
f2 = get_column(data, facet[1], "facet")
|
|
169
|
+
return f1, facet[0], f2, facet[1]
|
|
170
|
+
ser = facet_values(data, facet, "facet")
|
|
171
|
+
return ser, str(ser.name), None, None
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def kde(xg, grid, h):
|
|
175
|
+
"""Gaussian KDE of xg evaluated on grid with bandwidth h.
|
|
176
|
+
Shared by the density and violin renderers."""
|
|
177
|
+
import numpy as np
|
|
178
|
+
z = (grid[:, None] - xg[None, :]) / h
|
|
179
|
+
return (np.exp(-0.5 * z * z).sum(axis=1)
|
|
180
|
+
/ (len(xg) * h * float(np.sqrt(2 * np.pi))))
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def bw_nrd0(x):
|
|
184
|
+
"""Silverman rule-of-thumb KDE bandwidth, the default of R's
|
|
185
|
+
density(). R analog: stats::bw.nrd0()"""
|
|
186
|
+
import numpy as np
|
|
187
|
+
x = np.asarray(x, dtype=float)
|
|
188
|
+
x = x[np.isfinite(x)]
|
|
189
|
+
n = len(x)
|
|
190
|
+
if n < 2:
|
|
191
|
+
raise ValueError("bandwidth needs at least 2 data points")
|
|
192
|
+
sd = float(np.std(x, ddof=1))
|
|
193
|
+
iqr = float(np.subtract(*np.percentile(x, [75, 25])))
|
|
194
|
+
lo = min(sd, iqr / 1.34)
|
|
195
|
+
if lo == 0:
|
|
196
|
+
lo = sd or abs(float(x[0])) or 1.0
|
|
197
|
+
return 0.9 * lo * n ** -0.2
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def band_width(x, bw_iter=10, n=512):
|
|
201
|
+
"""Violin bandwidth: start at bw_nrd0 and widen by 10% per
|
|
202
|
+
iteration until the density curve has at most one direction
|
|
203
|
+
change (a single peak) or bw_iter iterations pass.
|
|
204
|
+
R analog: .band.width() (zzz.R)"""
|
|
205
|
+
import numpy as np
|
|
206
|
+
x = np.asarray(x, dtype=float)
|
|
207
|
+
x = x[np.isfinite(x)]
|
|
208
|
+
bw = bw_nrd0(x)
|
|
209
|
+
for _ in range(int(bw_iter)):
|
|
210
|
+
grid = np.linspace(x.min() - 3 * bw, x.max() + 3 * bw, n)
|
|
211
|
+
xd = np.diff(kde(x, grid, bw))
|
|
212
|
+
flips = int((np.sign(xd[1:]) != np.sign(xd[:-1])).sum())
|
|
213
|
+
if flips <= 1:
|
|
214
|
+
break
|
|
215
|
+
bw *= 1.1
|
|
216
|
+
return bw
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def pretty(lo, hi, n=5):
|
|
220
|
+
"""Nice tick values covering [lo, hi]. R analog: pretty()"""
|
|
221
|
+
if hi <= lo:
|
|
222
|
+
hi = lo + 1
|
|
223
|
+
raw = (hi - lo) / n
|
|
224
|
+
mag = 10 ** math.floor(math.log10(raw))
|
|
225
|
+
step = 10 * mag
|
|
226
|
+
for m in (1, 2, 5, 10):
|
|
227
|
+
if raw <= m * mag * (1 + 1e-10):
|
|
228
|
+
step = m * mag
|
|
229
|
+
break
|
|
230
|
+
start = math.floor(lo / step + 1e-10) * step
|
|
231
|
+
end = math.ceil(hi / step - 1e-10) * step
|
|
232
|
+
k = round((end - start) / step)
|
|
233
|
+
vals = [start + i * step for i in range(k + 1)]
|
|
234
|
+
# avoid -0.0 and float dust such as 0.30000000000000004
|
|
235
|
+
return [round(v, 10) + 0.0 for v in vals]
|