lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/pie_plotly.py
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
# pie_plotly.py — analog of piechart.plotly.R
|
|
2
|
+
#
|
|
3
|
+
# Renders a donut/pie from ALREADY-TABULATED input:
|
|
4
|
+
# 1-D pandas Series: index = slice names, values = counts/means
|
|
5
|
+
# -> one pie
|
|
6
|
+
# 2-D pandas DataFrame: rows = `by` levels, columns = slices
|
|
7
|
+
# -> a grid of pies, one per by level, group name in each hole
|
|
8
|
+
#
|
|
9
|
+
# Deliberate deviations from the R source:
|
|
10
|
+
# - The theme-dependent sequential fill branch (.scale.clr) is not
|
|
11
|
+
# ported; without a theme system, default fills are BASE_COLORS,
|
|
12
|
+
# the equivalent of lessR's default "colors" theme (hues).
|
|
13
|
+
# - R's defensive plotly_build() post-processing of slice borders
|
|
14
|
+
# works around R-plotly quirks; plotly.py sets marker.line
|
|
15
|
+
# directly, so the same result needs no post-build step.
|
|
16
|
+
|
|
17
|
+
import math
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
import plotly.graph_objects as go
|
|
22
|
+
|
|
23
|
+
from .utils import get_option
|
|
24
|
+
from .plotly_utils import (
|
|
25
|
+
BASE_COLORS, as_plotly_color, auto_text_color, make_trans,
|
|
26
|
+
plotly_style, to_hex,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
_LABEL_VALUES = ("%", "input", "prop", "off")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _pick_by_names_or_recycle(v, needed_names):
|
|
33
|
+
"""dict: pick by slice name; else recycle across slices.
|
|
34
|
+
R analog: .pick_by_names_or_recycle()"""
|
|
35
|
+
n = len(needed_names)
|
|
36
|
+
if isinstance(v, dict):
|
|
37
|
+
base = list(v.values()) or ["black"]
|
|
38
|
+
return [v.get(nm, base[i % len(base)])
|
|
39
|
+
for i, nm in enumerate(needed_names)]
|
|
40
|
+
if v is None:
|
|
41
|
+
return ["black"] * n
|
|
42
|
+
if not isinstance(v, (list, tuple)):
|
|
43
|
+
v = [v]
|
|
44
|
+
return [v[i % len(v)] for i in range(n)]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _num_str(vals, total, labels, digits_d, labels_decimals=None):
|
|
48
|
+
"""labels_decimals sets the decimal places; None falls back to
|
|
49
|
+
digits_d, except for "prop", whose values are proportions that
|
|
50
|
+
digits_d (0 for counts) would flatten to "0"."""
|
|
51
|
+
denom = max(total, 1e-12)
|
|
52
|
+
digits_d = (int(labels_decimals) if labels_decimals is not None
|
|
53
|
+
else (2 if labels == "prop" else digits_d))
|
|
54
|
+
if labels == "%":
|
|
55
|
+
return [f"{100 * v / denom:.{digits_d}f}%" for v in vals]
|
|
56
|
+
if labels == "prop":
|
|
57
|
+
return [f"{v / denom:.{digits_d}f}" for v in vals]
|
|
58
|
+
if labels == "input":
|
|
59
|
+
return [f"{v:.{digits_d}f}" for v in vals]
|
|
60
|
+
return ["" for _ in vals] # "off"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _slice_text(slices, num_str, labels):
|
|
64
|
+
if labels == "off":
|
|
65
|
+
return list(slices)
|
|
66
|
+
return [f"{s}<br>{n}" for s, n in zip(slices, num_str)]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _text_colors(labels_color, fills_rgba, panel_fill):
|
|
70
|
+
if labels_color is None or labels_color == "adjust":
|
|
71
|
+
return auto_text_color(fills_rgba, bg=panel_fill)
|
|
72
|
+
return [labels_color] * len(fills_rgba)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def pie_plotly(x, x_name=None, y_name=None, by_name=None, main=None,
|
|
76
|
+
fill=None, border=None, opacity=1.0,
|
|
77
|
+
hole=0.65, ncols=None,
|
|
78
|
+
labels=None, labels_position="in",
|
|
79
|
+
labels_color=None, labels_size=1.0,
|
|
80
|
+
labels_decimals=None,
|
|
81
|
+
digits_d=2,
|
|
82
|
+
group_labels=True, group_label_size=14,
|
|
83
|
+
style_opts=None):
|
|
84
|
+
|
|
85
|
+
if labels is None:
|
|
86
|
+
labels = "%" # R analog: match.arg default
|
|
87
|
+
if labels not in _LABEL_VALUES:
|
|
88
|
+
raise ValueError(f"labels must be one of {_LABEL_VALUES}")
|
|
89
|
+
|
|
90
|
+
two_d = isinstance(x, pd.DataFrame)
|
|
91
|
+
if not two_d and not isinstance(x, pd.Series):
|
|
92
|
+
raise TypeError("x must be a pandas Series (1-D) or "
|
|
93
|
+
"DataFrame (2-D, rows = by levels)")
|
|
94
|
+
|
|
95
|
+
if x_name is None:
|
|
96
|
+
x_name = (x.columns.name if two_d else x.index.name) or "x"
|
|
97
|
+
if y_name is None:
|
|
98
|
+
y_name = "Count" if two_d else (x.name or "Count")
|
|
99
|
+
if two_d and by_name is None:
|
|
100
|
+
by_name = x.index.name or "Group"
|
|
101
|
+
if style_opts is None:
|
|
102
|
+
style_opts = plotly_style()
|
|
103
|
+
if fill is None:
|
|
104
|
+
fill = BASE_COLORS
|
|
105
|
+
if border is None:
|
|
106
|
+
border = "transparent"
|
|
107
|
+
|
|
108
|
+
title_size = round(16 * get_option("main_size", 1))
|
|
109
|
+
alpha_fill = float(opacity)
|
|
110
|
+
if not math.isfinite(alpha_fill):
|
|
111
|
+
alpha_fill = 1.0
|
|
112
|
+
alpha_fill = max(0.0, min(1.0, alpha_fill))
|
|
113
|
+
|
|
114
|
+
# None = the default (R's labels_position %||% "in")
|
|
115
|
+
pos_in = labels_position is None or \
|
|
116
|
+
str(labels_position).lower() == "in"
|
|
117
|
+
txt_pos = "inside" if pos_in else "outside"
|
|
118
|
+
val_spec = f":.{max(0, int(digits_d))}f"
|
|
119
|
+
|
|
120
|
+
def title_layout():
|
|
121
|
+
if main:
|
|
122
|
+
return dict(text=main, y=0.94, yanchor="top",
|
|
123
|
+
font=dict(
|
|
124
|
+
size=title_size,
|
|
125
|
+
color=to_hex(get_option("lab_color",
|
|
126
|
+
"black"))))
|
|
127
|
+
return None
|
|
128
|
+
|
|
129
|
+
def panel_bg(fig):
|
|
130
|
+
# simulated panel background; only when not white
|
|
131
|
+
bg_plot = str(to_hex(style_opts["panel_fill"])).upper()
|
|
132
|
+
bg_paper = str(to_hex(style_opts["window_fill"])).upper()
|
|
133
|
+
if bg_plot != "#FFFFFF" or bg_paper != "#FFFFFF":
|
|
134
|
+
fig.update_layout(paper_bgcolor=bg_paper)
|
|
135
|
+
fig.add_shape(type="rect", xref="paper", yref="paper",
|
|
136
|
+
x0=0, y0=0, x1=1, y1=1, layer="below",
|
|
137
|
+
fillcolor=bg_plot, line=dict(width=0))
|
|
138
|
+
|
|
139
|
+
# --- SINGLE PIE: 1-D --------------------------------------------
|
|
140
|
+
if not two_d:
|
|
141
|
+
slices = [str(s) for s in x.index]
|
|
142
|
+
values = x.to_numpy(dtype=float)
|
|
143
|
+
tot = np.nansum(values)
|
|
144
|
+
|
|
145
|
+
fill_vec = _pick_by_names_or_recycle(fill, slices)
|
|
146
|
+
border_vec = _pick_by_names_or_recycle(border, slices)
|
|
147
|
+
|
|
148
|
+
num_str = _num_str(values, tot, labels, digits_d,
|
|
149
|
+
labels_decimals)
|
|
150
|
+
text_vec = _slice_text(slices, num_str, labels)
|
|
151
|
+
|
|
152
|
+
txt_size = round(12 * 1.38 * labels_size *
|
|
153
|
+
get_option("axis_size", 0.9))
|
|
154
|
+
|
|
155
|
+
fills_rgba = make_trans(fill_vec, alpha_fill)
|
|
156
|
+
txt_colors = _text_colors(labels_color, fills_rgba,
|
|
157
|
+
style_opts["panel_fill"])
|
|
158
|
+
|
|
159
|
+
overall_pct = values / (tot if tot > 0 else 1)
|
|
160
|
+
|
|
161
|
+
# inside labels need room: clamp an over-large hole
|
|
162
|
+
hole_use = 0.62 if (pos_in and hole > 0.62) else hole
|
|
163
|
+
|
|
164
|
+
font = dict(color=txt_colors, size=txt_size)
|
|
165
|
+
fig = go.Figure(go.Pie(
|
|
166
|
+
labels=slices,
|
|
167
|
+
values=values,
|
|
168
|
+
sort=False,
|
|
169
|
+
direction="clockwise",
|
|
170
|
+
hole=hole_use,
|
|
171
|
+
domain=dict(x=[0, 1], y=[0.03, 0.93]),
|
|
172
|
+
text=text_vec,
|
|
173
|
+
textinfo="text",
|
|
174
|
+
textposition=txt_pos,
|
|
175
|
+
insidetextorientation="radial",
|
|
176
|
+
textfont=font,
|
|
177
|
+
insidetextfont=font,
|
|
178
|
+
outsidetextfont=font,
|
|
179
|
+
automargin=not pos_in,
|
|
180
|
+
marker=dict(
|
|
181
|
+
colors=fills_rgba,
|
|
182
|
+
line=dict(color=as_plotly_color(border_vec),
|
|
183
|
+
width=1.6),
|
|
184
|
+
),
|
|
185
|
+
customdata=overall_pct,
|
|
186
|
+
hovertemplate=(
|
|
187
|
+
f"{x_name}: %{{label}}"
|
|
188
|
+
f"<br>{y_name}: %{{value{val_spec}}}"
|
|
189
|
+
"<br>% of total: %{customdata:.2%}"
|
|
190
|
+
"<extra></extra>"),
|
|
191
|
+
showlegend=False,
|
|
192
|
+
))
|
|
193
|
+
|
|
194
|
+
fig.update_layout(
|
|
195
|
+
uniformtext=dict(minsize=8, mode="show"),
|
|
196
|
+
margin=dict(t=round(title_size * 2.2),
|
|
197
|
+
r=20, b=30, l=20),
|
|
198
|
+
title=title_layout(),
|
|
199
|
+
)
|
|
200
|
+
panel_bg(fig)
|
|
201
|
+
return fig
|
|
202
|
+
|
|
203
|
+
# --- GROUPED INPUT (2-D): PIE GRID ------------------------------
|
|
204
|
+
groups = [str(g) for g in x.index]
|
|
205
|
+
slices = [str(c) for c in x.columns]
|
|
206
|
+
mat = x.to_numpy(dtype=float)
|
|
207
|
+
grand_total = np.nansum(mat)
|
|
208
|
+
|
|
209
|
+
k = len(groups)
|
|
210
|
+
nc = (math.ceil(math.sqrt(k)) if ncols is None or ncols < 1
|
|
211
|
+
else int(ncols))
|
|
212
|
+
nr = math.ceil(k / nc)
|
|
213
|
+
|
|
214
|
+
txt_size = round(12 * labels_size * get_option("axis_size", 0.9))
|
|
215
|
+
by_title = by_name if by_name else "Group"
|
|
216
|
+
|
|
217
|
+
fig = go.Figure()
|
|
218
|
+
|
|
219
|
+
for i, grp in enumerate(groups):
|
|
220
|
+
col, row = i % nc, i // nc
|
|
221
|
+
x0, x1 = col / nc, (col + 1) / nc
|
|
222
|
+
y0, y1 = 1 - (row + 1) / nr, 1 - row / nr
|
|
223
|
+
shrink = 0.92
|
|
224
|
+
y_mid, y_half = (y0 + y1) / 2, (y1 - y0) * shrink / 2
|
|
225
|
+
dom = dict(x=[x0, x1], y=[y_mid - y_half, y_mid + y_half])
|
|
226
|
+
|
|
227
|
+
vals = mat[i].copy()
|
|
228
|
+
vals[~np.isfinite(vals)] = np.nan
|
|
229
|
+
val_sum = np.nansum(vals)
|
|
230
|
+
overall_pct = vals / (grand_total if grand_total > 0 else 1)
|
|
231
|
+
|
|
232
|
+
num_str = _num_str(vals, val_sum, labels, digits_d,
|
|
233
|
+
labels_decimals)
|
|
234
|
+
text_vec = _slice_text(slices, num_str, labels)
|
|
235
|
+
|
|
236
|
+
fills_this = make_trans(
|
|
237
|
+
_pick_by_names_or_recycle(fill, slices), alpha_fill)
|
|
238
|
+
borders_this = _pick_by_names_or_recycle(border, slices)
|
|
239
|
+
txt_colors = _text_colors(labels_color, fills_this,
|
|
240
|
+
style_opts["panel_fill"])
|
|
241
|
+
|
|
242
|
+
font = dict(color=txt_colors, size=txt_size)
|
|
243
|
+
fig.add_trace(go.Pie(
|
|
244
|
+
labels=slices,
|
|
245
|
+
values=vals,
|
|
246
|
+
name=f"{by_title}: {grp}",
|
|
247
|
+
legendgroup="pies",
|
|
248
|
+
sort=False,
|
|
249
|
+
direction="clockwise",
|
|
250
|
+
hole=hole,
|
|
251
|
+
domain=dom,
|
|
252
|
+
text=text_vec,
|
|
253
|
+
textinfo="text",
|
|
254
|
+
textposition=txt_pos,
|
|
255
|
+
insidetextorientation="radial",
|
|
256
|
+
textfont=font,
|
|
257
|
+
insidetextfont=font,
|
|
258
|
+
outsidetextfont=font,
|
|
259
|
+
automargin=not pos_in,
|
|
260
|
+
marker=dict(
|
|
261
|
+
colors=fills_this,
|
|
262
|
+
line=dict(color=as_plotly_color(borders_this),
|
|
263
|
+
width=1),
|
|
264
|
+
),
|
|
265
|
+
customdata=overall_pct,
|
|
266
|
+
hovertemplate=(
|
|
267
|
+
f"{by_title}: {grp}"
|
|
268
|
+
f"<br>{x_name}: %{{label}}"
|
|
269
|
+
f"<br>{y_name}: %{{value{val_spec}}}"
|
|
270
|
+
f"<br>% of {grp}: %{{percent}}"
|
|
271
|
+
"<br>% of total: %{customdata:.2%}"
|
|
272
|
+
"<extra></extra>"),
|
|
273
|
+
showlegend=False,
|
|
274
|
+
))
|
|
275
|
+
|
|
276
|
+
if group_labels and hole > 0:
|
|
277
|
+
fig.add_annotation(
|
|
278
|
+
x=(dom["x"][0] + dom["x"][1]) / 2,
|
|
279
|
+
y=(dom["y"][0] + dom["y"][1]) / 2,
|
|
280
|
+
xref="paper", yref="paper",
|
|
281
|
+
text=grp, showarrow=False,
|
|
282
|
+
xanchor="center", yanchor="middle",
|
|
283
|
+
font=dict(size=group_label_size, color="#666666"),
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
fig.update_layout(
|
|
287
|
+
uniformtext=dict(minsize=10, mode="show"),
|
|
288
|
+
margin=dict(t=round(title_size * 2.2), r=20, b=20, l=20),
|
|
289
|
+
title=title_layout(),
|
|
290
|
+
)
|
|
291
|
+
panel_bg(fig)
|
|
292
|
+
return fig
|
lessPy/pivot.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# pivot.py — analog of pivot.R (the aggregation core).
|
|
2
|
+
#
|
|
3
|
+
# pivot(): aggregate a numeric variable over the categories of one
|
|
4
|
+
# or more `by` grouping variables, computing one or more summary
|
|
5
|
+
# statistics, or tabulate frequencies. The result is a long-form
|
|
6
|
+
# DataFrame with the by columns, an n (and na) count, and one
|
|
7
|
+
# {variable}_{stat} column per statistic — lessR's pivot output.
|
|
8
|
+
#
|
|
9
|
+
# Ported: the aggregation over by groups (sum, mean, median, min,
|
|
10
|
+
# max, sd, var, IQR, mad), the n/na counts and show_n, all group
|
|
11
|
+
# combinations (empty cells kept, as R's drop=FALSE), NA groups,
|
|
12
|
+
# sort= by the statistic, and the one- and two-way frequency
|
|
13
|
+
# table (compute="table"). Not ported: by_cols wide cross-tabs,
|
|
14
|
+
# table_prop row/col proportions, quantiles, and skew/kurtosis.
|
|
15
|
+
#
|
|
16
|
+
# The interface uses string names (compute="mean", variable=,
|
|
17
|
+
# by=), the lessPy convention.
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
|
|
22
|
+
from .utils import get_column
|
|
23
|
+
|
|
24
|
+
# statistic name -> (column-name abbreviation, aggregator).
|
|
25
|
+
# pandas quantile is linear (R type 7); std/var use n-1; all skip
|
|
26
|
+
# NaN, so na_remove is the default. R analog: pivot.R fun.vec
|
|
27
|
+
_STAT = {
|
|
28
|
+
"sum": ("sum", lambda s: s.sum()),
|
|
29
|
+
"mean": ("mean", lambda s: s.mean()),
|
|
30
|
+
"median": ("mdn", lambda s: s.median()),
|
|
31
|
+
"min": ("min", lambda s: s.min()),
|
|
32
|
+
"max": ("max", lambda s: s.max()),
|
|
33
|
+
"sd": ("sd", lambda s: s.std(ddof=1)),
|
|
34
|
+
"var": ("var", lambda s: s.var(ddof=1)),
|
|
35
|
+
"IQR": ("IQR", lambda s: s.quantile(0.75)
|
|
36
|
+
- s.quantile(0.25)),
|
|
37
|
+
"mad": ("mad", lambda s:
|
|
38
|
+
1.4826 * (s - s.median()).abs().median()),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def pivot(data, compute, variable=None, by=None, filter=None,
|
|
43
|
+
show_n=True, na_remove=True, sort=None, digits_d=None,
|
|
44
|
+
quiet=False):
|
|
45
|
+
"""Aggregate a numeric variable over by groups, or tabulate
|
|
46
|
+
frequencies. compute is a statistic name or a list of names
|
|
47
|
+
("mean", ["mean","sd"], "table"); variable is the numeric
|
|
48
|
+
column to aggregate; by is the grouping column(s). Returns a
|
|
49
|
+
long-form DataFrame. R analog: pivot()"""
|
|
50
|
+
if filter is not None:
|
|
51
|
+
data = data.query(filter)
|
|
52
|
+
computes = [compute] if isinstance(compute, str) \
|
|
53
|
+
else list(compute)
|
|
54
|
+
by = ([by] if isinstance(by, str)
|
|
55
|
+
else list(by) if by is not None else [])
|
|
56
|
+
if sort is not None and sort not in ("+", "-"):
|
|
57
|
+
raise ValueError('sort: "+" or "-"')
|
|
58
|
+
|
|
59
|
+
if "table" in computes:
|
|
60
|
+
if len(computes) > 1:
|
|
61
|
+
raise ValueError('compute="table" cannot be combined '
|
|
62
|
+
"with other statistics")
|
|
63
|
+
if not by:
|
|
64
|
+
raise ValueError('compute="table" needs by=')
|
|
65
|
+
return _pivot_table(data, by, show_n)
|
|
66
|
+
|
|
67
|
+
if variable is None:
|
|
68
|
+
raise ValueError("variable= is required: the numeric "
|
|
69
|
+
"column to aggregate")
|
|
70
|
+
unknown = [c for c in computes if c not in _STAT]
|
|
71
|
+
if unknown:
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"unknown compute {unknown}; use "
|
|
74
|
+
f"{', '.join(_STAT)}, or \"table\"")
|
|
75
|
+
v = get_column(data, variable, "variable")
|
|
76
|
+
if not pd.api.types.is_numeric_dtype(v):
|
|
77
|
+
raise TypeError(
|
|
78
|
+
f"the variable to aggregate '{variable}' is "
|
|
79
|
+
f"{v.dtype}: it must be numeric. Put categorical "
|
|
80
|
+
"variables in by=.")
|
|
81
|
+
if not by:
|
|
82
|
+
raise ValueError("by= is required: the grouping "
|
|
83
|
+
"column(s)")
|
|
84
|
+
return _pivot_agg(data, variable, by, computes, show_n, sort)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _cat_frame(data, by):
|
|
88
|
+
"""Copy of data with each by column made an ordered Categorical
|
|
89
|
+
(sorted levels), so groupby keeps every level combination and
|
|
90
|
+
the NA group, as R's factor()/drop=FALSE."""
|
|
91
|
+
g = data.copy()
|
|
92
|
+
for b in by:
|
|
93
|
+
col = g[b]
|
|
94
|
+
cats = sorted(pd.unique(col.dropna()),
|
|
95
|
+
key=lambda x: (str(type(x)), x))
|
|
96
|
+
g[b] = pd.Categorical(col, categories=cats)
|
|
97
|
+
return g
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _pivot_agg(data, variable, by, computes, show_n, sort):
|
|
101
|
+
g = _cat_frame(data, by)
|
|
102
|
+
grp = g.groupby(by, observed=False, dropna=False, sort=True)
|
|
103
|
+
v = grp[variable]
|
|
104
|
+
out = pd.DataFrame({
|
|
105
|
+
"n": v.apply(lambda s: int(s.notna().sum())),
|
|
106
|
+
"na": v.apply(lambda s: int(s.isna().sum()))})
|
|
107
|
+
for c in computes:
|
|
108
|
+
abbr, fn = _STAT[c]
|
|
109
|
+
out[f"{variable}_{abbr}"] = v.apply(fn)
|
|
110
|
+
out = out.reset_index()
|
|
111
|
+
# R aggregate row order: the FIRST by var varies fastest, so
|
|
112
|
+
# sort by the by columns in reverse listing order, NA last
|
|
113
|
+
out = out.sort_values(by[::-1], na_position="last",
|
|
114
|
+
kind="stable").reset_index(drop=True)
|
|
115
|
+
if sort is not None:
|
|
116
|
+
stat_col = out.columns[-1]
|
|
117
|
+
out = out.sort_values(
|
|
118
|
+
stat_col, ascending=(sort == "+"),
|
|
119
|
+
na_position="last", kind="stable"
|
|
120
|
+
).reset_index(drop=True)
|
|
121
|
+
if not show_n:
|
|
122
|
+
out = out.drop(columns=["n", "na"])
|
|
123
|
+
return out
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _pivot_table(data, by, show_n):
|
|
127
|
+
if len(by) == 1:
|
|
128
|
+
b = by[0]
|
|
129
|
+
col = data[b]
|
|
130
|
+
cats = sorted(pd.unique(col.dropna()),
|
|
131
|
+
key=lambda x: (str(type(x)), x))
|
|
132
|
+
cc = pd.Categorical(col, categories=cats)
|
|
133
|
+
n = pd.Series(cc, name=b).value_counts(
|
|
134
|
+
dropna=False, sort=False)
|
|
135
|
+
# value_counts on Categorical excludes NaN; add it back
|
|
136
|
+
n = n.reindex(cats)
|
|
137
|
+
na_count = int(col.isna().sum())
|
|
138
|
+
idx = list(cats)
|
|
139
|
+
counts = [int(n[c]) for c in cats]
|
|
140
|
+
if na_count:
|
|
141
|
+
idx.append(np.nan)
|
|
142
|
+
counts.append(na_count)
|
|
143
|
+
total = sum(counts)
|
|
144
|
+
out = pd.DataFrame({b: idx, "n": counts})
|
|
145
|
+
out["Prop"] = np.round(np.array(counts) / total, 2)
|
|
146
|
+
return out
|
|
147
|
+
if len(by) > 2:
|
|
148
|
+
raise ValueError('compute="table" supports one or two '
|
|
149
|
+
"by variables")
|
|
150
|
+
# two-way: long-form counts, second by var first, as R
|
|
151
|
+
b1, b2 = by
|
|
152
|
+
g = _cat_frame(data, by)
|
|
153
|
+
ct = (g.groupby([b1, b2], observed=False, dropna=False)
|
|
154
|
+
.size().reset_index(name="n"))
|
|
155
|
+
ct = ct[[b2, b1, "n"]]
|
|
156
|
+
ct = ct.sort_values([b1, b2], na_position="last",
|
|
157
|
+
kind="stable").reset_index(drop=True)
|
|
158
|
+
return ct
|