lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Chart.py
ADDED
|
@@ -0,0 +1,1055 @@
|
|
|
1
|
+
# Chart.py — analog of Chart.R (and the data-prep parts of
|
|
2
|
+
# bc.zmain.R / agg.R that feed .bc.plotly)
|
|
3
|
+
#
|
|
4
|
+
# The R Chart() is 1800+ lines; most of that is R-specific machinery
|
|
5
|
+
# with no Python counterpart and is deliberately absent here:
|
|
6
|
+
# - non-standard evaluation of variable names (Python interface is
|
|
7
|
+
# strings naming DataFrame columns)
|
|
8
|
+
# - legacy/renamed parameter migration (new package, no legacy)
|
|
9
|
+
# - base-R / lattice rendering paths and PDF graphics devices
|
|
10
|
+
# (plotly-only port)
|
|
11
|
+
# What carries over is the pipeline: resolve variables -> filter ->
|
|
12
|
+
# tabulate or aggregate -> sort -> render.
|
|
13
|
+
#
|
|
14
|
+
# All seven forms are implemented: bar, pie, dot, radar, bubble,
|
|
15
|
+
# treemap, icicle — plus "sunburst", which (as in R) is an alias
|
|
16
|
+
# for pie; pie with by= renders as a sunburst via the hier path.
|
|
17
|
+
#
|
|
18
|
+
# facet= draws one panel per facet level: a grid of panels for
|
|
19
|
+
# dot, radar, bubble (1-D), and the hier forms; the pie grid for
|
|
20
|
+
# a plain pie (facet becomes the grouping, as in R); and for bar
|
|
21
|
+
# the Trellis chart — stacked panels of horizontal count bars,
|
|
22
|
+
# the plotly port of R's lattice rendering.
|
|
23
|
+
|
|
24
|
+
import math
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
import pandas as pd
|
|
28
|
+
from pandas.api.types import is_numeric_dtype
|
|
29
|
+
|
|
30
|
+
from .bc_plotly import bc_facet_plotly, bc_plotly
|
|
31
|
+
from .plt_add import plt_add
|
|
32
|
+
from .bubble_plotly import bubble_plotly
|
|
33
|
+
from .dot_plotly import dot_plotly
|
|
34
|
+
from .hier_plotly import (
|
|
35
|
+
hier_aggregate, hier_color_resolve, hier_plotly,
|
|
36
|
+
)
|
|
37
|
+
from .pie_plotly import pie_plotly
|
|
38
|
+
from .radar_plotly import radar_plotly
|
|
39
|
+
from .plotly_utils import build_title, font_scaled
|
|
40
|
+
from .stats_out import chart_stats, resolve_quiet
|
|
41
|
+
from .utils import (
|
|
42
|
+
STAT_FUN, STAT_LBL, category_order as _category_order,
|
|
43
|
+
facet_values, get_column as _get_column, get_option, pretty,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# color theme -> HCL sequential palette family (R analog: .get_fill)
|
|
47
|
+
_THEME_PALETTE = {
|
|
48
|
+
"colors": "blues", "dodgerblue": "blues", "blue": "blues",
|
|
49
|
+
"lightbronze": "blues",
|
|
50
|
+
"gray": "grays", "white": "grays", "light": "grays",
|
|
51
|
+
"gold": "browns", "brown": "browns", "sienna": "browns",
|
|
52
|
+
"orange": "rusts",
|
|
53
|
+
"darkred": "reds", "red": "reds", "rose": "reds",
|
|
54
|
+
"slatered": "reds",
|
|
55
|
+
"darkgreen": "greens", "green": "greens",
|
|
56
|
+
"purple": "violets",
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _theme_fill(theme, n):
|
|
61
|
+
"""A theme's fill: an n-color sequential palette in the theme's
|
|
62
|
+
hue. R analog: the theme branch of the fill assignment."""
|
|
63
|
+
pal = _THEME_PALETTE.get(theme)
|
|
64
|
+
if pal is None:
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"unknown theme '{theme}'; one of: "
|
|
67
|
+
f"{', '.join(sorted(_THEME_PALETTE))}")
|
|
68
|
+
from .getColors import getColors
|
|
69
|
+
cols = getColors(pal, n=max(1, n), quiet=True)
|
|
70
|
+
return [c[:7] if len(c) == 9 else c for c in cols] # drop alpha
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _facet_table(x_s, y_s, by_s, stat, is_agg, x_order, by_order,
|
|
74
|
+
proportion=False):
|
|
75
|
+
"""Frequency or stat table for one facet level, on the global
|
|
76
|
+
category set(s) so panels stay aligned; absent cells fill 0,
|
|
77
|
+
as in R's xtabs. Series without by, DataFrame (by x cats)
|
|
78
|
+
with. proportion normalizes the per-panel counts (no y, no by)
|
|
79
|
+
to sum to 1, the faceted stat_x="proportion". Used by the
|
|
80
|
+
faceted radar and bubble paths."""
|
|
81
|
+
if by_s is None:
|
|
82
|
+
if y_s is None:
|
|
83
|
+
t = x_s.groupby(x_s, observed=True).size() \
|
|
84
|
+
.reindex(x_order, fill_value=0)
|
|
85
|
+
if proportion:
|
|
86
|
+
tot = t.sum()
|
|
87
|
+
t = t / tot if tot > 0 else t.astype(float)
|
|
88
|
+
elif is_agg and stat is None:
|
|
89
|
+
t = (pd.Series(y_s.values, index=x_s.values)
|
|
90
|
+
.reindex(x_order))
|
|
91
|
+
else:
|
|
92
|
+
t = STAT_FUN[stat](
|
|
93
|
+
y_s.groupby(x_s, observed=True)).reindex(x_order)
|
|
94
|
+
return t.fillna(0)
|
|
95
|
+
|
|
96
|
+
if y_s is None:
|
|
97
|
+
t = pd.crosstab(by_s, x_s)
|
|
98
|
+
elif is_agg and stat is None:
|
|
99
|
+
t = (pd.DataFrame({"by": by_s, "x": x_s, "y": y_s})
|
|
100
|
+
.pivot(index="by", columns="x", values="y"))
|
|
101
|
+
else:
|
|
102
|
+
df = pd.DataFrame({"by": by_s, "x": x_s, "y": y_s})
|
|
103
|
+
t = STAT_FUN[stat](
|
|
104
|
+
df.groupby(["by", "x"], observed=True)["y"]
|
|
105
|
+
).unstack("x")
|
|
106
|
+
return (t.reindex(index=by_order, columns=x_order)
|
|
107
|
+
.fillna(0))
|
|
108
|
+
|
|
109
|
+
_FORMS = ("bar", "radar", "bubble", "dot", "pie", "icicle", "treemap")
|
|
110
|
+
_STATS = ("mean", "sum", "sd", "deviation", "min", "median", "max")
|
|
111
|
+
_SORTS = ("0", "-", "+")
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _sort_1d(tbl, sort):
|
|
115
|
+
if sort == "-":
|
|
116
|
+
return tbl.sort_values(ascending=False)
|
|
117
|
+
if sort == "+":
|
|
118
|
+
return tbl.sort_values(ascending=True)
|
|
119
|
+
return tbl
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _sort_2d(tbl, sort):
|
|
123
|
+
"""Order x categories (columns) by their totals across groups."""
|
|
124
|
+
if sort == "0":
|
|
125
|
+
return tbl
|
|
126
|
+
totals = tbl.sum(axis=0)
|
|
127
|
+
asc = sort == "+"
|
|
128
|
+
return tbl[totals.sort_values(ascending=asc).index]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _dot_origin_grid(vals, origin_in=None, is_counts=None):
|
|
132
|
+
"""Value-axis origin and grid ticks for a dot chart. Counts
|
|
133
|
+
anchor at 0; continuous data start one step below the first
|
|
134
|
+
tick unless the values hug zero. R analog: .dot_origin_grid()"""
|
|
135
|
+
fv = [float(v) for v in vals if np.isfinite(v)]
|
|
136
|
+
if not fv:
|
|
137
|
+
return (0 if origin_in is None else origin_in,
|
|
138
|
+
pretty(0, 1))
|
|
139
|
+
|
|
140
|
+
if is_counts is None:
|
|
141
|
+
is_counts = all(v >= 0 and v.is_integer() for v in fv)
|
|
142
|
+
|
|
143
|
+
origin = origin_in
|
|
144
|
+
if origin is None:
|
|
145
|
+
if is_counts:
|
|
146
|
+
origin = 0
|
|
147
|
+
else:
|
|
148
|
+
fv2 = [-v for v in fv] if all(v < 0 for v in fv) else fv
|
|
149
|
+
mn_v, mx_v = min(fv2), max(fv2)
|
|
150
|
+
if mn_v > 0 and (mx_v - mn_v) / mn_v <= 2.40:
|
|
151
|
+
origin = mn_v
|
|
152
|
+
|
|
153
|
+
lo = min(fv) if origin is None else min(origin, min(fv))
|
|
154
|
+
gridT = pretty(lo, max(lo, max(fv)))
|
|
155
|
+
# nudge one step below first tick for continuous data, not counts
|
|
156
|
+
if origin_in is None and not is_counts and len(gridT) > 1:
|
|
157
|
+
step = gridT[1] - gridT[0]
|
|
158
|
+
origin = gridT[0] - step
|
|
159
|
+
gridT = pretty(min(origin, min(fv)),
|
|
160
|
+
max(origin, max(fv)))
|
|
161
|
+
return origin, gridT
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
165
|
+
form="bar",
|
|
166
|
+
n_row=None, n_col=None,
|
|
167
|
+
hole=0.65,
|
|
168
|
+
radius=0.50, power=0.5,
|
|
169
|
+
pt_size=1, origin_x=None, origin_y=None,
|
|
170
|
+
segments_x=None, segments_y=None,
|
|
171
|
+
stat=None, stat_x="count",
|
|
172
|
+
horiz=False, sort="0", beside=False, stack100=False,
|
|
173
|
+
gap=None, scale_y=None, break_x=None,
|
|
174
|
+
fill=None, color=None, transparency=None,
|
|
175
|
+
fill_split=None, fill_scaled=False, fill_chroma=75,
|
|
176
|
+
theme=None,
|
|
177
|
+
labels=None, labels_position=None,
|
|
178
|
+
labels_color=None, labels_size=None, labels_decimals=None,
|
|
179
|
+
legend_title=None, legend_position=None,
|
|
180
|
+
legend_labels=None, legend_horiz=False,
|
|
181
|
+
legend_size=None, legend_abbrev=None, legend_adjust=0,
|
|
182
|
+
add=None, x1=None, y1=None, x2=None, y2=None,
|
|
183
|
+
xlab=None, ylab=None, main=None,
|
|
184
|
+
rotate_x=0, rotate_y=0,
|
|
185
|
+
axis_fmt="K", axis_x_pre="", axis_y_pre="",
|
|
186
|
+
digits_d=None, quiet=None):
|
|
187
|
+
"""General analytic view of one categorical variable x,
|
|
188
|
+
optionally crossed with a second categorical variable (by=) or
|
|
189
|
+
summarizing a numerical variable (y= with stat=). facet= names
|
|
190
|
+
one more categorical variable (or a list of them, combined
|
|
191
|
+
into one panel grid) that splits the chart into one panel per
|
|
192
|
+
level; n_row/n_col lay those panels out as a grid.
|
|
193
|
+
|
|
194
|
+
The parameter order follows R's Chart(), so the second
|
|
195
|
+
positional argument is y, the numerical variable the chart
|
|
196
|
+
aggregates: Chart("Dept", "Salary", stat="mean", data=d). A
|
|
197
|
+
second CATEGORICAL variable is named -- by="Gender".
|
|
198
|
+
|
|
199
|
+
labels_position defaults to auto-tuning for a bar chart: each
|
|
200
|
+
value label sits inside its bar, or moves outside when the bar
|
|
201
|
+
is too small to hold it. Pass "in" or "out" to force one.
|
|
202
|
+
|
|
203
|
+
stack100 rescales each bar to sum to 1.0, so a stacked bar
|
|
204
|
+
chart shows the composition of every x category on a common
|
|
205
|
+
0-100% scale. Requires by=. labels="input" then displays the
|
|
206
|
+
underlying counts, as in R.
|
|
207
|
+
|
|
208
|
+
gap sets the spacing between bars in units of bar width (R's
|
|
209
|
+
barplot space=): one value, or (within, between) with beside=.
|
|
210
|
+
scale_y is (min, max, n_intervals) for the value axis, giving
|
|
211
|
+
n_intervals + 1 ticks (R's axTicks(axp=)). break_x breaks
|
|
212
|
+
category labels at their spaces onto separate lines; a "~" in
|
|
213
|
+
a label is a non-breaking space.
|
|
214
|
+
|
|
215
|
+
Value-axis tick labels follow the axis_fmt policies (default
|
|
216
|
+
"K": "60K" for thousands, commas past 9999); axis_x_pre /
|
|
217
|
+
axis_y_pre prepend a per-axis prefix (e.g. "$").
|
|
218
|
+
|
|
219
|
+
Variables are strings naming columns of the DataFrame `data`.
|
|
220
|
+
Returns a plotly Figure; call .show() to display from a script.
|
|
221
|
+
"""
|
|
222
|
+
|
|
223
|
+
# ----- validate parameters ------------------------------------
|
|
224
|
+
if form == "sunburst": # R analog: Chart.R line ~100
|
|
225
|
+
form = "pie"
|
|
226
|
+
if form not in _FORMS:
|
|
227
|
+
raise ValueError(f"form must be one of {_FORMS}")
|
|
228
|
+
|
|
229
|
+
# hierarchical routing: treemap/icicle always; pie with by=
|
|
230
|
+
# nests the by rings inside the x wedges as a sunburst
|
|
231
|
+
# R analog: Chart.R hier dispatch
|
|
232
|
+
hier_type = None
|
|
233
|
+
if form in ("treemap", "icicle"):
|
|
234
|
+
hier_type = form
|
|
235
|
+
elif form == "pie" and by is not None:
|
|
236
|
+
hier_type = "sunburst"
|
|
237
|
+
if hier_type is not None and stat == "deviation":
|
|
238
|
+
raise ValueError('stat="deviation" is not meaningful for '
|
|
239
|
+
"hierarchical charts: negative values "
|
|
240
|
+
"cannot form part-of-whole areas")
|
|
241
|
+
|
|
242
|
+
if form == "dot" and by is not None:
|
|
243
|
+
raise ValueError("The by variable is not meaningful for "
|
|
244
|
+
"dot charts. Do a bar chart.")
|
|
245
|
+
if isinstance(y, (list, tuple)) and form != "dot":
|
|
246
|
+
raise ValueError('Multiple y variables are only supported '
|
|
247
|
+
'for form="dot" (paired dot chart)')
|
|
248
|
+
if sort not in _SORTS:
|
|
249
|
+
raise ValueError('sort must be "0" (none), "-" (descending) '
|
|
250
|
+
'or "+" (ascending)')
|
|
251
|
+
if stat is not None and stat not in _STATS:
|
|
252
|
+
raise ValueError(f"stat must be one of {_STATS}")
|
|
253
|
+
if stat_x not in ("count", "proportion"):
|
|
254
|
+
raise ValueError('stat_x must be "count" or "proportion"')
|
|
255
|
+
if labels_position not in (None, "in", "out"):
|
|
256
|
+
raise ValueError('labels_position must be "in" or "out"')
|
|
257
|
+
if axis_fmt not in ("K", ",", ".", ""):
|
|
258
|
+
raise ValueError('axis_fmt must be "K", ",", "." or ""')
|
|
259
|
+
if stack100:
|
|
260
|
+
if form != "bar":
|
|
261
|
+
raise ValueError('stack100 rescales the bars of a bar '
|
|
262
|
+
'chart, so it needs form="bar"')
|
|
263
|
+
if facet is not None:
|
|
264
|
+
raise ValueError(
|
|
265
|
+
"stack100 rescales each bar within its by groups, "
|
|
266
|
+
"and a faceted bar chart (Trellis) has no by "
|
|
267
|
+
"variable. Use by= instead of facet=.")
|
|
268
|
+
if by is None and stat_x == "proportion":
|
|
269
|
+
raise ValueError(
|
|
270
|
+
'stack100 without a by variable is the same as '
|
|
271
|
+
'stat_x="proportion": specify one, not both')
|
|
272
|
+
if scale_y is not None:
|
|
273
|
+
if len(scale_y) != 3:
|
|
274
|
+
raise ValueError(
|
|
275
|
+
"scale_y is (min, max, n_intervals) for the value "
|
|
276
|
+
"axis: three values, giving n_intervals + 1 ticks")
|
|
277
|
+
if float(scale_y[1]) <= float(scale_y[0]):
|
|
278
|
+
raise ValueError("scale_y max must exceed scale_y min")
|
|
279
|
+
if int(scale_y[2]) < 1:
|
|
280
|
+
raise ValueError("scale_y needs at least one interval")
|
|
281
|
+
# R analog: Chart.R break_x <- !horiz && rotate_x == 0
|
|
282
|
+
if break_x is None:
|
|
283
|
+
break_x = not horiz and rotate_x == 0
|
|
284
|
+
# the legend keys the by= levels of a two-variable bar chart,
|
|
285
|
+
# which is where R defines these; say so rather than no-op
|
|
286
|
+
_leg = {"legend_title": legend_title,
|
|
287
|
+
"legend_position": legend_position,
|
|
288
|
+
"legend_labels": legend_labels,
|
|
289
|
+
"legend_horiz": legend_horiz or None,
|
|
290
|
+
"legend_size": legend_size,
|
|
291
|
+
"legend_abbrev": legend_abbrev,
|
|
292
|
+
"legend_adjust": legend_adjust or None}
|
|
293
|
+
_leg_set = [k for k, v in _leg.items() if v is not None]
|
|
294
|
+
if _leg_set:
|
|
295
|
+
if form != "bar":
|
|
296
|
+
raise ValueError(
|
|
297
|
+
f"{', '.join(_leg_set)} style the legend of a "
|
|
298
|
+
'two-variable bar chart, so they need form="bar"')
|
|
299
|
+
if by is None:
|
|
300
|
+
raise ValueError(
|
|
301
|
+
f"{', '.join(_leg_set)} style the legend that a by "
|
|
302
|
+
"variable creates, so they need by=")
|
|
303
|
+
if facet is not None:
|
|
304
|
+
raise ValueError(
|
|
305
|
+
f"{', '.join(_leg_set)} apply to a single-panel bar "
|
|
306
|
+
"chart; a faceted (Trellis) bar chart has no by "
|
|
307
|
+
"variable and no legend")
|
|
308
|
+
|
|
309
|
+
if (n_row is not None or n_col is not None) and facet is None:
|
|
310
|
+
raise ValueError("n_row and n_col lay out facet panels, "
|
|
311
|
+
"so they require a facet variable")
|
|
312
|
+
if add is not None and (form != "bar"
|
|
313
|
+
or facet is not None):
|
|
314
|
+
raise ValueError(
|
|
315
|
+
"add= annotations apply to a single-panel bar "
|
|
316
|
+
"chart in Chart()")
|
|
317
|
+
if data is None:
|
|
318
|
+
raise ValueError(
|
|
319
|
+
"data= is required: a pandas DataFrame containing the "
|
|
320
|
+
"named columns. (lessR's default of a data frame named "
|
|
321
|
+
"d in the global environment has no Python analog.)")
|
|
322
|
+
if stat is not None and y is None:
|
|
323
|
+
raise ValueError(
|
|
324
|
+
"stat requires a numerical y variable to transform, "
|
|
325
|
+
"e.g., y='Salary'")
|
|
326
|
+
|
|
327
|
+
if facet is not None:
|
|
328
|
+
if isinstance(y, (list, tuple)):
|
|
329
|
+
raise ValueError(
|
|
330
|
+
"Multiple y variables (paired dot chart) combined "
|
|
331
|
+
"with facet= is not supported. Use a single y "
|
|
332
|
+
"variable with facet=, or omit facet=.")
|
|
333
|
+
if stat == "deviation":
|
|
334
|
+
raise ValueError("deviation for stat is not "
|
|
335
|
+
"meaningful with a facet variable")
|
|
336
|
+
if form == "bubble" and by is not None:
|
|
337
|
+
raise ValueError(
|
|
338
|
+
"The facet option for bubble charts applies only "
|
|
339
|
+
"to a single categorical variable (no by "
|
|
340
|
+
"variable)")
|
|
341
|
+
if stat_x == "proportion" and y is not None:
|
|
342
|
+
raise ValueError(
|
|
343
|
+
'stat_x="proportion" applies to counts of x, so a '
|
|
344
|
+
"y variable does not apply")
|
|
345
|
+
if form == "bar":
|
|
346
|
+
# Trellis bar chart: panels of counts of the original
|
|
347
|
+
# data, as in R (.bcParamValid / .bar.lattice)
|
|
348
|
+
if sort != "0":
|
|
349
|
+
raise ValueError("Sort not applicable to Trellis "
|
|
350
|
+
"(faceted bar) charts")
|
|
351
|
+
if stat is not None:
|
|
352
|
+
raise ValueError(
|
|
353
|
+
"Only the original data work with Trellis "
|
|
354
|
+
"plots, no data aggregation with parameter "
|
|
355
|
+
"stat. Use by instead of facet.")
|
|
356
|
+
if y is not None:
|
|
357
|
+
raise ValueError(
|
|
358
|
+
"The faceted bar chart displays counts of x, "
|
|
359
|
+
"so a y variable does not apply. Use by "
|
|
360
|
+
"instead of facet.")
|
|
361
|
+
if by is not None:
|
|
362
|
+
raise ValueError(
|
|
363
|
+
"by and facet do not combine for bar charts: "
|
|
364
|
+
"each facet panel displays the counts of x. "
|
|
365
|
+
"Use one or the other.")
|
|
366
|
+
|
|
367
|
+
# ----- resolve variables and filter ---------------------------
|
|
368
|
+
if filter is not None:
|
|
369
|
+
data = data.query(filter)
|
|
370
|
+
|
|
371
|
+
# paired dot chart: x=labels, y=[col1, col2, ...], displayed
|
|
372
|
+
# directly, never aggregated. R analog: Chart.R paired-dot path
|
|
373
|
+
if isinstance(y, (list, tuple)):
|
|
374
|
+
if stat is not None:
|
|
375
|
+
raise ValueError("stat does not apply to a paired dot "
|
|
376
|
+
"chart; the values display directly")
|
|
377
|
+
x_ser = _get_column(data, x, "x")
|
|
378
|
+
y_cols = [_get_column(data, yi, "y") for yi in y]
|
|
379
|
+
sub = pd.concat([x_ser] + y_cols, axis=1).dropna()
|
|
380
|
+
cats = sub[x].astype(str).tolist()
|
|
381
|
+
ydf = sub[list(y)]
|
|
382
|
+
|
|
383
|
+
# sort= orders rows by row mean across all series
|
|
384
|
+
if sort != "0":
|
|
385
|
+
order = (ydf.mean(axis=1)
|
|
386
|
+
.sort_values(ascending=(sort == "+")).index)
|
|
387
|
+
cats = sub.loc[order, x].astype(str).tolist()
|
|
388
|
+
ydf = ydf.loc[order]
|
|
389
|
+
|
|
390
|
+
origin, gridT = _dot_origin_grid(
|
|
391
|
+
ydf.to_numpy(dtype=float).ravel(), origin_in=origin_x)
|
|
392
|
+
if origin is None:
|
|
393
|
+
origin = 0
|
|
394
|
+
if main is None:
|
|
395
|
+
main = build_title(x, y_name=" & ".join(y))
|
|
396
|
+
elif main == "":
|
|
397
|
+
main = None
|
|
398
|
+
return dot_plotly(
|
|
399
|
+
cats, ydf,
|
|
400
|
+
pt_size=pt_size,
|
|
401
|
+
x_lab="" if xlab is None else xlab, y_lab=x,
|
|
402
|
+
digits_d=2 if digits_d is None else digits_d,
|
|
403
|
+
pt_opacity=1 - (get_option("trans_pt_fill", 0.10)
|
|
404
|
+
if transparency is None
|
|
405
|
+
else transparency),
|
|
406
|
+
gridT=gridT, origin_x=origin, main=main,
|
|
407
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
408
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
409
|
+
segments_x=(True if segments_x is None
|
|
410
|
+
else segments_x),
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
x_ser = _get_column(data, x, "x")
|
|
414
|
+
by_ser = _get_column(data, by, "by") if by is not None else None
|
|
415
|
+
y_ser = _get_column(data, y, "y") if y is not None else None
|
|
416
|
+
|
|
417
|
+
# multiple facet variables define the panel grid, not extra
|
|
418
|
+
# levels: flatten to one " / " interaction, as in R
|
|
419
|
+
if facet is None:
|
|
420
|
+
fac_cols, facet_name = None, None
|
|
421
|
+
elif isinstance(facet, (list, tuple)):
|
|
422
|
+
fac_cols = [_get_column(data, f, "facet") for f in facet]
|
|
423
|
+
facet_name = ", ".join(facet)
|
|
424
|
+
elif isinstance(facet, str):
|
|
425
|
+
fac_cols = [_get_column(data, facet, "facet")]
|
|
426
|
+
facet_name = facet
|
|
427
|
+
else:
|
|
428
|
+
# computed facet values: a Series/array aligned with data,
|
|
429
|
+
# the Python analog of R's facet expression
|
|
430
|
+
ser = facet_values(data, facet, "facet")
|
|
431
|
+
fac_cols = [ser]
|
|
432
|
+
facet_name = str(ser.name)
|
|
433
|
+
|
|
434
|
+
# casewise deletion over the variables in the analysis
|
|
435
|
+
used = [s for s in (x_ser, by_ser, y_ser) if s is not None]
|
|
436
|
+
used += fac_cols or []
|
|
437
|
+
keep = ~pd.concat(used, axis=1).isna().any(axis=1)
|
|
438
|
+
x_ser = x_ser[keep]
|
|
439
|
+
if by_ser is not None:
|
|
440
|
+
by_ser = by_ser[keep]
|
|
441
|
+
if y_ser is not None:
|
|
442
|
+
y_ser = y_ser[keep]
|
|
443
|
+
if fac_cols is None:
|
|
444
|
+
facet_ser = None
|
|
445
|
+
elif len(fac_cols) == 1:
|
|
446
|
+
facet_ser = fac_cols[0][keep]
|
|
447
|
+
else:
|
|
448
|
+
facet_ser = fac_cols[0][keep].astype(str)
|
|
449
|
+
for c in fac_cols[1:]:
|
|
450
|
+
facet_ser = facet_ser + " / " + c[keep].astype(str)
|
|
451
|
+
|
|
452
|
+
x_order = _category_order(x_ser)
|
|
453
|
+
by_order = _category_order(by_ser) if by_ser is not None else None
|
|
454
|
+
facet_order = (_category_order(facet_ser)
|
|
455
|
+
if facet_ser is not None else None)
|
|
456
|
+
|
|
457
|
+
# theme=: a sequential palette in the theme's hue over the fill
|
|
458
|
+
# elements (the by groups if present, else the x categories)
|
|
459
|
+
if theme is not None and fill is None:
|
|
460
|
+
n_fill = len(by_order) if by_order is not None \
|
|
461
|
+
else len(x_order)
|
|
462
|
+
fill = _theme_fill(theme, n_fill)
|
|
463
|
+
|
|
464
|
+
# explicit panel-grid layout: n_col wins, else derive from
|
|
465
|
+
# n_row; None keeps each renderer's own default (single-column
|
|
466
|
+
# Trellis bar; up-to-3-column grid for radar/bubble/dot/hier)
|
|
467
|
+
n_col_use = None
|
|
468
|
+
if facet_order is not None and (n_row is not None
|
|
469
|
+
or n_col is not None):
|
|
470
|
+
if n_col is not None:
|
|
471
|
+
n_col_use = max(1, int(n_col))
|
|
472
|
+
else:
|
|
473
|
+
n_col_use = max(1, math.ceil(len(facet_order)
|
|
474
|
+
/ int(n_row)))
|
|
475
|
+
|
|
476
|
+
# facet panels leave less room for value labels (Chart.R:134)
|
|
477
|
+
if facet_ser is not None and labels_size is None:
|
|
478
|
+
labels_size = 0.85
|
|
479
|
+
|
|
480
|
+
# facet= on a plain pie renders the pie grid, one pie per facet
|
|
481
|
+
# level: the facet becomes the grouping, as in R. (Sunburst
|
|
482
|
+
# routing for pie was already decided on by= above.)
|
|
483
|
+
# the grid re-uses by= to group the panels, but the variable is
|
|
484
|
+
# still a facet, so the title reads "across", not "by"
|
|
485
|
+
facet_ttl = None
|
|
486
|
+
if form == "pie" and hier_type is None and facet_ser is not None:
|
|
487
|
+
facet_ttl = facet_name
|
|
488
|
+
by = facet_name
|
|
489
|
+
by_ser = facet_ser
|
|
490
|
+
by_order = facet_order
|
|
491
|
+
facet_ser = None
|
|
492
|
+
facet_name = None
|
|
493
|
+
facet_order = None
|
|
494
|
+
|
|
495
|
+
# radar polygons need >= 3 axes and, with by, >= 2 groups with
|
|
496
|
+
# every cell occupied. R analog: Chart.R radar checks
|
|
497
|
+
if form == "radar":
|
|
498
|
+
if len(x_order) < 3:
|
|
499
|
+
raise ValueError(
|
|
500
|
+
"radar: the categorical variable x must have at "
|
|
501
|
+
"least 3 levels to form a polygon. Found "
|
|
502
|
+
f"{len(x_order)} levels for {x}.")
|
|
503
|
+
if by_ser is not None:
|
|
504
|
+
if len(by_order) < 2:
|
|
505
|
+
raise ValueError(
|
|
506
|
+
"radar: the by variable must have at least 2 "
|
|
507
|
+
"levels to define multiple polygons. Found "
|
|
508
|
+
f"{len(by_order)} levels for {by}.")
|
|
509
|
+
cells = (pd.crosstab(by_ser, x_ser)
|
|
510
|
+
.reindex(index=by_order, columns=x_order,
|
|
511
|
+
fill_value=0))
|
|
512
|
+
n_zero = int((cells == 0).sum().sum())
|
|
513
|
+
if n_zero > 0:
|
|
514
|
+
raise ValueError(
|
|
515
|
+
"radar: one or more cells are empty, with 0 "
|
|
516
|
+
f"entries ({n_zero} found). Radar polygons "
|
|
517
|
+
"assume each group has a value at every axis. "
|
|
518
|
+
"Use a larger sample, reduce the number of "
|
|
519
|
+
"levels, or choose a different chart.")
|
|
520
|
+
|
|
521
|
+
# ----- pre-aggregated? (each x [,by] combination unique) ------
|
|
522
|
+
# R analog: is.agg logic in Chart.R
|
|
523
|
+
n_x = x_ser.nunique()
|
|
524
|
+
if by_ser is None:
|
|
525
|
+
is_agg = n_x >= len(x_ser)
|
|
526
|
+
else:
|
|
527
|
+
is_agg = n_x * by_ser.nunique() >= len(by_ser)
|
|
528
|
+
|
|
529
|
+
# a dot chart of counts needs repeated categories to count;
|
|
530
|
+
# unique-x data needs the values themselves
|
|
531
|
+
if form == "dot" and y_ser is None and is_agg:
|
|
532
|
+
raise ValueError("Need to specify a numerical variable "
|
|
533
|
+
"for y, y='NAME'")
|
|
534
|
+
|
|
535
|
+
if y_ser is not None:
|
|
536
|
+
# y is the numerical variable the bars measure. The second
|
|
537
|
+
# positional argument is y, as in R, so name a second
|
|
538
|
+
# categorical variable explicitly with by=.
|
|
539
|
+
if not is_numeric_dtype(y_ser):
|
|
540
|
+
raise ValueError(
|
|
541
|
+
f"y='{y}' is not a numerical variable, and y is the "
|
|
542
|
+
"variable the chart aggregates. To stratify by a "
|
|
543
|
+
f"second categorical variable, name it: by='{y}'")
|
|
544
|
+
if is_agg and stat is not None:
|
|
545
|
+
raise ValueError(
|
|
546
|
+
"The data are a summary table, so do not specify "
|
|
547
|
+
"stat: the aggregation has already been done")
|
|
548
|
+
if not is_agg and stat is None:
|
|
549
|
+
raise ValueError(
|
|
550
|
+
"The data are not a summary (pivot) table, and you "
|
|
551
|
+
f"have a numerical variable y='{y}', so specify "
|
|
552
|
+
'stat to define the aggregation, e.g., stat="mean"')
|
|
553
|
+
|
|
554
|
+
# accompanying statistics: frequencies, or the stat of y
|
|
555
|
+
if not resolve_quiet(quiet):
|
|
556
|
+
y_out = (y_ser if y_ser is not None
|
|
557
|
+
and stat in STAT_FUN else None)
|
|
558
|
+
if y_ser is None or y_out is not None:
|
|
559
|
+
print("\n".join(chart_stats(
|
|
560
|
+
x_ser, by_ser, y_out, stat, x, by, y,
|
|
561
|
+
2 if digits_d is None else digits_d)))
|
|
562
|
+
|
|
563
|
+
# labels default for aggregated data; for counts leave labels
|
|
564
|
+
# None so bc_plotly shows the value with % lines in hover
|
|
565
|
+
# R analog: Chart.R line ~716
|
|
566
|
+
if labels is None and (stat is not None or
|
|
567
|
+
(is_agg and y_ser is not None)):
|
|
568
|
+
labels = "%" if beside else "input"
|
|
569
|
+
|
|
570
|
+
# ----- hierarchical forms aggregate per nesting level ---------
|
|
571
|
+
# from the raw columns, not from the flat table built below
|
|
572
|
+
if hier_type is not None:
|
|
573
|
+
if digits_d is None:
|
|
574
|
+
digits_d = 0 if y_ser is None else 2
|
|
575
|
+
agg = hier_aggregate(x_ser, by=by_ser, y=y_ser, stat=stat,
|
|
576
|
+
facet=facet_ser,
|
|
577
|
+
x_name=x, by_name=by, y_name=y,
|
|
578
|
+
facet_name=facet_name,
|
|
579
|
+
facet_order=facet_order)
|
|
580
|
+
fill_vec = hier_color_resolve(x_order, fill)
|
|
581
|
+
if main is None:
|
|
582
|
+
main = build_title(x, by_name=by, y_name=y, stat=stat,
|
|
583
|
+
facet_name=facet_name)
|
|
584
|
+
elif main == "":
|
|
585
|
+
main = None
|
|
586
|
+
# Chart resolves the labels default before hier sees it:
|
|
587
|
+
# "%" for counts (match.arg, Chart.R line 233); the shared
|
|
588
|
+
# logic above already switched stat/pre-aggregated data to
|
|
589
|
+
# "input". Parents carry value 0, so "%"/texttemplate modes
|
|
590
|
+
# keep the inner rings meaningful.
|
|
591
|
+
return hier_plotly(
|
|
592
|
+
agg, fill_vec, type=hier_type,
|
|
593
|
+
x_name=x, by_name=by, facet_name=facet_name,
|
|
594
|
+
main=main, border=color, digits_d=digits_d,
|
|
595
|
+
labels="%" if labels is None else labels,
|
|
596
|
+
labels_color=("white" if labels_color is None
|
|
597
|
+
else labels_color),
|
|
598
|
+
labels_size=(0.75 if labels_size is None
|
|
599
|
+
else labels_size),
|
|
600
|
+
n_col=n_col_use,
|
|
601
|
+
)
|
|
602
|
+
|
|
603
|
+
# ----- faceted forms: one panel per facet level ----------------
|
|
604
|
+
# (hier forms handled above; pie facet became the by grouping)
|
|
605
|
+
|
|
606
|
+
if facet_ser is not None and form == "bar":
|
|
607
|
+
# Trellis bar chart: horizontal count bars per panel,
|
|
608
|
+
# the plotly port of R's .bar.lattice rendering
|
|
609
|
+
tbl = (pd.crosstab(facet_ser, x_ser)
|
|
610
|
+
.reindex(index=facet_order, columns=x_order,
|
|
611
|
+
fill_value=0))
|
|
612
|
+
tbl.index.name = facet_name
|
|
613
|
+
tbl.columns.name = x
|
|
614
|
+
return bc_facet_plotly(
|
|
615
|
+
tbl, x_name=x, facet_name=facet_name,
|
|
616
|
+
x_lab=xlab, y_lab=ylab, fill=fill,
|
|
617
|
+
border="off" if color is None else color,
|
|
618
|
+
opacity=(None if transparency is None
|
|
619
|
+
else 1 - transparency),
|
|
620
|
+
proportion=stat_x == "proportion",
|
|
621
|
+
digits_d=(digits_d if digits_d is not None
|
|
622
|
+
else (0 if stat_x == "count" else 2)),
|
|
623
|
+
n_col=n_col_use or 1,
|
|
624
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
625
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
626
|
+
main=main,
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
if facet_ser is not None and form == "radar":
|
|
630
|
+
if stat_x == "proportion" and by_ser is not None:
|
|
631
|
+
raise NotImplementedError(
|
|
632
|
+
'stat_x="proportion" with by= is not yet ported')
|
|
633
|
+
facets = {
|
|
634
|
+
str(lv): _facet_table(
|
|
635
|
+
x_ser[facet_ser == lv],
|
|
636
|
+
None if y_ser is None else y_ser[facet_ser == lv],
|
|
637
|
+
None if by_ser is None else by_ser[facet_ser == lv],
|
|
638
|
+
stat, is_agg, x_order, by_order,
|
|
639
|
+
proportion=stat_x == "proportion")
|
|
640
|
+
for lv in facet_order
|
|
641
|
+
}
|
|
642
|
+
cap = {"mean": "Mean", "sum": "Sum", "median": "Median",
|
|
643
|
+
"min": "Min", "max": "Max", "sd": "SD"}
|
|
644
|
+
if y_ser is None:
|
|
645
|
+
val_label = "Proportion" if stat_x == "proportion" \
|
|
646
|
+
else "Count"
|
|
647
|
+
elif stat is not None:
|
|
648
|
+
val_label = f"{cap[stat]} {y}"
|
|
649
|
+
else:
|
|
650
|
+
val_label = y
|
|
651
|
+
trans_use = (0.4 if transparency is None and by is not None
|
|
652
|
+
else (transparency or 0))
|
|
653
|
+
if digits_d is None:
|
|
654
|
+
digits_d = (2 if (y_ser is not None
|
|
655
|
+
or stat_x == "proportion") else 0)
|
|
656
|
+
if main is None:
|
|
657
|
+
main = build_title(
|
|
658
|
+
x, by_name=by,
|
|
659
|
+
y_name=("Proportion" if stat_x == "proportion"
|
|
660
|
+
else y),
|
|
661
|
+
stat=stat, facet_name=facet_name)
|
|
662
|
+
elif main == "":
|
|
663
|
+
main = None
|
|
664
|
+
return radar_plotly(
|
|
665
|
+
facets=facets, facet_name=facet_name,
|
|
666
|
+
x_name=x, by_name=by, main=main,
|
|
667
|
+
fill=fill, opacity=1 - trans_use,
|
|
668
|
+
digits_d=digits_d, val_label=val_label,
|
|
669
|
+
n_col=n_col_use,
|
|
670
|
+
)
|
|
671
|
+
|
|
672
|
+
if facet_ser is not None and form == "bubble":
|
|
673
|
+
# one 1-D panel per facet level, all on the full category
|
|
674
|
+
# set so the panels stay aligned (by= was rejected above)
|
|
675
|
+
facet_tbls = {
|
|
676
|
+
str(lv): _facet_table(
|
|
677
|
+
x_ser[facet_ser == lv],
|
|
678
|
+
None if y_ser is None else y_ser[facet_ser == lv],
|
|
679
|
+
None, stat, is_agg, x_order, None,
|
|
680
|
+
proportion=stat_x == "proportion")
|
|
681
|
+
for lv in facet_order
|
|
682
|
+
}
|
|
683
|
+
prop = stat_x == "proportion"
|
|
684
|
+
if digits_d is None:
|
|
685
|
+
digits_d = 2 if (y_ser is not None or prop) else 0
|
|
686
|
+
y_name_b = ("Proportion" if prop
|
|
687
|
+
else ("Count" if y is None else y))
|
|
688
|
+
if main is None:
|
|
689
|
+
main = build_title(x, y_name=y_name_b,
|
|
690
|
+
stat=stat, facet_name=facet_name)
|
|
691
|
+
elif main == "":
|
|
692
|
+
main = None
|
|
693
|
+
return bubble_plotly(
|
|
694
|
+
x_name=x, y_name=y_name_b,
|
|
695
|
+
x_lab=x if xlab is None else xlab,
|
|
696
|
+
main=main, fill=fill,
|
|
697
|
+
border="black" if color is None else color,
|
|
698
|
+
opacity=(None if transparency is None
|
|
699
|
+
else 1 - transparency),
|
|
700
|
+
power=power, radius=radius,
|
|
701
|
+
digits_d=digits_d,
|
|
702
|
+
labels="input" if (prop and labels is None) else labels,
|
|
703
|
+
labels_position=labels_position,
|
|
704
|
+
labels_color=labels_color,
|
|
705
|
+
labels_size=labels_size,
|
|
706
|
+
facet_tbls=facet_tbls, facet_name=facet_name,
|
|
707
|
+
n_col=n_col_use,
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
if facet_ser is not None and form == "dot":
|
|
711
|
+
# aggregate and sort per facet panel; a panel shows only
|
|
712
|
+
# its own categories. R analog: Chart.R faceted dot prep
|
|
713
|
+
prop = stat_x == "proportion"
|
|
714
|
+
cats_l, vals_l, fac_l = [], [], []
|
|
715
|
+
for lv in facet_order:
|
|
716
|
+
m = facet_ser == lv
|
|
717
|
+
xs = x_ser[m].astype(str)
|
|
718
|
+
if y_ser is None: # counts per panel
|
|
719
|
+
t = xs.groupby(xs).size()
|
|
720
|
+
if prop:
|
|
721
|
+
tot = t.sum()
|
|
722
|
+
t = t / tot if tot > 0 else t.astype(float)
|
|
723
|
+
elif not is_agg and stat is not None:
|
|
724
|
+
t = STAT_FUN[stat](y_ser[m].groupby(xs))
|
|
725
|
+
else: # pre-aggregated rows
|
|
726
|
+
t = pd.Series(y_ser[m].to_numpy(dtype=float),
|
|
727
|
+
index=xs.values)
|
|
728
|
+
if sort != "0":
|
|
729
|
+
t = t.sort_values(ascending=(sort == "+"))
|
|
730
|
+
cats_l += [str(c) for c in t.index]
|
|
731
|
+
vals_l += [float(v) for v in t.to_numpy()]
|
|
732
|
+
fac_l += [str(lv)] * len(t)
|
|
733
|
+
|
|
734
|
+
val_lab = (y if y is not None
|
|
735
|
+
else ("Proportion of " + x if prop
|
|
736
|
+
else f"Count of {x}"))
|
|
737
|
+
if horiz:
|
|
738
|
+
orientation = "h"
|
|
739
|
+
x_lab_arg = xlab if xlab is not None else val_lab
|
|
740
|
+
y_lab_arg = ylab if ylab is not None else x
|
|
741
|
+
else:
|
|
742
|
+
orientation = "v"
|
|
743
|
+
x_lab_arg = xlab if xlab is not None else x
|
|
744
|
+
y_lab_arg = ylab if ylab is not None else val_lab
|
|
745
|
+
|
|
746
|
+
origin, gridT = _dot_origin_grid(
|
|
747
|
+
vals_l,
|
|
748
|
+
origin_in=origin_x if horiz else origin_y,
|
|
749
|
+
is_counts=True if y_ser is None else None)
|
|
750
|
+
if digits_d is None:
|
|
751
|
+
digits_d = 2 if (y_ser is not None or prop) else 0
|
|
752
|
+
if main is None:
|
|
753
|
+
main = build_title(
|
|
754
|
+
x, y_name="Proportion" if prop else y,
|
|
755
|
+
stat=stat, facet_name=facet_name)
|
|
756
|
+
elif main == "":
|
|
757
|
+
main = None
|
|
758
|
+
return dot_plotly(
|
|
759
|
+
cats_l, vals_l, orientation=orientation,
|
|
760
|
+
fill=fill, border=color, pt_size=pt_size,
|
|
761
|
+
x_lab=x_lab_arg, y_lab=y_lab_arg,
|
|
762
|
+
digits_d=digits_d,
|
|
763
|
+
pt_opacity=1 - (get_option("trans_pt_fill", 0.10)
|
|
764
|
+
if transparency is None
|
|
765
|
+
else transparency),
|
|
766
|
+
gridT=gridT, origin_x=origin, main=main,
|
|
767
|
+
facet=fac_l, facet_name=facet_name, n_col=n_col_use,
|
|
768
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
769
|
+
axis_y_pre=axis_y_pre,
|
|
770
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
771
|
+
segments_x=True if segments_x is None else segments_x,
|
|
772
|
+
segments_y=True if segments_y is None else segments_y,
|
|
773
|
+
)
|
|
774
|
+
|
|
775
|
+
# user-supplied axis label, before the pipeline computes a
|
|
776
|
+
# default into ylab (the bubble matrix labels its y-axis with
|
|
777
|
+
# the by variable, not the value label)
|
|
778
|
+
ylab_user = ylab
|
|
779
|
+
digits_d_user = digits_d
|
|
780
|
+
|
|
781
|
+
# ----- build the table the renderer receives ------------------
|
|
782
|
+
if y_ser is None: # counts of x
|
|
783
|
+
if stat_x == "proportion" and by_ser is not None:
|
|
784
|
+
raise NotImplementedError(
|
|
785
|
+
'stat_x="proportion" with by= is not yet ported')
|
|
786
|
+
|
|
787
|
+
if by_ser is None:
|
|
788
|
+
tbl = (x_ser.groupby(x_ser).size()
|
|
789
|
+
.reindex(x_order, fill_value=0))
|
|
790
|
+
if stat_x == "proportion":
|
|
791
|
+
tbl = tbl / tbl.sum()
|
|
792
|
+
y_name = "Proportion"
|
|
793
|
+
else:
|
|
794
|
+
y_name = "Count"
|
|
795
|
+
tbl.index.name = x
|
|
796
|
+
tbl.name = y_name
|
|
797
|
+
if ylab is None:
|
|
798
|
+
ylab = f"{y_name} of {x}"
|
|
799
|
+
if digits_d is None:
|
|
800
|
+
digits_d = 0 if stat_x == "count" else 2
|
|
801
|
+
else:
|
|
802
|
+
tbl = pd.crosstab(by_ser, x_ser)
|
|
803
|
+
tbl = tbl.reindex(index=by_order, columns=x_order,
|
|
804
|
+
fill_value=0)
|
|
805
|
+
y_name = "Count"
|
|
806
|
+
if ylab is None:
|
|
807
|
+
ylab = f"Count of {x}"
|
|
808
|
+
if digits_d is None:
|
|
809
|
+
digits_d = 0
|
|
810
|
+
|
|
811
|
+
elif stat == "deviation": # R analog: .agg()
|
|
812
|
+
if by_ser is not None:
|
|
813
|
+
raise ValueError("deviation for stat is not meaningful "
|
|
814
|
+
"with a by variable")
|
|
815
|
+
means = y_ser.groupby(x_ser).mean().reindex(x_order)
|
|
816
|
+
# deviation from the unweighted mean of the group means,
|
|
817
|
+
# so each group counts equally regardless of its n
|
|
818
|
+
tbl = means - means.mean()
|
|
819
|
+
y_name = y
|
|
820
|
+
if ylab is None:
|
|
821
|
+
ylab = f"{STAT_LBL[stat]} of {y}"
|
|
822
|
+
if digits_d is None:
|
|
823
|
+
digits_d = 2
|
|
824
|
+
|
|
825
|
+
else: # y with stat (or agg)
|
|
826
|
+
fun = STAT_FUN[stat] if stat is not None else None
|
|
827
|
+
if by_ser is None:
|
|
828
|
+
if is_agg and stat is None:
|
|
829
|
+
tbl = pd.Series(y_ser.values, index=x_ser.values)
|
|
830
|
+
tbl = tbl.reindex(x_order)
|
|
831
|
+
else:
|
|
832
|
+
tbl = fun(y_ser.groupby(x_ser)).reindex(x_order)
|
|
833
|
+
else:
|
|
834
|
+
df = pd.DataFrame({"by": by_ser, "x": x_ser,
|
|
835
|
+
"y": y_ser})
|
|
836
|
+
if is_agg and stat is None:
|
|
837
|
+
tbl = df.pivot(index="by", columns="x", values="y")
|
|
838
|
+
else:
|
|
839
|
+
tbl = fun(df.groupby(["by", "x"])["y"]).unstack("x")
|
|
840
|
+
tbl = tbl.reindex(index=by_order, columns=x_order)
|
|
841
|
+
if not np.isfinite(tbl.to_numpy(dtype=float)).all():
|
|
842
|
+
raise ValueError(
|
|
843
|
+
"The summary table of the transformed data has "
|
|
844
|
+
"missing or non-finite values, likely because some "
|
|
845
|
+
"cells have too few (or no) data values to compute "
|
|
846
|
+
"the specified statistic")
|
|
847
|
+
y_name = y
|
|
848
|
+
if ylab is None:
|
|
849
|
+
# pre-aggregated data label with the variable name
|
|
850
|
+
# itself, as in R (Chart.R dot path, x.lab.dot)
|
|
851
|
+
ylab = (f"{STAT_LBL[stat]} of {y}"
|
|
852
|
+
if stat is not None else y)
|
|
853
|
+
if digits_d is None:
|
|
854
|
+
digits_d = 2
|
|
855
|
+
|
|
856
|
+
# ----- stack100: rescale each bar to sum to 1.0 ----------------
|
|
857
|
+
# R analog: bc.main.R x <- prop.table(x, 2), which normalizes
|
|
858
|
+
# within each COLUMN of the by-by-x table, i.e. within each bar.
|
|
859
|
+
# The counts are kept for the value labels, which R displays
|
|
860
|
+
# instead of the proportions when labels="input".
|
|
861
|
+
counts_tbl = None
|
|
862
|
+
if stack100:
|
|
863
|
+
counts_tbl = tbl.copy()
|
|
864
|
+
if isinstance(tbl, pd.DataFrame):
|
|
865
|
+
totals = tbl.sum(axis=0)
|
|
866
|
+
if (totals == 0).any():
|
|
867
|
+
zero = list(totals.index[totals == 0])
|
|
868
|
+
raise ValueError(
|
|
869
|
+
"stack100 divides each bar by its total, and "
|
|
870
|
+
f"these categories of {x} have no data: {zero}")
|
|
871
|
+
tbl = tbl.div(totals, axis=1)
|
|
872
|
+
else:
|
|
873
|
+
total = tbl.sum()
|
|
874
|
+
if total == 0:
|
|
875
|
+
raise ValueError("stack100 divides each bar by its "
|
|
876
|
+
"total, and all counts are zero")
|
|
877
|
+
tbl = tbl / total
|
|
878
|
+
y_name = "Proportion"
|
|
879
|
+
if ylab_user is None:
|
|
880
|
+
if isinstance(counts_tbl, pd.Series):
|
|
881
|
+
ylab = f"Proportion of {x}"
|
|
882
|
+
elif beside:
|
|
883
|
+
ylab = "Percentage"
|
|
884
|
+
else:
|
|
885
|
+
ylab = f"Cell % within {x} by {by}"
|
|
886
|
+
if digits_d_user is None:
|
|
887
|
+
digits_d = 2
|
|
888
|
+
|
|
889
|
+
if form != "radar": # a radar's axis order stays fixed
|
|
890
|
+
tbl = (_sort_1d(tbl, sort) if isinstance(tbl, pd.Series)
|
|
891
|
+
else _sort_2d(tbl, sort))
|
|
892
|
+
if counts_tbl is not None: # keep the counts aligned
|
|
893
|
+
counts_tbl = (counts_tbl.reindex(tbl.index)
|
|
894
|
+
if isinstance(tbl, pd.Series)
|
|
895
|
+
else counts_tbl.reindex(index=tbl.index,
|
|
896
|
+
columns=tbl.columns))
|
|
897
|
+
|
|
898
|
+
# labels_decimals passes straight through: each renderer applies
|
|
899
|
+
# its own per-mode default when None, which is what R's plotly
|
|
900
|
+
# path does. R's BASE path instead defaults to 0 for an
|
|
901
|
+
# aggregated y (bc.main.R), so the two differ there, as in R.
|
|
902
|
+
|
|
903
|
+
# ----- render --------------------------------------------------
|
|
904
|
+
opacity = None if transparency is None else 1 - transparency
|
|
905
|
+
|
|
906
|
+
if form == "bubble":
|
|
907
|
+
if main is None:
|
|
908
|
+
main = build_title(x, by_name=by, y_name=y_name,
|
|
909
|
+
stat=stat)
|
|
910
|
+
elif main == "":
|
|
911
|
+
main = None
|
|
912
|
+
return bubble_plotly(
|
|
913
|
+
tbl,
|
|
914
|
+
x_name=x, y_name=y_name, by_name=by,
|
|
915
|
+
x_lab=x if xlab is None else xlab,
|
|
916
|
+
y_lab=ylab_user, # None -> by name (2-D only)
|
|
917
|
+
main=main,
|
|
918
|
+
fill=fill,
|
|
919
|
+
border="black" if color is None else color,
|
|
920
|
+
opacity=opacity,
|
|
921
|
+
power=power, radius=radius,
|
|
922
|
+
digits_d=digits_d,
|
|
923
|
+
labels=labels, labels_position=labels_position,
|
|
924
|
+
labels_color=labels_color,
|
|
925
|
+
labels_decimals=labels_decimals,
|
|
926
|
+
labels_size=0.90 if labels_size is None
|
|
927
|
+
else labels_size,
|
|
928
|
+
)
|
|
929
|
+
|
|
930
|
+
if form == "radar":
|
|
931
|
+
# hover value label, R analog: .radar_aggregate() y.label
|
|
932
|
+
# ("Count", "Mean Salary", or the bare variable name)
|
|
933
|
+
cap = {"mean": "Mean", "sum": "Sum", "median": "Median",
|
|
934
|
+
"min": "Min", "max": "Max", "sd": "SD",
|
|
935
|
+
"deviation": "Deviation"}
|
|
936
|
+
if y_ser is None:
|
|
937
|
+
val_label = y_name # "Count"/"Proportion"
|
|
938
|
+
elif stat is not None:
|
|
939
|
+
val_label = f"{cap[stat]} {y}"
|
|
940
|
+
else:
|
|
941
|
+
val_label = y
|
|
942
|
+
# groups overlap, so default to translucent fills with by=
|
|
943
|
+
trans_use = (0.4 if transparency is None and by is not None
|
|
944
|
+
else (transparency or 0))
|
|
945
|
+
if main is None:
|
|
946
|
+
main = build_title(x, by_name=by, y_name=y_name,
|
|
947
|
+
stat=stat)
|
|
948
|
+
elif main == "":
|
|
949
|
+
main = None
|
|
950
|
+
return radar_plotly(
|
|
951
|
+
tbl,
|
|
952
|
+
x_name=x, by_name=by, main=main,
|
|
953
|
+
fill=fill, opacity=1 - trans_use,
|
|
954
|
+
digits_d=digits_d, val_label=val_label,
|
|
955
|
+
)
|
|
956
|
+
|
|
957
|
+
if form == "dot":
|
|
958
|
+
# tbl is always a Series here (by= was rejected above)
|
|
959
|
+
cats = [str(c) for c in tbl.index]
|
|
960
|
+
vals = tbl.to_numpy(dtype=float)
|
|
961
|
+
origin, gridT = _dot_origin_grid(
|
|
962
|
+
vals,
|
|
963
|
+
origin_in=origin_x if horiz else origin_y,
|
|
964
|
+
is_counts=True if y_ser is None else None)
|
|
965
|
+
cat_lab = x if xlab is None else xlab
|
|
966
|
+
val_lab = ylab # set with tbl above
|
|
967
|
+
if main is None:
|
|
968
|
+
main = build_title(x, y_name=y_name, stat=stat)
|
|
969
|
+
elif main == "":
|
|
970
|
+
main = None
|
|
971
|
+
return dot_plotly(
|
|
972
|
+
cats, vals,
|
|
973
|
+
orientation="h" if horiz else "v",
|
|
974
|
+
fill=fill, border=color,
|
|
975
|
+
pt_size=pt_size,
|
|
976
|
+
x_lab=val_lab if horiz else cat_lab,
|
|
977
|
+
y_lab=cat_lab if horiz else val_lab,
|
|
978
|
+
digits_d=digits_d,
|
|
979
|
+
pt_opacity=1 - (get_option("trans_pt_fill", 0.10)
|
|
980
|
+
if transparency is None
|
|
981
|
+
else transparency),
|
|
982
|
+
gridT=gridT, origin_x=origin, main=main,
|
|
983
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
984
|
+
axis_y_pre=axis_y_pre,
|
|
985
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
986
|
+
segments_x=True if segments_x is None else segments_x,
|
|
987
|
+
segments_y=True if segments_y is None else segments_y,
|
|
988
|
+
)
|
|
989
|
+
|
|
990
|
+
if form == "pie":
|
|
991
|
+
# R analog: .build_chart_title() — auto-title unless the
|
|
992
|
+
# caller supplies main; main="" suppresses the title
|
|
993
|
+
if main is None:
|
|
994
|
+
main = build_title(
|
|
995
|
+
x, by_name=None if facet_ttl else by,
|
|
996
|
+
facet_name=facet_ttl, y_name=y_name, stat=stat)
|
|
997
|
+
elif main == "":
|
|
998
|
+
main = None
|
|
999
|
+
return pie_plotly(
|
|
1000
|
+
tbl,
|
|
1001
|
+
x_name=x, y_name=y_name, by_name=by, main=main,
|
|
1002
|
+
fill=fill,
|
|
1003
|
+
border="transparent" if color is None else color,
|
|
1004
|
+
opacity=1.0 if opacity is None else opacity,
|
|
1005
|
+
hole=hole,
|
|
1006
|
+
labels=labels, labels_position=labels_position,
|
|
1007
|
+
labels_color=labels_color,
|
|
1008
|
+
labels_decimals=labels_decimals,
|
|
1009
|
+
labels_size=1.0 if labels_size is None else labels_size,
|
|
1010
|
+
digits_d=digits_d,
|
|
1011
|
+
)
|
|
1012
|
+
|
|
1013
|
+
# value-keyed bar fills: a two-color split (fill_split) or an
|
|
1014
|
+
# HCL gradient by distance from the split (fill_scaled). 1-D
|
|
1015
|
+
# bars only (a Series), as in R.
|
|
1016
|
+
if fill_scaled and by is not None:
|
|
1017
|
+
raise ValueError("fill_scaled applies only without a by "
|
|
1018
|
+
"variable")
|
|
1019
|
+
if (fill_split is not None or fill_scaled) \
|
|
1020
|
+
and isinstance(tbl, pd.Series):
|
|
1021
|
+
from .getColors import bar_fill_colors
|
|
1022
|
+
fill = bar_fill_colors(tbl.to_numpy(dtype=float), fill,
|
|
1023
|
+
fill_split, fill_scaled, fill_chroma)
|
|
1024
|
+
|
|
1025
|
+
fig = bc_plotly(
|
|
1026
|
+
tbl,
|
|
1027
|
+
x_name=x, y_name=y_name, by_name=by,
|
|
1028
|
+
x_lab=xlab if xlab is not None else x,
|
|
1029
|
+
y_lab=ylab,
|
|
1030
|
+
fill=fill,
|
|
1031
|
+
border="off" if color is None else color,
|
|
1032
|
+
opacity=opacity,
|
|
1033
|
+
beside=beside, horiz=horiz,
|
|
1034
|
+
counts=counts_tbl,
|
|
1035
|
+
gap=gap, scale_y=scale_y, break_x=break_x,
|
|
1036
|
+
digits_d=digits_d, main=main,
|
|
1037
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
1038
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
1039
|
+
axis_y_pre=axis_y_pre,
|
|
1040
|
+
labels=labels, labels_position=labels_position,
|
|
1041
|
+
labels_size=0.90 if labels_size is None else labels_size,
|
|
1042
|
+
labels_color=labels_color,
|
|
1043
|
+
labels_decimals=labels_decimals,
|
|
1044
|
+
legend_title=legend_title, legend_position=legend_position,
|
|
1045
|
+
legend_labels=legend_labels, legend_horiz=legend_horiz,
|
|
1046
|
+
legend_size=legend_size, legend_abbrev=legend_abbrev,
|
|
1047
|
+
legend_adjust=legend_adjust,
|
|
1048
|
+
)
|
|
1049
|
+
if add is not None:
|
|
1050
|
+
plt_add(fig, add, x1=x1, x2=x2, y1=y1, y2=y2)
|
|
1051
|
+
return fig
|
|
1052
|
+
|
|
1053
|
+
|
|
1054
|
+
# font_size= scales all text of the returned figure
|
|
1055
|
+
Chart = font_scaled(Chart)
|