lessPython 0.1.1__py3-none-any.whl → 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/Chart.py +754 -118
- lessPy/Correlation.py +12 -1
- lessPy/Flows.py +5 -1
- lessPy/X.py +1 -1
- lessPy/XY.py +95 -95
- lessPy/bc_items_plotly.py +195 -0
- lessPy/bc_plotly.py +97 -33
- lessPy/dn_plotly.py +1 -1
- lessPy/dot_plotly.py +112 -41
- lessPy/freq_poly_plotly.py +1 -1
- lessPy/getColors.py +6 -1
- lessPy/hier_plotly.py +50 -7
- lessPy/hs_plotly.py +3 -3
- lessPy/pie_plotly.py +17 -7
- lessPy/pivot.py +6 -3
- lessPy/plotly_utils.py +69 -15
- lessPy/plt_plotly.py +15 -3
- lessPy/plt_time.py +1 -1
- lessPy/profile_plotly.py +121 -0
- lessPy/stats_out.py +382 -27
- lessPy/utils.py +54 -14
- lessPy/vbs_plotly.py +14 -10
- {lesspython-0.1.1.dist-info → lesspython-0.2.0.dist-info}/METADATA +4 -2
- {lesspython-0.1.1.dist-info → lesspython-0.2.0.dist-info}/RECORD +27 -25
- {lesspython-0.1.1.dist-info → lesspython-0.2.0.dist-info}/WHEEL +1 -1
- {lesspython-0.1.1.dist-info → lesspython-0.2.0.dist-info}/licenses/LICENSE +0 -0
- {lesspython-0.1.1.dist-info → lesspython-0.2.0.dist-info}/top_level.txt +0 -0
lessPy/Chart.py
CHANGED
|
@@ -28,6 +28,9 @@ import pandas as pd
|
|
|
28
28
|
from pandas.api.types import is_numeric_dtype
|
|
29
29
|
|
|
30
30
|
from .bc_plotly import bc_facet_plotly, bc_plotly
|
|
31
|
+
from .bc_items_plotly import (
|
|
32
|
+
bc_items_plotly, items_fill, items_stats_lines, items_table,
|
|
33
|
+
)
|
|
31
34
|
from .plt_add import plt_add
|
|
32
35
|
from .bubble_plotly import bubble_plotly
|
|
33
36
|
from .dot_plotly import dot_plotly
|
|
@@ -35,6 +38,8 @@ from .hier_plotly import (
|
|
|
35
38
|
hier_aggregate, hier_color_resolve, hier_plotly,
|
|
36
39
|
)
|
|
37
40
|
from .pie_plotly import pie_plotly
|
|
41
|
+
from .plt_plotly import plt_plotly
|
|
42
|
+
from .profile_plotly import profile_facet_plotly
|
|
38
43
|
from .radar_plotly import radar_plotly
|
|
39
44
|
from .plotly_utils import build_title, font_scaled
|
|
40
45
|
from .stats_out import chart_stats, resolve_quiet
|
|
@@ -106,7 +111,8 @@ def _facet_table(x_s, y_s, by_s, stat, is_agg, x_order, by_order,
|
|
|
106
111
|
return (t.reindex(index=by_order, columns=x_order)
|
|
107
112
|
.fillna(0))
|
|
108
113
|
|
|
109
|
-
_FORMS = ("bar", "radar", "bubble", "dot", "pie", "icicle", "treemap"
|
|
114
|
+
_FORMS = ("bar", "radar", "bubble", "dot", "pie", "icicle", "treemap",
|
|
115
|
+
"profile")
|
|
110
116
|
_STATS = ("mean", "sum", "sd", "deviation", "min", "median", "max")
|
|
111
117
|
_SORTS = ("0", "-", "+")
|
|
112
118
|
|
|
@@ -161,21 +167,236 @@ def _dot_origin_grid(vals, origin_in=None, is_counts=None):
|
|
|
161
167
|
return origin, gridT
|
|
162
168
|
|
|
163
169
|
|
|
170
|
+
def _chart_items(data, items, y, by, facet, form, stat, sort,
|
|
171
|
+
sort_miss, stack100, horiz, fill, theme, labels,
|
|
172
|
+
labels_size, gap, labels_cut, xlab, ylab, main,
|
|
173
|
+
legend_title,
|
|
174
|
+
n_col, n_row, axis_fmt, axis_x_pre, rotate_x,
|
|
175
|
+
rotate_y, quiet):
|
|
176
|
+
"""The multi-item stacked chart and its faceted form. R analog:
|
|
177
|
+
the multiple-x branch of Chart.R, .bc.main() multi, and
|
|
178
|
+
.bar.stackedLattice()"""
|
|
179
|
+
if by is not None:
|
|
180
|
+
raise ValueError(
|
|
181
|
+
"by is not available for multiple x variables: the "
|
|
182
|
+
"items claim position and the responses color, so a by "
|
|
183
|
+
"variable has no encoding left. Use facet, which draws "
|
|
184
|
+
"each group in a panel of its own.")
|
|
185
|
+
if facet is not None and form != "bar":
|
|
186
|
+
raise ValueError(
|
|
187
|
+
'facet with multiple x variables requires form="bar"')
|
|
188
|
+
if form == "bubble":
|
|
189
|
+
raise NotImplementedError(
|
|
190
|
+
"the bubble plot frequency matrix of multiple x "
|
|
191
|
+
'variables is not yet ported; use form="bar"')
|
|
192
|
+
if form != "bar":
|
|
193
|
+
raise ValueError(
|
|
194
|
+
'Multiple x variables only available for form="bar"')
|
|
195
|
+
if y is not None or stat is not None:
|
|
196
|
+
raise ValueError(
|
|
197
|
+
"A multi-item chart counts the responses to each item, "
|
|
198
|
+
"so y and stat do not apply")
|
|
199
|
+
if not horiz:
|
|
200
|
+
raise NotImplementedError(
|
|
201
|
+
"the vertical multi-item chart is not yet ported")
|
|
202
|
+
cols = [_get_column(data, it, "x") for it in items]
|
|
203
|
+
frq, wm, numeric = items_table(cols, items)
|
|
204
|
+
fill_use = (fill if isinstance(fill, (list, tuple))
|
|
205
|
+
else [fill] * len(frq.index) if fill is not None
|
|
206
|
+
else items_fill(theme, len(frq.index)))
|
|
207
|
+
leg = "Responses" if legend_title is None else legend_title
|
|
208
|
+
show = not resolve_quiet(quiet)
|
|
209
|
+
|
|
210
|
+
if facet is None:
|
|
211
|
+
# ordered by mean response, the largest at the top: the bars
|
|
212
|
+
# build upward from the bottom, so ascending unless sort
|
|
213
|
+
# says otherwise
|
|
214
|
+
if sort_miss:
|
|
215
|
+
sort = "+"
|
|
216
|
+
order = list(items)
|
|
217
|
+
if sort != "0":
|
|
218
|
+
order = list(wm.sort_values(ascending=(sort == "+"),
|
|
219
|
+
kind="stable").index)
|
|
220
|
+
if show:
|
|
221
|
+
# the table lists the items top of the chart first
|
|
222
|
+
print("\n".join(_items_print(
|
|
223
|
+
frq[order[::-1]], wm[order[::-1]], numeric)))
|
|
224
|
+
return bc_items_plotly(
|
|
225
|
+
frq, order, fill_use,
|
|
226
|
+
labels="%" if labels is None else labels,
|
|
227
|
+
labels_size=labels_size, stack100=stack100,
|
|
228
|
+
labels_cut=0.04 if labels_cut is None else labels_cut,
|
|
229
|
+
x_lab="" if xlab is None else xlab,
|
|
230
|
+
y_lab="" if ylab is None else ylab,
|
|
231
|
+
legend_title=leg, main=main or None,
|
|
232
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
233
|
+
rotate_x=rotate_x, rotate_y=rotate_y)
|
|
234
|
+
|
|
235
|
+
# faceted: a panel is a fraction of the width of the single
|
|
236
|
+
# chart, too narrow for segment labels, and the paneled chart
|
|
237
|
+
# keeps the order given so the panels read alike
|
|
238
|
+
if labels is not None and labels != "off":
|
|
239
|
+
raise ValueError(
|
|
240
|
+
"labels are not drawn on a faceted multi-item chart. Each "
|
|
241
|
+
"panel is a fraction of the width of the single chart, so "
|
|
242
|
+
"its segments are too narrow for a legible label. The "
|
|
243
|
+
"counts for each panel are reported at the console.")
|
|
244
|
+
if gap is not None:
|
|
245
|
+
raise ValueError(
|
|
246
|
+
"gap is not available for a faceted multi-item chart")
|
|
247
|
+
if not sort_miss and sort != "0":
|
|
248
|
+
raise ValueError(
|
|
249
|
+
"sort is not available for a faceted multi-item chart: "
|
|
250
|
+
"the panels keep the items in the order given, so they "
|
|
251
|
+
"read alike")
|
|
252
|
+
if isinstance(facet, (list, tuple)):
|
|
253
|
+
fcols = [_get_column(data, f, "facet") for f in facet]
|
|
254
|
+
fac = fcols[0].astype(str)
|
|
255
|
+
for c in fcols[1:]:
|
|
256
|
+
fac = fac + " / " + c.astype(str)
|
|
257
|
+
facet_name = ", ".join(facet)
|
|
258
|
+
else:
|
|
259
|
+
fac = _get_column(data, facet, "facet")
|
|
260
|
+
facet_name = facet
|
|
261
|
+
keep = fac.notna()
|
|
262
|
+
facet_order = _category_order(fac[keep])
|
|
263
|
+
resp = list(frq.index)
|
|
264
|
+
tbls = {}
|
|
265
|
+
for lv in facet_order:
|
|
266
|
+
m = keep & (fac == lv)
|
|
267
|
+
f_lv, w_lv, _ = items_table([c[m] for c in cols], items) \
|
|
268
|
+
if all(c[m].notna().any() for c in cols) else (
|
|
269
|
+
pd.DataFrame(0, index=resp, columns=items), None, None)
|
|
270
|
+
f_lv = f_lv.reindex(index=resp, fill_value=0)
|
|
271
|
+
if show:
|
|
272
|
+
if w_lv is None:
|
|
273
|
+
w_lv = pd.Series(np.nan, index=items)
|
|
274
|
+
print("\n".join(_items_print(
|
|
275
|
+
f_lv, w_lv, numeric,
|
|
276
|
+
f"Frequencies of Responses by Variable, "
|
|
277
|
+
f"{facet_name}: {lv}")))
|
|
278
|
+
t = f_lv.astype(float)
|
|
279
|
+
if stack100:
|
|
280
|
+
tot = t.sum(axis=0)
|
|
281
|
+
t = t.div(tot.where(tot != 0), axis=1).fillna(0)
|
|
282
|
+
tbls[lv] = t[items[::-1]] # first item at the top
|
|
283
|
+
tbl = pd.DataFrame([tbls[lv].sum(axis=0) for lv in facet_order],
|
|
284
|
+
index=facet_order)
|
|
285
|
+
tbl.index.name = facet_name
|
|
286
|
+
n_col_use = (max(1, int(n_col)) if n_col is not None
|
|
287
|
+
else max(1, math.ceil(len(facet_order) / int(n_row)))
|
|
288
|
+
if n_row is not None else 1)
|
|
289
|
+
return bc_facet_plotly(
|
|
290
|
+
tbl, x_name="", facet_name=facet_name,
|
|
291
|
+
x_lab="" if xlab is None else xlab,
|
|
292
|
+
y_lab="" if ylab is None else ylab,
|
|
293
|
+
fill=fill_use, border="off",
|
|
294
|
+
digits_d=2 if stack100 else 0, n_col=n_col_use,
|
|
295
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
296
|
+
rotate_x=rotate_x, rotate_y=rotate_y, main=main,
|
|
297
|
+
by_tbls=tbls, by_name=leg, beside=False,
|
|
298
|
+
val_name=("Proportion" if stack100 else "Count"),
|
|
299
|
+
legend_title=leg)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _items_print(frq, wm, numeric, title=None):
|
|
303
|
+
return items_stats_lines(
|
|
304
|
+
frq, wm, title or "Frequencies of Responses by Variable",
|
|
305
|
+
numeric)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _profile_fill(theme):
|
|
309
|
+
"""One hue per series from the theme, as R's profile does:
|
|
310
|
+
.color_range(.get_fill(theme), n), the qualitative hues for the
|
|
311
|
+
default theme."""
|
|
312
|
+
from .plotly_utils import BASE_COLORS
|
|
313
|
+
if theme is None or theme == "colors":
|
|
314
|
+
return list(BASE_COLORS)
|
|
315
|
+
return _theme_fill(theme, 8)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _max_dd(vals):
|
|
319
|
+
"""Most decimal digits among the first 200 values as R format()
|
|
320
|
+
writes them, 7 significant digits. R analog: .max.dd()"""
|
|
321
|
+
mx = 0
|
|
322
|
+
for v in list(vals)[:200]:
|
|
323
|
+
if v is None or not np.isfinite(v):
|
|
324
|
+
continue
|
|
325
|
+
txt = f"{float(v):.7g}"
|
|
326
|
+
if "." in txt and "e" not in txt:
|
|
327
|
+
mx = max(mx, len(txt) - txt.index(".") - 1)
|
|
328
|
+
return mx
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _dot_diff_lines(cats, ydf, x_name):
|
|
332
|
+
"""The difference a paired dot plot presents, listed in the
|
|
333
|
+
order drawn, top of the plot first. R analog: the dif.pair
|
|
334
|
+
listing of Chart.R's paired dot path."""
|
|
335
|
+
difs = (ydf.iloc[:, 1] - ydf.iloc[:, 0]).to_numpy(dtype=float)
|
|
336
|
+
dd = min(_max_dd(np.r_[ydf.iloc[:, 0], ydf.iloc[:, 1]]) + 1, 7)
|
|
337
|
+
ny = len(difs)
|
|
338
|
+
mx_i = len(str(ny))
|
|
339
|
+
mx_d = max(len(f"{v:.{dd}f}") for v in difs)
|
|
340
|
+
mx_f = max([5] + [len(c) for c in cats])
|
|
341
|
+
lines = ["", f"{ydf.columns[1]} - {ydf.columns[0]}",
|
|
342
|
+
f"{'n'.rjust(mx_i)} {' diff'.rjust(mx_d)} {x_name}",
|
|
343
|
+
"-" * (mx_i + mx_d + mx_f + 2)]
|
|
344
|
+
shown = (range(1, ny + 1) if ny <= 20
|
|
345
|
+
else list(range(1, 11)) + list(range(ny - 9, ny + 1)))
|
|
346
|
+
for i in range(1, ny + 1):
|
|
347
|
+
k = ny - i
|
|
348
|
+
if i in shown:
|
|
349
|
+
lines.append(f"{str(i).rjust(mx_i)} "
|
|
350
|
+
f"{f'{difs[k]:.{dd}f}'.rjust(mx_d)} {cats[k]}")
|
|
351
|
+
return lines + [""]
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _dot_paired(cats, ydf, sort, sort_miss, origin_x, show_diff,
|
|
355
|
+
x_name, **render):
|
|
356
|
+
"""Order a multi-series dot plot and draw it. Two series are a
|
|
357
|
+
pair, read for the gap between them, so by default they are
|
|
358
|
+
ordered by that difference, ascending so the largest positive
|
|
359
|
+
difference is drawn at the top, and the value axis begins at
|
|
360
|
+
zero. More series are ordered by their row mean when sort is
|
|
361
|
+
given. R analog: Chart.R paired dot path."""
|
|
362
|
+
dif_pair = ydf.shape[1] == 2
|
|
363
|
+
if dif_pair and sort_miss:
|
|
364
|
+
sort = "+"
|
|
365
|
+
if sort != "0":
|
|
366
|
+
key = (ydf.iloc[:, 1] - ydf.iloc[:, 0] if dif_pair
|
|
367
|
+
else ydf.mean(axis=1))
|
|
368
|
+
order = np.argsort(key.to_numpy(dtype=float)
|
|
369
|
+
* (-1 if sort == "-" else 1),
|
|
370
|
+
kind="stable")
|
|
371
|
+
cats = [cats[i] for i in order]
|
|
372
|
+
ydf = ydf.iloc[order]
|
|
373
|
+
if dif_pair and show_diff:
|
|
374
|
+
print("\n".join(_dot_diff_lines(cats, ydf, x_name)))
|
|
375
|
+
origin, gridT = _dot_origin_grid(
|
|
376
|
+
ydf.to_numpy(dtype=float).ravel(),
|
|
377
|
+
origin_in=0 if (dif_pair and origin_x is None) else origin_x)
|
|
378
|
+
if origin is None:
|
|
379
|
+
origin = 0
|
|
380
|
+
return dot_plotly(cats, ydf, gridT=gridT, origin_x=origin,
|
|
381
|
+
**render)
|
|
382
|
+
|
|
383
|
+
|
|
164
384
|
def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
165
385
|
form="bar",
|
|
166
386
|
n_row=None, n_col=None,
|
|
167
|
-
hole=0.
|
|
387
|
+
hole=0.62,
|
|
168
388
|
radius=0.50, power=0.5,
|
|
169
389
|
pt_size=1, origin_x=None, origin_y=None,
|
|
170
390
|
segments_x=None, segments_y=None,
|
|
171
391
|
stat=None, stat_x="count",
|
|
172
|
-
horiz=
|
|
392
|
+
horiz=None, sort=None, beside=False, stack100=False,
|
|
173
393
|
gap=None, scale_y=None, break_x=None,
|
|
174
394
|
fill=None, color=None, transparency=None,
|
|
175
395
|
fill_split=None, fill_scaled=False, fill_chroma=75,
|
|
176
396
|
theme=None,
|
|
177
397
|
labels=None, labels_position=None,
|
|
178
398
|
labels_color=None, labels_size=None, labels_decimals=None,
|
|
399
|
+
labels_cut=None, segments=None,
|
|
179
400
|
legend_title=None, legend_position=None,
|
|
180
401
|
legend_labels=None, legend_horiz=False,
|
|
181
402
|
legend_size=None, legend_abbrev=None, legend_adjust=0,
|
|
@@ -216,6 +437,20 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
216
437
|
"K": "60K" for thousands, commas past 9999); axis_x_pre /
|
|
217
438
|
axis_y_pre prepend a per-axis prefix (e.g. "$").
|
|
218
439
|
|
|
440
|
+
form="dot" is Cleveland's dot plot: categories on the vertical
|
|
441
|
+
axis (horiz defaults to True for this form alone), the value
|
|
442
|
+
axis from zero, no title unless main= is given. With by=, one
|
|
443
|
+
series of dots per level; with exactly two levels, or a pair
|
|
444
|
+
of y variables (y=["Pre", "Post"]), the pair is joined by a
|
|
445
|
+
segment and ordered by its difference, largest positive at
|
|
446
|
+
the top. sort="0" keeps the data order instead.
|
|
447
|
+
|
|
448
|
+
form="profile" plots one point per category of x and connects
|
|
449
|
+
the points across the categories; with by= it draws one
|
|
450
|
+
profile per level, the interaction plot of a two-way ANOVA.
|
|
451
|
+
origin_y sets where the value axis begins; facet= draws one
|
|
452
|
+
panel per level on a shared value scale.
|
|
453
|
+
|
|
219
454
|
Variables are strings naming columns of the DataFrame `data`.
|
|
220
455
|
Returns a plotly Figure; call .show() to display from a script.
|
|
221
456
|
"""
|
|
@@ -226,6 +461,62 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
226
461
|
if form not in _FORMS:
|
|
227
462
|
raise ValueError(f"form must be one of {_FORMS}")
|
|
228
463
|
|
|
464
|
+
# Cleveland's dot plot places the categories on the vertical
|
|
465
|
+
# axis so their labels read across. A horizontal chart lists
|
|
466
|
+
# its first category at the bottom, so sort is inverted to keep
|
|
467
|
+
# "-" meaning largest at the top. R analog: Chart.R horiz.miss
|
|
468
|
+
sort_miss = sort is None
|
|
469
|
+
if sort is None:
|
|
470
|
+
sort = "0"
|
|
471
|
+
multi_x = isinstance(x, (list, tuple))
|
|
472
|
+
if horiz is None:
|
|
473
|
+
# the dot plot and the multi-item chart list their categories
|
|
474
|
+
# down the vertical axis
|
|
475
|
+
horiz = form == "dot" or multi_x
|
|
476
|
+
if horiz and sort in ("+", "-"):
|
|
477
|
+
sort = "+" if sort == "-" else "-"
|
|
478
|
+
|
|
479
|
+
# segments joins the points of a profile; the dot chart's stems
|
|
480
|
+
# are set along its value axis
|
|
481
|
+
if segments is not None and form != "profile":
|
|
482
|
+
raise ValueError(
|
|
483
|
+
"segments joins the points of a profile chart, so it "
|
|
484
|
+
'needs form="profile"; for the segments of a dot chart '
|
|
485
|
+
"set segments_x or segments_y")
|
|
486
|
+
if labels_cut is not None and form != "bar":
|
|
487
|
+
raise ValueError(
|
|
488
|
+
"labels_cut leaves off the value labels of small bar "
|
|
489
|
+
'segments, so it needs form="bar"')
|
|
490
|
+
|
|
491
|
+
# the value axis of a profile is y, so origin_x has nothing to set
|
|
492
|
+
if form == "profile" and origin_x is not None:
|
|
493
|
+
raise ValueError(
|
|
494
|
+
"origin_x does not apply to a profile chart: its value "
|
|
495
|
+
"axis is y, so set origin_y")
|
|
496
|
+
|
|
497
|
+
# Only the segments parameter along the value axis applies. The
|
|
498
|
+
# dot plot's orientation decides which one that is, so say which
|
|
499
|
+
# parameter to set instead rather than let the given one go
|
|
500
|
+
# inert. An unfaceted plot of several series is always
|
|
501
|
+
# horizontal.
|
|
502
|
+
# R analog: Chart.R .seg_note
|
|
503
|
+
if form == "dot":
|
|
504
|
+
dot_h = horiz or (facet is None and (
|
|
505
|
+
by is not None or isinstance(y, (list, tuple))))
|
|
506
|
+
given, instead, cat_ax, seg_ax = (
|
|
507
|
+
("segments_y", "segments_x", "y", "x") if dot_h
|
|
508
|
+
else ("segments_x", "segments_y", "x", "y"))
|
|
509
|
+
if (segments_y if dot_h else segments_x) is not None:
|
|
510
|
+
print(f">>> {given} does not apply to this dot plot.\n"
|
|
511
|
+
f" The categories are on the {cat_ax}-axis, so "
|
|
512
|
+
"the line segments run along the "
|
|
513
|
+
f"{seg_ax}-axis.\n"
|
|
514
|
+
f" To control them, set {instead}.\n")
|
|
515
|
+
if dot_h:
|
|
516
|
+
segments_y = None
|
|
517
|
+
else:
|
|
518
|
+
segments_x = None
|
|
519
|
+
|
|
229
520
|
# hierarchical routing: treemap/icicle always; pie with by=
|
|
230
521
|
# nests the by rings inside the x wedges as a sunburst
|
|
231
522
|
# R analog: Chart.R hier dispatch
|
|
@@ -234,14 +525,33 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
234
525
|
hier_type = form
|
|
235
526
|
elif form == "pie" and by is not None:
|
|
236
527
|
hier_type = "sunburst"
|
|
528
|
+
if isinstance(by, (list, tuple)) and len(by) > 1 \
|
|
529
|
+
and hier_type is None:
|
|
530
|
+
raise ValueError(
|
|
531
|
+
f"Only one by variable is permitted for a {form} chart, "
|
|
532
|
+
"but more than one specified.\n\n"
|
|
533
|
+
"The groups of by are overlaid within the display, one "
|
|
534
|
+
"color\n to each. That one channel is carried by the "
|
|
535
|
+
"first variable,\n so a second has no encoding left by "
|
|
536
|
+
"which to separate its\n levels.\n"
|
|
537
|
+
"Multiple by variables apply only to the hierarchical\n"
|
|
538
|
+
" charts -- pie/sunburst, treemap, and icicle -- where\n"
|
|
539
|
+
" nesting accepts depth and each added variable is one\n"
|
|
540
|
+
" level deeper.\n"
|
|
541
|
+
f"To stratify a {form} chart by a second variable, use\n"
|
|
542
|
+
" facet, which draws each group in a panel of its own.")
|
|
543
|
+
# hierarchical forms: each further by variable is one level
|
|
544
|
+
# deeper; the first plays the part of a single by elsewhere
|
|
545
|
+
by_more = []
|
|
546
|
+
if isinstance(by, (list, tuple)):
|
|
547
|
+
by_more = list(by[1:])
|
|
548
|
+
by = by[0] if by else None
|
|
549
|
+
by_label = ", ".join([by] + by_more) if by is not None else None
|
|
237
550
|
if hier_type is not None and stat == "deviation":
|
|
238
551
|
raise ValueError('stat="deviation" is not meaningful for '
|
|
239
552
|
"hierarchical charts: negative values "
|
|
240
553
|
"cannot form part-of-whole areas")
|
|
241
554
|
|
|
242
|
-
if form == "dot" and by is not None:
|
|
243
|
-
raise ValueError("The by variable is not meaningful for "
|
|
244
|
-
"dot charts. Do a bar chart.")
|
|
245
555
|
if isinstance(y, (list, tuple)) and form != "dot":
|
|
246
556
|
raise ValueError('Multiple y variables are only supported '
|
|
247
557
|
'for form="dot" (paired dot chart)')
|
|
@@ -260,11 +570,11 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
260
570
|
if form != "bar":
|
|
261
571
|
raise ValueError('stack100 rescales the bars of a bar '
|
|
262
572
|
'chart, so it needs form="bar"')
|
|
263
|
-
if facet is not None:
|
|
573
|
+
if facet is not None and by is None:
|
|
264
574
|
raise ValueError(
|
|
265
|
-
"stack100 rescales each bar
|
|
266
|
-
"
|
|
267
|
-
"
|
|
575
|
+
"stack100 rescales each bar to show how the levels "
|
|
576
|
+
"of a by variable divide it, so a faceted bar chart "
|
|
577
|
+
"needs by= for it")
|
|
268
578
|
if by is None and stat_x == "proportion":
|
|
269
579
|
raise ValueError(
|
|
270
580
|
'stack100 without a by variable is the same as '
|
|
@@ -291,6 +601,10 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
291
601
|
"legend_abbrev": legend_abbrev,
|
|
292
602
|
"legend_adjust": legend_adjust or None}
|
|
293
603
|
_leg_set = [k for k, v in _leg.items() if v is not None]
|
|
604
|
+
# a dot plot of several series heads its legend with legend_title
|
|
605
|
+
if form == "dot" and (by is not None
|
|
606
|
+
or isinstance(y, (list, tuple))):
|
|
607
|
+
_leg_set = [k for k in _leg_set if k != "legend_title"]
|
|
294
608
|
if _leg_set:
|
|
295
609
|
if form != "bar":
|
|
296
610
|
raise ValueError(
|
|
@@ -300,7 +614,7 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
300
614
|
raise ValueError(
|
|
301
615
|
f"{', '.join(_leg_set)} style the legend that a by "
|
|
302
616
|
"variable creates, so they need by=")
|
|
303
|
-
if facet is not None:
|
|
617
|
+
if facet is not None and _leg_set != ["legend_title"]:
|
|
304
618
|
raise ValueError(
|
|
305
619
|
f"{', '.join(_leg_set)} apply to a single-panel bar "
|
|
306
620
|
"chart; a faceted (Trellis) bar chart has no by "
|
|
@@ -348,62 +662,90 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
348
662
|
if sort != "0":
|
|
349
663
|
raise ValueError("Sort not applicable to Trellis "
|
|
350
664
|
"(faceted bar) charts")
|
|
351
|
-
if
|
|
352
|
-
raise
|
|
353
|
-
"
|
|
354
|
-
"plots, no data aggregation with parameter "
|
|
355
|
-
"stat. Use by instead of facet.")
|
|
356
|
-
if y is not None:
|
|
357
|
-
raise ValueError(
|
|
358
|
-
"The faceted bar chart displays counts of x, "
|
|
359
|
-
"so a y variable does not apply. Use by "
|
|
360
|
-
"instead of facet.")
|
|
361
|
-
if by is not None:
|
|
362
|
-
raise ValueError(
|
|
363
|
-
"by and facet do not combine for bar charts: "
|
|
364
|
-
"each facet panel displays the counts of x. "
|
|
365
|
-
"Use one or the other.")
|
|
665
|
+
if stat_x == "proportion" and by is not None:
|
|
666
|
+
raise NotImplementedError(
|
|
667
|
+
'stat_x="proportion" with by= is not yet ported')
|
|
366
668
|
|
|
367
669
|
# ----- resolve variables and filter ---------------------------
|
|
368
670
|
if filter is not None:
|
|
369
671
|
data = data.query(filter)
|
|
370
672
|
|
|
371
|
-
#
|
|
372
|
-
#
|
|
673
|
+
# row_names: the data frame's row labels as the categorical x,
|
|
674
|
+
# in their own order rather than alphabetical; the axis label
|
|
675
|
+
# is dropped unless given. R analog: Chart.R row_names
|
|
676
|
+
if (isinstance(x, str) and x in ("row_names", "row.names")
|
|
677
|
+
and x not in data.columns):
|
|
678
|
+
names = data.index.astype(str)
|
|
679
|
+
data = data.copy()
|
|
680
|
+
data[x] = pd.Categorical(
|
|
681
|
+
names, categories=list(dict.fromkeys(names)))
|
|
682
|
+
if xlab is None:
|
|
683
|
+
xlab = ""
|
|
684
|
+
|
|
685
|
+
# a list of x variables that share one response scale: the
|
|
686
|
+
# multi-item stacked chart. Its bands claim position and color
|
|
687
|
+
# the responses within them, so by has no encoding left; facet
|
|
688
|
+
# panels remain. R analog: Chart.R multiple-x path
|
|
689
|
+
if multi_x:
|
|
690
|
+
return _chart_items(
|
|
691
|
+
data, list(x), y=y, by=by, facet=facet, form=form,
|
|
692
|
+
stat=stat, sort=sort, sort_miss=sort_miss,
|
|
693
|
+
stack100=stack100, horiz=horiz, fill=fill, theme=theme,
|
|
694
|
+
labels=labels, labels_size=labels_size, gap=gap,
|
|
695
|
+
labels_cut=labels_cut,
|
|
696
|
+
xlab=xlab, ylab=ylab, main=main,
|
|
697
|
+
legend_title=legend_title, n_col=n_col, n_row=n_row,
|
|
698
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
699
|
+
rotate_x=rotate_x, rotate_y=rotate_y, quiet=quiet)
|
|
700
|
+
|
|
701
|
+
# paired dot chart: x=labels, y=[col1, col2, ...]. Displayed
|
|
702
|
+
# directly when x identifies the cases; with a repeated x each
|
|
703
|
+
# column is aggregated over x by stat, which is then required.
|
|
704
|
+
# R analog: Chart.R paired-dot path and its multi-column y stat
|
|
373
705
|
if isinstance(y, (list, tuple)):
|
|
374
|
-
if stat is not None:
|
|
375
|
-
raise ValueError("stat does not apply to a paired dot "
|
|
376
|
-
"chart; the values display directly")
|
|
377
706
|
x_ser = _get_column(data, x, "x")
|
|
378
707
|
y_cols = [_get_column(data, yi, "y") for yi in y]
|
|
379
708
|
sub = pd.concat([x_ser] + y_cols, axis=1).dropna()
|
|
380
|
-
cats = sub[x].astype(str).tolist()
|
|
381
709
|
ydf = sub[list(y)]
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
710
|
+
y_lbl = " & ".join(y)
|
|
711
|
+
if sub[x].duplicated().any():
|
|
712
|
+
if stat is None:
|
|
713
|
+
raise ValueError(
|
|
714
|
+
"The data are not a summary (pivot) table, and "
|
|
715
|
+
f"you have numerical variables, y = {y_lbl}, so "
|
|
716
|
+
"specify stat to define the aggregation, e.g., "
|
|
717
|
+
'stat="mean"')
|
|
718
|
+
if stat not in STAT_FUN and stat != "deviation":
|
|
719
|
+
raise ValueError(f"stat must be one of {_STATS}")
|
|
720
|
+
grp = ydf.groupby(sub[x], observed=True, sort=True)
|
|
721
|
+
if stat == "deviation":
|
|
722
|
+
ydf = grp.mean()
|
|
723
|
+
ydf = ydf - ydf.mean()
|
|
724
|
+
else:
|
|
725
|
+
ydf = STAT_FUN[stat](grp)
|
|
726
|
+
ydf = ydf.reindex([c for c in _category_order(sub[x])
|
|
727
|
+
if c in ydf.index])
|
|
728
|
+
cats = [str(c) for c in ydf.index]
|
|
729
|
+
val_lab = f"{STAT_LBL[stat]} of {y_lbl}"
|
|
730
|
+
else:
|
|
731
|
+
if stat is not None:
|
|
732
|
+
raise ValueError(
|
|
733
|
+
"The data are a summary table, so do not specify "
|
|
734
|
+
"stat: the aggregation has already been done")
|
|
735
|
+
cats = sub[x].astype(str).tolist()
|
|
736
|
+
val_lab = y_lbl
|
|
737
|
+
x_lab_dot = (xlab if xlab else ylab if ylab else val_lab)
|
|
738
|
+
return _dot_paired(
|
|
739
|
+
cats, ydf.reset_index(drop=True), sort, sort_miss,
|
|
740
|
+
origin_x, show_diff=not resolve_quiet(quiet), x_name=x,
|
|
741
|
+
fill=fill, border=color, pt_size=pt_size,
|
|
742
|
+
x_lab=x_lab_dot, y_lab=x,
|
|
402
743
|
digits_d=2 if digits_d is None else digits_d,
|
|
403
744
|
pt_opacity=1 - (get_option("trans_pt_fill", 0.10)
|
|
404
745
|
if transparency is None
|
|
405
746
|
else transparency),
|
|
406
|
-
|
|
747
|
+
main=None if not main else main,
|
|
748
|
+
legend_title=legend_title,
|
|
407
749
|
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
408
750
|
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
409
751
|
segments_x=(True if segments_x is None
|
|
@@ -412,6 +754,7 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
412
754
|
|
|
413
755
|
x_ser = _get_column(data, x, "x")
|
|
414
756
|
by_ser = _get_column(data, by, "by") if by is not None else None
|
|
757
|
+
by_more_sers = [_get_column(data, b, "by") for b in by_more]
|
|
415
758
|
y_ser = _get_column(data, y, "y") if y is not None else None
|
|
416
759
|
|
|
417
760
|
# multiple facet variables define the panel grid, not extra
|
|
@@ -431,11 +774,17 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
431
774
|
fac_cols = [ser]
|
|
432
775
|
facet_name = str(ser.name)
|
|
433
776
|
|
|
777
|
+
# count missing x before the removal below, or it is always 0
|
|
778
|
+
# R analog: Chart.R n.miss.x
|
|
779
|
+
n_miss_x = int(x_ser.isna().sum())
|
|
780
|
+
|
|
434
781
|
# casewise deletion over the variables in the analysis
|
|
435
782
|
used = [s for s in (x_ser, by_ser, y_ser) if s is not None]
|
|
783
|
+
used += by_more_sers
|
|
436
784
|
used += fac_cols or []
|
|
437
785
|
keep = ~pd.concat(used, axis=1).isna().any(axis=1)
|
|
438
786
|
x_ser = x_ser[keep]
|
|
787
|
+
by_more_sers = [b[keep] for b in by_more_sers]
|
|
439
788
|
if by_ser is not None:
|
|
440
789
|
by_ser = by_ser[keep]
|
|
441
790
|
if y_ser is not None:
|
|
@@ -451,6 +800,17 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
451
800
|
|
|
452
801
|
x_order = _category_order(x_ser)
|
|
453
802
|
by_order = _category_order(by_ser) if by_ser is not None else None
|
|
803
|
+
|
|
804
|
+
# A numeric x with many distinct values yields one category per
|
|
805
|
+
# value, usually a distribution in disguise. Advisory only: no
|
|
806
|
+
# count of levels reliably separates categorical from
|
|
807
|
+
# continuous, so the call is honored. R analog: Chart.R n.x > 12
|
|
808
|
+
if (y_ser is None and pd.api.types.is_numeric_dtype(x_ser)
|
|
809
|
+
and len(x_order) > 12):
|
|
810
|
+
print(f">>> {x} is numeric with {len(x_order)} distinct "
|
|
811
|
+
"values, so each\n becomes its own category. For "
|
|
812
|
+
"the distribution of a continuous\n variable: "
|
|
813
|
+
f'X("{x}")\n')
|
|
454
814
|
facet_order = (_category_order(facet_ser)
|
|
455
815
|
if facet_ser is not None else None)
|
|
456
816
|
|
|
@@ -523,6 +883,10 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
523
883
|
n_x = x_ser.nunique()
|
|
524
884
|
if by_ser is None:
|
|
525
885
|
is_agg = n_x >= len(x_ser)
|
|
886
|
+
elif by_more_sers:
|
|
887
|
+
# each cell of the nesting appears once in a summary table
|
|
888
|
+
cells = pd.concat([x_ser, by_ser] + by_more_sers, axis=1)
|
|
889
|
+
is_agg = not cells.astype(str).duplicated().any()
|
|
526
890
|
else:
|
|
527
891
|
is_agg = n_x * by_ser.nunique() >= len(by_ser)
|
|
528
892
|
|
|
@@ -555,32 +919,52 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
555
919
|
if not resolve_quiet(quiet):
|
|
556
920
|
y_out = (y_ser if y_ser is not None
|
|
557
921
|
and stat in STAT_FUN else None)
|
|
558
|
-
if y_ser is None or y_out is not None:
|
|
922
|
+
if by_more_sers and (y_ser is None or y_out is not None):
|
|
923
|
+
from .stats_out import nested_stats
|
|
924
|
+
print("\n".join(nested_stats(
|
|
925
|
+
[x_ser, by_ser] + by_more_sers, [x, by] + by_more,
|
|
926
|
+
y_out, stat, y,
|
|
927
|
+
2 if digits_d is None else digits_d)))
|
|
928
|
+
elif y_ser is None or y_out is not None:
|
|
929
|
+
var_lbl = data.attrs.get("variable_labels", {}) or {}
|
|
559
930
|
print("\n".join(chart_stats(
|
|
560
931
|
x_ser, by_ser, y_out, stat, x, by, y,
|
|
561
|
-
2 if digits_d is None else digits_d
|
|
932
|
+
2 if digits_d is None else digits_d,
|
|
933
|
+
stack100=stack100, n_miss=n_miss_x,
|
|
934
|
+
x_lbl=var_lbl.get(x), y_lbl=var_lbl.get(y),
|
|
935
|
+
facet_ser=facet_ser, facet_name=facet_name,
|
|
936
|
+
form=form)))
|
|
562
937
|
|
|
563
938
|
# labels default for aggregated data; for counts leave labels
|
|
564
939
|
# None so bc_plotly shows the value with % lines in hover
|
|
565
940
|
# R analog: Chart.R line ~716
|
|
566
941
|
if labels is None and (stat is not None or
|
|
567
942
|
(is_agg and y_ser is not None)):
|
|
568
|
-
|
|
943
|
+
# the bars carry a statistic, so the label is its value;
|
|
944
|
+
# beside only arranges the bars, and a percentage of a sum
|
|
945
|
+
# of means is not a quantity
|
|
946
|
+
labels = "input"
|
|
569
947
|
|
|
570
948
|
# ----- hierarchical forms aggregate per nesting level ---------
|
|
571
949
|
# from the raw columns, not from the flat table built below
|
|
572
950
|
if hier_type is not None:
|
|
573
951
|
if digits_d is None:
|
|
574
952
|
digits_d = 0 if y_ser is None else 2
|
|
575
|
-
agg = hier_aggregate(x_ser,
|
|
953
|
+
agg = hier_aggregate(x_ser,
|
|
954
|
+
by=([by_ser] + by_more_sers if by_more
|
|
955
|
+
else by_ser),
|
|
956
|
+
y=y_ser, stat=stat,
|
|
576
957
|
facet=facet_ser,
|
|
577
|
-
x_name=x,
|
|
958
|
+
x_name=x,
|
|
959
|
+
by_name=[by] + by_more if by_more else by,
|
|
960
|
+
y_name=y,
|
|
578
961
|
facet_name=facet_name,
|
|
579
962
|
facet_order=facet_order)
|
|
580
|
-
fill_vec = hier_color_resolve(x_order, fill
|
|
963
|
+
fill_vec = hier_color_resolve(x_order, fill, x=x_ser,
|
|
964
|
+
y=y_ser, stat=stat)
|
|
581
965
|
if main is None:
|
|
582
|
-
main = build_title(x, by_name=
|
|
583
|
-
facet_name=facet_name)
|
|
966
|
+
main = build_title(x, by_name=by_label, y_name=y,
|
|
967
|
+
stat=stat, facet_name=facet_name)
|
|
584
968
|
elif main == "":
|
|
585
969
|
main = None
|
|
586
970
|
# Chart resolves the labels default before hier sees it:
|
|
@@ -604,26 +988,85 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
604
988
|
# (hier forms handled above; pie facet became the by grouping)
|
|
605
989
|
|
|
606
990
|
if facet_ser is not None and form == "bar":
|
|
607
|
-
# Trellis bar chart: horizontal
|
|
608
|
-
#
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
991
|
+
# Trellis bar chart: horizontal bars per panel, the plotly
|
|
992
|
+
# port of R's .bar.lattice rendering. Counts of x, or the
|
|
993
|
+
# stat of y aggregated within each panel; a by variable
|
|
994
|
+
# divides each bar into its levels, as the bar chart of a
|
|
995
|
+
# single panel does, drawn within every panel
|
|
996
|
+
agg_y = y_ser is not None and stat is not None
|
|
997
|
+
if y_ser is not None and not agg_y and not is_agg:
|
|
998
|
+
raise ValueError(
|
|
999
|
+
"The data are not a summary (pivot) table, so "
|
|
1000
|
+
'specify stat to define the aggregation, e.g., '
|
|
1001
|
+
'stat="mean"')
|
|
1002
|
+
|
|
1003
|
+
def cell_tbl(m):
|
|
1004
|
+
xs = x_ser[m]
|
|
1005
|
+
if by_ser is None:
|
|
1006
|
+
if y_ser is None:
|
|
1007
|
+
t = xs.groupby(xs, observed=True).size()
|
|
1008
|
+
elif agg_y:
|
|
1009
|
+
t = STAT_FUN[stat](y_ser[m].groupby(xs,
|
|
1010
|
+
observed=True))
|
|
1011
|
+
else:
|
|
1012
|
+
t = y_ser[m].groupby(xs, observed=True).first()
|
|
1013
|
+
return t.reindex(x_order)
|
|
1014
|
+
df = pd.DataFrame({"b": by_ser[m], "x": xs,
|
|
1015
|
+
"y": 1.0 if y_ser is None
|
|
1016
|
+
else y_ser[m].astype(float)})
|
|
1017
|
+
g = df.groupby(["b", "x"], observed=True)["y"]
|
|
1018
|
+
t = (g.sum() if y_ser is None
|
|
1019
|
+
else STAT_FUN[stat](g) if agg_y else g.first())
|
|
1020
|
+
t = t.unstack("x").reindex(index=by_order,
|
|
1021
|
+
columns=x_order)
|
|
1022
|
+
if y_ser is None:
|
|
1023
|
+
t = t.fillna(0)
|
|
1024
|
+
if stack100:
|
|
1025
|
+
# each bar scaled to its own total compares
|
|
1026
|
+
# composition rather than magnitude
|
|
1027
|
+
tot = t.sum(axis=0)
|
|
1028
|
+
t = t.div(tot.where(tot != 0), axis=1)
|
|
1029
|
+
return t
|
|
1030
|
+
|
|
1031
|
+
tbls = {lv: cell_tbl(facet_ser == lv) for lv in facet_order}
|
|
1032
|
+
if by_ser is None:
|
|
1033
|
+
tbl = pd.DataFrame([tbls[lv] for lv in facet_order],
|
|
1034
|
+
index=facet_order)
|
|
1035
|
+
by_tbls = None
|
|
1036
|
+
else:
|
|
1037
|
+
# the panel totals stand in for the shape of the grid
|
|
1038
|
+
tbl = pd.DataFrame([tbls[lv].sum(axis=0)
|
|
1039
|
+
for lv in facet_order],
|
|
1040
|
+
index=facet_order)
|
|
1041
|
+
by_tbls = tbls
|
|
612
1042
|
tbl.index.name = facet_name
|
|
613
1043
|
tbl.columns.name = x
|
|
1044
|
+
if stack100:
|
|
1045
|
+
val_name = f"Proportion within {x}"
|
|
1046
|
+
elif agg_y:
|
|
1047
|
+
val_name = f"{STAT_LBL[stat]} of {y}"
|
|
1048
|
+
elif y_ser is not None:
|
|
1049
|
+
val_name = y
|
|
1050
|
+
elif by_ser is not None:
|
|
1051
|
+
val_name = "Count"
|
|
1052
|
+
else:
|
|
1053
|
+
val_name = None # Count/Proportion of x
|
|
614
1054
|
return bc_facet_plotly(
|
|
615
1055
|
tbl, x_name=x, facet_name=facet_name,
|
|
616
1056
|
x_lab=xlab, y_lab=ylab, fill=fill,
|
|
617
1057
|
border="off" if color is None else color,
|
|
618
1058
|
opacity=(None if transparency is None
|
|
619
1059
|
else 1 - transparency),
|
|
620
|
-
proportion=stat_x == "proportion",
|
|
1060
|
+
proportion=stat_x == "proportion" and by_ser is None,
|
|
621
1061
|
digits_d=(digits_d if digits_d is not None
|
|
622
|
-
else
|
|
1062
|
+
else 2 if (y_ser is not None or stack100
|
|
1063
|
+
or stat_x != "count") else 0),
|
|
623
1064
|
n_col=n_col_use or 1,
|
|
624
1065
|
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
625
1066
|
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
626
1067
|
main=main,
|
|
1068
|
+
by_tbls=by_tbls, by_name=by, beside=beside,
|
|
1069
|
+
val_name=val_name, legend_title=legend_title,
|
|
627
1070
|
)
|
|
628
1071
|
|
|
629
1072
|
if facet_ser is not None and form == "radar":
|
|
@@ -707,54 +1150,145 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
707
1150
|
n_col=n_col_use,
|
|
708
1151
|
)
|
|
709
1152
|
|
|
1153
|
+
if facet_ser is not None and form == "profile":
|
|
1154
|
+
# one profile per group within every panel, on the categories
|
|
1155
|
+
# and value scale all panels share. Without y the value is a
|
|
1156
|
+
# count; with y and no stat the values are supplied directly,
|
|
1157
|
+
# one per cell. R analog: Chart.R faceted profile, drawn by
|
|
1158
|
+
# .plt.profile.facet() in base R; here a plotly grid
|
|
1159
|
+
agg_y = not is_agg and stat is not None
|
|
1160
|
+
if y_ser is not None and not agg_y and not is_agg:
|
|
1161
|
+
raise ValueError(
|
|
1162
|
+
"The data are not a summary (pivot) table, so "
|
|
1163
|
+
'specify stat to define the aggregation, e.g., '
|
|
1164
|
+
'stat="mean"')
|
|
1165
|
+
series = ([str(b) for b in by_order] if by_ser is not None
|
|
1166
|
+
else [None])
|
|
1167
|
+
x_lv = [str(c) for c in x_order]
|
|
1168
|
+
pos = {c: k + 1 for k, c in enumerate(x_lv)}
|
|
1169
|
+
df = pd.DataFrame({
|
|
1170
|
+
"f": facet_ser.astype(str), "x": x_ser.astype(str),
|
|
1171
|
+
"g": (by_ser.astype(str) if by_ser is not None
|
|
1172
|
+
else "\0"),
|
|
1173
|
+
"y": 1.0 if y_ser is None else y_ser.astype(float)})
|
|
1174
|
+
g = df.groupby(["f", "g", "x"], observed=True)["y"]
|
|
1175
|
+
t = (g.sum() if y_ser is None
|
|
1176
|
+
else STAT_FUN[stat](g) if agg_y else g.mean())
|
|
1177
|
+
cells = {}
|
|
1178
|
+
for (f, gg, xx), v in t.items():
|
|
1179
|
+
nm = None if by_ser is None else gg
|
|
1180
|
+
xs, ys = cells.setdefault(f, {}).setdefault(nm, ([], []))
|
|
1181
|
+
xs.append(pos[xx])
|
|
1182
|
+
ys.append(float(v))
|
|
1183
|
+
for f in cells: # draw left to right
|
|
1184
|
+
for nm, (xs, ys) in cells[f].items():
|
|
1185
|
+
o = np.argsort(xs)
|
|
1186
|
+
cells[f][nm] = ([xs[k] for k in o], [ys[k] for k in o])
|
|
1187
|
+
vals = t.to_numpy(dtype=float)
|
|
1188
|
+
lo, hi = float(np.nanmin(vals)), float(np.nanmax(vals))
|
|
1189
|
+
# the panels share one value scale, so the origin widens it
|
|
1190
|
+
if origin_y is not None:
|
|
1191
|
+
lo, hi = min(lo, origin_y), max(hi, origin_y)
|
|
1192
|
+
y_tick = pretty(lo, hi)
|
|
1193
|
+
if digits_d is None:
|
|
1194
|
+
digits_d = 0 if y_ser is None else 2
|
|
1195
|
+
from .plotly_utils import axis_format
|
|
1196
|
+
y_lab = (ylab if ylab is not None
|
|
1197
|
+
else "Count" if y_ser is None
|
|
1198
|
+
else f"{STAT_LBL[stat]} of {y}" if agg_y else y)
|
|
1199
|
+
return profile_facet_plotly(
|
|
1200
|
+
cells, x_lv, series, y_tick,
|
|
1201
|
+
axis_format(y_tick, digits_d, axis_fmt, axis_y_pre),
|
|
1202
|
+
connect=True if segments is None else bool(segments),
|
|
1203
|
+
x_lab=x if xlab is None else xlab, y_lab=y_lab,
|
|
1204
|
+
by_name=by,
|
|
1205
|
+
fill=(fill if isinstance(fill, (list, tuple))
|
|
1206
|
+
else [fill] if fill is not None
|
|
1207
|
+
else _profile_fill(theme)),
|
|
1208
|
+
pt_size=1.5 * pt_size,
|
|
1209
|
+
facet_levels=[str(f) for f in facet_order],
|
|
1210
|
+
facet_name=facet_name, n_col=n_col_use,
|
|
1211
|
+
main=None if not main else main, rotate_x=rotate_x)
|
|
1212
|
+
|
|
710
1213
|
if facet_ser is not None and form == "dot":
|
|
711
1214
|
# aggregate and sort per facet panel; a panel shows only
|
|
712
|
-
# its own categories.
|
|
1215
|
+
# its own categories. With by=, one series per level of by in
|
|
1216
|
+
# every panel, on the full x-by-panel grid so a combination
|
|
1217
|
+
# absent from the data is an empty cell. R analog: Chart.R
|
|
1218
|
+
# faceted dot prep and its by reshape
|
|
713
1219
|
prop = stat_x == "proportion"
|
|
1220
|
+
agg_y = not is_agg and stat is not None
|
|
714
1221
|
cats_l, vals_l, fac_l = [], [], []
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
1222
|
+
if by_ser is not None:
|
|
1223
|
+
if prop:
|
|
1224
|
+
raise ValueError(
|
|
1225
|
+
'stat_x="proportion" with by= is not yet ported')
|
|
1226
|
+
rows = []
|
|
1227
|
+
for lv in facet_order:
|
|
1228
|
+
m = facet_ser == lv
|
|
1229
|
+
df = pd.DataFrame({"x": x_ser[m].astype(str),
|
|
1230
|
+
"by": by_ser[m].astype(str),
|
|
1231
|
+
"y": (1.0 if y_ser is None
|
|
1232
|
+
else y_ser[m].astype(float))})
|
|
1233
|
+
g = df.groupby(["x", "by"], observed=True)["y"]
|
|
1234
|
+
t = (g.sum() if y_ser is None
|
|
1235
|
+
else STAT_FUN[stat](g) if agg_y else g.first())
|
|
1236
|
+
wide = (t.unstack("by")
|
|
1237
|
+
.reindex(index=[str(c) for c in x_order],
|
|
1238
|
+
columns=[str(b) for b in by_order]))
|
|
1239
|
+
rows.append(wide)
|
|
1240
|
+
cats_l += list(wide.index)
|
|
1241
|
+
fac_l += [str(lv)] * len(wide)
|
|
1242
|
+
vals_l = pd.concat(rows, ignore_index=True)
|
|
1243
|
+
all_vals = vals_l.to_numpy(dtype=float).ravel()
|
|
1244
|
+
else:
|
|
1245
|
+
for lv in facet_order:
|
|
1246
|
+
m = facet_ser == lv
|
|
1247
|
+
xs = x_ser[m].astype(str)
|
|
1248
|
+
if y_ser is None: # counts per panel
|
|
1249
|
+
t = xs.groupby(xs).size()
|
|
1250
|
+
if prop:
|
|
1251
|
+
tot = t.sum()
|
|
1252
|
+
t = t / tot if tot > 0 else t.astype(float)
|
|
1253
|
+
elif agg_y:
|
|
1254
|
+
t = STAT_FUN[stat](y_ser[m].groupby(xs))
|
|
1255
|
+
else: # pre-aggregated rows
|
|
1256
|
+
t = pd.Series(y_ser[m].to_numpy(dtype=float),
|
|
1257
|
+
index=xs.values)
|
|
1258
|
+
if sort != "0":
|
|
1259
|
+
t = t.sort_values(ascending=(sort == "+"),
|
|
1260
|
+
kind="stable")
|
|
1261
|
+
cats_l += [str(c) for c in t.index]
|
|
1262
|
+
vals_l += [float(v) for v in t.to_numpy()]
|
|
1263
|
+
fac_l += [str(lv)] * len(t)
|
|
1264
|
+
all_vals = vals_l
|
|
1265
|
+
|
|
1266
|
+
# an aggregated value is labeled by its statistic, as the
|
|
1267
|
+
# unfaceted chart is: "Mean of Salary", not "Salary"
|
|
1268
|
+
if y_ser is None:
|
|
1269
|
+
val_lab = ("Count" if by_ser is not None
|
|
1270
|
+
else f"Proportion of {x}" if prop
|
|
1271
|
+
else f"Count of {x}")
|
|
1272
|
+
else:
|
|
1273
|
+
val_lab = f"{STAT_LBL[stat]} of {y}" if agg_y else y
|
|
737
1274
|
if horiz:
|
|
738
1275
|
orientation = "h"
|
|
739
|
-
x_lab_arg = xlab if xlab is not None
|
|
740
|
-
|
|
1276
|
+
x_lab_arg = (xlab if xlab is not None
|
|
1277
|
+
else ylab if ylab is not None else val_lab)
|
|
1278
|
+
y_lab_arg = x
|
|
741
1279
|
else:
|
|
742
1280
|
orientation = "v"
|
|
743
1281
|
x_lab_arg = xlab if xlab is not None else x
|
|
744
1282
|
y_lab_arg = ylab if ylab is not None else val_lab
|
|
745
1283
|
|
|
1284
|
+
# the length of the segment carries the value, so the value
|
|
1285
|
+
# axis begins at zero unless an origin is specified
|
|
1286
|
+
org_in = origin_x if horiz else origin_y
|
|
746
1287
|
origin, gridT = _dot_origin_grid(
|
|
747
|
-
|
|
748
|
-
origin_in=origin_x if horiz else origin_y,
|
|
1288
|
+
all_vals, origin_in=0 if org_in is None else org_in,
|
|
749
1289
|
is_counts=True if y_ser is None else None)
|
|
750
1290
|
if digits_d is None:
|
|
751
1291
|
digits_d = 2 if (y_ser is not None or prop) else 0
|
|
752
|
-
if main is None:
|
|
753
|
-
main = build_title(
|
|
754
|
-
x, y_name="Proportion" if prop else y,
|
|
755
|
-
stat=stat, facet_name=facet_name)
|
|
756
|
-
elif main == "":
|
|
757
|
-
main = None
|
|
758
1292
|
return dot_plotly(
|
|
759
1293
|
cats_l, vals_l, orientation=orientation,
|
|
760
1294
|
fill=fill, border=color, pt_size=pt_size,
|
|
@@ -763,7 +1297,10 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
763
1297
|
pt_opacity=1 - (get_option("trans_pt_fill", 0.10)
|
|
764
1298
|
if transparency is None
|
|
765
1299
|
else transparency),
|
|
766
|
-
gridT=gridT, origin_x=origin,
|
|
1300
|
+
gridT=gridT, origin_x=origin,
|
|
1301
|
+
main=None if not main else main,
|
|
1302
|
+
legend_title=(legend_title if legend_title is not None
|
|
1303
|
+
else by),
|
|
767
1304
|
facet=fac_l, facet_name=facet_name, n_col=n_col_use,
|
|
768
1305
|
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
769
1306
|
axis_y_pre=axis_y_pre,
|
|
@@ -838,7 +1375,11 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
838
1375
|
else:
|
|
839
1376
|
tbl = fun(df.groupby(["by", "x"])["y"]).unstack("x")
|
|
840
1377
|
tbl = tbl.reindex(index=by_order, columns=x_order)
|
|
841
|
-
|
|
1378
|
+
# a profile or dot plot of several groups draws an empty
|
|
1379
|
+
# cell as a missing point; the other forms need every cell
|
|
1380
|
+
gaps_ok = form in ("profile", "dot") and by_ser is not None
|
|
1381
|
+
if not gaps_ok and not np.isfinite(
|
|
1382
|
+
tbl.to_numpy(dtype=float)).all():
|
|
842
1383
|
raise ValueError(
|
|
843
1384
|
"The summary table of the transformed data has "
|
|
844
1385
|
"missing or non-finite values, likely because some "
|
|
@@ -882,7 +1423,9 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
882
1423
|
elif beside:
|
|
883
1424
|
ylab = "Percentage"
|
|
884
1425
|
else:
|
|
885
|
-
|
|
1426
|
+
# the axis holds proportions, so it is named for them;
|
|
1427
|
+
# the legend names by, so it is not repeated here
|
|
1428
|
+
ylab = f"Proportion within {x}"
|
|
886
1429
|
if digits_d_user is None:
|
|
887
1430
|
digits_d = 2
|
|
888
1431
|
|
|
@@ -954,32 +1497,120 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
954
1497
|
digits_d=digits_d, val_label=val_label,
|
|
955
1498
|
)
|
|
956
1499
|
|
|
1500
|
+
if form == "profile":
|
|
1501
|
+
# one point per category of x, connected across the
|
|
1502
|
+
# categories, one profile per level of by: the interaction
|
|
1503
|
+
# plot of the analysis of variance. The connecting segments
|
|
1504
|
+
# are the purpose of the form. R analog: Chart.R profile ->
|
|
1505
|
+
# .plt.main(cat.x=TRUE, segments=TRUE) -> plt.plotly()
|
|
1506
|
+
if isinstance(tbl, pd.DataFrame):
|
|
1507
|
+
groups = [(str(b), tbl.columns, tbl.loc[b])
|
|
1508
|
+
for b in tbl.index]
|
|
1509
|
+
else:
|
|
1510
|
+
groups = [(None, tbl.index, tbl)]
|
|
1511
|
+
cats = [str(c) for c in groups[0][1]]
|
|
1512
|
+
pos = list(range(1, len(cats) + 1))
|
|
1513
|
+
# an empty cell stays NaN, which plotly draws as a gap
|
|
1514
|
+
groups = [(nm, pos, vals.to_numpy(dtype=float).tolist())
|
|
1515
|
+
for nm, _, vals in groups]
|
|
1516
|
+
vals = tbl.to_numpy(dtype=float).ravel()
|
|
1517
|
+
lo, hi = float(np.nanmin(vals)), float(np.nanmax(vals))
|
|
1518
|
+
# the value axis of a profile is y, so origin_y sets where
|
|
1519
|
+
# it begins, as it does for the dot chart
|
|
1520
|
+
if origin_y is not None:
|
|
1521
|
+
lo, hi = min(lo, origin_y), max(hi, origin_y)
|
|
1522
|
+
axT2 = pretty(lo, hi)
|
|
1523
|
+
from .plotly_utils import axis_format
|
|
1524
|
+
y_lab = (ylab_user if ylab_user is not None
|
|
1525
|
+
else "Count" if y_ser is None and stat_x == "count"
|
|
1526
|
+
else ylab)
|
|
1527
|
+
fig = plt_plotly(
|
|
1528
|
+
groups, by_name=by,
|
|
1529
|
+
fill=(fill if isinstance(fill, (list, tuple))
|
|
1530
|
+
else [fill] if fill is not None
|
|
1531
|
+
else _profile_fill(theme)),
|
|
1532
|
+
pt_size=1.5 * pt_size,
|
|
1533
|
+
x_lab=x if xlab is None else xlab, y_lab=y_lab,
|
|
1534
|
+
ax={"axT1": pos, "axL1": cats, "axT2": axT2,
|
|
1535
|
+
"axL2": axis_format(axT2, digits_d, axis_fmt,
|
|
1536
|
+
axis_y_pre)},
|
|
1537
|
+
gridT1=pos, gridT2=axT2,
|
|
1538
|
+
# no automatic title: it would only repeat the axis labels
|
|
1539
|
+
main=None if not main else main,
|
|
1540
|
+
digits_d=digits_d,
|
|
1541
|
+
connect=True if segments is None else bool(segments),
|
|
1542
|
+
pt_opacity=1,
|
|
1543
|
+
style_opts=None)
|
|
1544
|
+
pad = 0.04 * (axT2[-1] - axT2[0])
|
|
1545
|
+
fig.update_yaxes(range=[axT2[0] - pad, axT2[-1] + pad])
|
|
1546
|
+
fig.update_xaxes(range=[0.5, len(cats) + 0.5])
|
|
1547
|
+
if rotate_x:
|
|
1548
|
+
fig.update_xaxes(tickangle=-rotate_x)
|
|
1549
|
+
if rotate_y:
|
|
1550
|
+
fig.update_yaxes(tickangle=-rotate_y)
|
|
1551
|
+
return fig
|
|
1552
|
+
|
|
957
1553
|
if form == "dot":
|
|
958
|
-
|
|
1554
|
+
quiet_use = resolve_quiet(quiet)
|
|
1555
|
+
pt_op = 1 - (get_option("trans_pt_fill", 0.10)
|
|
1556
|
+
if transparency is None else transparency)
|
|
1557
|
+
if isinstance(tbl, pd.DataFrame):
|
|
1558
|
+
# by=: one series of dots per level of by, drawn as the
|
|
1559
|
+
# paired geometry, one column per level. A two-level by
|
|
1560
|
+
# with a stat already lists its difference as the Diff
|
|
1561
|
+
# row of the summary table. R analog: Chart.R dot by
|
|
1562
|
+
# reshape into the paired path
|
|
1563
|
+
ydf = tbl.T.reindex([c for c in x_order if c in tbl.columns])
|
|
1564
|
+
ydf.columns = [str(c) for c in ydf.columns]
|
|
1565
|
+
cats = [str(c) for c in ydf.index]
|
|
1566
|
+
val_lab = (xlab if xlab else ylab_user if ylab_user
|
|
1567
|
+
else "Count" if y_ser is None else ylab)
|
|
1568
|
+
return _dot_paired(
|
|
1569
|
+
cats, ydf.reset_index(drop=True), sort, sort_miss,
|
|
1570
|
+
origin_x,
|
|
1571
|
+
show_diff=not quiet_use and y_ser is None,
|
|
1572
|
+
x_name=x,
|
|
1573
|
+
fill=fill, border=color, pt_size=pt_size,
|
|
1574
|
+
x_lab=val_lab, y_lab=x, digits_d=digits_d,
|
|
1575
|
+
pt_opacity=pt_op,
|
|
1576
|
+
main=None if not main else main,
|
|
1577
|
+
legend_title=(legend_title if legend_title is not None
|
|
1578
|
+
else by),
|
|
1579
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
1580
|
+
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
1581
|
+
segments_x=(True if segments_x is None
|
|
1582
|
+
else segments_x),
|
|
1583
|
+
)
|
|
1584
|
+
|
|
959
1585
|
cats = [str(c) for c in tbl.index]
|
|
960
1586
|
vals = tbl.to_numpy(dtype=float)
|
|
1587
|
+
# the length of the segment carries the value, so the value
|
|
1588
|
+
# axis begins at zero unless an origin is specified
|
|
1589
|
+
org_in = origin_x if horiz else origin_y
|
|
961
1590
|
origin, gridT = _dot_origin_grid(
|
|
962
|
-
vals,
|
|
963
|
-
origin_in=origin_x if horiz else origin_y,
|
|
1591
|
+
vals, origin_in=0 if org_in is None else org_in,
|
|
964
1592
|
is_counts=True if y_ser is None else None)
|
|
965
|
-
cat_lab = x if xlab is None else xlab
|
|
966
1593
|
val_lab = ylab # set with tbl above
|
|
967
|
-
if
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
1594
|
+
if horiz:
|
|
1595
|
+
# the value axis takes the plotted quantity, from xlab or
|
|
1596
|
+
# the statistic; the category axis takes x's own name
|
|
1597
|
+
x_lab_arg = xlab if xlab is not None else val_lab
|
|
1598
|
+
y_lab_arg = x
|
|
1599
|
+
else:
|
|
1600
|
+
x_lab_arg = xlab if xlab is not None else x
|
|
1601
|
+
y_lab_arg = val_lab
|
|
1602
|
+
# the value axis already names the plotted quantity, so an
|
|
1603
|
+
# unrequested title would only repeat it
|
|
971
1604
|
return dot_plotly(
|
|
972
1605
|
cats, vals,
|
|
973
1606
|
orientation="h" if horiz else "v",
|
|
974
1607
|
fill=fill, border=color,
|
|
975
1608
|
pt_size=pt_size,
|
|
976
|
-
x_lab=
|
|
977
|
-
y_lab=cat_lab if horiz else val_lab,
|
|
1609
|
+
x_lab=x_lab_arg, y_lab=y_lab_arg,
|
|
978
1610
|
digits_d=digits_d,
|
|
979
|
-
pt_opacity=
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
gridT=gridT, origin_x=origin, main=main,
|
|
1611
|
+
pt_opacity=pt_op,
|
|
1612
|
+
gridT=gridT, origin_x=origin,
|
|
1613
|
+
main=None if not main else main,
|
|
983
1614
|
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
984
1615
|
axis_y_pre=axis_y_pre,
|
|
985
1616
|
rotate_x=rotate_x, rotate_y=rotate_y,
|
|
@@ -1041,6 +1672,11 @@ def Chart(x, y=None, data=None, filter=None, by=None, facet=None,
|
|
|
1041
1672
|
labels_size=0.90 if labels_size is None else labels_size,
|
|
1042
1673
|
labels_color=labels_color,
|
|
1043
1674
|
labels_decimals=labels_decimals,
|
|
1675
|
+
# applied only when given; R cuts by share, which a single
|
|
1676
|
+
# series of a statistic does not carry, nor outside labels
|
|
1677
|
+
labels_cut=(None if (labels_position == "out" or (
|
|
1678
|
+
isinstance(tbl, pd.Series) and y_ser is not None))
|
|
1679
|
+
else labels_cut),
|
|
1044
1680
|
legend_title=legend_title, legend_position=legend_position,
|
|
1045
1681
|
legend_labels=legend_labels, legend_horiz=legend_horiz,
|
|
1046
1682
|
legend_size=legend_size, legend_abbrev=legend_abbrev,
|