lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/VariableLabels.py
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# VariableLabels.py — analog of VariableLabels.R.
|
|
2
|
+
#
|
|
3
|
+
# VariableLabels(): get or set the variable labels (and optional
|
|
4
|
+
# units) of a data frame. lessPy stores this metadata in the pandas
|
|
5
|
+
# attrs of the frame -- data.attrs["variable_labels"] and
|
|
6
|
+
# data.attrs["variable_units"], each a {column: text} mapping --
|
|
7
|
+
# which is where details() reads it.
|
|
8
|
+
#
|
|
9
|
+
# R's VariableLabels relies on global objects (the label table `l`,
|
|
10
|
+
# the data frame `d`) and non-standard evaluation of a bare variable
|
|
11
|
+
# name; those modes do not translate. This port keeps the name and
|
|
12
|
+
# purpose but takes the frame explicitly:
|
|
13
|
+
# VariableLabels(data) -> return / show the labels
|
|
14
|
+
# VariableLabels(data, {"Salary": ...}) -> set from a dict
|
|
15
|
+
# VariableLabels(data, "labels.csv") -> set from a file
|
|
16
|
+
# Setting updates data.attrs in place and also returns the resulting
|
|
17
|
+
# labels as a DataFrame (the single return shape for both modes).
|
|
18
|
+
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
from .utils import get_option
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def VariableLabels(data, labels=None, units=None, quiet=None):
|
|
25
|
+
"""Get or set a data frame's variable labels. With only data,
|
|
26
|
+
return (and print) the current labels as a DataFrame. With
|
|
27
|
+
labels as a {column: label} dict or a CSV/Excel file path
|
|
28
|
+
(column, label, optional unit per row, no header), attach them
|
|
29
|
+
to data.attrs and return the resulting labels. units is an
|
|
30
|
+
optional {column: unit} dict when setting from a dict.
|
|
31
|
+
R analog: VariableLabels()"""
|
|
32
|
+
if not isinstance(data, pd.DataFrame):
|
|
33
|
+
raise TypeError("data must be a pandas DataFrame")
|
|
34
|
+
if quiet is None:
|
|
35
|
+
quiet = get_option("quiet", False)
|
|
36
|
+
|
|
37
|
+
if labels is None:
|
|
38
|
+
return _get(data, quiet)
|
|
39
|
+
return _set(data, labels, units, quiet)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _get(data, quiet):
|
|
43
|
+
lab = data.attrs.get("variable_labels")
|
|
44
|
+
if not lab:
|
|
45
|
+
if not quiet:
|
|
46
|
+
print("\nNo variable labels present\n")
|
|
47
|
+
return _frame({}, {})
|
|
48
|
+
frame = _frame(lab, data.attrs.get("variable_units", {}))
|
|
49
|
+
if not quiet:
|
|
50
|
+
_show(frame)
|
|
51
|
+
return frame
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _set(data, labels, units, quiet):
|
|
55
|
+
if isinstance(labels, str):
|
|
56
|
+
labels, file_units = _read_file(labels)
|
|
57
|
+
if file_units:
|
|
58
|
+
units = file_units
|
|
59
|
+
if not isinstance(labels, dict):
|
|
60
|
+
raise TypeError(
|
|
61
|
+
"labels must be a {column: label} dict, a CSV/Excel "
|
|
62
|
+
"file path, or None to display the current labels")
|
|
63
|
+
|
|
64
|
+
unknown = [k for k in labels if k not in data.columns]
|
|
65
|
+
if unknown:
|
|
66
|
+
raise ValueError(
|
|
67
|
+
f"not column(s) of data: {unknown}. Columns: "
|
|
68
|
+
f"{', '.join(map(str, data.columns))}")
|
|
69
|
+
|
|
70
|
+
lab = dict(data.attrs.get("variable_labels", {}))
|
|
71
|
+
lab.update({str(k): str(v) for k, v in labels.items()})
|
|
72
|
+
data.attrs["variable_labels"] = lab
|
|
73
|
+
if units:
|
|
74
|
+
unt = dict(data.attrs.get("variable_units", {}))
|
|
75
|
+
unt.update({str(k): str(v) for k, v in units.items()})
|
|
76
|
+
data.attrs["variable_units"] = unt
|
|
77
|
+
|
|
78
|
+
frame = _frame(lab, data.attrs.get("variable_units", {}))
|
|
79
|
+
if not quiet:
|
|
80
|
+
_show(frame)
|
|
81
|
+
return frame
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _frame(labels, units):
|
|
85
|
+
"""Build the labels DataFrame: index of column names, a 'label'
|
|
86
|
+
column, and a 'unit' column when any units are present."""
|
|
87
|
+
names = list(labels.keys())
|
|
88
|
+
data = {"label": [labels[n] for n in names]}
|
|
89
|
+
if units:
|
|
90
|
+
data["unit"] = [units.get(n, "") for n in names]
|
|
91
|
+
return pd.DataFrame(data, index=pd.Index(names, name="variable"))
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _show(frame):
|
|
95
|
+
print()
|
|
96
|
+
for name, row in frame.iterrows():
|
|
97
|
+
unit = f" ({row['unit']})" if "unit" in frame.columns \
|
|
98
|
+
and row["unit"] else ""
|
|
99
|
+
print(f"{name}: {row['label']}{unit}")
|
|
100
|
+
print()
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _read_file(path):
|
|
104
|
+
"""Read labels from a headerless file: column name, label, and
|
|
105
|
+
an optional unit per row. Returns (labels, units) dicts."""
|
|
106
|
+
if path.endswith(".xlsx"):
|
|
107
|
+
raw = pd.read_excel(path, header=None, index_col=0)
|
|
108
|
+
else:
|
|
109
|
+
raw = pd.read_csv(path, header=None, index_col=0)
|
|
110
|
+
names = [str(k) for k in raw.index]
|
|
111
|
+
labels = {n: str(raw.iloc[i, 0]) for i, n in enumerate(names)}
|
|
112
|
+
units = {}
|
|
113
|
+
if raw.shape[1] >= 2:
|
|
114
|
+
for i, n in enumerate(names):
|
|
115
|
+
u = raw.iloc[i, 1]
|
|
116
|
+
units[n] = "" if pd.isna(u) else str(u)
|
|
117
|
+
if not any(units.values()):
|
|
118
|
+
units = {}
|
|
119
|
+
return labels, units
|
lessPy/X.py
ADDED
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
# X.py — analog of X.R
|
|
2
|
+
#
|
|
3
|
+
# X(): the one-variable analytic view — the distribution of a
|
|
4
|
+
# single NUMERICAL variable. Categorical variables belong to
|
|
5
|
+
# Chart(). As with Chart(), the pipeline is ported, not the lines:
|
|
6
|
+
# X.R's NSE, legacy parameters, base-R/lattice paths, and PDF
|
|
7
|
+
# device code have no Python counterpart.
|
|
8
|
+
#
|
|
9
|
+
# Implemented forms: "histogram" (default), "freq_poly",
|
|
10
|
+
# "density", and the VBS family ("violin", "box", "strip", "bs",
|
|
11
|
+
# "vbs") via the designed vbs_plotly renderer (lessR renders VBS
|
|
12
|
+
# through lattice only, so that renderer is a design, not a
|
|
13
|
+
# translation), with by= support throughout and facet= for every
|
|
14
|
+
# form: VBS (one band per level), histogram (~ .bar.lattice),
|
|
15
|
+
# and density and freq_poly (~ .plt.dist.facet). by= combines
|
|
16
|
+
# with facet=: one series per group within each panel or band.
|
|
17
|
+
# All forms of X.R are ported.
|
|
18
|
+
#
|
|
19
|
+
# facet= orthogonality (July 2026, ~ the R facet unification):
|
|
20
|
+
# facet= takes a column name, a list of two names (a row x
|
|
21
|
+
# column grid — rows the second variable; VBS instead stacks
|
|
22
|
+
# one band section per second-level, all on the shared x axis),
|
|
23
|
+
# or an aligned Series/array of computed values (the analog of
|
|
24
|
+
# R's facet expression). Density facets take the single-panel
|
|
25
|
+
# embellishments per panel (kind=, show_histogram, rug) when
|
|
26
|
+
# by= is absent, as R's .plt.dist.facet.
|
|
27
|
+
#
|
|
28
|
+
# Also ported (July 2026): cumulate= (with reg=) and counts=
|
|
29
|
+
# for the histogram; kind= (normal curve + Shapiro-Wilk
|
|
30
|
+
# console report), show_histogram/fill_hist, and rug (rug
|
|
31
|
+
# styling implies the density form, X.R:136-138) for the
|
|
32
|
+
# density — per panel with facet=, as in R (by= raises,
|
|
33
|
+
# except show_histogram which quietly does not draw).
|
|
34
|
+
# VBS: box_adj (+a, b) medcouple-adjusted fences, bw_iter
|
|
35
|
+
# violin bandwidth search (now the default bandwidth, as R),
|
|
36
|
+
# out_cut/ID/ID_size outlier labels above the strip.
|
|
37
|
+
# n_row/n_col lay the histogram/density/freq_poly facets out
|
|
38
|
+
# as a grid (lattice bottom-up fill); not for the VBS bands.
|
|
39
|
+
# Not ported: aspect= (panel aspect is figure sizing in
|
|
40
|
+
# plotly), the axis-format/margin family, add= annotations,
|
|
41
|
+
# themes.
|
|
42
|
+
|
|
43
|
+
import math
|
|
44
|
+
|
|
45
|
+
import numpy as np
|
|
46
|
+
import pandas as pd
|
|
47
|
+
from scipy import stats as sps
|
|
48
|
+
|
|
49
|
+
from .dn_plotly import dn_plotly
|
|
50
|
+
from .freq_poly_plotly import freq_poly_plotly
|
|
51
|
+
from .hs_plotly import hs_plotly
|
|
52
|
+
from .plotly_utils import BASE_COLORS, axis_format, font_scaled
|
|
53
|
+
from .plt_add import plt_add
|
|
54
|
+
from .stats_out import resolve_quiet, x_stats
|
|
55
|
+
from .vbs_plotly import vbs_plotly
|
|
56
|
+
from .utils import (
|
|
57
|
+
bw_nrd0, category_order, get_column, get_option, pretty,
|
|
58
|
+
resolve_facet,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# the VBS family: single-element v/b/s, and the composites. "vb"
|
|
62
|
+
# and "vs" are the two-element combos R reaches only through the
|
|
63
|
+
# (deprecated) vbs_plot= subset string; here they are just forms.
|
|
64
|
+
_VBS_FORMS = ("violin", "box", "strip", "vb", "vs", "bs", "vbs")
|
|
65
|
+
_FORMS = ("histogram", "freq_poly", "density") + _VBS_FORMS
|
|
66
|
+
_STATS = ("count", "proportion", "density")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _breaks_from_args(fx, bin_start, bin_width, bin_end, breaks):
|
|
70
|
+
"""Bin edges: explicit bin_* arguments win; otherwise Sturges
|
|
71
|
+
via pretty(), the policy of R's hist()."""
|
|
72
|
+
lo, hi = float(fx.min()), float(fx.max())
|
|
73
|
+
if bin_width is not None or bin_start is not None \
|
|
74
|
+
or bin_end is not None:
|
|
75
|
+
if bin_width is None:
|
|
76
|
+
nb = max(1, math.ceil(math.log2(len(fx))) + 1)
|
|
77
|
+
step = pretty(lo, hi, nb)
|
|
78
|
+
bin_width = step[1] - step[0]
|
|
79
|
+
start = lo if bin_start is None else float(bin_start)
|
|
80
|
+
end = hi if bin_end is None else float(bin_end)
|
|
81
|
+
edges = [start]
|
|
82
|
+
while edges[-1] < end - 1e-12:
|
|
83
|
+
edges.append(edges[-1] + float(bin_width))
|
|
84
|
+
if edges[-1] < hi: # cover the max even past bin_end
|
|
85
|
+
edges.append(edges[-1] + float(bin_width))
|
|
86
|
+
return edges
|
|
87
|
+
if isinstance(breaks, (list, tuple, np.ndarray)):
|
|
88
|
+
return [float(b) for b in breaks]
|
|
89
|
+
if isinstance(breaks, (int, float)):
|
|
90
|
+
return pretty(lo, hi, int(breaks))
|
|
91
|
+
if str(breaks).lower() != "sturges":
|
|
92
|
+
raise ValueError('breaks must be "Sturges", a number of '
|
|
93
|
+
"bins, or a list of edges")
|
|
94
|
+
nb = max(1, math.ceil(math.log2(len(fx))) + 1)
|
|
95
|
+
return pretty(lo, hi, nb)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _x_finish(fig, rotate_x, rotate_y, scale_x, axis_fmt,
|
|
99
|
+
axis_x_pre, digits_d):
|
|
100
|
+
"""Post-render axis adjustments shared by every X() form:
|
|
101
|
+
scale_x explicit ticks/range, rotate_x/rotate_y tick
|
|
102
|
+
angles. R analogs: scale_x, rotate_x, rotate_y."""
|
|
103
|
+
if scale_x is not None:
|
|
104
|
+
tv = np.linspace(float(scale_x[0]), float(scale_x[1]),
|
|
105
|
+
int(scale_x[2]))
|
|
106
|
+
fig.update_xaxes(
|
|
107
|
+
tickvals=tv,
|
|
108
|
+
ticktext=axis_format(tv, digits_d, axis_fmt,
|
|
109
|
+
axis_x_pre),
|
|
110
|
+
range=[float(scale_x[0]), float(scale_x[1])])
|
|
111
|
+
if rotate_x:
|
|
112
|
+
fig.update_xaxes(tickangle=-float(rotate_x))
|
|
113
|
+
if rotate_y:
|
|
114
|
+
fig.update_yaxes(tickangle=-float(rotate_y))
|
|
115
|
+
return fig
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def X(x, by=None, facet=None, data=None, filter=None,
|
|
119
|
+
form="histogram",
|
|
120
|
+
stat="count",
|
|
121
|
+
fill=None, color=None,
|
|
122
|
+
bin_start=None, bin_width=None, bin_end=None,
|
|
123
|
+
breaks="Sturges",
|
|
124
|
+
counts=False, cumulate="off", reg="snow2",
|
|
125
|
+
bandwidth=None, adjust=1, full_curve=True, area_fill="on",
|
|
126
|
+
kind="general", show_histogram=True,
|
|
127
|
+
fill_normal=None, color_normal="gray20", fill_hist=None,
|
|
128
|
+
rug=False, color_rug="black", size_rug=0.5,
|
|
129
|
+
position="overlay",
|
|
130
|
+
bw=None, bw_iter=10, vbs_ratio=1.1, vbs_pt_fill="black",
|
|
131
|
+
violin_fill=None, box_fill=None,
|
|
132
|
+
vbs_mean=False, fences=False, k=1.5,
|
|
133
|
+
box_adj=False, a=-4, b=3,
|
|
134
|
+
out_cut=0, ID=None, ID_size=0.6,
|
|
135
|
+
jitter_x=None, jitter_y=None,
|
|
136
|
+
pt_size=None, out_size=None, out_shape="circle", pt_shape=None,
|
|
137
|
+
n_row=None, n_col=None,
|
|
138
|
+
add=None, x1=None, y1=None, x2=None, y2=None,
|
|
139
|
+
axis_fmt="K", axis_x_pre="", axis_y_pre="",
|
|
140
|
+
rotate_x=0, rotate_y=0, scale_x=None,
|
|
141
|
+
xlab=None, ylab=None, main=None, digits_d=None,
|
|
142
|
+
quiet=None):
|
|
143
|
+
"""Analytic view of the distribution of one numerical variable,
|
|
144
|
+
optionally grouped (by=). Variables are strings naming columns
|
|
145
|
+
of the DataFrame `data`. Returns a plotly Figure.
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
if form not in _FORMS:
|
|
149
|
+
raise ValueError(f"form must be one of {_FORMS}")
|
|
150
|
+
if stat not in _STATS:
|
|
151
|
+
raise ValueError(f"stat must be one of {_STATS}")
|
|
152
|
+
if data is None:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
"data= is required: a pandas DataFrame containing the "
|
|
155
|
+
"named columns")
|
|
156
|
+
if stat == "density":
|
|
157
|
+
if form == "freq_poly":
|
|
158
|
+
stat = "count" # density is not a freq poly stat (X.R)
|
|
159
|
+
else:
|
|
160
|
+
form = "density" # R analog: X.R stat="density"
|
|
161
|
+
|
|
162
|
+
if cumulate not in ("off", "on", "both"):
|
|
163
|
+
raise ValueError('cumulate: "off", "on", or "both"')
|
|
164
|
+
if kind not in ("general", "normal", "both"):
|
|
165
|
+
raise ValueError('kind: "general", "normal", or "both"')
|
|
166
|
+
# a rug, or rug styling, implies a density plot (X.R:136-138)
|
|
167
|
+
if color_rug != "black" or size_rug != 0.5:
|
|
168
|
+
rug = True
|
|
169
|
+
if rug and form != "density":
|
|
170
|
+
form = "density"
|
|
171
|
+
if cumulate != "off" and (by is not None
|
|
172
|
+
or facet is not None):
|
|
173
|
+
raise ValueError(
|
|
174
|
+
"cumulate applies to a single histogram: "
|
|
175
|
+
"no by= or facet=")
|
|
176
|
+
if counts and (by is not None or facet is not None):
|
|
177
|
+
raise ValueError(
|
|
178
|
+
"counts labels apply to a single histogram: "
|
|
179
|
+
"no by= or facet=")
|
|
180
|
+
if (kind != "general" or rug) and by is not None:
|
|
181
|
+
raise ValueError(
|
|
182
|
+
"kind and rug draw per-panel curves of a single "
|
|
183
|
+
"series: no by=")
|
|
184
|
+
if (out_cut > 0 or box_adj) and form not in _VBS_FORMS:
|
|
185
|
+
raise ValueError(
|
|
186
|
+
"out_cut and box_adj apply to the VBS forms: "
|
|
187
|
+
'violin, box, strip, "vb", "vs", "bs", "vbs"')
|
|
188
|
+
|
|
189
|
+
if axis_fmt not in ("K", ",", ".", ""):
|
|
190
|
+
raise ValueError('axis_fmt: "K", ",", ".", or ""')
|
|
191
|
+
if scale_x is not None and facet is not None:
|
|
192
|
+
raise ValueError(
|
|
193
|
+
"scale_x applies to a single panel: no facet=")
|
|
194
|
+
|
|
195
|
+
if add is not None:
|
|
196
|
+
# R draws X() annotations in hst.main only, n.by == 1
|
|
197
|
+
if form != "histogram" or by is not None \
|
|
198
|
+
or facet is not None:
|
|
199
|
+
raise ValueError(
|
|
200
|
+
"add= annotations apply to a single-panel "
|
|
201
|
+
"histogram in X()")
|
|
202
|
+
|
|
203
|
+
# facet grid layout (the lattice n_row/n_col); aspect= is
|
|
204
|
+
# not ported — panel aspect is figure sizing in plotly
|
|
205
|
+
if (n_row is not None or n_col is not None) and facet is None:
|
|
206
|
+
raise ValueError("n_row and n_col lay out facet panels: "
|
|
207
|
+
"specify facet=")
|
|
208
|
+
if ((n_row is not None or n_col is not None)
|
|
209
|
+
and form in _VBS_FORMS):
|
|
210
|
+
raise ValueError(
|
|
211
|
+
"the plotly VBS draws facet levels as bands on one "
|
|
212
|
+
"panel, so n_row and n_col do not apply")
|
|
213
|
+
|
|
214
|
+
if filter is not None:
|
|
215
|
+
data = data.query(filter)
|
|
216
|
+
|
|
217
|
+
x_ser = get_column(data, x, "x")
|
|
218
|
+
if not pd.api.types.is_numeric_dtype(x_ser):
|
|
219
|
+
raise TypeError(
|
|
220
|
+
f"X() analyzes the distribution of a numerical "
|
|
221
|
+
f"variable, but '{x}' is {x_ser.dtype}. For a "
|
|
222
|
+
"categorical variable use Chart().")
|
|
223
|
+
by_ser = get_column(data, by, "by") if by is not None else None
|
|
224
|
+
(facet_ser, facet_name,
|
|
225
|
+
facet2_ser, facet2_name) = resolve_facet(data, facet, "X")
|
|
226
|
+
|
|
227
|
+
used = [s for s in (x_ser, by_ser, facet_ser, facet2_ser)
|
|
228
|
+
if s is not None]
|
|
229
|
+
keep = ~pd.concat(used, axis=1).isna().any(axis=1)
|
|
230
|
+
x_ser = x_ser[keep]
|
|
231
|
+
if by_ser is not None:
|
|
232
|
+
by_ser = by_ser[keep]
|
|
233
|
+
if facet_ser is not None:
|
|
234
|
+
facet_ser = facet_ser[keep]
|
|
235
|
+
if facet2_ser is not None:
|
|
236
|
+
facet2_ser = facet2_ser[keep]
|
|
237
|
+
|
|
238
|
+
fx = x_ser.to_numpy(dtype=float)
|
|
239
|
+
fx = fx[np.isfinite(fx)]
|
|
240
|
+
if len(fx) == 0:
|
|
241
|
+
raise ValueError(f"'{x}' has no finite values")
|
|
242
|
+
|
|
243
|
+
if not resolve_quiet(quiet): # accompanying statistics
|
|
244
|
+
print("\n".join(x_stats(
|
|
245
|
+
x_ser.to_numpy(dtype=float), x,
|
|
246
|
+
2 if digits_d is None else digits_d)))
|
|
247
|
+
if form == "density" and kind in ("normal", "both"):
|
|
248
|
+
if 2 < len(fx) < 5000: # R dn.main.R range
|
|
249
|
+
W, p = sps.shapiro(fx)
|
|
250
|
+
print("\nNull hypothesis is a normal population")
|
|
251
|
+
print("Shapiro-Wilk normality test: "
|
|
252
|
+
f"W = {W:.4f}, p-value = {p:.4f}")
|
|
253
|
+
else:
|
|
254
|
+
print("\nSample size out of range for the "
|
|
255
|
+
"Shapiro-Wilk normality test")
|
|
256
|
+
|
|
257
|
+
# ----- violin / box / strip (VBS) ----------------------------
|
|
258
|
+
if form in _VBS_FORMS:
|
|
259
|
+
vbs_plot = {"violin": "v", "box": "b",
|
|
260
|
+
"strip": "s"}.get(form, form)
|
|
261
|
+
fin = np.isfinite(x_ser.to_numpy(dtype=float))
|
|
262
|
+
by_arr = by_order = None
|
|
263
|
+
facet_arr = facet_order = None
|
|
264
|
+
facet2_arr = facet2_order = None
|
|
265
|
+
if facet_ser is not None:
|
|
266
|
+
facet_arr = facet_ser.to_numpy()[fin]
|
|
267
|
+
facet_order = category_order(facet_ser)
|
|
268
|
+
# per-panel box hues, ~ .plt.fill "hues" for facets,
|
|
269
|
+
# unless box_fill= names the fill(s) explicitly
|
|
270
|
+
if box_fill is None:
|
|
271
|
+
box_fill = [BASE_COLORS[i % len(BASE_COLORS)]
|
|
272
|
+
for i in range(len(facet_order))]
|
|
273
|
+
if facet2_ser is not None:
|
|
274
|
+
facet2_arr = facet2_ser.to_numpy()[fin]
|
|
275
|
+
facet2_order = category_order(facet2_ser)
|
|
276
|
+
if by_ser is not None:
|
|
277
|
+
# by= colors points per group; vbs_pt_fill does not
|
|
278
|
+
# apply, as in the X.R n.by > 1 branch
|
|
279
|
+
by_arr = by_ser.to_numpy()[fin]
|
|
280
|
+
by_order = category_order(by_ser)
|
|
281
|
+
n_g = len(by_order)
|
|
282
|
+
pt_fill = [BASE_COLORS[i % len(BASE_COLORS)]
|
|
283
|
+
for i in range(n_g)]
|
|
284
|
+
pt_color = pt_fill
|
|
285
|
+
pt_trans = get_option("trans_pt_fill", 0.10)
|
|
286
|
+
# strip-point colors per vbs_pt_fill, X.R VBS section
|
|
287
|
+
elif vbs_pt_fill == "black":
|
|
288
|
+
pt_fill, pt_color, pt_trans = "black", "black", 0.10
|
|
289
|
+
elif vbs_pt_fill == "default":
|
|
290
|
+
pt_fill, pt_color = "black", "black"
|
|
291
|
+
pt_trans = get_option("trans_pt_fill", 0.10)
|
|
292
|
+
else:
|
|
293
|
+
pt_fill = vbs_pt_fill
|
|
294
|
+
pt_color = get_option("pt_color", "#324E5C")
|
|
295
|
+
pt_trans = get_option("trans_pt_fill", 0.10)
|
|
296
|
+
if fill is not None:
|
|
297
|
+
pt_fill = fill
|
|
298
|
+
if color is not None:
|
|
299
|
+
pt_color = color
|
|
300
|
+
ids = None
|
|
301
|
+
if out_cut > 0: # outlier labels
|
|
302
|
+
ids = np.asarray(
|
|
303
|
+
get_column(data, ID, "ID")[keep]
|
|
304
|
+
if ID is not None
|
|
305
|
+
else x_ser.index).astype(str)[fin]
|
|
306
|
+
return _x_finish(vbs_plotly(
|
|
307
|
+
fx, x_name=x, vbs_plot=vbs_plot,
|
|
308
|
+
by=by_arr, by_order=by_order, by_name=by,
|
|
309
|
+
facet=facet_arr, facet_order=facet_order,
|
|
310
|
+
facet_name=facet_name,
|
|
311
|
+
facet2=facet2_arr, facet2_order=facet2_order,
|
|
312
|
+
facet2_name=facet2_name,
|
|
313
|
+
violin_fill=violin_fill, box_fill=box_fill,
|
|
314
|
+
bw=bw, vbs_ratio=vbs_ratio,
|
|
315
|
+
pt_fill=pt_fill, pt_color=pt_color, pt_trans=pt_trans,
|
|
316
|
+
pt_size=pt_size, out_size=out_size,
|
|
317
|
+
out_shape=out_shape, shape=pt_shape,
|
|
318
|
+
vbs_mean=vbs_mean, fences=fences, k_iqr=k,
|
|
319
|
+
box_adj=box_adj, a=a, b=b, bw_iter=bw_iter,
|
|
320
|
+
out_cut=out_cut, ids=ids, ID_size=ID_size,
|
|
321
|
+
jitter_x=jitter_x, jitter_y=jitter_y,
|
|
322
|
+
x_lab=x if xlab is None else xlab, main=main,
|
|
323
|
+
digits_d=2 if digits_d is None else digits_d,
|
|
324
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
325
|
+
), rotate_x, rotate_y, scale_x, axis_fmt, axis_x_pre,
|
|
326
|
+
2 if digits_d is None else digits_d)
|
|
327
|
+
|
|
328
|
+
# facet for histogram/density: aligned array + level order
|
|
329
|
+
facet_arr = (facet_ser.to_numpy()
|
|
330
|
+
if facet_ser is not None else None)
|
|
331
|
+
f_order = (category_order(facet_ser)
|
|
332
|
+
if facet_ser is not None else None)
|
|
333
|
+
facet2_arr = (facet2_ser.to_numpy()
|
|
334
|
+
if facet2_ser is not None else None)
|
|
335
|
+
f2_order = (category_order(facet2_ser)
|
|
336
|
+
if facet2_ser is not None else None)
|
|
337
|
+
# resolve the panel grid: explicit n_col wins, else derive
|
|
338
|
+
# from n_row; default one column, the established layout
|
|
339
|
+
if f_order is not None:
|
|
340
|
+
if n_col is None:
|
|
341
|
+
n_col = (math.ceil(len(f_order) / int(n_row))
|
|
342
|
+
if n_row is not None else 1)
|
|
343
|
+
n_col = max(1, int(n_col))
|
|
344
|
+
else:
|
|
345
|
+
n_col = 1
|
|
346
|
+
|
|
347
|
+
# ----- frequency polygon -------------------------------------
|
|
348
|
+
if form == "freq_poly":
|
|
349
|
+
edges = _breaks_from_args(fx, bin_start, bin_width,
|
|
350
|
+
bin_end, breaks)
|
|
351
|
+
return _x_finish(freq_poly_plotly(
|
|
352
|
+
x_ser, by=by_ser, x_name=x, by_name=by,
|
|
353
|
+
facet=facet_arr, facet_order=f_order,
|
|
354
|
+
facet_name=facet_name,
|
|
355
|
+
facet2=facet2_arr, facet2_order=f2_order,
|
|
356
|
+
facet2_name=facet2_name,
|
|
357
|
+
breaks=edges, proportion=stat == "proportion",
|
|
358
|
+
fill=fill,
|
|
359
|
+
fill_area=area_fill not in ("off", "transparent"),
|
|
360
|
+
x_lab=x if xlab is None else xlab, y_lab=ylab,
|
|
361
|
+
main=main, n_col=n_col,
|
|
362
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
363
|
+
axis_y_pre=axis_y_pre,
|
|
364
|
+
digits_d=3 if digits_d is None else digits_d,
|
|
365
|
+
), rotate_x, rotate_y, scale_x, axis_fmt, axis_x_pre,
|
|
366
|
+
3 if digits_d is None else digits_d)
|
|
367
|
+
|
|
368
|
+
# ----- density ---------------------------------------------
|
|
369
|
+
if form == "density":
|
|
370
|
+
bw = bandwidth if bandwidth is not None else bw_nrd0(fx)
|
|
371
|
+
show_h = show_histogram and by is None
|
|
372
|
+
return _x_finish(dn_plotly(
|
|
373
|
+
x_ser, by=by_ser, x_name=x, by_name=by,
|
|
374
|
+
facet=facet_arr, facet_order=f_order,
|
|
375
|
+
facet_name=facet_name,
|
|
376
|
+
facet2=facet2_arr, facet2_order=f2_order,
|
|
377
|
+
facet2_name=facet2_name,
|
|
378
|
+
fill=fill,
|
|
379
|
+
x_lab=x if xlab is None else xlab,
|
|
380
|
+
y_lab="Density" if ylab is None else ylab,
|
|
381
|
+
main=main,
|
|
382
|
+
bw=bw, adjust=adjust, full_curve=full_curve,
|
|
383
|
+
fill_area=area_fill not in ("off", "transparent"),
|
|
384
|
+
kind=kind, fill_normal=fill_normal,
|
|
385
|
+
color_normal=color_normal,
|
|
386
|
+
show_histogram=show_h, fill_hist=fill_hist,
|
|
387
|
+
hist_edges=(_breaks_from_args(fx, bin_start,
|
|
388
|
+
bin_width, bin_end,
|
|
389
|
+
breaks)
|
|
390
|
+
if show_h else None),
|
|
391
|
+
rug=rug, color_rug=color_rug, size_rug=size_rug,
|
|
392
|
+
n_col=n_col,
|
|
393
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
394
|
+
), rotate_x, rotate_y, scale_x, axis_fmt, axis_x_pre,
|
|
395
|
+
3 if digits_d is None else digits_d)
|
|
396
|
+
|
|
397
|
+
# ----- histogram ---------------------------------------------
|
|
398
|
+
edges = _breaks_from_args(fx, bin_start, bin_width, bin_end,
|
|
399
|
+
breaks)
|
|
400
|
+
proportion = stat == "proportion"
|
|
401
|
+
if ylab is None:
|
|
402
|
+
prefix = "Proportion of" if proportion else "Count of"
|
|
403
|
+
ylab = f"{prefix} {x}"
|
|
404
|
+
fig = hs_plotly(
|
|
405
|
+
x_ser, by=by_ser, x_name=x, by_name=by,
|
|
406
|
+
facet=facet_arr, facet_order=f_order, facet_name=facet_name,
|
|
407
|
+
facet2=facet2_arr, facet2_order=f2_order,
|
|
408
|
+
facet2_name=facet2_name,
|
|
409
|
+
breaks=edges, freq=True, proportion=proportion,
|
|
410
|
+
fill=fill, border=color,
|
|
411
|
+
cumulate=cumulate, reg=reg, counts=counts,
|
|
412
|
+
x_lab=x if xlab is None else xlab, y_lab=ylab,
|
|
413
|
+
digits_d=2 if digits_d is None else digits_d,
|
|
414
|
+
position=position, main=main, n_col=n_col,
|
|
415
|
+
axis_fmt=axis_fmt, axis_x_pre=axis_x_pre,
|
|
416
|
+
axis_y_pre=axis_y_pre,
|
|
417
|
+
)
|
|
418
|
+
_x_finish(fig, rotate_x, rotate_y, scale_x, axis_fmt,
|
|
419
|
+
axis_x_pre, 2 if digits_d is None else digits_d)
|
|
420
|
+
if add is not None:
|
|
421
|
+
plt_add(fig, add, x1=x1, x2=x2, y1=y1, y2=y2)
|
|
422
|
+
return fig
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
# font_size= scales all text of the returned figure
|
|
426
|
+
X = font_scaled(X)
|