lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/pie_plotly.py ADDED
@@ -0,0 +1,292 @@
1
+ # pie_plotly.py — analog of piechart.plotly.R
2
+ #
3
+ # Renders a donut/pie from ALREADY-TABULATED input:
4
+ # 1-D pandas Series: index = slice names, values = counts/means
5
+ # -> one pie
6
+ # 2-D pandas DataFrame: rows = `by` levels, columns = slices
7
+ # -> a grid of pies, one per by level, group name in each hole
8
+ #
9
+ # Deliberate deviations from the R source:
10
+ # - The theme-dependent sequential fill branch (.scale.clr) is not
11
+ # ported; without a theme system, default fills are BASE_COLORS,
12
+ # the equivalent of lessR's default "colors" theme (hues).
13
+ # - R's defensive plotly_build() post-processing of slice borders
14
+ # works around R-plotly quirks; plotly.py sets marker.line
15
+ # directly, so the same result needs no post-build step.
16
+
17
+ import math
18
+
19
+ import numpy as np
20
+ import pandas as pd
21
+ import plotly.graph_objects as go
22
+
23
+ from .utils import get_option
24
+ from .plotly_utils import (
25
+ BASE_COLORS, as_plotly_color, auto_text_color, make_trans,
26
+ plotly_style, to_hex,
27
+ )
28
+
29
+ _LABEL_VALUES = ("%", "input", "prop", "off")
30
+
31
+
32
+ def _pick_by_names_or_recycle(v, needed_names):
33
+ """dict: pick by slice name; else recycle across slices.
34
+ R analog: .pick_by_names_or_recycle()"""
35
+ n = len(needed_names)
36
+ if isinstance(v, dict):
37
+ base = list(v.values()) or ["black"]
38
+ return [v.get(nm, base[i % len(base)])
39
+ for i, nm in enumerate(needed_names)]
40
+ if v is None:
41
+ return ["black"] * n
42
+ if not isinstance(v, (list, tuple)):
43
+ v = [v]
44
+ return [v[i % len(v)] for i in range(n)]
45
+
46
+
47
+ def _num_str(vals, total, labels, digits_d, labels_decimals=None):
48
+ """labels_decimals sets the decimal places; None falls back to
49
+ digits_d, except for "prop", whose values are proportions that
50
+ digits_d (0 for counts) would flatten to "0"."""
51
+ denom = max(total, 1e-12)
52
+ digits_d = (int(labels_decimals) if labels_decimals is not None
53
+ else (2 if labels == "prop" else digits_d))
54
+ if labels == "%":
55
+ return [f"{100 * v / denom:.{digits_d}f}%" for v in vals]
56
+ if labels == "prop":
57
+ return [f"{v / denom:.{digits_d}f}" for v in vals]
58
+ if labels == "input":
59
+ return [f"{v:.{digits_d}f}" for v in vals]
60
+ return ["" for _ in vals] # "off"
61
+
62
+
63
+ def _slice_text(slices, num_str, labels):
64
+ if labels == "off":
65
+ return list(slices)
66
+ return [f"{s}<br>{n}" for s, n in zip(slices, num_str)]
67
+
68
+
69
+ def _text_colors(labels_color, fills_rgba, panel_fill):
70
+ if labels_color is None or labels_color == "adjust":
71
+ return auto_text_color(fills_rgba, bg=panel_fill)
72
+ return [labels_color] * len(fills_rgba)
73
+
74
+
75
+ def pie_plotly(x, x_name=None, y_name=None, by_name=None, main=None,
76
+ fill=None, border=None, opacity=1.0,
77
+ hole=0.65, ncols=None,
78
+ labels=None, labels_position="in",
79
+ labels_color=None, labels_size=1.0,
80
+ labels_decimals=None,
81
+ digits_d=2,
82
+ group_labels=True, group_label_size=14,
83
+ style_opts=None):
84
+
85
+ if labels is None:
86
+ labels = "%" # R analog: match.arg default
87
+ if labels not in _LABEL_VALUES:
88
+ raise ValueError(f"labels must be one of {_LABEL_VALUES}")
89
+
90
+ two_d = isinstance(x, pd.DataFrame)
91
+ if not two_d and not isinstance(x, pd.Series):
92
+ raise TypeError("x must be a pandas Series (1-D) or "
93
+ "DataFrame (2-D, rows = by levels)")
94
+
95
+ if x_name is None:
96
+ x_name = (x.columns.name if two_d else x.index.name) or "x"
97
+ if y_name is None:
98
+ y_name = "Count" if two_d else (x.name or "Count")
99
+ if two_d and by_name is None:
100
+ by_name = x.index.name or "Group"
101
+ if style_opts is None:
102
+ style_opts = plotly_style()
103
+ if fill is None:
104
+ fill = BASE_COLORS
105
+ if border is None:
106
+ border = "transparent"
107
+
108
+ title_size = round(16 * get_option("main_size", 1))
109
+ alpha_fill = float(opacity)
110
+ if not math.isfinite(alpha_fill):
111
+ alpha_fill = 1.0
112
+ alpha_fill = max(0.0, min(1.0, alpha_fill))
113
+
114
+ # None = the default (R's labels_position %||% "in")
115
+ pos_in = labels_position is None or \
116
+ str(labels_position).lower() == "in"
117
+ txt_pos = "inside" if pos_in else "outside"
118
+ val_spec = f":.{max(0, int(digits_d))}f"
119
+
120
+ def title_layout():
121
+ if main:
122
+ return dict(text=main, y=0.94, yanchor="top",
123
+ font=dict(
124
+ size=title_size,
125
+ color=to_hex(get_option("lab_color",
126
+ "black"))))
127
+ return None
128
+
129
+ def panel_bg(fig):
130
+ # simulated panel background; only when not white
131
+ bg_plot = str(to_hex(style_opts["panel_fill"])).upper()
132
+ bg_paper = str(to_hex(style_opts["window_fill"])).upper()
133
+ if bg_plot != "#FFFFFF" or bg_paper != "#FFFFFF":
134
+ fig.update_layout(paper_bgcolor=bg_paper)
135
+ fig.add_shape(type="rect", xref="paper", yref="paper",
136
+ x0=0, y0=0, x1=1, y1=1, layer="below",
137
+ fillcolor=bg_plot, line=dict(width=0))
138
+
139
+ # --- SINGLE PIE: 1-D --------------------------------------------
140
+ if not two_d:
141
+ slices = [str(s) for s in x.index]
142
+ values = x.to_numpy(dtype=float)
143
+ tot = np.nansum(values)
144
+
145
+ fill_vec = _pick_by_names_or_recycle(fill, slices)
146
+ border_vec = _pick_by_names_or_recycle(border, slices)
147
+
148
+ num_str = _num_str(values, tot, labels, digits_d,
149
+ labels_decimals)
150
+ text_vec = _slice_text(slices, num_str, labels)
151
+
152
+ txt_size = round(12 * 1.38 * labels_size *
153
+ get_option("axis_size", 0.9))
154
+
155
+ fills_rgba = make_trans(fill_vec, alpha_fill)
156
+ txt_colors = _text_colors(labels_color, fills_rgba,
157
+ style_opts["panel_fill"])
158
+
159
+ overall_pct = values / (tot if tot > 0 else 1)
160
+
161
+ # inside labels need room: clamp an over-large hole
162
+ hole_use = 0.62 if (pos_in and hole > 0.62) else hole
163
+
164
+ font = dict(color=txt_colors, size=txt_size)
165
+ fig = go.Figure(go.Pie(
166
+ labels=slices,
167
+ values=values,
168
+ sort=False,
169
+ direction="clockwise",
170
+ hole=hole_use,
171
+ domain=dict(x=[0, 1], y=[0.03, 0.93]),
172
+ text=text_vec,
173
+ textinfo="text",
174
+ textposition=txt_pos,
175
+ insidetextorientation="radial",
176
+ textfont=font,
177
+ insidetextfont=font,
178
+ outsidetextfont=font,
179
+ automargin=not pos_in,
180
+ marker=dict(
181
+ colors=fills_rgba,
182
+ line=dict(color=as_plotly_color(border_vec),
183
+ width=1.6),
184
+ ),
185
+ customdata=overall_pct,
186
+ hovertemplate=(
187
+ f"{x_name}: %{{label}}"
188
+ f"<br>{y_name}: %{{value{val_spec}}}"
189
+ "<br>% of total: %{customdata:.2%}"
190
+ "<extra></extra>"),
191
+ showlegend=False,
192
+ ))
193
+
194
+ fig.update_layout(
195
+ uniformtext=dict(minsize=8, mode="show"),
196
+ margin=dict(t=round(title_size * 2.2),
197
+ r=20, b=30, l=20),
198
+ title=title_layout(),
199
+ )
200
+ panel_bg(fig)
201
+ return fig
202
+
203
+ # --- GROUPED INPUT (2-D): PIE GRID ------------------------------
204
+ groups = [str(g) for g in x.index]
205
+ slices = [str(c) for c in x.columns]
206
+ mat = x.to_numpy(dtype=float)
207
+ grand_total = np.nansum(mat)
208
+
209
+ k = len(groups)
210
+ nc = (math.ceil(math.sqrt(k)) if ncols is None or ncols < 1
211
+ else int(ncols))
212
+ nr = math.ceil(k / nc)
213
+
214
+ txt_size = round(12 * labels_size * get_option("axis_size", 0.9))
215
+ by_title = by_name if by_name else "Group"
216
+
217
+ fig = go.Figure()
218
+
219
+ for i, grp in enumerate(groups):
220
+ col, row = i % nc, i // nc
221
+ x0, x1 = col / nc, (col + 1) / nc
222
+ y0, y1 = 1 - (row + 1) / nr, 1 - row / nr
223
+ shrink = 0.92
224
+ y_mid, y_half = (y0 + y1) / 2, (y1 - y0) * shrink / 2
225
+ dom = dict(x=[x0, x1], y=[y_mid - y_half, y_mid + y_half])
226
+
227
+ vals = mat[i].copy()
228
+ vals[~np.isfinite(vals)] = np.nan
229
+ val_sum = np.nansum(vals)
230
+ overall_pct = vals / (grand_total if grand_total > 0 else 1)
231
+
232
+ num_str = _num_str(vals, val_sum, labels, digits_d,
233
+ labels_decimals)
234
+ text_vec = _slice_text(slices, num_str, labels)
235
+
236
+ fills_this = make_trans(
237
+ _pick_by_names_or_recycle(fill, slices), alpha_fill)
238
+ borders_this = _pick_by_names_or_recycle(border, slices)
239
+ txt_colors = _text_colors(labels_color, fills_this,
240
+ style_opts["panel_fill"])
241
+
242
+ font = dict(color=txt_colors, size=txt_size)
243
+ fig.add_trace(go.Pie(
244
+ labels=slices,
245
+ values=vals,
246
+ name=f"{by_title}: {grp}",
247
+ legendgroup="pies",
248
+ sort=False,
249
+ direction="clockwise",
250
+ hole=hole,
251
+ domain=dom,
252
+ text=text_vec,
253
+ textinfo="text",
254
+ textposition=txt_pos,
255
+ insidetextorientation="radial",
256
+ textfont=font,
257
+ insidetextfont=font,
258
+ outsidetextfont=font,
259
+ automargin=not pos_in,
260
+ marker=dict(
261
+ colors=fills_this,
262
+ line=dict(color=as_plotly_color(borders_this),
263
+ width=1),
264
+ ),
265
+ customdata=overall_pct,
266
+ hovertemplate=(
267
+ f"{by_title}: {grp}"
268
+ f"<br>{x_name}: %{{label}}"
269
+ f"<br>{y_name}: %{{value{val_spec}}}"
270
+ f"<br>% of {grp}: %{{percent}}"
271
+ "<br>% of total: %{customdata:.2%}"
272
+ "<extra></extra>"),
273
+ showlegend=False,
274
+ ))
275
+
276
+ if group_labels and hole > 0:
277
+ fig.add_annotation(
278
+ x=(dom["x"][0] + dom["x"][1]) / 2,
279
+ y=(dom["y"][0] + dom["y"][1]) / 2,
280
+ xref="paper", yref="paper",
281
+ text=grp, showarrow=False,
282
+ xanchor="center", yanchor="middle",
283
+ font=dict(size=group_label_size, color="#666666"),
284
+ )
285
+
286
+ fig.update_layout(
287
+ uniformtext=dict(minsize=10, mode="show"),
288
+ margin=dict(t=round(title_size * 2.2), r=20, b=20, l=20),
289
+ title=title_layout(),
290
+ )
291
+ panel_bg(fig)
292
+ return fig
lessPy/pivot.py ADDED
@@ -0,0 +1,158 @@
1
+ # pivot.py — analog of pivot.R (the aggregation core).
2
+ #
3
+ # pivot(): aggregate a numeric variable over the categories of one
4
+ # or more `by` grouping variables, computing one or more summary
5
+ # statistics, or tabulate frequencies. The result is a long-form
6
+ # DataFrame with the by columns, an n (and na) count, and one
7
+ # {variable}_{stat} column per statistic — lessR's pivot output.
8
+ #
9
+ # Ported: the aggregation over by groups (sum, mean, median, min,
10
+ # max, sd, var, IQR, mad), the n/na counts and show_n, all group
11
+ # combinations (empty cells kept, as R's drop=FALSE), NA groups,
12
+ # sort= by the statistic, and the one- and two-way frequency
13
+ # table (compute="table"). Not ported: by_cols wide cross-tabs,
14
+ # table_prop row/col proportions, quantiles, and skew/kurtosis.
15
+ #
16
+ # The interface uses string names (compute="mean", variable=,
17
+ # by=), the lessPy convention.
18
+
19
+ import numpy as np
20
+ import pandas as pd
21
+
22
+ from .utils import get_column
23
+
24
+ # statistic name -> (column-name abbreviation, aggregator).
25
+ # pandas quantile is linear (R type 7); std/var use n-1; all skip
26
+ # NaN, so na_remove is the default. R analog: pivot.R fun.vec
27
+ _STAT = {
28
+ "sum": ("sum", lambda s: s.sum()),
29
+ "mean": ("mean", lambda s: s.mean()),
30
+ "median": ("mdn", lambda s: s.median()),
31
+ "min": ("min", lambda s: s.min()),
32
+ "max": ("max", lambda s: s.max()),
33
+ "sd": ("sd", lambda s: s.std(ddof=1)),
34
+ "var": ("var", lambda s: s.var(ddof=1)),
35
+ "IQR": ("IQR", lambda s: s.quantile(0.75)
36
+ - s.quantile(0.25)),
37
+ "mad": ("mad", lambda s:
38
+ 1.4826 * (s - s.median()).abs().median()),
39
+ }
40
+
41
+
42
+ def pivot(data, compute, variable=None, by=None, filter=None,
43
+ show_n=True, na_remove=True, sort=None, digits_d=None,
44
+ quiet=False):
45
+ """Aggregate a numeric variable over by groups, or tabulate
46
+ frequencies. compute is a statistic name or a list of names
47
+ ("mean", ["mean","sd"], "table"); variable is the numeric
48
+ column to aggregate; by is the grouping column(s). Returns a
49
+ long-form DataFrame. R analog: pivot()"""
50
+ if filter is not None:
51
+ data = data.query(filter)
52
+ computes = [compute] if isinstance(compute, str) \
53
+ else list(compute)
54
+ by = ([by] if isinstance(by, str)
55
+ else list(by) if by is not None else [])
56
+ if sort is not None and sort not in ("+", "-"):
57
+ raise ValueError('sort: "+" or "-"')
58
+
59
+ if "table" in computes:
60
+ if len(computes) > 1:
61
+ raise ValueError('compute="table" cannot be combined '
62
+ "with other statistics")
63
+ if not by:
64
+ raise ValueError('compute="table" needs by=')
65
+ return _pivot_table(data, by, show_n)
66
+
67
+ if variable is None:
68
+ raise ValueError("variable= is required: the numeric "
69
+ "column to aggregate")
70
+ unknown = [c for c in computes if c not in _STAT]
71
+ if unknown:
72
+ raise ValueError(
73
+ f"unknown compute {unknown}; use "
74
+ f"{', '.join(_STAT)}, or \"table\"")
75
+ v = get_column(data, variable, "variable")
76
+ if not pd.api.types.is_numeric_dtype(v):
77
+ raise TypeError(
78
+ f"the variable to aggregate '{variable}' is "
79
+ f"{v.dtype}: it must be numeric. Put categorical "
80
+ "variables in by=.")
81
+ if not by:
82
+ raise ValueError("by= is required: the grouping "
83
+ "column(s)")
84
+ return _pivot_agg(data, variable, by, computes, show_n, sort)
85
+
86
+
87
+ def _cat_frame(data, by):
88
+ """Copy of data with each by column made an ordered Categorical
89
+ (sorted levels), so groupby keeps every level combination and
90
+ the NA group, as R's factor()/drop=FALSE."""
91
+ g = data.copy()
92
+ for b in by:
93
+ col = g[b]
94
+ cats = sorted(pd.unique(col.dropna()),
95
+ key=lambda x: (str(type(x)), x))
96
+ g[b] = pd.Categorical(col, categories=cats)
97
+ return g
98
+
99
+
100
+ def _pivot_agg(data, variable, by, computes, show_n, sort):
101
+ g = _cat_frame(data, by)
102
+ grp = g.groupby(by, observed=False, dropna=False, sort=True)
103
+ v = grp[variable]
104
+ out = pd.DataFrame({
105
+ "n": v.apply(lambda s: int(s.notna().sum())),
106
+ "na": v.apply(lambda s: int(s.isna().sum()))})
107
+ for c in computes:
108
+ abbr, fn = _STAT[c]
109
+ out[f"{variable}_{abbr}"] = v.apply(fn)
110
+ out = out.reset_index()
111
+ # R aggregate row order: the FIRST by var varies fastest, so
112
+ # sort by the by columns in reverse listing order, NA last
113
+ out = out.sort_values(by[::-1], na_position="last",
114
+ kind="stable").reset_index(drop=True)
115
+ if sort is not None:
116
+ stat_col = out.columns[-1]
117
+ out = out.sort_values(
118
+ stat_col, ascending=(sort == "+"),
119
+ na_position="last", kind="stable"
120
+ ).reset_index(drop=True)
121
+ if not show_n:
122
+ out = out.drop(columns=["n", "na"])
123
+ return out
124
+
125
+
126
+ def _pivot_table(data, by, show_n):
127
+ if len(by) == 1:
128
+ b = by[0]
129
+ col = data[b]
130
+ cats = sorted(pd.unique(col.dropna()),
131
+ key=lambda x: (str(type(x)), x))
132
+ cc = pd.Categorical(col, categories=cats)
133
+ n = pd.Series(cc, name=b).value_counts(
134
+ dropna=False, sort=False)
135
+ # value_counts on Categorical excludes NaN; add it back
136
+ n = n.reindex(cats)
137
+ na_count = int(col.isna().sum())
138
+ idx = list(cats)
139
+ counts = [int(n[c]) for c in cats]
140
+ if na_count:
141
+ idx.append(np.nan)
142
+ counts.append(na_count)
143
+ total = sum(counts)
144
+ out = pd.DataFrame({b: idx, "n": counts})
145
+ out["Prop"] = np.round(np.array(counts) / total, 2)
146
+ return out
147
+ if len(by) > 2:
148
+ raise ValueError('compute="table" supports one or two '
149
+ "by variables")
150
+ # two-way: long-form counts, second by var first, as R
151
+ b1, b2 = by
152
+ g = _cat_frame(data, by)
153
+ ct = (g.groupby([b1, b2], observed=False, dropna=False)
154
+ .size().reset_index(name="n"))
155
+ ct = ct[[b2, b1, "n"]]
156
+ ct = ct.sort_values([b1, b2], na_position="last",
157
+ kind="stable").reset_index(drop=True)
158
+ return ct