lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/utils.py ADDED
@@ -0,0 +1,235 @@
1
+ # utils.py — analog of the general utilities in lessR zzz.R
2
+ #
3
+ # Includes the option (style) system that replaces R's getOption()
4
+ # calls, number formatting (.fmt), and an R-style pretty() for axis
5
+ # tick values, which lessR computes upstream of the render functions.
6
+
7
+ import math
8
+
9
+ # style settings; R analog: options() set by lessR style()
10
+ _OPTIONS = {
11
+ "main_size": 1.0,
12
+ "lab_size": 1.0,
13
+ "axis_size": 0.9,
14
+ "axis_color": "black",
15
+ "lab_color": "black",
16
+ "grid_color": "gray90",
17
+ "grid_col": "gray85",
18
+ "grid_lwd": 0.5,
19
+ "grid_lty": None,
20
+ "axis_lwd": 1,
21
+ "panel_border": "#808080",
22
+ "panel_lwd": 1,
23
+ "panel_fill": "white",
24
+ "window_fill": "white",
25
+ "digits_d": 2,
26
+ "pt_color": "#324E5C", # rgb(50,78,92), zzz_on.R
27
+ "trans_pt_fill": 0.10,
28
+ "segment_color": "gray40",
29
+ "fit_color": "#5C4032", # rgb(92,64,50), zzz_on.R
30
+ "fit_lwd": 2,
31
+ "se_fill": "#1A1A1A19",
32
+ "ellipse_fill": "#92806F28",
33
+ "ellipse_color": "gray20",
34
+ "ellipse_lwd": 1,
35
+ "violin_fill": "#7485975A",
36
+ "violin_color": "gray15",
37
+ "box_fill": "#419BD2", # rgb(65,155,210), zzz_on.R
38
+ "box_color": "gray15",
39
+ "out_fill": "#8B1A1A", # firebrick4
40
+ "out_color": "#8B1A1A",
41
+ "out2_fill": "#EE2C2C", # firebrick2
42
+ "out2_color": "#EE2C2C",
43
+ "strip_fill": "#7F7F7F37",
44
+ "strip_color": "gray40",
45
+ "strip_text_color": "gray15",
46
+ }
47
+
48
+
49
+ def get_option(name, default=None):
50
+ val = _OPTIONS.get(name, default)
51
+ return default if val is None else val
52
+
53
+
54
+ def set_option(name, value):
55
+ _OPTIONS[name] = value
56
+
57
+
58
+ def fmt(k, d=None):
59
+ """Format one number with d decimal digits. R analog: .fmt()"""
60
+ if d is None:
61
+ d = get_option("digits_d", 2)
62
+ return f"{k:.{d}f}"
63
+
64
+
65
+ # aggregation functions for stat=; sd uses n-1 as in R
66
+ # R analog: .stat_fun()
67
+ STAT_FUN = {
68
+ "mean": lambda s: s.mean(),
69
+ "sum": lambda s: s.sum(),
70
+ "sd": lambda s: s.std(ddof=1),
71
+ "min": lambda s: s.min(),
72
+ "median": lambda s: s.median(),
73
+ "max": lambda s: s.max(),
74
+ }
75
+
76
+ # display label for a stat= value; R analog: .stat_lbl()
77
+ STAT_LBL = {
78
+ "sum": "Sum",
79
+ "mean": "Mean",
80
+ "sd": "Standard Deviation",
81
+ "deviation": "Mean Deviation",
82
+ "min": "Minimum",
83
+ "median": "Median",
84
+ "max": "Maximum",
85
+ }
86
+
87
+
88
+ def category_order(s):
89
+ """Category order: declared order for a pandas Categorical,
90
+ else sorted unique values -- the same convention as an R factor
91
+ built by table(). Shared by Chart() and XY()."""
92
+ import pandas as pd
93
+ if isinstance(s.dtype, pd.CategoricalDtype):
94
+ return [c for c in s.cat.categories if c in set(s.dropna())]
95
+ return sorted(s.dropna().unique().tolist())
96
+
97
+
98
+ def get_column(data, name, arg):
99
+ """Resolve a string column name against a DataFrame, with the
100
+ errors the string interface implies. Shared by Chart() and X()."""
101
+ if not isinstance(name, str):
102
+ raise TypeError(
103
+ f"{arg} must be a string naming a column of data, "
104
+ f"not {type(name).__name__}. (Python has no equivalent "
105
+ "of R's unquoted variable names.)")
106
+ if name not in data.columns:
107
+ raise KeyError(
108
+ f"{arg}='{name}' is not a column of data. "
109
+ f"Columns: {', '.join(map(str, data.columns))}")
110
+ return data[name]
111
+
112
+
113
+ def facet_values(data, values, arg):
114
+ """Resolve a computed facet -- a pandas Series or numpy array
115
+ of values aligned with data, the Python analog of R's facet
116
+ expression. Length-checked against data, as R checks the
117
+ expression length. Returns an aligned Series."""
118
+ import numpy as np
119
+ import pandas as pd
120
+ if isinstance(values, pd.Series):
121
+ if len(values) != len(data):
122
+ raise ValueError(
123
+ f"data has {len(data)} rows, but the {arg} Series "
124
+ f"has {len(values)} values")
125
+ name = values.name if values.name else arg
126
+ return pd.Series(values.to_numpy(), index=data.index,
127
+ name=name)
128
+ if isinstance(values, np.ndarray):
129
+ if len(values) != len(data):
130
+ raise ValueError(
131
+ f"data has {len(data)} rows, but the {arg} array "
132
+ f"has {len(values)} values")
133
+ return pd.Series(values, index=data.index, name=arg)
134
+ raise TypeError(
135
+ f"{arg} must be a string naming a column of data, a "
136
+ f"list of such names, or a pandas Series / numpy array "
137
+ f"of computed values, not {type(values).__name__}")
138
+
139
+
140
+ def resolve_facet(data, facet, fun):
141
+ """Resolve facet= into up to two aligned Series:
142
+ (facet1, facet1_name, facet2, facet2_name). Accepts a column
143
+ name, a list/tuple of names (facet1 = panel columns, facet2 =
144
+ panel rows; more than two uses the first two, with a message),
145
+ or a Series/array of computed values (facet_values). R analog:
146
+ the --- resolve facet --- block of X.R / XY.R."""
147
+ if facet is None:
148
+ return None, None, None, None
149
+ if isinstance(facet, str):
150
+ return get_column(data, facet, "facet"), facet, None, None
151
+ if isinstance(facet, (list, tuple)):
152
+ if not all(isinstance(f, str) for f in facet):
153
+ raise TypeError(
154
+ "a facet list gives column names; for computed "
155
+ "values pass a single pandas Series or numpy "
156
+ "array")
157
+ if len(facet) == 0:
158
+ raise ValueError(
159
+ "facet names a categorical variable for its "
160
+ "panels")
161
+ if len(facet) > 2:
162
+ print(f"facet has {len(facet)} variables. {fun}() "
163
+ "uses the first two: "
164
+ f"{facet[0]}, {facet[1]}.")
165
+ f1 = get_column(data, facet[0], "facet")
166
+ if len(facet) == 1:
167
+ return f1, facet[0], None, None
168
+ f2 = get_column(data, facet[1], "facet")
169
+ return f1, facet[0], f2, facet[1]
170
+ ser = facet_values(data, facet, "facet")
171
+ return ser, str(ser.name), None, None
172
+
173
+
174
+ def kde(xg, grid, h):
175
+ """Gaussian KDE of xg evaluated on grid with bandwidth h.
176
+ Shared by the density and violin renderers."""
177
+ import numpy as np
178
+ z = (grid[:, None] - xg[None, :]) / h
179
+ return (np.exp(-0.5 * z * z).sum(axis=1)
180
+ / (len(xg) * h * float(np.sqrt(2 * np.pi))))
181
+
182
+
183
+ def bw_nrd0(x):
184
+ """Silverman rule-of-thumb KDE bandwidth, the default of R's
185
+ density(). R analog: stats::bw.nrd0()"""
186
+ import numpy as np
187
+ x = np.asarray(x, dtype=float)
188
+ x = x[np.isfinite(x)]
189
+ n = len(x)
190
+ if n < 2:
191
+ raise ValueError("bandwidth needs at least 2 data points")
192
+ sd = float(np.std(x, ddof=1))
193
+ iqr = float(np.subtract(*np.percentile(x, [75, 25])))
194
+ lo = min(sd, iqr / 1.34)
195
+ if lo == 0:
196
+ lo = sd or abs(float(x[0])) or 1.0
197
+ return 0.9 * lo * n ** -0.2
198
+
199
+
200
+ def band_width(x, bw_iter=10, n=512):
201
+ """Violin bandwidth: start at bw_nrd0 and widen by 10% per
202
+ iteration until the density curve has at most one direction
203
+ change (a single peak) or bw_iter iterations pass.
204
+ R analog: .band.width() (zzz.R)"""
205
+ import numpy as np
206
+ x = np.asarray(x, dtype=float)
207
+ x = x[np.isfinite(x)]
208
+ bw = bw_nrd0(x)
209
+ for _ in range(int(bw_iter)):
210
+ grid = np.linspace(x.min() - 3 * bw, x.max() + 3 * bw, n)
211
+ xd = np.diff(kde(x, grid, bw))
212
+ flips = int((np.sign(xd[1:]) != np.sign(xd[:-1])).sum())
213
+ if flips <= 1:
214
+ break
215
+ bw *= 1.1
216
+ return bw
217
+
218
+
219
+ def pretty(lo, hi, n=5):
220
+ """Nice tick values covering [lo, hi]. R analog: pretty()"""
221
+ if hi <= lo:
222
+ hi = lo + 1
223
+ raw = (hi - lo) / n
224
+ mag = 10 ** math.floor(math.log10(raw))
225
+ step = 10 * mag
226
+ for m in (1, 2, 5, 10):
227
+ if raw <= m * mag * (1 + 1e-10):
228
+ step = m * mag
229
+ break
230
+ start = math.floor(lo / step + 1e-10) * step
231
+ end = math.ceil(hi / step - 1e-10) * step
232
+ k = round((end - start) / step)
233
+ vals = [start + i * step for i in range(k + 1)]
234
+ # avoid -0.0 and float dust such as 0.30000000000000004
235
+ return [round(v, 10) + 0.0 for v in vals]