lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/__init__.py ADDED
@@ -0,0 +1,60 @@
1
+ # Python analog of NAMESPACE: names imported here are the public API.
2
+
3
+ from .bc_plotly import bc_plotly
4
+ from .bubble_plotly import bubble_plotly
5
+ from .Chart import Chart
6
+ from .dn_plotly import dn_plotly
7
+ from .dot_plotly import dot_plotly
8
+ from .freq_poly_plotly import freq_poly_plotly
9
+ from .hier_plotly import hier_plotly
10
+ from .hs_plotly import hs_plotly
11
+ from .Logit import Logit
12
+ from .pie_plotly import pie_plotly
13
+ from .plt_plotly import plt_plotly
14
+ from .radar_plotly import radar_plotly
15
+ from .Regression import Regression
16
+ from .utils import get_option, set_option
17
+ from .vbs_plotly import vbs_plotly
18
+ from .X import X
19
+ from .ANOVA import ANOVA
20
+ from .ttest import ttest
21
+ from .datasets import datasets, read_data
22
+ from .reshape import reshape_long, reshape_wide
23
+ from .pivot import pivot
24
+ from .corEFA import corEFA
25
+ from .corCFA import corCFA
26
+ from .corScree import corScree
27
+ from .corReorder import corReorder
28
+ from .corProp import corProp
29
+ from .Correlation import Correlation
30
+ from .corReflect import corReflect
31
+ from .corRead import corRead
32
+ from .Prop_test import Prop_test
33
+ from .corPrint import corPrint
34
+ from .Flows import Flows
35
+ from .date_infer import date_infer, format_date_labels
36
+ from .rename import rename
37
+ from .showColors import showColors
38
+ from .getColors import getColors
39
+ from .simCLT import simCLT
40
+ from .simMeans import simMeans
41
+ from .simFlips import simFlips
42
+ from .simCImean import simCImean
43
+ from .order_by import order_by
44
+ from .prob_norm import prob_norm
45
+ from .prob_znorm import prob_znorm
46
+ from .prob_tcut import prob_tcut
47
+ from .details import details
48
+ from .VariableLabels import VariableLabels
49
+ from .XY import XY
50
+
51
+ __version__ = "0.1.0"
52
+
53
+ __all__ = ["Chart", "X", "XY", "ANOVA", "ttest", "Regression", "Logit",
54
+ "read_data", "datasets", "reshape_long", "reshape_wide", "pivot", "corEFA", "corCFA", "corScree", "corReorder", "corProp", "Correlation", "corReflect", "corRead", "Prop_test", "corPrint", "Flows", "date_infer", "format_date_labels", "rename", "showColors", "getColors", "simCLT", "simMeans", "simFlips", "simCImean", "order_by", "prob_norm", "prob_znorm", "prob_tcut", "details",
55
+ "VariableLabels",
56
+ "bc_plotly", "bubble_plotly", "dn_plotly", "dot_plotly",
57
+ "freq_poly_plotly", "hier_plotly", "hs_plotly",
58
+ "pie_plotly", "plt_plotly", "radar_plotly",
59
+ "vbs_plotly",
60
+ "get_option", "set_option"]
lessPy/anova_rmd.py ADDED
@@ -0,0 +1,227 @@
1
+ # anova_rmd.py — Quarto (.qmd) report for ANOVA(), the analog of
2
+ # R's Rmd= for ANOVA (av.Rmd.R). Unlike the Regression report,
3
+ # av.Rmd.R uses no external prose templates and no toggles
4
+ # (explain/results are hardcoded on), so this generator is
5
+ # correspondingly simple. Same design choices as the other lessPy
6
+ # reports: Quarto target, direct per-section emitters, live code
7
+ # chunks for tables and plots (recomputed from the result object
8
+ # / data), narrative from R's prose. Deviation from av.Rmd.R: the
9
+ # cell-means plot is included (R's ANOVA report has no graphics).
10
+
11
+ from pathlib import Path
12
+
13
+ from .reg_rmd import _chunk, _maybe_render, _xAnd
14
+ from .utils import fmt
15
+
16
+
17
+ def anova_rmd(result, y_name, facs, formula_str, design,
18
+ data_cols, rmd, rmd_data, rmd_format, rmd_browser):
19
+ """Write a Quarto report reproducing the ANOVA. Returns the
20
+ path of the .qmd file written."""
21
+ Y = y_name
22
+ X = _xAnd(facs)
23
+ d = result.digits_d
24
+ data_name = "d"
25
+ read_expr = (f'pd.read_csv("{rmd_data}")' if rmd_data
26
+ else 'pd.read_csv("your_data.csv")')
27
+ call = f'ANOVA("{formula_str}", data={data_name})'
28
+
29
+ tx = _front_matter(Y)
30
+ tx += [
31
+ "The purpose of this analysis is the analysis of "
32
+ f"variance of the values of {Y} for the different "
33
+ f"levels of {X}.", ""]
34
+
35
+ # data
36
+ tx += ["## The Data", "",
37
+ "Read the data into a pandas data frame.", ""]
38
+ tx += _chunk(["import pandas as pd",
39
+ f"{data_name} = {read_expr}", data_name],
40
+ echo=True)
41
+ tx += ["Data from the following variables are available for "
42
+ f"analysis: {_xAnd(data_cols)}.", ""]
43
+
44
+ # analysis
45
+ tx += ["## Analysis of Variance", "",
46
+ "Obtain the analysis with the _lessPy_ function "
47
+ "_ANOVA()_.", "", _design_sentence(design, Y, facs),
48
+ ""]
49
+ tx += _chunk(["import pandas as pd",
50
+ "from lessPy import ANOVA",
51
+ f"r = {call}"], echo=True, output=False)
52
+
53
+ # background
54
+ tx += ["### Background", "",
55
+ "The output begins with a specification of the "
56
+ "variables in the model and a brief description of "
57
+ "the data.", "",
58
+ f"The response variable is {Y}. The "
59
+ f"{'factor' if len(facs) == 1 else 'factors'} "
60
+ f"{X} define the groups. Of {result.n_obs} cases, "
61
+ f"{result.n_keep} are retained for analysis.", ""]
62
+
63
+ tx += _sec_descriptive(result, Y, facs, design, d)
64
+ tx += _sec_summary(Y, X)
65
+ tx += _sec_effects(result, facs, design, d)
66
+ tx += _sec_pairwise(result, X, design, facs)
67
+ tx += _sec_residuals()
68
+
69
+ path = Path(rmd)
70
+ if path.suffix.lower() != ".qmd":
71
+ path = path.with_suffix(".qmd")
72
+ path.write_text("\n".join(tx))
73
+ print(f"\nQuarto report written: {path}")
74
+ _maybe_render(path, rmd_format, rmd_browser)
75
+ return path
76
+
77
+
78
+ def _front_matter(Y):
79
+ return [
80
+ "---",
81
+ f'title: "ANOVA of {Y}"',
82
+ "format:",
83
+ " html:",
84
+ " toc: true",
85
+ " toc-depth: 4",
86
+ " embed-resources: true",
87
+ "engine: jupyter",
88
+ "---",
89
+ "",
90
+ "------",
91
+ ""]
92
+
93
+
94
+ def _design_sentence(design, Y, facs):
95
+ if design == "oneway":
96
+ return (f"This is a one-way ANOVA of {Y} with treatment "
97
+ f"factor {facs[0]}.")
98
+ if design == "two-between":
99
+ return (f"This is a two-way between groups ANOVA of {Y} "
100
+ f"with treatment factors {_xAnd(facs)}.")
101
+ return (f"This is a randomized blocks ANOVA of {Y} with one "
102
+ f"treatment factor, {facs[0]}, and one blocking "
103
+ f"factor, {facs[1]}.")
104
+
105
+
106
+ def _sec_descriptive(result, Y, facs, design, d):
107
+ out = ["### Descriptive Statistics", "",
108
+ "The descriptive statistics of "
109
+ f"{Y} for the different levels of {_xAnd(facs)} "
110
+ "begin the analysis.", ""]
111
+ if design == "oneway":
112
+ out += _chunk(["r.descriptive"], echo=False)
113
+ out += [f"Grand Mean: {fmt(result.grand_mean, d + 1)}",
114
+ ""]
115
+ out += _plot_chunk(result, "means")
116
+ return out
117
+
118
+ f1, f2 = facs
119
+ if design == "two-between":
120
+ out += ["First, the number of data values in each "
121
+ "cell.", ""]
122
+ out += _chunk(
123
+ [f'd.pivot_table(index="{f2}", columns="{f1}", '
124
+ f'values="{Y}", aggfunc="size")'], echo=False)
125
+ out += ["The cell means follow.", ""]
126
+ out += _chunk(
127
+ [f'd.pivot_table(index="{f2}", columns="{f1}", '
128
+ f'values="{Y}", aggfunc="mean")'], echo=False)
129
+ out += ["The marginal means of each factor assist in "
130
+ "interpreting any main effects.", ""]
131
+ out += _chunk(
132
+ [f'd.groupby("{f1}")["{Y}"].mean()'], echo=False)
133
+ out += _chunk(
134
+ [f'd.groupby("{f2}")["{Y}"].mean()'], echo=False)
135
+ out += [f"The grand mean of all the data is "
136
+ f"{fmt(result.grand_mean, d + 1)}.", ""]
137
+ if design == "two-between":
138
+ out += ["The variation in each cell, its standard "
139
+ "deviation, follows.", ""]
140
+ out += _chunk(
141
+ [f'd.pivot_table(index="{f2}", columns="{f1}", '
142
+ f'values="{Y}", aggfunc="std")'], echo=False)
143
+ out += _plot_chunk(result, "interaction")
144
+ else:
145
+ out += _plot_chunk(result, "data")
146
+ out += _plot_chunk(result, "fitted")
147
+ return out
148
+
149
+
150
+ def _sec_summary(Y, X):
151
+ return ["### Summary Table", "",
152
+ "The analysis of variance (ANOVA) partitions the "
153
+ f"total sum of squares for {Y} into the residual "
154
+ "variability, $\\sum e^2_i$, and the sum of squares "
155
+ f"for {X}. The ANOVA table displays these sources "
156
+ "of variation.", "",
157
+ *_chunk(["r.anova"], echo=False)]
158
+
159
+
160
+ def _sec_effects(result, facs, design, d):
161
+ out = ["### Effects", "",
162
+ "The ANOVA table gives the significance test for the "
163
+ "factor; also of interest is the size of the effect.",
164
+ ""]
165
+ e = result.effects
166
+ if design == "oneway":
167
+ out += [
168
+ f"R Squared: {fmt(e['R_squared'], 3)} \n"
169
+ f"R Sq Adjusted: {fmt(e['R_sq_adjusted'], 3)} \n"
170
+ f"Omega Squared: {fmt(e['omega_squared'], 3)} \n"
171
+ f"Cohen's f: {fmt(e['cohen_f'], 3)}", ""]
172
+ elif design == "two-between":
173
+ f1, f2 = facs
174
+ out += [
175
+ f"Partial Omega Squared for {f1}: "
176
+ f"{fmt(e['omega_sq_' + f1], 3)} \n"
177
+ f"Partial Omega Squared for {f2}: "
178
+ f"{fmt(e['omega_sq_' + f2], 3)} \n"
179
+ f"Partial Omega Squared for {f1} & {f2}: "
180
+ f"{fmt(e['omega_sq_interaction'], 3)}", ""]
181
+ else:
182
+ f1, f2 = facs
183
+ out += [
184
+ f"Partial Omega Squared for {f1}: "
185
+ f"{fmt(e['omega_sq_' + f1], 3)} \n"
186
+ f"Partial Intraclass Correlation for {f2}: "
187
+ f"{fmt(e['intraclass_' + f2], 3)}", ""]
188
+ return out
189
+
190
+
191
+ def _sec_pairwise(result, X, design, facs):
192
+ out = ["### Pairwise Differences", "",
193
+ f"A significant effect shows that {X} impacts the "
194
+ "response, but do all levels differ or just some? "
195
+ "The Tukey pairwise comparisons, adjusted for the "
196
+ "overall significance level, are the search for "
197
+ "Honestly Significant Differences (HSD).", ""]
198
+ if result.tukey is None:
199
+ return out
200
+ if design == "oneway":
201
+ out += _chunk(["r.tukey"], echo=False)
202
+ else:
203
+ f1 = facs[0]
204
+ out += [f"Factor {f1}:", ""]
205
+ out += _chunk([f'r.tukey["{f1}"]'], echo=False)
206
+ if design == "two-between":
207
+ f2 = facs[1]
208
+ out += [f"Factor {f2}:", ""]
209
+ out += _chunk([f'r.tukey["{f2}"]'], echo=False)
210
+ out += ["Cell means:", ""]
211
+ out += _chunk(['r.tukey["cells"]'], echo=False)
212
+ return out
213
+
214
+
215
+ def _sec_residuals():
216
+ return ["### Residuals", "",
217
+ "ANOVA is a form of regression: each data value has "
218
+ "a fitted value, its cell mean, and the difference "
219
+ "is the residual. The cases with the largest "
220
+ "standardized residuals are listed first.", "",
221
+ *_chunk(["r.residuals"], echo=False)]
222
+
223
+
224
+ def _plot_chunk(result, key):
225
+ if key not in result.plots:
226
+ return []
227
+ return _chunk([f'r.plots["{key}"]'], echo=False)