lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/__init__.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# Python analog of NAMESPACE: names imported here are the public API.
|
|
2
|
+
|
|
3
|
+
from .bc_plotly import bc_plotly
|
|
4
|
+
from .bubble_plotly import bubble_plotly
|
|
5
|
+
from .Chart import Chart
|
|
6
|
+
from .dn_plotly import dn_plotly
|
|
7
|
+
from .dot_plotly import dot_plotly
|
|
8
|
+
from .freq_poly_plotly import freq_poly_plotly
|
|
9
|
+
from .hier_plotly import hier_plotly
|
|
10
|
+
from .hs_plotly import hs_plotly
|
|
11
|
+
from .Logit import Logit
|
|
12
|
+
from .pie_plotly import pie_plotly
|
|
13
|
+
from .plt_plotly import plt_plotly
|
|
14
|
+
from .radar_plotly import radar_plotly
|
|
15
|
+
from .Regression import Regression
|
|
16
|
+
from .utils import get_option, set_option
|
|
17
|
+
from .vbs_plotly import vbs_plotly
|
|
18
|
+
from .X import X
|
|
19
|
+
from .ANOVA import ANOVA
|
|
20
|
+
from .ttest import ttest
|
|
21
|
+
from .datasets import datasets, read_data
|
|
22
|
+
from .reshape import reshape_long, reshape_wide
|
|
23
|
+
from .pivot import pivot
|
|
24
|
+
from .corEFA import corEFA
|
|
25
|
+
from .corCFA import corCFA
|
|
26
|
+
from .corScree import corScree
|
|
27
|
+
from .corReorder import corReorder
|
|
28
|
+
from .corProp import corProp
|
|
29
|
+
from .Correlation import Correlation
|
|
30
|
+
from .corReflect import corReflect
|
|
31
|
+
from .corRead import corRead
|
|
32
|
+
from .Prop_test import Prop_test
|
|
33
|
+
from .corPrint import corPrint
|
|
34
|
+
from .Flows import Flows
|
|
35
|
+
from .date_infer import date_infer, format_date_labels
|
|
36
|
+
from .rename import rename
|
|
37
|
+
from .showColors import showColors
|
|
38
|
+
from .getColors import getColors
|
|
39
|
+
from .simCLT import simCLT
|
|
40
|
+
from .simMeans import simMeans
|
|
41
|
+
from .simFlips import simFlips
|
|
42
|
+
from .simCImean import simCImean
|
|
43
|
+
from .order_by import order_by
|
|
44
|
+
from .prob_norm import prob_norm
|
|
45
|
+
from .prob_znorm import prob_znorm
|
|
46
|
+
from .prob_tcut import prob_tcut
|
|
47
|
+
from .details import details
|
|
48
|
+
from .VariableLabels import VariableLabels
|
|
49
|
+
from .XY import XY
|
|
50
|
+
|
|
51
|
+
__version__ = "0.1.0"
|
|
52
|
+
|
|
53
|
+
__all__ = ["Chart", "X", "XY", "ANOVA", "ttest", "Regression", "Logit",
|
|
54
|
+
"read_data", "datasets", "reshape_long", "reshape_wide", "pivot", "corEFA", "corCFA", "corScree", "corReorder", "corProp", "Correlation", "corReflect", "corRead", "Prop_test", "corPrint", "Flows", "date_infer", "format_date_labels", "rename", "showColors", "getColors", "simCLT", "simMeans", "simFlips", "simCImean", "order_by", "prob_norm", "prob_znorm", "prob_tcut", "details",
|
|
55
|
+
"VariableLabels",
|
|
56
|
+
"bc_plotly", "bubble_plotly", "dn_plotly", "dot_plotly",
|
|
57
|
+
"freq_poly_plotly", "hier_plotly", "hs_plotly",
|
|
58
|
+
"pie_plotly", "plt_plotly", "radar_plotly",
|
|
59
|
+
"vbs_plotly",
|
|
60
|
+
"get_option", "set_option"]
|
lessPy/anova_rmd.py
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
# anova_rmd.py — Quarto (.qmd) report for ANOVA(), the analog of
|
|
2
|
+
# R's Rmd= for ANOVA (av.Rmd.R). Unlike the Regression report,
|
|
3
|
+
# av.Rmd.R uses no external prose templates and no toggles
|
|
4
|
+
# (explain/results are hardcoded on), so this generator is
|
|
5
|
+
# correspondingly simple. Same design choices as the other lessPy
|
|
6
|
+
# reports: Quarto target, direct per-section emitters, live code
|
|
7
|
+
# chunks for tables and plots (recomputed from the result object
|
|
8
|
+
# / data), narrative from R's prose. Deviation from av.Rmd.R: the
|
|
9
|
+
# cell-means plot is included (R's ANOVA report has no graphics).
|
|
10
|
+
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from .reg_rmd import _chunk, _maybe_render, _xAnd
|
|
14
|
+
from .utils import fmt
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def anova_rmd(result, y_name, facs, formula_str, design,
|
|
18
|
+
data_cols, rmd, rmd_data, rmd_format, rmd_browser):
|
|
19
|
+
"""Write a Quarto report reproducing the ANOVA. Returns the
|
|
20
|
+
path of the .qmd file written."""
|
|
21
|
+
Y = y_name
|
|
22
|
+
X = _xAnd(facs)
|
|
23
|
+
d = result.digits_d
|
|
24
|
+
data_name = "d"
|
|
25
|
+
read_expr = (f'pd.read_csv("{rmd_data}")' if rmd_data
|
|
26
|
+
else 'pd.read_csv("your_data.csv")')
|
|
27
|
+
call = f'ANOVA("{formula_str}", data={data_name})'
|
|
28
|
+
|
|
29
|
+
tx = _front_matter(Y)
|
|
30
|
+
tx += [
|
|
31
|
+
"The purpose of this analysis is the analysis of "
|
|
32
|
+
f"variance of the values of {Y} for the different "
|
|
33
|
+
f"levels of {X}.", ""]
|
|
34
|
+
|
|
35
|
+
# data
|
|
36
|
+
tx += ["## The Data", "",
|
|
37
|
+
"Read the data into a pandas data frame.", ""]
|
|
38
|
+
tx += _chunk(["import pandas as pd",
|
|
39
|
+
f"{data_name} = {read_expr}", data_name],
|
|
40
|
+
echo=True)
|
|
41
|
+
tx += ["Data from the following variables are available for "
|
|
42
|
+
f"analysis: {_xAnd(data_cols)}.", ""]
|
|
43
|
+
|
|
44
|
+
# analysis
|
|
45
|
+
tx += ["## Analysis of Variance", "",
|
|
46
|
+
"Obtain the analysis with the _lessPy_ function "
|
|
47
|
+
"_ANOVA()_.", "", _design_sentence(design, Y, facs),
|
|
48
|
+
""]
|
|
49
|
+
tx += _chunk(["import pandas as pd",
|
|
50
|
+
"from lessPy import ANOVA",
|
|
51
|
+
f"r = {call}"], echo=True, output=False)
|
|
52
|
+
|
|
53
|
+
# background
|
|
54
|
+
tx += ["### Background", "",
|
|
55
|
+
"The output begins with a specification of the "
|
|
56
|
+
"variables in the model and a brief description of "
|
|
57
|
+
"the data.", "",
|
|
58
|
+
f"The response variable is {Y}. The "
|
|
59
|
+
f"{'factor' if len(facs) == 1 else 'factors'} "
|
|
60
|
+
f"{X} define the groups. Of {result.n_obs} cases, "
|
|
61
|
+
f"{result.n_keep} are retained for analysis.", ""]
|
|
62
|
+
|
|
63
|
+
tx += _sec_descriptive(result, Y, facs, design, d)
|
|
64
|
+
tx += _sec_summary(Y, X)
|
|
65
|
+
tx += _sec_effects(result, facs, design, d)
|
|
66
|
+
tx += _sec_pairwise(result, X, design, facs)
|
|
67
|
+
tx += _sec_residuals()
|
|
68
|
+
|
|
69
|
+
path = Path(rmd)
|
|
70
|
+
if path.suffix.lower() != ".qmd":
|
|
71
|
+
path = path.with_suffix(".qmd")
|
|
72
|
+
path.write_text("\n".join(tx))
|
|
73
|
+
print(f"\nQuarto report written: {path}")
|
|
74
|
+
_maybe_render(path, rmd_format, rmd_browser)
|
|
75
|
+
return path
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _front_matter(Y):
|
|
79
|
+
return [
|
|
80
|
+
"---",
|
|
81
|
+
f'title: "ANOVA of {Y}"',
|
|
82
|
+
"format:",
|
|
83
|
+
" html:",
|
|
84
|
+
" toc: true",
|
|
85
|
+
" toc-depth: 4",
|
|
86
|
+
" embed-resources: true",
|
|
87
|
+
"engine: jupyter",
|
|
88
|
+
"---",
|
|
89
|
+
"",
|
|
90
|
+
"------",
|
|
91
|
+
""]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _design_sentence(design, Y, facs):
|
|
95
|
+
if design == "oneway":
|
|
96
|
+
return (f"This is a one-way ANOVA of {Y} with treatment "
|
|
97
|
+
f"factor {facs[0]}.")
|
|
98
|
+
if design == "two-between":
|
|
99
|
+
return (f"This is a two-way between groups ANOVA of {Y} "
|
|
100
|
+
f"with treatment factors {_xAnd(facs)}.")
|
|
101
|
+
return (f"This is a randomized blocks ANOVA of {Y} with one "
|
|
102
|
+
f"treatment factor, {facs[0]}, and one blocking "
|
|
103
|
+
f"factor, {facs[1]}.")
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _sec_descriptive(result, Y, facs, design, d):
|
|
107
|
+
out = ["### Descriptive Statistics", "",
|
|
108
|
+
"The descriptive statistics of "
|
|
109
|
+
f"{Y} for the different levels of {_xAnd(facs)} "
|
|
110
|
+
"begin the analysis.", ""]
|
|
111
|
+
if design == "oneway":
|
|
112
|
+
out += _chunk(["r.descriptive"], echo=False)
|
|
113
|
+
out += [f"Grand Mean: {fmt(result.grand_mean, d + 1)}",
|
|
114
|
+
""]
|
|
115
|
+
out += _plot_chunk(result, "means")
|
|
116
|
+
return out
|
|
117
|
+
|
|
118
|
+
f1, f2 = facs
|
|
119
|
+
if design == "two-between":
|
|
120
|
+
out += ["First, the number of data values in each "
|
|
121
|
+
"cell.", ""]
|
|
122
|
+
out += _chunk(
|
|
123
|
+
[f'd.pivot_table(index="{f2}", columns="{f1}", '
|
|
124
|
+
f'values="{Y}", aggfunc="size")'], echo=False)
|
|
125
|
+
out += ["The cell means follow.", ""]
|
|
126
|
+
out += _chunk(
|
|
127
|
+
[f'd.pivot_table(index="{f2}", columns="{f1}", '
|
|
128
|
+
f'values="{Y}", aggfunc="mean")'], echo=False)
|
|
129
|
+
out += ["The marginal means of each factor assist in "
|
|
130
|
+
"interpreting any main effects.", ""]
|
|
131
|
+
out += _chunk(
|
|
132
|
+
[f'd.groupby("{f1}")["{Y}"].mean()'], echo=False)
|
|
133
|
+
out += _chunk(
|
|
134
|
+
[f'd.groupby("{f2}")["{Y}"].mean()'], echo=False)
|
|
135
|
+
out += [f"The grand mean of all the data is "
|
|
136
|
+
f"{fmt(result.grand_mean, d + 1)}.", ""]
|
|
137
|
+
if design == "two-between":
|
|
138
|
+
out += ["The variation in each cell, its standard "
|
|
139
|
+
"deviation, follows.", ""]
|
|
140
|
+
out += _chunk(
|
|
141
|
+
[f'd.pivot_table(index="{f2}", columns="{f1}", '
|
|
142
|
+
f'values="{Y}", aggfunc="std")'], echo=False)
|
|
143
|
+
out += _plot_chunk(result, "interaction")
|
|
144
|
+
else:
|
|
145
|
+
out += _plot_chunk(result, "data")
|
|
146
|
+
out += _plot_chunk(result, "fitted")
|
|
147
|
+
return out
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _sec_summary(Y, X):
|
|
151
|
+
return ["### Summary Table", "",
|
|
152
|
+
"The analysis of variance (ANOVA) partitions the "
|
|
153
|
+
f"total sum of squares for {Y} into the residual "
|
|
154
|
+
"variability, $\\sum e^2_i$, and the sum of squares "
|
|
155
|
+
f"for {X}. The ANOVA table displays these sources "
|
|
156
|
+
"of variation.", "",
|
|
157
|
+
*_chunk(["r.anova"], echo=False)]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _sec_effects(result, facs, design, d):
|
|
161
|
+
out = ["### Effects", "",
|
|
162
|
+
"The ANOVA table gives the significance test for the "
|
|
163
|
+
"factor; also of interest is the size of the effect.",
|
|
164
|
+
""]
|
|
165
|
+
e = result.effects
|
|
166
|
+
if design == "oneway":
|
|
167
|
+
out += [
|
|
168
|
+
f"R Squared: {fmt(e['R_squared'], 3)} \n"
|
|
169
|
+
f"R Sq Adjusted: {fmt(e['R_sq_adjusted'], 3)} \n"
|
|
170
|
+
f"Omega Squared: {fmt(e['omega_squared'], 3)} \n"
|
|
171
|
+
f"Cohen's f: {fmt(e['cohen_f'], 3)}", ""]
|
|
172
|
+
elif design == "two-between":
|
|
173
|
+
f1, f2 = facs
|
|
174
|
+
out += [
|
|
175
|
+
f"Partial Omega Squared for {f1}: "
|
|
176
|
+
f"{fmt(e['omega_sq_' + f1], 3)} \n"
|
|
177
|
+
f"Partial Omega Squared for {f2}: "
|
|
178
|
+
f"{fmt(e['omega_sq_' + f2], 3)} \n"
|
|
179
|
+
f"Partial Omega Squared for {f1} & {f2}: "
|
|
180
|
+
f"{fmt(e['omega_sq_interaction'], 3)}", ""]
|
|
181
|
+
else:
|
|
182
|
+
f1, f2 = facs
|
|
183
|
+
out += [
|
|
184
|
+
f"Partial Omega Squared for {f1}: "
|
|
185
|
+
f"{fmt(e['omega_sq_' + f1], 3)} \n"
|
|
186
|
+
f"Partial Intraclass Correlation for {f2}: "
|
|
187
|
+
f"{fmt(e['intraclass_' + f2], 3)}", ""]
|
|
188
|
+
return out
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _sec_pairwise(result, X, design, facs):
|
|
192
|
+
out = ["### Pairwise Differences", "",
|
|
193
|
+
f"A significant effect shows that {X} impacts the "
|
|
194
|
+
"response, but do all levels differ or just some? "
|
|
195
|
+
"The Tukey pairwise comparisons, adjusted for the "
|
|
196
|
+
"overall significance level, are the search for "
|
|
197
|
+
"Honestly Significant Differences (HSD).", ""]
|
|
198
|
+
if result.tukey is None:
|
|
199
|
+
return out
|
|
200
|
+
if design == "oneway":
|
|
201
|
+
out += _chunk(["r.tukey"], echo=False)
|
|
202
|
+
else:
|
|
203
|
+
f1 = facs[0]
|
|
204
|
+
out += [f"Factor {f1}:", ""]
|
|
205
|
+
out += _chunk([f'r.tukey["{f1}"]'], echo=False)
|
|
206
|
+
if design == "two-between":
|
|
207
|
+
f2 = facs[1]
|
|
208
|
+
out += [f"Factor {f2}:", ""]
|
|
209
|
+
out += _chunk([f'r.tukey["{f2}"]'], echo=False)
|
|
210
|
+
out += ["Cell means:", ""]
|
|
211
|
+
out += _chunk(['r.tukey["cells"]'], echo=False)
|
|
212
|
+
return out
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _sec_residuals():
|
|
216
|
+
return ["### Residuals", "",
|
|
217
|
+
"ANOVA is a form of regression: each data value has "
|
|
218
|
+
"a fitted value, its cell mean, and the difference "
|
|
219
|
+
"is the residual. The cases with the largest "
|
|
220
|
+
"standardized residuals are listed first.", "",
|
|
221
|
+
*_chunk(["r.residuals"], echo=False)]
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _plot_chunk(result, key):
|
|
225
|
+
if key not in result.plots:
|
|
226
|
+
return []
|
|
227
|
+
return _chunk([f'r.plots["{key}"]'], echo=False)
|