lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/reg_rmd.py
ADDED
|
@@ -0,0 +1,754 @@
|
|
|
1
|
+
# reg_rmd.py — generate a Quarto (.qmd) reproducible report for
|
|
2
|
+
# Regression(), the Python analog of R's Rmd= (reg.Rmd.R + the
|
|
3
|
+
# inst/Rmd/reg/*.txt prose templates + reg.RmdParse.R).
|
|
4
|
+
#
|
|
5
|
+
# The document is a full tutorial-style narrative with the same
|
|
6
|
+
# section structure as lessR: Introduction, Data, Model, Fit,
|
|
7
|
+
# Relations, Influence, Prediction, Validity. The explain,
|
|
8
|
+
# interpret, and results toggles gate the prose exactly as in R.
|
|
9
|
+
#
|
|
10
|
+
# Deviations from the R generator, by design:
|
|
11
|
+
# - Target is Quarto (.qmd) with Python chunks calling lessPy,
|
|
12
|
+
# not R Markdown (.Rmd) with R chunks.
|
|
13
|
+
# - Sections are emitted directly in Python rather than through
|
|
14
|
+
# the .txt-file + .RmdParse backtick-symbol engine; the output
|
|
15
|
+
# prose is the same, the indirection is not reproduced. So
|
|
16
|
+
# Rmd_custom / Rmd_dir / Rmd_labels (custom .txt templates) are
|
|
17
|
+
# not ported.
|
|
18
|
+
# - Tables and plots render live from the result object in code
|
|
19
|
+
# chunks (r.estimates, r.plots[...]); narrative numbers are
|
|
20
|
+
# baked at generation time from the same object. The analysis
|
|
21
|
+
# (tables, figures) is fully reproducible; the narrative
|
|
22
|
+
# numbers are fixed to the generating run.
|
|
23
|
+
# - The "Tests of Multiple Coefficients" subsection (R's
|
|
24
|
+
# explain + n.pred>2 block) uses Nest(), which is not ported,
|
|
25
|
+
# so it is omitted.
|
|
26
|
+
|
|
27
|
+
import shutil
|
|
28
|
+
import subprocess
|
|
29
|
+
import webbrowser
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
import numpy as np
|
|
33
|
+
|
|
34
|
+
from .utils import fmt
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# ------------------------------------------------------------
|
|
38
|
+
# narrative helpers, ports of xP / xNum / xAnd / xU (xP.R etc.)
|
|
39
|
+
# ------------------------------------------------------------
|
|
40
|
+
|
|
41
|
+
def _xP(x, d):
|
|
42
|
+
"""Format a number, thousands separated, d decimals. ~ xP()"""
|
|
43
|
+
if x is None or (isinstance(x, float) and np.isnan(x)):
|
|
44
|
+
return ""
|
|
45
|
+
return f"{float(x):,.{d}f}"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
_WORDS = ["none", "one", "two", "three", "four", "five", "six",
|
|
49
|
+
"seven", "eight", "nine", "ten", "eleven", "twelve"]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _xNum(x):
|
|
53
|
+
"""Small counts as words, larger as integers. ~ xNum()"""
|
|
54
|
+
x = int(round(x))
|
|
55
|
+
return _WORDS[x] if 0 <= x <= 12 else str(x)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _xAnd(seq):
|
|
59
|
+
"""Join with commas and a final "and". ~ xAnd()"""
|
|
60
|
+
seq = list(seq)
|
|
61
|
+
if not seq:
|
|
62
|
+
return ""
|
|
63
|
+
if len(seq) == 1:
|
|
64
|
+
return str(seq[0])
|
|
65
|
+
if len(seq) == 2:
|
|
66
|
+
return f"{seq[0]} and {seq[1]}"
|
|
67
|
+
return ", ".join(str(s) for s in seq[:-1]) + f", and {seq[-1]}"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _xU(s):
|
|
71
|
+
"""Capitalize the first letter. ~ xU()"""
|
|
72
|
+
return s[:1].upper() + s[1:] if s else s
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# ------------------------------------------------------------
|
|
76
|
+
# code-chunk emitter
|
|
77
|
+
# ------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
def _chunk(lines, echo, output=True):
|
|
80
|
+
"""A Quarto python code chunk. echo shows the code (the code=
|
|
81
|
+
toggle); output=False runs the chunk but hides its output."""
|
|
82
|
+
out = ["```{python}"]
|
|
83
|
+
if not echo:
|
|
84
|
+
out.append("#| echo: false")
|
|
85
|
+
if not output:
|
|
86
|
+
out.append("#| output: false")
|
|
87
|
+
out += lines
|
|
88
|
+
out.append("```")
|
|
89
|
+
out.append("")
|
|
90
|
+
return out
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _hr():
|
|
94
|
+
return ["", "------", ""]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# ============================================================
|
|
98
|
+
# the generator
|
|
99
|
+
# ============================================================
|
|
100
|
+
|
|
101
|
+
def reg_rmd(result, y_name, pred_names, formula_str, data_cols,
|
|
102
|
+
rmd, rmd_data, rmd_format, rmd_browser,
|
|
103
|
+
results, explain, interpret, code,
|
|
104
|
+
n_res_rows, n_pred_rows, res_sort, digits_d):
|
|
105
|
+
"""Write a Quarto report reproducing the regression analysis.
|
|
106
|
+
Returns the path of the .qmd file written."""
|
|
107
|
+
n_pred = len(pred_names)
|
|
108
|
+
d = digits_d
|
|
109
|
+
est = result.estimates
|
|
110
|
+
coefs = est["Estimate"]
|
|
111
|
+
pvals = est["p-value"]
|
|
112
|
+
cilb, ciub = est["Lower 95%"], est["Upper 95%"]
|
|
113
|
+
anova = result.anova
|
|
114
|
+
fitd = result.fit
|
|
115
|
+
|
|
116
|
+
# context strings, ~ reg.Rmd.R symbol setup
|
|
117
|
+
Y = y_name
|
|
118
|
+
X = _xAnd(pred_names)
|
|
119
|
+
pl = "s" if n_pred > 1 else ""
|
|
120
|
+
et_c = "Each " if n_pred > 1 else "The"
|
|
121
|
+
cnst = (", with the values of all remaining predictor "
|
|
122
|
+
"variables held constant" if n_pred > 1 else "")
|
|
123
|
+
mult = f"through $b_{n_pred}$" if n_pred > 1 else ""
|
|
124
|
+
|
|
125
|
+
data_name = "d"
|
|
126
|
+
read_expr = (f'pd.read_csv("{rmd_data}")' if rmd_data
|
|
127
|
+
else 'pd.read_csv("your_data.csv")')
|
|
128
|
+
call = f'Regression("{formula_str}", data={data_name})'
|
|
129
|
+
|
|
130
|
+
tx = []
|
|
131
|
+
tx += _front_matter(Y, X, n_pred, results, explain,
|
|
132
|
+
interpret, code)
|
|
133
|
+
tx += _setup_chunk(read_expr, data_name, call, code)
|
|
134
|
+
tx += _sec_intro(Y, X, pl)
|
|
135
|
+
tx += _sec_data(read_expr, data_name, data_cols, code)
|
|
136
|
+
tx += _sec_model(Y, X, pred_names, pl, et_c, cnst, mult,
|
|
137
|
+
coefs, pvals, cilb, ciub, result, d, n_pred,
|
|
138
|
+
results, explain, interpret, code)
|
|
139
|
+
tx += _sec_fit(Y, anova, fitd, d, n_pred, results, explain)
|
|
140
|
+
tx += _sec_relations(Y, X, pred_names, pl, result, fitd, d,
|
|
141
|
+
n_pred, results, explain, interpret,
|
|
142
|
+
code)
|
|
143
|
+
if n_res_rows and n_res_rows > 0 and result.residuals is not None:
|
|
144
|
+
tx += _sec_influence(Y, result, res_sort, d, code)
|
|
145
|
+
if (n_pred_rows and n_pred_rows > 0 and n_pred <= 6
|
|
146
|
+
and result.predictions is not None):
|
|
147
|
+
tx += _sec_prediction(Y, result, d, n_pred, code)
|
|
148
|
+
tx += _sec_validity(code)
|
|
149
|
+
|
|
150
|
+
path = Path(rmd)
|
|
151
|
+
if path.suffix.lower() != ".qmd":
|
|
152
|
+
path = path.with_suffix(".qmd")
|
|
153
|
+
path.write_text("\n".join(tx))
|
|
154
|
+
print(f"\nQuarto report written: {path}")
|
|
155
|
+
|
|
156
|
+
_maybe_render(path, rmd_format, rmd_browser)
|
|
157
|
+
return path
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# ------------------------------------------------------------
|
|
161
|
+
# front matter and setup
|
|
162
|
+
# ------------------------------------------------------------
|
|
163
|
+
|
|
164
|
+
def _front_matter(Y, X, n_pred, results, explain, interpret, code):
|
|
165
|
+
title = (f"Multiple Regression of {Y}" if n_pred > 1
|
|
166
|
+
else f"Regression of {Y} on {X}")
|
|
167
|
+
return [
|
|
168
|
+
"---",
|
|
169
|
+
f'title: "{title}"',
|
|
170
|
+
"format:",
|
|
171
|
+
" html:",
|
|
172
|
+
" toc: true",
|
|
173
|
+
" toc-depth: 4",
|
|
174
|
+
" embed-resources: true",
|
|
175
|
+
"engine: jupyter",
|
|
176
|
+
"---",
|
|
177
|
+
"",
|
|
178
|
+
"------",
|
|
179
|
+
"",
|
|
180
|
+
f"_Output Options: explain={explain}, "
|
|
181
|
+
f"interpret={interpret}, results={results}, "
|
|
182
|
+
f"code={code}_",
|
|
183
|
+
"",
|
|
184
|
+
"------",
|
|
185
|
+
""]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _setup_chunk(read_expr, data_name, call, code):
|
|
189
|
+
# runs the analysis once; output hidden, r reused per section
|
|
190
|
+
return _chunk(
|
|
191
|
+
["import pandas as pd",
|
|
192
|
+
"from lessPy import Regression",
|
|
193
|
+
f"{data_name} = {read_expr}",
|
|
194
|
+
f"r = {call}"],
|
|
195
|
+
echo=code, output=False)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
# ------------------------------------------------------------
|
|
199
|
+
# 1 - Introduction
|
|
200
|
+
# ------------------------------------------------------------
|
|
201
|
+
|
|
202
|
+
def _sec_intro(Y, X, pl):
|
|
203
|
+
return [
|
|
204
|
+
"## Introduction",
|
|
205
|
+
"",
|
|
206
|
+
f"The variable of primary interest is the _response "
|
|
207
|
+
f"variable_ {Y}. The purpose of this analysis is to "
|
|
208
|
+
f"account for the values of {Y} from the information "
|
|
209
|
+
f"provided by the values of the _predictor "
|
|
210
|
+
f"variable{pl}_, {X}.",
|
|
211
|
+
"",
|
|
212
|
+
"This regression analysis is an example of _supervised "
|
|
213
|
+
"machine learning_. In the language of machine learning, "
|
|
214
|
+
"refer to the response variable as the _target_ and each "
|
|
215
|
+
"predictor variable as a _feature_. The machine learns "
|
|
216
|
+
f"the relationship between the target {Y} and the "
|
|
217
|
+
f"feature{pl} {X} from the analysis of the training "
|
|
218
|
+
"data. Express this learning in the form of a regression "
|
|
219
|
+
"model.",
|
|
220
|
+
""]
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# ------------------------------------------------------------
|
|
224
|
+
# 2 - Data
|
|
225
|
+
# ------------------------------------------------------------
|
|
226
|
+
|
|
227
|
+
def _sec_data(read_expr, data_name, data_cols, code):
|
|
228
|
+
out = ["## Data", "",
|
|
229
|
+
"Read the data into a pandas data frame.", ""]
|
|
230
|
+
out += _chunk([f"{data_name} = {read_expr}", data_name],
|
|
231
|
+
echo=code)
|
|
232
|
+
out += [
|
|
233
|
+
"The corresponding data values for the variables in the "
|
|
234
|
+
"model comprise the _training data_, from which the "
|
|
235
|
+
"model is estimated.",
|
|
236
|
+
"",
|
|
237
|
+
"Data from the following variables are available for "
|
|
238
|
+
f"analysis: {_xAnd(data_cols)}.",
|
|
239
|
+
""]
|
|
240
|
+
return out
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# ------------------------------------------------------------
|
|
244
|
+
# 3 - Model
|
|
245
|
+
# ------------------------------------------------------------
|
|
246
|
+
|
|
247
|
+
def _sec_model(Y, X, pred_names, pl, et_c, cnst, mult, coefs,
|
|
248
|
+
pvals, cilb, ciub, result, d, n_pred, results,
|
|
249
|
+
explain, interpret, code):
|
|
250
|
+
out = ["## Model", "", "### Specified Model", "",
|
|
251
|
+
f"Express {Y} as a linear function of {_xNum(n_pred)} "
|
|
252
|
+
f"predictor variable{pl}: {X}. Within the context of "
|
|
253
|
+
"the model, indicate the response variable with a Y "
|
|
254
|
+
f"subscripted by the variable name, $Y_{{{Y}}}$. "
|
|
255
|
+
"Identify each predictor variable according to a "
|
|
256
|
+
"subscript to an X. From the training data compute "
|
|
257
|
+
f"$\\hat Y_{{{Y}}}$, the _fitted value_ of the "
|
|
258
|
+
"response variable from the model for a specific set "
|
|
259
|
+
f"of values for {X}.",
|
|
260
|
+
"",
|
|
261
|
+
_eq_model_b(Y, pred_names),
|
|
262
|
+
"",
|
|
263
|
+
f"The _intercept_, $b_0$, indicates the fitted value "
|
|
264
|
+
f"of {Y} for values of {X} all equal to zero. "
|
|
265
|
+
f"{et_c} _slope coefficient_ $b_1$ {mult} is the "
|
|
266
|
+
"average change in the value of response variable, "
|
|
267
|
+
f"{Y}, for a one-unit increase in the value of the "
|
|
268
|
+
f"corresponding predictor variable{cnst}.",
|
|
269
|
+
"",
|
|
270
|
+
"Estimate the coefficients with ordinary least "
|
|
271
|
+
"squares (OLS), which minimizes the sum of the "
|
|
272
|
+
"squared residuals, $\\sum e^2_i$, across all the "
|
|
273
|
+
"rows of the training data.",
|
|
274
|
+
"",
|
|
275
|
+
_eq_resid(),
|
|
276
|
+
"",
|
|
277
|
+
"Accomplish the estimation with the _lessPy_ function "
|
|
278
|
+
"_Regression()_. The analysis was run in the setup "
|
|
279
|
+
"chunk above; its output is examined section by "
|
|
280
|
+
"section below.",
|
|
281
|
+
""]
|
|
282
|
+
|
|
283
|
+
# background: cases presented / retained
|
|
284
|
+
out += [
|
|
285
|
+
f"Of the {result.n_obs} cases presented for analysis, "
|
|
286
|
+
f"{result.n_keep} are retained. The number of deleted "
|
|
287
|
+
f"cases due to missing data is "
|
|
288
|
+
f"{result.n_obs - result.n_keep}.",
|
|
289
|
+
"", "### Estimated Model", "",
|
|
290
|
+
"The analysis begins with the estimation of each sample "
|
|
291
|
+
"regression coefficient, $b_j$, what the estimation "
|
|
292
|
+
"algorithm learns from the training data. Of greater "
|
|
293
|
+
"interest is each corresponding population value, "
|
|
294
|
+
"$\\beta_j$, in the _population model_.",
|
|
295
|
+
"",
|
|
296
|
+
_eq_model_beta(Y, pred_names),
|
|
297
|
+
"",
|
|
298
|
+
"Each _t_-test evaluates the _null hypothesis_ that the "
|
|
299
|
+
"corresponding population regression coefficient is 0.",
|
|
300
|
+
"",
|
|
301
|
+
"$$H_0: \\beta_j=0$$",
|
|
302
|
+
"$$H_1: \\beta_j \\ne 0$$",
|
|
303
|
+
""]
|
|
304
|
+
if results:
|
|
305
|
+
out += _chunk(["r.estimates"], echo=code)
|
|
306
|
+
out += [
|
|
307
|
+
"This estimated model is the linear function with "
|
|
308
|
+
"estimated numeric coefficients that yield a fitted "
|
|
309
|
+
f"value of {Y}, from the provided data value{pl} of {X}.",
|
|
310
|
+
"",
|
|
311
|
+
_eq_model_est(Y, pred_names, coefs, d),
|
|
312
|
+
""]
|
|
313
|
+
if interpret:
|
|
314
|
+
out += _intr_ci(Y, pred_names, pvals, cilb, ciub, d,
|
|
315
|
+
n_pred, cnst)
|
|
316
|
+
return out
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _eq_model_b(Y, pred_names):
|
|
320
|
+
s = f"$$\\hat Y_{{{Y}}} = b_0 + b_1 X_{{{pred_names[0]}}}"
|
|
321
|
+
for i, nm in enumerate(pred_names[1:], start=2):
|
|
322
|
+
s += f" + b_{i} X_{{{nm}}}"
|
|
323
|
+
return s + "$$"
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _eq_model_beta(Y, pred_names):
|
|
327
|
+
s = (f"$$\\hat Y_{{{Y}}} = \\beta_0 + \\beta_1 "
|
|
328
|
+
f"X_{{{pred_names[0]}}}")
|
|
329
|
+
for i, nm in enumerate(pred_names[1:], start=2):
|
|
330
|
+
s += f" + \\beta_{i} X_{{{nm}}}"
|
|
331
|
+
return s + "$$"
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _eq_resid():
|
|
335
|
+
return "$$e_i = Y_i - \\hat Y_i$$"
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _eq_model_est(Y, pred_names, coefs, d):
|
|
339
|
+
s = f"$$\\hat Y_{{{Y}}} = {_xP(coefs.iloc[0], d)}"
|
|
340
|
+
for i, nm in enumerate(pred_names, start=1):
|
|
341
|
+
b = coefs.iloc[i]
|
|
342
|
+
sign = "+" if b >= 0 else "-"
|
|
343
|
+
s += f" {sign} {_xP(abs(b), d)} X_{{{nm}}}"
|
|
344
|
+
return s + "$$"
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _intr_ci(Y, pred_names, pvals, cilb, ciub, d, n_pred, cnst):
|
|
348
|
+
out = []
|
|
349
|
+
ns = [pred_names[i] for i in range(n_pred)
|
|
350
|
+
if pvals.iloc[i + 1] > 0.05]
|
|
351
|
+
sig = [pred_names[i] for i in range(n_pred)
|
|
352
|
+
if pvals.iloc[i + 1] <= 0.05]
|
|
353
|
+
if ns:
|
|
354
|
+
pl2 = "s" if len(ns) > 1 else ""
|
|
355
|
+
has = "have _p_-values" if len(ns) > 1 else "has a _p_-value"
|
|
356
|
+
the = "each " if len(ns) > 1 else "the "
|
|
357
|
+
out += [
|
|
358
|
+
f"{_xU(_xNum(len(ns)))} predictor variable{pl2} "
|
|
359
|
+
f"{has} larger than $\\alpha$ = 0.05: "
|
|
360
|
+
f"_{_xAnd(ns)}_. {_xU(the)}null hypothesis of no "
|
|
361
|
+
"relationship could not be rejected, so there is a "
|
|
362
|
+
f"reasonable possibility that {the}predictor "
|
|
363
|
+
f"variable may not contribute to explaining the "
|
|
364
|
+
f"values of {Y}.",
|
|
365
|
+
""]
|
|
366
|
+
if sig:
|
|
367
|
+
pl3 = "s" if len(sig) > 1 else ""
|
|
368
|
+
t1 = ("These predictor variables each have "
|
|
369
|
+
if len(sig) > 1 else "This predictor variable has ")
|
|
370
|
+
t2 = "their" if len(sig) > 1 else "its"
|
|
371
|
+
t3 = ("these corresponding slope coefficients"
|
|
372
|
+
if len(sig) > 1 else "this corresponding slope "
|
|
373
|
+
"coefficient")
|
|
374
|
+
out += [
|
|
375
|
+
f"{t1}a _p_-value less than or equal to $\\alpha$ = "
|
|
376
|
+
f"0.05: _{_xAnd(sig)}_. To extend the results beyond "
|
|
377
|
+
"this sample to the population, interpret the "
|
|
378
|
+
f"meaning of {t3} in terms of {t2} confidence "
|
|
379
|
+
f"interval{pl3}.",
|
|
380
|
+
""]
|
|
381
|
+
for nm in sig:
|
|
382
|
+
j = pred_names.index(nm)
|
|
383
|
+
remain = [p for p in pred_names if p != nm]
|
|
384
|
+
hold = (f", with the values of {_xAnd(remain)} held "
|
|
385
|
+
"constant" if n_pred > 1 else "")
|
|
386
|
+
bullet = f"- _{nm}_: " if n_pred > 1 else ""
|
|
387
|
+
out += [
|
|
388
|
+
f"{bullet}With 95% confidence, for each "
|
|
389
|
+
f"additional unit of {nm}, on average, {Y} "
|
|
390
|
+
"changes somewhere between "
|
|
391
|
+
f"{_xP(cilb.iloc[j + 1], d)} to "
|
|
392
|
+
f"{_xP(ciub.iloc[j + 1], d)}{hold}.",
|
|
393
|
+
""]
|
|
394
|
+
return out
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# ------------------------------------------------------------
|
|
398
|
+
# 4 - Fit
|
|
399
|
+
# ------------------------------------------------------------
|
|
400
|
+
|
|
401
|
+
def _sec_fit(Y, anova, fitd, d, n_pred, results, explain):
|
|
402
|
+
m_ss = anova.loc["Model", "Sum Sq"]
|
|
403
|
+
r_ss = anova.loc["Residuals", "Sum Sq"]
|
|
404
|
+
t_ss = anova.loc[Y, "Sum Sq"]
|
|
405
|
+
r_ms = anova.loc["Residuals", "Mean Sq"]
|
|
406
|
+
t_ms = anova.loc[Y, "Mean Sq"]
|
|
407
|
+
r_df = int(anova.loc["Residuals", "df"])
|
|
408
|
+
t_df = int(anova.loc[Y, "df"])
|
|
409
|
+
tcut = float(-_tppf(r_df))
|
|
410
|
+
out = ["## Fit", "", "### Partitioning Variance", ""]
|
|
411
|
+
if explain:
|
|
412
|
+
out += [
|
|
413
|
+
"The analysis of fit evaluates how well the model "
|
|
414
|
+
f"accounts for the variability of {Y}. The core "
|
|
415
|
+
"measure of variability is the _sum of squares_, "
|
|
416
|
+
"_SS_. The analysis of variance (ANOVA) partitions "
|
|
417
|
+
f"the Total sum of squares of {Y} into the Residual "
|
|
418
|
+
"variability, $\\sum e^2_i$, and the Model sum of "
|
|
419
|
+
"squares. The larger the explained variability "
|
|
420
|
+
"relative to the unexplained, the better the model "
|
|
421
|
+
"fits the data.",
|
|
422
|
+
""]
|
|
423
|
+
if results:
|
|
424
|
+
out += _chunk(["r.anova"], echo=False)
|
|
425
|
+
out += [
|
|
426
|
+
_eq_decomp(Y, m_ss, r_ss, t_ss, d),
|
|
427
|
+
"",
|
|
428
|
+
"The total variation is that which is explained by "
|
|
429
|
+
"the model, and that which is _not_ explained.",
|
|
430
|
+
""]
|
|
431
|
+
out += ["### Fit Indices", ""]
|
|
432
|
+
if results:
|
|
433
|
+
out += _chunk(["r.fit"], echo=False)
|
|
434
|
+
out += [
|
|
435
|
+
"#### Standard Deviation of Residuals", "",
|
|
436
|
+
"The _standard deviation of the residuals_, $s_e$, "
|
|
437
|
+
f"the square root of the mean square of the "
|
|
438
|
+
f"residuals, assesses the variability of {Y} about "
|
|
439
|
+
"the fitted values.",
|
|
440
|
+
"",
|
|
441
|
+
_eq_se(r_ms, fitd["se"], d),
|
|
442
|
+
"",
|
|
443
|
+
f"To interpret $s_e$ = {_xP(fitd['se'], d)}, consider "
|
|
444
|
+
"the estimated range of 95% of the values of a "
|
|
445
|
+
"normally distributed variable, from the 2.5% cutoff "
|
|
446
|
+
f"of the _t_-distribution for df={r_df}: "
|
|
447
|
+
f"{_xP(tcut, 3)}.",
|
|
448
|
+
"",
|
|
449
|
+
_eq_se_range(tcut, fitd["se"], fitd["resid_range"], d),
|
|
450
|
+
"",
|
|
451
|
+
"#### $R^2$ Family", "",
|
|
452
|
+
f"$R^2$ is the proportion of the variability of {Y} "
|
|
453
|
+
"accounted for by the model.",
|
|
454
|
+
"",
|
|
455
|
+
_eq_r2(Y, r_ss, t_ss, fitd["Rsq"], d),
|
|
456
|
+
"",
|
|
457
|
+
"Use the adjusted version, $R^2_{adj}$, to compare "
|
|
458
|
+
"models with different numbers of predictors, which "
|
|
459
|
+
"guards against overfitting.",
|
|
460
|
+
"",
|
|
461
|
+
_eq_r2adj(Y, r_ms, t_ms, fitd["Rsq_adj"], d),
|
|
462
|
+
"",
|
|
463
|
+
f"Compare $R^2$ = {_xP(fitd['Rsq'], 3)} to the "
|
|
464
|
+
f"adjusted $R^2_{{adj}}$ = {_xP(fitd['Rsq_adj'], 3)}, "
|
|
465
|
+
f"a difference of {_xP(fitd['Rsq'] - fitd['Rsq_adj'], 3)}. "
|
|
466
|
+
"A large difference indicates too many predictors "
|
|
467
|
+
"for the available data.",
|
|
468
|
+
"",
|
|
469
|
+
"To generalize to prediction accuracy on _new_ data, "
|
|
470
|
+
"evaluate the model with the _predictive residual_ "
|
|
471
|
+
"(PRESS): estimate the model with each case deleted, "
|
|
472
|
+
"predict that case, and sum the squared predictive "
|
|
473
|
+
"residuals.",
|
|
474
|
+
"",
|
|
475
|
+
_eq_r2press(Y, fitd["PRESS"], t_ss, fitd["Rsq_PRESS"],
|
|
476
|
+
d),
|
|
477
|
+
"",
|
|
478
|
+
"Because an estimated model overfits its training "
|
|
479
|
+
f"data, $R^2_{{PRESS}}$ = {_xP(fitd['Rsq_PRESS'], 3)} "
|
|
480
|
+
"is lower than both $R^2$ and $R^2_{adj}$, and is "
|
|
481
|
+
"the more appropriate value for how well the model "
|
|
482
|
+
"predicts beyond the training data.",
|
|
483
|
+
""]
|
|
484
|
+
return out
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _eq_decomp(Y, m, r, t, d):
|
|
488
|
+
return (f"$$SS_{{{Y}}} = SS_{{Model}} + SS_{{Residual}} = "
|
|
489
|
+
f"{_xP(m, d)} + {_xP(r, d)} = {_xP(t, d)}$$")
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def _eq_se(r_ms, se, d):
|
|
493
|
+
return (f"$$s_e = \\sqrt{{MS_{{Residual}}}} = "
|
|
494
|
+
f"\\sqrt{{{_xP(r_ms, d)}}} = {_xP(se, d)}$$")
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def _eq_se_range(tcut, se, rng, d):
|
|
498
|
+
return ("$$95\\% \\; Range: 2 * t_{cutoff} * s_e = "
|
|
499
|
+
f"2 * {_xP(tcut, 3)} * {_xP(se, d)} = {_xP(rng, d)}$$")
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _eq_r2(Y, r_ss, t_ss, rsq, d):
|
|
503
|
+
return (f"$$R^2 = 1 - \\frac{{SS_{{Residual}}}}{{SS_{{{Y}}}}} "
|
|
504
|
+
f"= 1 - \\frac{{{_xP(r_ss, d)}}}{{{_xP(t_ss, d)}}} = "
|
|
505
|
+
f"{_xP(rsq, 3)}$$")
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _eq_r2adj(Y, r_ms, t_ms, rsqadj, d):
|
|
509
|
+
return (f"$$R^2_{{adj}} = 1 - "
|
|
510
|
+
f"\\frac{{MS_{{Residual}}}}{{MS_{{{Y}}}}} = 1 - "
|
|
511
|
+
f"\\frac{{{_xP(r_ms, d)}}}{{{_xP(t_ms, d)}}} = "
|
|
512
|
+
f"{_xP(rsqadj, 3)}$$")
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _eq_r2press(Y, press, t_ss, rsqpress, d):
|
|
516
|
+
return (f"$$R^2_{{PRESS}} = 1 - "
|
|
517
|
+
f"\\frac{{SS_{{PRE}}}}{{SS_{{{Y}}}}} = 1 - "
|
|
518
|
+
f"\\frac{{{_xP(press, d)}}}{{{_xP(t_ss, d)}}} = "
|
|
519
|
+
f"{_xP(rsqpress, 3)}$$")
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
# ------------------------------------------------------------
|
|
523
|
+
# 5 - Relations
|
|
524
|
+
# ------------------------------------------------------------
|
|
525
|
+
|
|
526
|
+
def _sec_relations(Y, X, pred_names, pl, result, fitd, d, n_pred,
|
|
527
|
+
results, explain, interpret, code):
|
|
528
|
+
out = ["## Relations", ""]
|
|
529
|
+
if n_pred == 1:
|
|
530
|
+
out += ["### Scatter Plot", ""]
|
|
531
|
+
cor = float(np.sign(result.estimates["Estimate"].iloc[1])
|
|
532
|
+
* np.sqrt(fitd["Rsq"]))
|
|
533
|
+
out += [
|
|
534
|
+
f"How do the variables relate? The correlation of "
|
|
535
|
+
f"the response variable {Y} with the predictor "
|
|
536
|
+
f"{X} should be relatively high. In the training "
|
|
537
|
+
f"data, $r$ = {_xP(cor, 3)}.",
|
|
538
|
+
"",
|
|
539
|
+
"Visually summarize the relation with a scatterplot "
|
|
540
|
+
"and its least-squares line.",
|
|
541
|
+
""]
|
|
542
|
+
out += _chunk(['r.plots["scatter"]'], echo=code)
|
|
543
|
+
else:
|
|
544
|
+
out += ["### Scatter Plot Matrix", ""]
|
|
545
|
+
out += [
|
|
546
|
+
f"How do the variables relate? The correlations of "
|
|
547
|
+
f"the response variable {Y} with the predictor "
|
|
548
|
+
f"variables {X} should be relatively high, and the "
|
|
549
|
+
"correlations of the predictors with each other "
|
|
550
|
+
"relatively small.",
|
|
551
|
+
"",
|
|
552
|
+
"The scatterplots between each pair of variables "
|
|
553
|
+
"appear in a _scatterplot matrix_, each with its "
|
|
554
|
+
"best-fitting line below the diagonal and the "
|
|
555
|
+
"corresponding correlation above the diagonal.",
|
|
556
|
+
""]
|
|
557
|
+
out += _chunk(['r.plots["scatter_matrix"]'], echo=code)
|
|
558
|
+
out += _relations_collinear(X, result, d, code, results,
|
|
559
|
+
interpret)
|
|
560
|
+
return out
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _relations_collinear(X, result, d, code, results, interpret):
|
|
564
|
+
out = ["### Collinearity", "",
|
|
565
|
+
"The collinearity analysis assesses the extent that "
|
|
566
|
+
f"the predictor variables -- {X} -- linearly depend "
|
|
567
|
+
"upon each other. Collinear predictors have large "
|
|
568
|
+
"standard errors and unstable estimates. The "
|
|
569
|
+
"_tolerance_ of a predictor is $1 - R^2_j$ from "
|
|
570
|
+
"regressing it on the other predictors; it should be "
|
|
571
|
+
"high, at least above about 0.20. The _variance "
|
|
572
|
+
"inflation factor_ (VIF) is its reciprocal, and "
|
|
573
|
+
"should be below about 5.",
|
|
574
|
+
""]
|
|
575
|
+
if results:
|
|
576
|
+
out += _chunk(
|
|
577
|
+
["import pandas as pd",
|
|
578
|
+
'pd.DataFrame({"Tolerance": r.tolerance, '
|
|
579
|
+
'"VIF": r.vif}, index=r.estimates.index[1:])'],
|
|
580
|
+
echo=code)
|
|
581
|
+
if interpret:
|
|
582
|
+
out += _intr_tolerance(result, d)
|
|
583
|
+
out += ["### Subset Models", "",
|
|
584
|
+
"Especially when collinearity is present, can a "
|
|
585
|
+
"simpler model be about as effective? Each row below "
|
|
586
|
+
"is a different model: a 1 means the predictor is in "
|
|
587
|
+
"the model, a 0 means it is excluded.",
|
|
588
|
+
""]
|
|
589
|
+
if results:
|
|
590
|
+
out += _chunk(["r.subsets"], echo=code)
|
|
591
|
+
out += [
|
|
592
|
+
"The goal is _parsimony_: the most explanatory power, "
|
|
593
|
+
"assessed with $R^2_{adj}$, from the fewest predictors. "
|
|
594
|
+
"This subset analysis is descriptive only; its inferential "
|
|
595
|
+
"statistics are no longer valid, so a revised model "
|
|
596
|
+
"requires cross-validation on new data.",
|
|
597
|
+
""]
|
|
598
|
+
return out
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def _intr_tolerance(result, d):
|
|
602
|
+
tol = np.asarray(result.tolerance, dtype=float)
|
|
603
|
+
names = list(result.estimates.index[1:])
|
|
604
|
+
lo = [names[i] for i in range(len(tol)) if tol[i] < 0.20]
|
|
605
|
+
out = []
|
|
606
|
+
if lo:
|
|
607
|
+
hh = ("s have tolerances" if len(lo) > 1
|
|
608
|
+
else " has a tolerance")
|
|
609
|
+
out.append(
|
|
610
|
+
f"Collinearity is indicated. {_xU(_xNum(len(lo)))} "
|
|
611
|
+
f"variable{hh} less than the cutoff of 0.20: "
|
|
612
|
+
f"{_xAnd(lo)}.")
|
|
613
|
+
if len(lo) == 0:
|
|
614
|
+
out.append("No collinearity exists according to the "
|
|
615
|
+
"tolerance cutoff of 0.20.")
|
|
616
|
+
j = int(np.argmin(tol))
|
|
617
|
+
out.append(
|
|
618
|
+
f"The predictor variable with the lowest tolerance is "
|
|
619
|
+
f"{names[j]} at {_xP(tol[j], 3)}.")
|
|
620
|
+
return out + [""]
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
# ------------------------------------------------------------
|
|
624
|
+
# 6 - Influence
|
|
625
|
+
# ------------------------------------------------------------
|
|
626
|
+
|
|
627
|
+
_SORTCOL = {"cooks": "cooks", "rstudent": "rstdnt",
|
|
628
|
+
"dffits": "dffits"}
|
|
629
|
+
_SORTLBL = {"cooks": "Cook's distances",
|
|
630
|
+
"rstudent": "Studentized residuals",
|
|
631
|
+
"dffits": "dffits values"}
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def _sec_influence(Y, result, res_sort, d, code):
|
|
635
|
+
out = ["## Influence", "",
|
|
636
|
+
"Which cases (rows of data) contribute the most to "
|
|
637
|
+
"the lack of fit? For each case the analysis reports "
|
|
638
|
+
"the residual $e = Y - \\hat Y$, the externally "
|
|
639
|
+
"Studentized residual (_rstudent_), the standardized "
|
|
640
|
+
"change in fit (_dffits_), and Cook's Distance "
|
|
641
|
+
"(_cooks_), the aggregate influence of the case on "
|
|
642
|
+
"all the fitted values.",
|
|
643
|
+
""]
|
|
644
|
+
out += _chunk(["r.residuals"], echo=code)
|
|
645
|
+
col = _SORTCOL.get(res_sort, "cooks")
|
|
646
|
+
if col in result.residuals.columns:
|
|
647
|
+
vals = result.residuals[col].to_numpy(dtype=float)[:5]
|
|
648
|
+
top = ", ".join(_xP(v, 3) for v in vals)
|
|
649
|
+
out += [
|
|
650
|
+
f"The largest {_SORTLBL.get(res_sort, 'values')}: "
|
|
651
|
+
f"{top}.",
|
|
652
|
+
""]
|
|
653
|
+
if res_sort == "cooks":
|
|
654
|
+
labels = result.residuals.index[
|
|
655
|
+
result.residuals["cooks"] > 1]
|
|
656
|
+
if len(labels):
|
|
657
|
+
out.append(
|
|
658
|
+
"The following cases exceed the informal "
|
|
659
|
+
"Cook's Distance cutoff of 1: "
|
|
660
|
+
f"{_xAnd([f'Row {i}' for i in labels])}.")
|
|
661
|
+
else:
|
|
662
|
+
out.append("No cases have a Cook's Distance "
|
|
663
|
+
"larger than 1 in this analysis.")
|
|
664
|
+
out.append("")
|
|
665
|
+
return out
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
# ------------------------------------------------------------
|
|
669
|
+
# 7 - Prediction
|
|
670
|
+
# ------------------------------------------------------------
|
|
671
|
+
|
|
672
|
+
def _sec_prediction(Y, result, d, n_pred, code):
|
|
673
|
+
out = ["## Prediction", "", "### Prediction Error", "",
|
|
674
|
+
"Prediction moves beyond the training sample to "
|
|
675
|
+
f"predict {Y} from new values of the predictors. "
|
|
676
|
+
"Prediction error combines the modeling error $s_e$ "
|
|
677
|
+
"with the sampling error of a value on the regression "
|
|
678
|
+
"line, $s_{\\hat Y}$, into the _standard error of "
|
|
679
|
+
"prediction_.",
|
|
680
|
+
"",
|
|
681
|
+
"$$s_{\\hat Y_{p,i}} = \\sqrt{s^2_e + "
|
|
682
|
+
"s^2_{\\hat Y_{p,i}}}$$",
|
|
683
|
+
"",
|
|
684
|
+
"### From Training Data", "",
|
|
685
|
+
"Each row of data is treated _as if_ it were new, "
|
|
686
|
+
"with a predicted value, a standard error of "
|
|
687
|
+
"prediction, and a 95% prediction interval.",
|
|
688
|
+
""]
|
|
689
|
+
out += _chunk(["r.predictions"], echo=code)
|
|
690
|
+
w = result.predictions["width"]
|
|
691
|
+
imin, imax = w.idxmin(), w.idxmax()
|
|
692
|
+
out += [
|
|
693
|
+
"The width of the prediction intervals ranges from a "
|
|
694
|
+
f"minimum of {_xP(w.loc[imin], d)} for Row {imin} to a "
|
|
695
|
+
f"maximum of {_xP(w.loc[imax], d)} for Row {imax}.",
|
|
696
|
+
""]
|
|
697
|
+
return out
|
|
698
|
+
|
|
699
|
+
|
|
700
|
+
# ------------------------------------------------------------
|
|
701
|
+
# 8 - Validity
|
|
702
|
+
# ------------------------------------------------------------
|
|
703
|
+
|
|
704
|
+
def _sec_validity(code):
|
|
705
|
+
out = ["## Validity", "",
|
|
706
|
+
"The residuals should be independent, normal random "
|
|
707
|
+
"variables with a mean of zero and constant variance.",
|
|
708
|
+
"", "### Distribution of Residuals", "",
|
|
709
|
+
"For the inferential tests to be valid, the residuals "
|
|
710
|
+
"should be normally distributed.",
|
|
711
|
+
""]
|
|
712
|
+
out += _chunk(['r.plots["residuals_density"]'], echo=code)
|
|
713
|
+
out += ["### Fitted Values vs Residuals", "",
|
|
714
|
+
"The residuals should scatter randomly about 0 with "
|
|
715
|
+
"roughly constant variability across the fitted "
|
|
716
|
+
"values (_homoscedasticity_), free of any pattern.",
|
|
717
|
+
""]
|
|
718
|
+
out += _chunk(['r.plots["residuals_fitted"]'], echo=code)
|
|
719
|
+
return out
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
# ------------------------------------------------------------
|
|
723
|
+
# rendering
|
|
724
|
+
# ------------------------------------------------------------
|
|
725
|
+
|
|
726
|
+
def _tppf(df):
|
|
727
|
+
from scipy import stats as sps
|
|
728
|
+
return sps.t.ppf(0.025, df)
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def _maybe_render(path, rmd_format, rmd_browser):
|
|
732
|
+
"""Render the .qmd with the quarto CLI if it is installed and
|
|
733
|
+
a format other than "none" is requested; otherwise leave the
|
|
734
|
+
.qmd for the user to render."""
|
|
735
|
+
if rmd_format in (None, "none"):
|
|
736
|
+
return
|
|
737
|
+
if shutil.which("quarto") is None:
|
|
738
|
+
print(" [quarto CLI not found on PATH: render the "
|
|
739
|
+
f".qmd yourself with quarto render {path.name}]")
|
|
740
|
+
return
|
|
741
|
+
fmt_map = {"html": "html", "pdf": "pdf", "docx": "docx",
|
|
742
|
+
"word": "docx"}
|
|
743
|
+
to = fmt_map.get(rmd_format, "html")
|
|
744
|
+
try:
|
|
745
|
+
subprocess.run(["quarto", "render", str(path), "--to", to],
|
|
746
|
+
check=True)
|
|
747
|
+
except subprocess.CalledProcessError as e:
|
|
748
|
+
print(f" [quarto render failed: {e}]")
|
|
749
|
+
return
|
|
750
|
+
rendered = path.with_suffix("." + ("html" if to == "html"
|
|
751
|
+
else to))
|
|
752
|
+
print(f" rendered: {rendered}")
|
|
753
|
+
if rmd_browser and to == "html" and rendered.exists():
|
|
754
|
+
webbrowser.open(rendered.resolve().as_uri())
|