lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Regression.py
ADDED
|
@@ -0,0 +1,1491 @@
|
|
|
1
|
+
# Regression.py — analog of Regression.R (core numeric OLS)
|
|
2
|
+
#
|
|
3
|
+
# Regression(): least-squares regression with the lessR output
|
|
4
|
+
# pipeline — background, estimated model with confidence
|
|
5
|
+
# intervals, model fit (with the PRESS R-squared), sequential
|
|
6
|
+
# ANOVA with the aggregate Model row, collinearity, the
|
|
7
|
+
# residuals-and-influence listing, and prediction intervals —
|
|
8
|
+
# plus the plotly graphics: the simple-regression scatterplot
|
|
9
|
+
# with confidence and prediction bands (~ .reg5Plot), the
|
|
10
|
+
# distribution of residuals (~ .reg3dnResidual), and residuals
|
|
11
|
+
# vs fitted values with Cook's-distance flagging
|
|
12
|
+
# (~ .reg3resfitResidual). As with the views, the pipeline is
|
|
13
|
+
# ported, not the lines.
|
|
14
|
+
#
|
|
15
|
+
# The model is a formula string, "Y ~ X1 + X2" ("Y ~ ." takes
|
|
16
|
+
# every other numeric column), the Python analog of the R
|
|
17
|
+
# formula. Expression terms such as log(Years) or I(Years^2)
|
|
18
|
+
# are materialized as data columns (~ .formula_expr); top-level
|
|
19
|
+
# interaction/crossing operators (: * ^) are not ported.
|
|
20
|
+
# Categorical predictors become treatment-coded
|
|
21
|
+
# indicator variables (VarLevel columns, first level the
|
|
22
|
+
# reference, ~ model.matrix); with exactly one covariate and
|
|
23
|
+
# one factor, the ANOVA reports term-level Type II sums of
|
|
24
|
+
# squares — each term adjusted for the other, matching R's
|
|
25
|
+
# .reg1ancova as corrected July 2026 (its earlier row
|
|
26
|
+
# replacement mislabeled the factor's SS). The
|
|
27
|
+
# the full Regression.R analysis is ported. Rmd= generates a
|
|
28
|
+
# Quarto (.qmd) report rather than R Markdown (see reg_rmd.py).
|
|
29
|
+
# quiet= does not exist, as in R: Regression() always prints.
|
|
30
|
+
#
|
|
31
|
+
# Numerics through statsmodels OLS (imported lazily, as
|
|
32
|
+
# plt_forecast does); influence measures from OLSInfluence.
|
|
33
|
+
# Returns a RegressionResults object holding the tables as
|
|
34
|
+
# DataFrames and the plotly figures in .plots — figures are
|
|
35
|
+
# not auto-shown (no R graphics device to open).
|
|
36
|
+
|
|
37
|
+
import math
|
|
38
|
+
import re
|
|
39
|
+
|
|
40
|
+
import numpy as np
|
|
41
|
+
import pandas as pd
|
|
42
|
+
import plotly.graph_objects as go
|
|
43
|
+
from scipy import stats as sps
|
|
44
|
+
|
|
45
|
+
from .dn_plotly import dn_plotly
|
|
46
|
+
from .plt_mat_plotly import scatter_matrix
|
|
47
|
+
from .reg_rmd import reg_rmd
|
|
48
|
+
from .plotly_utils import (
|
|
49
|
+
BASE_COLORS, as_plotly_color, axis_format, axis_num,
|
|
50
|
+
make_trans, plot_border, plotly_style, to_hex, x_grid)
|
|
51
|
+
from .utils import (
|
|
52
|
+
category_order, fmt, get_column, get_option, pretty)
|
|
53
|
+
from .X import _breaks_from_args
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _dash(n):
|
|
57
|
+
return "-" * n
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _getdigits(values, min_digits=3):
|
|
61
|
+
"""Decimal digits for output: one more than the largest
|
|
62
|
+
number of decimal places in the response, at least
|
|
63
|
+
min_digits. R analog: .getdigits()"""
|
|
64
|
+
dmax = 0
|
|
65
|
+
for v in np.asarray(values, dtype=float)[:500]:
|
|
66
|
+
s = f"{v:.10f}".rstrip("0")
|
|
67
|
+
if "." in s:
|
|
68
|
+
dmax = max(dmax, len(s.split(".")[1]))
|
|
69
|
+
if dmax >= 8:
|
|
70
|
+
break
|
|
71
|
+
return max(min_digits, min(dmax + 1, 8))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _rescalable(s):
|
|
75
|
+
"""A variable is rescaled only if numeric with more than two
|
|
76
|
+
distinct values, so binary and indicator columns pass through
|
|
77
|
+
unchanged. ~ Regression.R unq.x > 2 && is.numeric"""
|
|
78
|
+
return (pd.api.types.is_numeric_dtype(s)
|
|
79
|
+
and s.dropna().nunique() > 2)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _rescale(v, kind, digits_d):
|
|
83
|
+
"""Rescale a numeric vector, missing values ignored, rounded
|
|
84
|
+
to digits_d. R analog: rescale()
|
|
85
|
+
z (x - mean) / sd (sd with n-1)
|
|
86
|
+
center x - mean
|
|
87
|
+
0to1 (x - min) / (max - min)
|
|
88
|
+
robust (x - median) / IQR"""
|
|
89
|
+
if kind == "z":
|
|
90
|
+
out = (v - np.nanmean(v)) / np.nanstd(v, ddof=1)
|
|
91
|
+
elif kind == "center":
|
|
92
|
+
out = v - np.nanmean(v)
|
|
93
|
+
elif kind == "0to1":
|
|
94
|
+
lo, hi = np.nanmin(v), np.nanmax(v)
|
|
95
|
+
out = (v - lo) / (hi - lo)
|
|
96
|
+
else: # robust
|
|
97
|
+
q1, q3 = np.nanpercentile(v, [25, 75])
|
|
98
|
+
out = (v - np.nanmedian(v)) / (q3 - q1)
|
|
99
|
+
return np.round(out, digits_d)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _split_top(s, sep):
|
|
103
|
+
"""Split s on sep at parenthesis depth 0, so a separator
|
|
104
|
+
inside a function call is not a break."""
|
|
105
|
+
out, depth, start = [], 0, 0
|
|
106
|
+
for i, ch in enumerate(s):
|
|
107
|
+
if ch in "([":
|
|
108
|
+
depth += 1
|
|
109
|
+
elif ch in ")]":
|
|
110
|
+
depth -= 1
|
|
111
|
+
elif ch == sep and depth == 0:
|
|
112
|
+
out.append(s[start:i])
|
|
113
|
+
start = i + 1
|
|
114
|
+
out.append(s[start:])
|
|
115
|
+
return [t.strip() for t in out]
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _has_top(s, chars):
|
|
119
|
+
"""True if any char in chars appears at parenthesis depth 0."""
|
|
120
|
+
depth = 0
|
|
121
|
+
for ch in s:
|
|
122
|
+
if ch in "([":
|
|
123
|
+
depth += 1
|
|
124
|
+
elif ch in ")]":
|
|
125
|
+
depth -= 1
|
|
126
|
+
elif ch in chars and depth == 0:
|
|
127
|
+
return True
|
|
128
|
+
return False
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
_BARE = re.compile(r"^[A-Za-z_.][\w.]*$")
|
|
132
|
+
_EXPR_FUN = {"log": np.log, "log10": np.log10, "log2": np.log2,
|
|
133
|
+
"sqrt": np.sqrt, "exp": np.exp, "abs": np.abs,
|
|
134
|
+
"sin": np.sin, "cos": np.cos, "tan": np.tan}
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _materialize(term, data):
|
|
138
|
+
"""Evaluate a formula expression term (log(Years), sqrt(x),
|
|
139
|
+
I(Years^2), x/100 ...) into a numeric column named by the
|
|
140
|
+
term's text, as R's .formula_expr does with terms()
|
|
141
|
+
variables that are calls. Returns the (name, data) with the
|
|
142
|
+
new column added to a copy of data."""
|
|
143
|
+
py = re.sub(r"\bI\(", "(", term.replace("^", "**"))
|
|
144
|
+
ns = {c: data[c].to_numpy() for c in data.columns}
|
|
145
|
+
ns.update(_EXPR_FUN)
|
|
146
|
+
try:
|
|
147
|
+
val = eval(py, {"__builtins__": {}}, ns) # user's own model
|
|
148
|
+
except Exception as e:
|
|
149
|
+
raise ValueError(
|
|
150
|
+
f'cannot evaluate the formula term "{term}": {e}. '
|
|
151
|
+
"Name a data column, or compute the transformed "
|
|
152
|
+
"column first.")
|
|
153
|
+
data = data.copy()
|
|
154
|
+
data[term] = np.asarray(val, dtype=float)
|
|
155
|
+
return term, data
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _resolve_term(term, data):
|
|
159
|
+
"""A bare column name stays; an expression is materialized
|
|
160
|
+
(~ .formula_expr). A top-level interaction/crossing operator
|
|
161
|
+
(: * ^) is not ported."""
|
|
162
|
+
if _BARE.match(term) and term in data.columns:
|
|
163
|
+
return term, data
|
|
164
|
+
if _has_top(term, ":*^"):
|
|
165
|
+
raise NotImplementedError(
|
|
166
|
+
f'formula operator in "{term}" is not ported: list '
|
|
167
|
+
"predictors with +, and compute any interaction "
|
|
168
|
+
"column first (I(a*b) materializes a product term)")
|
|
169
|
+
return _materialize(term, data)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _parse_formula(my_formula, data):
|
|
173
|
+
""" "Y ~ X1 + X2" -> (response, [predictors], data). "Y ~ ."
|
|
174
|
+
takes every other numeric column; "Y ~ 1" is the null model.
|
|
175
|
+
Expression terms such as log(Years) or I(Years^2) are
|
|
176
|
+
materialized as data columns (~ .formula_expr); the returned
|
|
177
|
+
data carries those columns."""
|
|
178
|
+
if not isinstance(my_formula, str):
|
|
179
|
+
raise TypeError(
|
|
180
|
+
"the model is a formula string, such as "
|
|
181
|
+
'"Salary ~ Years + Pre"')
|
|
182
|
+
if my_formula.count("~") != 1:
|
|
183
|
+
raise ValueError(
|
|
184
|
+
'the formula has one "~": "Y ~ X1 + X2"')
|
|
185
|
+
lhs, rhs = (s.strip() for s in my_formula.split("~"))
|
|
186
|
+
if not lhs:
|
|
187
|
+
raise ValueError("the formula names the response "
|
|
188
|
+
'before the "~"')
|
|
189
|
+
y_name, data = _resolve_term(lhs, data)
|
|
190
|
+
if rhs == ".":
|
|
191
|
+
preds = [c for c in data.columns
|
|
192
|
+
if c != y_name
|
|
193
|
+
and pd.api.types.is_numeric_dtype(data[c])]
|
|
194
|
+
elif rhs in ("1", ""):
|
|
195
|
+
preds = []
|
|
196
|
+
else:
|
|
197
|
+
terms = _split_top(rhs, "+")
|
|
198
|
+
if any(not t for t in terms):
|
|
199
|
+
raise ValueError(f'cannot parse "{rhs}": list the '
|
|
200
|
+
"predictors separated by +")
|
|
201
|
+
preds = []
|
|
202
|
+
for t in terms:
|
|
203
|
+
nm, data = _resolve_term(t, data)
|
|
204
|
+
preds.append(nm)
|
|
205
|
+
return y_name, preds, data
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _expand_indicators(pred_names, pred_sers, note_fmt):
|
|
209
|
+
"""Treatment-coded indicator variables for the categorical
|
|
210
|
+
predictors, in formula position: the first level (declared
|
|
211
|
+
Categorical order, else sorted) is the reference, and each
|
|
212
|
+
other level becomes a 0/1 column named VarLevel, the naming
|
|
213
|
+
of R's model.matrix(). Returns (names, series, notes,
|
|
214
|
+
cat_names). R analog: the "construct the indicator
|
|
215
|
+
variables" block of Regression.R / Logit.R"""
|
|
216
|
+
from .utils import category_order
|
|
217
|
+
names, sers, notes, cat_names = [], [], [], []
|
|
218
|
+
term_map = {} # original predictor -> its columns
|
|
219
|
+
for nm, s in zip(pred_names, pred_sers):
|
|
220
|
+
if pd.api.types.is_numeric_dtype(s):
|
|
221
|
+
names.append(nm)
|
|
222
|
+
sers.append(s)
|
|
223
|
+
term_map[nm] = [nm]
|
|
224
|
+
continue
|
|
225
|
+
cat_names.append(nm)
|
|
226
|
+
notes.append(note_fmt.format(nm))
|
|
227
|
+
term_map[nm] = []
|
|
228
|
+
for lv in category_order(
|
|
229
|
+
s if isinstance(s.dtype, pd.CategoricalDtype)
|
|
230
|
+
else s.astype(str))[1:]:
|
|
231
|
+
dnm = f"{nm}{lv}"
|
|
232
|
+
names.append(dnm)
|
|
233
|
+
term_map[nm].append(dnm)
|
|
234
|
+
sers.append((s.astype(str) == str(lv))
|
|
235
|
+
.astype(float).rename(dnm))
|
|
236
|
+
return names, sers, notes, cat_names, term_map
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _best_subsets(Xd, yv, tot_ss, MSW, best_sub, nbest=10):
|
|
240
|
+
"""Best subset regressions over all predictor subsets,
|
|
241
|
+
keeping the nbest best of each size (the leaps() default),
|
|
242
|
+
scored by adjusted R-squared or Mallows' Cp on the full
|
|
243
|
+
model's error variance. Engine deviation: an exhaustive
|
|
244
|
+
all-subsets search replaces the leaps branch-and-bound —
|
|
245
|
+
same results, no size limit beyond practicality.
|
|
246
|
+
R analog: leaps::leaps() in .reg2Relations"""
|
|
247
|
+
from itertools import combinations
|
|
248
|
+
names = list(Xd.columns)
|
|
249
|
+
p = len(names)
|
|
250
|
+
if p > 15:
|
|
251
|
+
return None # 2^15 solves is the cap
|
|
252
|
+
n = len(yv)
|
|
253
|
+
Xc = Xd.to_numpy(dtype=float)
|
|
254
|
+
rows = []
|
|
255
|
+
for k in range(1, p + 1):
|
|
256
|
+
size_rows = []
|
|
257
|
+
for cols in combinations(range(p), k):
|
|
258
|
+
Xs = np.column_stack(
|
|
259
|
+
[np.ones(n), Xc[:, list(cols)]])
|
|
260
|
+
rss = float(((yv - Xs @ np.linalg.lstsq(
|
|
261
|
+
Xs, yv, rcond=None)[0]) ** 2).sum())
|
|
262
|
+
if best_sub == "adjr2":
|
|
263
|
+
crit = 1 - (rss / (n - k - 1)) \
|
|
264
|
+
/ (tot_ss / (n - 1))
|
|
265
|
+
else: # Mallows' Cp
|
|
266
|
+
crit = rss / MSW - (n - 2 * (k + 1))
|
|
267
|
+
size_rows.append((cols, crit))
|
|
268
|
+
size_rows.sort(key=lambda r: r[1],
|
|
269
|
+
reverse=best_sub == "adjr2")
|
|
270
|
+
rows += size_rows[:nbest]
|
|
271
|
+
lbl = "R2adj" if best_sub == "adjr2" else "Cp"
|
|
272
|
+
out = pd.DataFrame(
|
|
273
|
+
[[int(j in cols) for j in range(p)] + [crit, len(cols)]
|
|
274
|
+
for cols, crit in rows],
|
|
275
|
+
columns=names + [lbl, "X's"])
|
|
276
|
+
return out.sort_values(
|
|
277
|
+
lbl, ascending=best_sub != "adjr2",
|
|
278
|
+
kind="stable").reset_index(drop=True)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _prntbl(df, digits_d, int_cols=()):
|
|
282
|
+
"""Aligned text table with row labels, floats at digits_d.
|
|
283
|
+
Columns named in int_cols print as integers. R analog:
|
|
284
|
+
.prntbl()"""
|
|
285
|
+
show = pd.DataFrame(index=df.index.astype(str))
|
|
286
|
+
for c in df.columns:
|
|
287
|
+
col = df[c]
|
|
288
|
+
if c in int_cols:
|
|
289
|
+
show[c] = ["" if pd.isna(v) else str(int(round(v)))
|
|
290
|
+
for v in col]
|
|
291
|
+
elif pd.api.types.is_float_dtype(col):
|
|
292
|
+
show[c] = [fmt(v, digits_d) if pd.notna(v) else ""
|
|
293
|
+
for v in col]
|
|
294
|
+
else:
|
|
295
|
+
show[c] = col.astype(str)
|
|
296
|
+
return show.to_string()
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _int_vars(df, names):
|
|
300
|
+
"""Of the given data columns, those whose values are all whole
|
|
301
|
+
numbers -- shown without decimals, as the integer variables
|
|
302
|
+
they are (an integer column with a missing value elsewhere is
|
|
303
|
+
stored as float, so this is decided by value, not dtype)."""
|
|
304
|
+
out = []
|
|
305
|
+
for nm in names:
|
|
306
|
+
if nm not in df.columns:
|
|
307
|
+
continue
|
|
308
|
+
col = df[nm]
|
|
309
|
+
if not pd.api.types.is_numeric_dtype(col):
|
|
310
|
+
continue
|
|
311
|
+
v = col.to_numpy(dtype=float)
|
|
312
|
+
v = v[np.isfinite(v)]
|
|
313
|
+
if len(v) and np.all(v == np.floor(v)):
|
|
314
|
+
out.append(nm)
|
|
315
|
+
return out
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
class RegressionResults:
|
|
319
|
+
"""Numeric results and figures of Regression(): estimates,
|
|
320
|
+
fit, anova (DataFrames/dict), residuals and predictions
|
|
321
|
+
listings, and the plotly figures in .plots."""
|
|
322
|
+
|
|
323
|
+
def __init__(self, **kw):
|
|
324
|
+
self.__dict__.update(kw)
|
|
325
|
+
|
|
326
|
+
def __repr__(self):
|
|
327
|
+
return (f"<lessPy Regression: {self.formula}, "
|
|
328
|
+
f"n={self.n_keep}>")
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def Regression(my_formula, data=None, filter=None, digits_d=None,
|
|
332
|
+
brief=False,
|
|
333
|
+
n_res_rows=None, res_sort="cooks",
|
|
334
|
+
n_pred_rows=None, pred_sort="predint",
|
|
335
|
+
subsets=None, best_sub="adjr2",
|
|
336
|
+
cooks_cut=1,
|
|
337
|
+
X1_new=None, X2_new=None, X3_new=None,
|
|
338
|
+
X4_new=None, X5_new=None, X6_new=None,
|
|
339
|
+
kfold=0, seed=None,
|
|
340
|
+
new_scale="none", scale_response=False,
|
|
341
|
+
mod=None, mod_transf="center",
|
|
342
|
+
Rmd=None, Rmd_data=None, Rmd_format="html",
|
|
343
|
+
Rmd_browser=True,
|
|
344
|
+
results=True, explain=True, interpret=True,
|
|
345
|
+
code=True,
|
|
346
|
+
graphics=True):
|
|
347
|
+
"""Least-squares regression of a formula string,
|
|
348
|
+
"Y ~ X1 + X2", with the lessR analysis pipeline: estimates,
|
|
349
|
+
fit, ANOVA, collinearity, residuals and influence,
|
|
350
|
+
prediction intervals, and the regression graphics. Always
|
|
351
|
+
prints, as in R; returns a RegressionResults object with
|
|
352
|
+
the figures in .plots."""
|
|
353
|
+
import statsmodels.api as smapi
|
|
354
|
+
|
|
355
|
+
if data is None:
|
|
356
|
+
raise ValueError(
|
|
357
|
+
"data= is required: a pandas DataFrame containing "
|
|
358
|
+
"the model's variables")
|
|
359
|
+
if res_sort not in ("cooks", "rstudent", "dffits", "off"):
|
|
360
|
+
raise ValueError(
|
|
361
|
+
'res_sort: "cooks", "rstudent", "dffits", or "off"')
|
|
362
|
+
if pred_sort not in ("predint", "off"):
|
|
363
|
+
raise ValueError('pred_sort: "predint" or "off"')
|
|
364
|
+
if best_sub not in ("adjr2", "Cp"):
|
|
365
|
+
raise ValueError('best_sub: "adjr2" or "Cp"')
|
|
366
|
+
if new_scale not in ("none", "z", "center", "0to1", "robust"):
|
|
367
|
+
raise ValueError('new_scale: "none", "z", "center", '
|
|
368
|
+
'"0to1", or "robust"')
|
|
369
|
+
if mod_transf not in ("center", "z", "none"):
|
|
370
|
+
raise ValueError('mod_transf: "center", "z", or "none"')
|
|
371
|
+
if Rmd is not None:
|
|
372
|
+
if Rmd_format not in ("html", "pdf", "docx", "word",
|
|
373
|
+
"none"):
|
|
374
|
+
raise ValueError('Rmd_format: "html", "pdf", '
|
|
375
|
+
'"docx", or "none"')
|
|
376
|
+
if brief:
|
|
377
|
+
raise ValueError(
|
|
378
|
+
"a Quarto report needs the full analysis, so "
|
|
379
|
+
"Rmd= is not available with brief=True")
|
|
380
|
+
if filter is not None:
|
|
381
|
+
data = data.query(filter)
|
|
382
|
+
|
|
383
|
+
y_name, pred_names, data = _parse_formula(my_formula, data)
|
|
384
|
+
formula = (f"{y_name} ~ "
|
|
385
|
+
+ (" + ".join(pred_names) if pred_names else "1"))
|
|
386
|
+
n_pred = len(pred_names)
|
|
387
|
+
|
|
388
|
+
y_ser = get_column(data, y_name, "response")
|
|
389
|
+
if not pd.api.types.is_numeric_dtype(y_ser):
|
|
390
|
+
raise TypeError(
|
|
391
|
+
f"'{y_name}' is {y_ser.dtype}: the response of "
|
|
392
|
+
"Regression() is numeric. For a two-level "
|
|
393
|
+
"categorical response use Logit().")
|
|
394
|
+
pred_sers = [get_column(data, nm, "predictor")
|
|
395
|
+
for nm in pred_names]
|
|
396
|
+
|
|
397
|
+
# digits from the response, before any rescaling (~ R sets
|
|
398
|
+
# digits_d here, ahead of the new_scale block)
|
|
399
|
+
if digits_d is None:
|
|
400
|
+
digits_d = _getdigits(
|
|
401
|
+
y_ser.dropna().to_numpy(dtype=float))
|
|
402
|
+
d = digits_d
|
|
403
|
+
|
|
404
|
+
# new_scale: rescale the numeric model variables in place --
|
|
405
|
+
# predictors always, the response only when scale_response;
|
|
406
|
+
# binary and non-numeric variables pass through unchanged
|
|
407
|
+
# (~ Regression.R rescale block, rescale.R). For kfold the
|
|
408
|
+
# rescaling is per fold, done in _reg_kfold.
|
|
409
|
+
transf = None
|
|
410
|
+
rescale_lines = []
|
|
411
|
+
if new_scale != "none":
|
|
412
|
+
transf = {"z": "Standardized", "center": "Centered",
|
|
413
|
+
"0to1": "Min-Max (0 to 1)",
|
|
414
|
+
"robust": "Robust Version of Standardized"
|
|
415
|
+
}[new_scale]
|
|
416
|
+
if kfold == 0:
|
|
417
|
+
if scale_response and _rescalable(y_ser):
|
|
418
|
+
y_ser = pd.Series(
|
|
419
|
+
_rescale(y_ser.to_numpy(dtype=float),
|
|
420
|
+
new_scale, d),
|
|
421
|
+
index=y_ser.index, name=y_name)
|
|
422
|
+
pred_sers = [
|
|
423
|
+
pd.Series(_rescale(s.to_numpy(dtype=float),
|
|
424
|
+
new_scale, d),
|
|
425
|
+
index=s.index, name=s.name)
|
|
426
|
+
if _rescalable(s) else s
|
|
427
|
+
for s in pred_sers]
|
|
428
|
+
head = pd.concat(
|
|
429
|
+
[y_ser] + pred_sers, axis=1).head(6)
|
|
430
|
+
rescale_lines = (
|
|
431
|
+
["Rescaled Data, First Six Rows", ""]
|
|
432
|
+
+ head.to_string().split("\n") + [""])
|
|
433
|
+
|
|
434
|
+
# mod: moderation. Add the X*W interaction and refit
|
|
435
|
+
# Y ~ X + W + X*W. The two predictors are first centered (or
|
|
436
|
+
# standardized) per mod_transf, which curbs the collinearity
|
|
437
|
+
# the product term would otherwise introduce. ~ Regression.R
|
|
438
|
+
# mod block, .reg6mod
|
|
439
|
+
mod_info = None
|
|
440
|
+
if mod is not None:
|
|
441
|
+
if n_pred != 2:
|
|
442
|
+
raise ValueError(
|
|
443
|
+
"mod moderation currently needs exactly 2 "
|
|
444
|
+
"predictors")
|
|
445
|
+
if mod not in pred_names:
|
|
446
|
+
raise ValueError(
|
|
447
|
+
f"mod variable '{mod}' must be one of the two "
|
|
448
|
+
"predictors")
|
|
449
|
+
if any(not pd.api.types.is_numeric_dtype(s)
|
|
450
|
+
for s in pred_sers):
|
|
451
|
+
raise TypeError(
|
|
452
|
+
"both predictors of a moderation must be numeric")
|
|
453
|
+
cc = pd.concat([y_ser] + pred_sers,
|
|
454
|
+
axis=1).notna().all(axis=1)
|
|
455
|
+
if mod_transf != "none":
|
|
456
|
+
is_z = mod_transf == "z"
|
|
457
|
+
scaled = []
|
|
458
|
+
for s in pred_sers:
|
|
459
|
+
v = s.astype(float)
|
|
460
|
+
mu = v[cc].mean()
|
|
461
|
+
v = ((v - mu) / v[cc].std(ddof=1) if is_z
|
|
462
|
+
else v - mu)
|
|
463
|
+
scaled.append(pd.Series(v, index=s.index,
|
|
464
|
+
name=s.name))
|
|
465
|
+
pred_sers = scaled
|
|
466
|
+
w_name = mod
|
|
467
|
+
x_name = next(p for p in pred_names if p != mod)
|
|
468
|
+
xw_name = f"{w_name}.{x_name}"
|
|
469
|
+
inter = pd.Series(
|
|
470
|
+
pred_sers[0].to_numpy(dtype=float)
|
|
471
|
+
* pred_sers[1].to_numpy(dtype=float),
|
|
472
|
+
index=pred_sers[0].index, name=xw_name)
|
|
473
|
+
pred_names = pred_names + [xw_name]
|
|
474
|
+
pred_sers = pred_sers + [inter]
|
|
475
|
+
mod_info = {"x": x_name, "w": w_name, "xw": xw_name}
|
|
476
|
+
|
|
477
|
+
used = pd.concat([y_ser] + pred_sers, axis=1)
|
|
478
|
+
keep = ~used.isna().any(axis=1)
|
|
479
|
+
n_obs = len(data)
|
|
480
|
+
n_keep = int(keep.sum())
|
|
481
|
+
|
|
482
|
+
# categorical predictors become indicator variables; with
|
|
483
|
+
# exactly one covariate and one factor, the ANOVA reports
|
|
484
|
+
# Type II sums of squares (the ANCOVA table)
|
|
485
|
+
(pred_names, pred_sers_x, ind_notes, cat_names,
|
|
486
|
+
term_map) = _expand_indicators(
|
|
487
|
+
pred_names, [s[keep] for s in pred_sers],
|
|
488
|
+
">>> {0} is not numeric. "
|
|
489
|
+
"Converted to indicator variables.")
|
|
490
|
+
ancova = (n_pred == 2 and len(cat_names) == 1)
|
|
491
|
+
ancova_terms = ([nm for nm in term_map
|
|
492
|
+
if nm not in cat_names] + cat_names
|
|
493
|
+
if ancova else None)
|
|
494
|
+
ancova_info = None
|
|
495
|
+
if ancova:
|
|
496
|
+
# the covariate is the numeric term (maps to itself), the
|
|
497
|
+
# factor is the one categorical predictor; keep its
|
|
498
|
+
# original values, in level order, for the group plot
|
|
499
|
+
cov_name = next(k for k, v in term_map.items()
|
|
500
|
+
if v == [k])
|
|
501
|
+
fac_name = cat_names[0]
|
|
502
|
+
fac_series = next(s for s in pred_sers
|
|
503
|
+
if s.name == fac_name)[keep]
|
|
504
|
+
fac_series = (fac_series
|
|
505
|
+
if isinstance(fac_series.dtype,
|
|
506
|
+
pd.CategoricalDtype)
|
|
507
|
+
else fac_series.astype(str))
|
|
508
|
+
ancova_info = {
|
|
509
|
+
"cov": cov_name, "fac": fac_name,
|
|
510
|
+
"preds": list(term_map.keys()),
|
|
511
|
+
"levels": [str(lv) for lv in
|
|
512
|
+
category_order(fac_series)],
|
|
513
|
+
"fac_vals": fac_series.astype(str).to_numpy()}
|
|
514
|
+
# collinearity and subsets follow the count of listed
|
|
515
|
+
# predictors, before indicator expansion, as R
|
|
516
|
+
n_pred_orig = n_pred
|
|
517
|
+
n_pred = len(pred_names)
|
|
518
|
+
|
|
519
|
+
if n_keep < n_pred + 2:
|
|
520
|
+
raise ValueError(
|
|
521
|
+
f"only {n_keep} complete rows: too few to estimate "
|
|
522
|
+
f"{n_pred + 1} coefficients")
|
|
523
|
+
yv = y_ser[keep].to_numpy(dtype=float)
|
|
524
|
+
Xd = pd.DataFrame(
|
|
525
|
+
{nm: s.to_numpy(dtype=float)
|
|
526
|
+
for nm, s in zip(pred_names, pred_sers_x)},
|
|
527
|
+
index=y_ser[keep].index)
|
|
528
|
+
row_labels = Xd.index.astype(str)
|
|
529
|
+
|
|
530
|
+
if kfold and kfold > 0:
|
|
531
|
+
# cross-validation replaces the single-model analysis;
|
|
532
|
+
# graphics and the residual/prediction listings are off,
|
|
533
|
+
# exactly as R turns them off for kfold > 0
|
|
534
|
+
if n_pred == 0:
|
|
535
|
+
raise ValueError(
|
|
536
|
+
"kfold cross-validation needs at least one "
|
|
537
|
+
"predictor")
|
|
538
|
+
return _reg_kfold(yv, Xd, y_name, pred_names, n_keep,
|
|
539
|
+
formula, kfold, seed, d, smapi,
|
|
540
|
+
new_scale, scale_response)
|
|
541
|
+
|
|
542
|
+
X = smapi.add_constant(Xd, has_constant="add") \
|
|
543
|
+
if n_pred > 0 else pd.DataFrame(
|
|
544
|
+
{"const": np.ones(n_keep)}, index=Xd.index)
|
|
545
|
+
fit = smapi.OLS(yv, X).fit()
|
|
546
|
+
infl = fit.get_influence()
|
|
547
|
+
hat = infl.hat_matrix_diag
|
|
548
|
+
rstudent = infl.resid_studentized_external
|
|
549
|
+
dffits = infl.dffits[0]
|
|
550
|
+
cooks = infl.cooks_distance[0]
|
|
551
|
+
resid = np.asarray(fit.resid, dtype=float)
|
|
552
|
+
fitted = np.asarray(fit.fittedvalues, dtype=float)
|
|
553
|
+
df_res = int(fit.df_resid)
|
|
554
|
+
|
|
555
|
+
lines = list(rescale_lines) # console output
|
|
556
|
+
for note in ind_notes:
|
|
557
|
+
lines += [note, ""]
|
|
558
|
+
|
|
559
|
+
# ---------- BACKGROUND -------------------------------------
|
|
560
|
+
lines += ["", " BACKGROUND", ""]
|
|
561
|
+
for i, nm in enumerate([y_name] + pred_names):
|
|
562
|
+
if i == 0:
|
|
563
|
+
lbl = "Response Variable: "
|
|
564
|
+
elif n_pred > 1:
|
|
565
|
+
lbl = f"Predictor Variable {i}: "
|
|
566
|
+
else:
|
|
567
|
+
lbl = "Predictor Variable: "
|
|
568
|
+
lines.append(lbl + nm)
|
|
569
|
+
if transf is not None:
|
|
570
|
+
lines += ["", f"Data are {transf}"]
|
|
571
|
+
lines += ["",
|
|
572
|
+
f"Number of cases (rows) of data: {n_obs}",
|
|
573
|
+
f"Number of cases retained for analysis: "
|
|
574
|
+
f"{n_keep}"]
|
|
575
|
+
|
|
576
|
+
# ---------- BASIC ANALYSIS ---------------------------------
|
|
577
|
+
lines += ["", "", " BASIC ANALYSIS", ""]
|
|
578
|
+
|
|
579
|
+
# estimates with 95% confidence intervals, ~ .reg1modelBasic
|
|
580
|
+
ci = fit.conf_int(alpha=0.05)
|
|
581
|
+
est = pd.DataFrame({
|
|
582
|
+
"Estimate": fit.params,
|
|
583
|
+
"Std Err": fit.bse,
|
|
584
|
+
"t-value": fit.tvalues,
|
|
585
|
+
"p-value": fit.pvalues,
|
|
586
|
+
"Lower 95%": ci[0],
|
|
587
|
+
"Upper 95%": ci[1],
|
|
588
|
+
})
|
|
589
|
+
est.index = ["(Intercept)"] + pred_names
|
|
590
|
+
lines += [f"-- Estimated Model for {y_name}", ""]
|
|
591
|
+
buf = max(len(s) for s in est.index)
|
|
592
|
+
w = [max(9, max(len(fmt(v, d)) for v in est[c]) + 1)
|
|
593
|
+
for c in est.columns]
|
|
594
|
+
lines.append(" " * buf
|
|
595
|
+
+ f"{'Estimate':>{w[0] + 1}}"
|
|
596
|
+
+ f"{'Std Err':>{w[1] + 2}}"
|
|
597
|
+
+ f"{'t-value':>9}{'p-value':>9}"
|
|
598
|
+
+ f"{'Lower 95%':>{w[4] + 3}}"
|
|
599
|
+
+ f"{'Upper 95%':>{w[5] + 3}}")
|
|
600
|
+
for lbl, r in est.iterrows():
|
|
601
|
+
lines.append(
|
|
602
|
+
f"{lbl:<{buf}}"
|
|
603
|
+
+ f"{fmt(r['Estimate'], d):>{w[0] + 1}}"
|
|
604
|
+
+ f"{fmt(r['Std Err'], d):>{w[1] + 2}}"
|
|
605
|
+
+ f"{fmt(r['t-value'], 3):>9}"
|
|
606
|
+
+ f"{fmt(r['p-value'], 3):>9}"
|
|
607
|
+
+ f"{fmt(r['Lower 95%'], d):>{w[4] + 3}}"
|
|
608
|
+
+ f"{fmt(r['Upper 95%'], d):>{w[5] + 3}}")
|
|
609
|
+
|
|
610
|
+
# model fit, ~ .reg1fitBasic
|
|
611
|
+
tot_ss = float(((yv - yv.mean()) ** 2).sum())
|
|
612
|
+
sy = math.sqrt(tot_ss / (n_keep - 1))
|
|
613
|
+
se = math.sqrt(fit.scale)
|
|
614
|
+
tcut = -sps.t.ppf(0.025, df=df_res)
|
|
615
|
+
res_range = 2 * tcut * se
|
|
616
|
+
prs = resid / (1 - hat)
|
|
617
|
+
prs = np.where(hat >= 1 - 1e-13, np.nan, prs)
|
|
618
|
+
PRESS = float(np.nansum(prs ** 2))
|
|
619
|
+
Rsq_press = (np.nan if tot_ss == 0
|
|
620
|
+
else 1 - PRESS / tot_ss)
|
|
621
|
+
lines += ["", "-- Model Fit", "",
|
|
622
|
+
f"Standard deviation of {y_name}: {fmt(sy, d)}",
|
|
623
|
+
"",
|
|
624
|
+
f"Standard deviation of residuals: {fmt(se, d)}"
|
|
625
|
+
f" for df={df_res}",
|
|
626
|
+
f"95% range of residuals: {fmt(res_range, d)}"
|
|
627
|
+
f" = 2 * ({fmt(tcut, 3)} * {fmt(se, d)})"]
|
|
628
|
+
if n_pred > 0:
|
|
629
|
+
lines += ["",
|
|
630
|
+
f"R-squared: {fmt(fit.rsquared, 3)} "
|
|
631
|
+
f"Adjusted R-squared: "
|
|
632
|
+
f"{fmt(fit.rsquared_adj, 3)} "
|
|
633
|
+
f"PRESS R-squared: {fmt(Rsq_press, 3)}",
|
|
634
|
+
"",
|
|
635
|
+
"Null hypothesis of all 0 population slope "
|
|
636
|
+
"coefficients:",
|
|
637
|
+
f" F-statistic: {fmt(fit.fvalue, 3)} "
|
|
638
|
+
f"df: {int(fit.df_model)} and {df_res} "
|
|
639
|
+
f"p-value: {fmt(fit.f_pvalue, 3)}"]
|
|
640
|
+
|
|
641
|
+
# ANOVA: sequential (Type I) with the Model row,
|
|
642
|
+
# ~ .reg1anvBasic; for the ANCOVA case (one covariate, one
|
|
643
|
+
# factor) term-level Type II sums of squares, each term
|
|
644
|
+
# adjusted for the other
|
|
645
|
+
anova = None
|
|
646
|
+
MSW = fit.scale
|
|
647
|
+
res_ss = float(fit.ssr)
|
|
648
|
+
if n_pred > 0 and ancova:
|
|
649
|
+
rows = []
|
|
650
|
+
for t in ancova_terms:
|
|
651
|
+
others = [c for c in pred_names
|
|
652
|
+
if c not in term_map[t]]
|
|
653
|
+
Xr = smapi.add_constant(Xd[others],
|
|
654
|
+
has_constant="add")
|
|
655
|
+
ss = float(smapi.OLS(yv, Xr).fit().ssr) - res_ss
|
|
656
|
+
dft = len(term_map[t])
|
|
657
|
+
f_v = (ss / dft) / MSW
|
|
658
|
+
rows.append([t, dft, ss, ss / dft, f_v,
|
|
659
|
+
float(sps.f.sf(f_v, dft, df_res))])
|
|
660
|
+
anova = pd.DataFrame(
|
|
661
|
+
rows + [["Residuals", df_res, res_ss, MSW,
|
|
662
|
+
np.nan, np.nan]],
|
|
663
|
+
columns=["term", "df", "Sum Sq", "Mean Sq",
|
|
664
|
+
"F-value", "p-value"]).set_index("term")
|
|
665
|
+
lines += ["", "-- Analysis of Variance from Type II "
|
|
666
|
+
"Sums of Squares", ""]
|
|
667
|
+
anv_names = ancova_terms
|
|
668
|
+
elif n_pred > 0:
|
|
669
|
+
seq_ss = []
|
|
670
|
+
rss_prev = tot_ss
|
|
671
|
+
for i in range(1, n_pred + 1):
|
|
672
|
+
Xi = smapi.add_constant(Xd.iloc[:, :i],
|
|
673
|
+
has_constant="add")
|
|
674
|
+
rss_i = float(smapi.OLS(yv, Xi).fit().ssr)
|
|
675
|
+
seq_ss.append(rss_prev - rss_i)
|
|
676
|
+
rss_prev = rss_i
|
|
677
|
+
rows = []
|
|
678
|
+
for nm, ss in zip(pred_names, seq_ss):
|
|
679
|
+
f_v = ss / MSW
|
|
680
|
+
rows.append([nm, 1, ss, ss, f_v,
|
|
681
|
+
float(sps.f.sf(f_v, 1, df_res))])
|
|
682
|
+
mod_df = n_pred
|
|
683
|
+
mod_ss = sum(seq_ss)
|
|
684
|
+
mod_ms = mod_ss / mod_df
|
|
685
|
+
mod_f = mod_ms / MSW
|
|
686
|
+
mod_p = float(sps.f.sf(mod_f, mod_df, df_res))
|
|
687
|
+
anova = pd.DataFrame(
|
|
688
|
+
rows + [["Model", mod_df, mod_ss, mod_ms, mod_f,
|
|
689
|
+
mod_p],
|
|
690
|
+
["Residuals", df_res, res_ss, MSW,
|
|
691
|
+
np.nan, np.nan],
|
|
692
|
+
[y_name, n_keep - 1, tot_ss,
|
|
693
|
+
tot_ss / (n_keep - 1), np.nan, np.nan]],
|
|
694
|
+
columns=["term", "df", "Sum Sq", "Mean Sq",
|
|
695
|
+
"F-value", "p-value"]).set_index("term")
|
|
696
|
+
lines += ["", "-- Analysis of Variance", ""]
|
|
697
|
+
anv_names = pred_names
|
|
698
|
+
if n_pred > 0:
|
|
699
|
+
c1 = max(len(s) for s in anova.index)
|
|
700
|
+
wn = [max(9, max(len(fmt(v, d))
|
|
701
|
+
for v in anova[c].dropna()) + 1)
|
|
702
|
+
for c in ("Sum Sq", "Mean Sq", "F-value")]
|
|
703
|
+
lines.append(" " * c1 + f"{'df':>7}"
|
|
704
|
+
+ f"{'Sum Sq':>{wn[0]}}"
|
|
705
|
+
+ f"{'Mean Sq':>{wn[1]}}"
|
|
706
|
+
+ f"{'F-value':>{wn[2]}}"
|
|
707
|
+
+ f"{'p-value':>9}")
|
|
708
|
+
|
|
709
|
+
def anv_line(lbl, r, with_test=True):
|
|
710
|
+
t = (f"{lbl:<{c1}}{int(r['df']):>7}"
|
|
711
|
+
+ f"{fmt(r['Sum Sq'], d):>{wn[0]}}"
|
|
712
|
+
+ f"{fmt(r['Mean Sq'], d):>{wn[1]}}")
|
|
713
|
+
if with_test:
|
|
714
|
+
t += (f"{fmt(r['F-value'], d):>{wn[2]}}"
|
|
715
|
+
+ f"{fmt(r['p-value'], 3):>9}")
|
|
716
|
+
return t
|
|
717
|
+
|
|
718
|
+
for nm in anv_names:
|
|
719
|
+
lines.append(anv_line(nm, anova.loc[nm]))
|
|
720
|
+
if not ancova: # Type II has no total
|
|
721
|
+
lines.append("")
|
|
722
|
+
lines.append(anv_line("Model",
|
|
723
|
+
anova.loc["Model"]))
|
|
724
|
+
lines.append(anv_line("Residuals",
|
|
725
|
+
anova.loc["Residuals"],
|
|
726
|
+
with_test=False))
|
|
727
|
+
if not ancova:
|
|
728
|
+
lines.append(anv_line(y_name, anova.loc[y_name],
|
|
729
|
+
with_test=False))
|
|
730
|
+
|
|
731
|
+
# ---------- MODERATION ANALYSIS ----------------------------
|
|
732
|
+
# the simple slopes at the moderator's mean and +/-1 SD; as R,
|
|
733
|
+
# this section is generated with the graphics (~ .reg6mod)
|
|
734
|
+
if mod_info is not None and graphics:
|
|
735
|
+
lines += _moderation_lines(fit, Xd, mod_info, d)
|
|
736
|
+
|
|
737
|
+
# ---------- ANCOVA GROUP MODELS ----------------------------
|
|
738
|
+
# the interaction test and the per-level parallel-line
|
|
739
|
+
# equations; as R, generated with the graphics (~ .reg5ancova)
|
|
740
|
+
if ancova and graphics:
|
|
741
|
+
lines += _reg_ancova_models(fit, Xd, yv, y_name,
|
|
742
|
+
ancova_info, d, smapi)
|
|
743
|
+
|
|
744
|
+
# ---------- RELATIONS AMONG THE VARIABLES ------------------
|
|
745
|
+
# collinearity, ~ .reg2Relations (tolerance and VIF from the
|
|
746
|
+
# coefficient standard errors, the R computation), and the
|
|
747
|
+
# best-subset models
|
|
748
|
+
tol = vif = None
|
|
749
|
+
sub_df = None
|
|
750
|
+
if not brief and n_pred_orig > 1:
|
|
751
|
+
vif = np.array([
|
|
752
|
+
(Xd[nm].var(ddof=1) * (n_keep - 1)
|
|
753
|
+
* est.loc[nm, "Std Err"] ** 2) / MSW
|
|
754
|
+
for nm in pred_names])
|
|
755
|
+
tol = 1 / vif
|
|
756
|
+
lines += ["", "", " RELATIONS AMONG THE VARIABLES",
|
|
757
|
+
"", "-- Collinearity", ""]
|
|
758
|
+
c1 = max(len(s) for s in pred_names)
|
|
759
|
+
lines.append(" " * c1 + f"{'Tolerance':>11}"
|
|
760
|
+
+ f"{'VIF':>9}")
|
|
761
|
+
for nm, t_i, v_i in zip(pred_names, tol, vif):
|
|
762
|
+
lines.append(f"{nm:<{c1}}{fmt(t_i, 3):>11}"
|
|
763
|
+
+ f"{fmt(v_i, 3):>9}")
|
|
764
|
+
|
|
765
|
+
# best subsets: default on, subsets=n caps the listing
|
|
766
|
+
max_sublns = 50
|
|
767
|
+
do_subsets = True if subsets is None else subsets
|
|
768
|
+
if not isinstance(do_subsets, bool) \
|
|
769
|
+
and isinstance(do_subsets, (int, float)) \
|
|
770
|
+
and do_subsets > 1:
|
|
771
|
+
max_sublns = int(do_subsets)
|
|
772
|
+
do_subsets = True
|
|
773
|
+
if do_subsets:
|
|
774
|
+
sub_df = _best_subsets(Xd, yv, tot_ss, MSW,
|
|
775
|
+
best_sub)
|
|
776
|
+
if sub_df is not None:
|
|
777
|
+
lines += ["", "-- Best Subset Regression Models"]
|
|
778
|
+
if n_pred > 5:
|
|
779
|
+
lines.append("up to 10 subsets of each "
|
|
780
|
+
"number of predictors")
|
|
781
|
+
crit_lbl = sub_df.columns[-2]
|
|
782
|
+
xs_lbl = "X's"
|
|
783
|
+
wids = [max(4, len(nm) + 1) for nm in pred_names]
|
|
784
|
+
hdr = ("".join(f"{nm:>{w}}" for nm, w in
|
|
785
|
+
zip(pred_names, wids))
|
|
786
|
+
+ f"{crit_lbl:>9}{xs_lbl:>7}")
|
|
787
|
+
lines += ["", hdr]
|
|
788
|
+
shown = min(max_sublns, len(sub_df))
|
|
789
|
+
for i in range(shown):
|
|
790
|
+
if shown > 40 and (i + 1) % 30 == 0:
|
|
791
|
+
lines.append(hdr)
|
|
792
|
+
r = sub_df.iloc[i]
|
|
793
|
+
lines.append(
|
|
794
|
+
"".join(f"{int(r[nm]):>{w}}"
|
|
795
|
+
for nm, w in zip(pred_names,
|
|
796
|
+
wids))
|
|
797
|
+
+ f" {r[crit_lbl]:>8.3f}"
|
|
798
|
+
+ f" {int(r[xs_lbl]):>6}")
|
|
799
|
+
if len(sub_df) > max_sublns:
|
|
800
|
+
lines += ["",
|
|
801
|
+
f">>> Only first {shown} of "
|
|
802
|
+
f"{len(sub_df)} rows printed",
|
|
803
|
+
" To indicate more, add "
|
|
804
|
+
"subsets=n, where n is the number "
|
|
805
|
+
"of lines"]
|
|
806
|
+
lines += ["",
|
|
807
|
+
"[exhaustive search of all predictor "
|
|
808
|
+
"subsets]"]
|
|
809
|
+
|
|
810
|
+
# ---------- RESIDUALS AND INFLUENCE ------------------------
|
|
811
|
+
res_tbl = None
|
|
812
|
+
if brief and n_res_rows is None:
|
|
813
|
+
n_res_rows = 0
|
|
814
|
+
if n_res_rows is None:
|
|
815
|
+
n_res_rows = n_keep if n_keep < 20 else 20
|
|
816
|
+
if n_res_rows == "all":
|
|
817
|
+
n_res_rows = n_keep
|
|
818
|
+
n_res_rows = min(int(n_res_rows), n_keep)
|
|
819
|
+
|
|
820
|
+
if n_res_rows > 0:
|
|
821
|
+
res_tbl = pd.DataFrame(index=row_labels)
|
|
822
|
+
for nm in pred_names:
|
|
823
|
+
res_tbl[nm] = Xd[nm].to_numpy()
|
|
824
|
+
res_tbl[y_name] = yv
|
|
825
|
+
res_tbl["fitted"] = fitted
|
|
826
|
+
res_tbl["resid"] = resid
|
|
827
|
+
res_tbl["rstdnt"] = rstudent
|
|
828
|
+
res_tbl["dffits"] = dffits
|
|
829
|
+
res_tbl["cooks"] = np.round(cooks, 5)
|
|
830
|
+
if res_sort == "cooks":
|
|
831
|
+
res_tbl = res_tbl.sort_values(
|
|
832
|
+
"cooks", ascending=False)
|
|
833
|
+
elif res_sort == "rstudent":
|
|
834
|
+
res_tbl = res_tbl.reindex(
|
|
835
|
+
res_tbl["rstdnt"].abs().sort_values(
|
|
836
|
+
ascending=False).index)
|
|
837
|
+
elif res_sort == "dffits":
|
|
838
|
+
res_tbl = res_tbl.reindex(
|
|
839
|
+
res_tbl["dffits"].abs().sort_values(
|
|
840
|
+
ascending=False).index)
|
|
841
|
+
lines += ["", "", " RESIDUALS AND INFLUENCE", "",
|
|
842
|
+
"-- Data, Fitted, Residual, Studentized "
|
|
843
|
+
"Residual, Dffits, Cook's Distance"]
|
|
844
|
+
if res_sort == "cooks":
|
|
845
|
+
lines.append(" [sorted by Cook's Distance]")
|
|
846
|
+
elif res_sort == "rstudent":
|
|
847
|
+
lines.append(" [sorted by Studentized Residual,"
|
|
848
|
+
" ignoring + or - sign]")
|
|
849
|
+
elif res_sort == "dffits":
|
|
850
|
+
lines.append(" [sorted by dffits, ignoring + or"
|
|
851
|
+
" - sign]")
|
|
852
|
+
more = (" rows of data, or do n_res_rows=\"all\"]"
|
|
853
|
+
if n_res_rows < n_keep else "]")
|
|
854
|
+
lines.append(f" [n_res_rows = {n_res_rows}, out of "
|
|
855
|
+
f"{n_keep}{more}")
|
|
856
|
+
res_ints = _int_vars(res_tbl, list(pred_names) + [y_name])
|
|
857
|
+
lines += _prntbl(res_tbl.head(n_res_rows), d,
|
|
858
|
+
int_cols=res_ints).split("\n")
|
|
859
|
+
|
|
860
|
+
# ---------- PREDICTION ERROR -------------------------------
|
|
861
|
+
pred_tbl = None
|
|
862
|
+
new_data = X1_new is not None
|
|
863
|
+
if brief and n_pred_rows is None and not new_data:
|
|
864
|
+
n_pred_rows = 0
|
|
865
|
+
if n_pred_rows is None:
|
|
866
|
+
n_pred_rows = n_keep if n_keep < 25 else 10
|
|
867
|
+
if n_pred_rows == "all":
|
|
868
|
+
n_pred_rows = n_keep
|
|
869
|
+
|
|
870
|
+
if n_pred_rows > 0 or new_data:
|
|
871
|
+
if new_data: # X1_new..X6_new grid
|
|
872
|
+
grids = [np.atleast_1d(g) for g in
|
|
873
|
+
(X1_new, X2_new, X3_new, X4_new,
|
|
874
|
+
X5_new, X6_new)[:n_pred]
|
|
875
|
+
if g is not None]
|
|
876
|
+
if len(grids) != n_pred:
|
|
877
|
+
raise ValueError(
|
|
878
|
+
"specify new values for every predictor: "
|
|
879
|
+
f"X1_new ... X{n_pred}_new")
|
|
880
|
+
mesh = np.meshgrid(*grids, indexing="ij")
|
|
881
|
+
Xnew = pd.DataFrame(
|
|
882
|
+
{nm: m.ravel() for nm, m in
|
|
883
|
+
zip(pred_names, mesh)})
|
|
884
|
+
Xn = smapi.add_constant(Xnew, has_constant="add")
|
|
885
|
+
pr = fit.get_prediction(Xn)
|
|
886
|
+
base = Xnew.copy()
|
|
887
|
+
base[y_name] = ""
|
|
888
|
+
base.index = [""] * len(base)
|
|
889
|
+
else:
|
|
890
|
+
pr = fit.get_prediction(X)
|
|
891
|
+
base = pd.DataFrame(index=row_labels)
|
|
892
|
+
for nm in pred_names:
|
|
893
|
+
base[nm] = Xd[nm].to_numpy()
|
|
894
|
+
base[y_name] = yv
|
|
895
|
+
sf = pr.summary_frame(alpha=0.05)
|
|
896
|
+
s_pred = np.sqrt(fit.scale + pr.se_mean ** 2)
|
|
897
|
+
pred_tbl = base
|
|
898
|
+
pred_tbl["pred"] = sf["mean"].to_numpy()
|
|
899
|
+
pred_tbl["s_pred"] = s_pred
|
|
900
|
+
pred_tbl["pi.lwr"] = sf["obs_ci_lower"].to_numpy()
|
|
901
|
+
pred_tbl["pi.upr"] = sf["obs_ci_upper"].to_numpy()
|
|
902
|
+
pred_tbl["width"] = (pred_tbl["pi.upr"]
|
|
903
|
+
- pred_tbl["pi.lwr"])
|
|
904
|
+
if pred_sort == "predint":
|
|
905
|
+
pred_tbl = pred_tbl.sort_values("pi.lwr")
|
|
906
|
+
|
|
907
|
+
hdr = ["", "", " PREDICTION ERROR", "",
|
|
908
|
+
"-- Data, Predicted, Standard Error of "
|
|
909
|
+
"Prediction, 95% Prediction Intervals",
|
|
910
|
+
" [sorted by lower bound of prediction "
|
|
911
|
+
"interval]"]
|
|
912
|
+
if n_pred_rows < n_keep and not new_data:
|
|
913
|
+
hdr.append(' [to see all intervals add '
|
|
914
|
+
'n_pred_rows="all"]')
|
|
915
|
+
hdr.append(_dash(46))
|
|
916
|
+
lines += hdr
|
|
917
|
+
pred_ints = _int_vars(pred_tbl, list(pred_names)
|
|
918
|
+
+ [y_name])
|
|
919
|
+
if new_data or n_pred_rows >= len(pred_tbl):
|
|
920
|
+
lines += _prntbl(pred_tbl, d,
|
|
921
|
+
int_cols=pred_ints).split("\n")
|
|
922
|
+
else:
|
|
923
|
+
# three pieces, as R: around the widest interval
|
|
924
|
+
# when it falls in that half (else the start/end),
|
|
925
|
+
# and around the narrowest interval in the middle
|
|
926
|
+
widths = pred_tbl["width"].to_numpy()
|
|
927
|
+
nr = len(pred_tbl)
|
|
928
|
+
min_row = int(widths.argmin())
|
|
929
|
+
max_row = int(widths.argmax())
|
|
930
|
+
max_side = max_row < n_keep / 2
|
|
931
|
+
piece = max(1, round(n_pred_rows / 3))
|
|
932
|
+
pr2 = piece // 2
|
|
933
|
+
if max_side:
|
|
934
|
+
r1 = list(range(max(max_row - pr2, 0),
|
|
935
|
+
min(max_row + pr2 + 1, nr)))
|
|
936
|
+
r3 = list(range(max(nr - piece, 0), nr))
|
|
937
|
+
else:
|
|
938
|
+
r1 = list(range(0, min(piece, nr)))
|
|
939
|
+
r3 = list(range(max(max_row - pr2, 0),
|
|
940
|
+
min(max_row + pr2 + 1, nr)))
|
|
941
|
+
r2 = list(range(max(min_row - pr2, 0),
|
|
942
|
+
min(min_row + pr2 + 1, nr)))
|
|
943
|
+
body = _prntbl(pred_tbl, d,
|
|
944
|
+
int_cols=pred_ints).split("\n")
|
|
945
|
+
head_ln, rows_ln = body[0], body[1:]
|
|
946
|
+
lines.append(head_ln)
|
|
947
|
+
for k, rr in enumerate((r1, r2, r3)):
|
|
948
|
+
if k > 0:
|
|
949
|
+
lines.append("...")
|
|
950
|
+
for i in rr:
|
|
951
|
+
lines.append(rows_ln[i])
|
|
952
|
+
|
|
953
|
+
# ---------- graphics ---------------------------------------
|
|
954
|
+
plots = {}
|
|
955
|
+
if graphics and n_pred > 0 and n_res_rows > 0:
|
|
956
|
+
plots["residuals_density"] = _reg_dn_residual(resid)
|
|
957
|
+
plots["residuals_fitted"] = _reg_resfit(
|
|
958
|
+
fitted, resid, cooks, cooks_cut, row_labels, d)
|
|
959
|
+
if graphics and n_pred == 1:
|
|
960
|
+
pi_df = (pred_tbl if (pred_tbl is not None
|
|
961
|
+
and not new_data) else None)
|
|
962
|
+
plots["scatter"] = _reg_scatter(
|
|
963
|
+
Xd.iloc[:, 0].to_numpy(), yv, pred_names[0],
|
|
964
|
+
y_name, fit, smapi, d,
|
|
965
|
+
with_bands=pi_df is not None)
|
|
966
|
+
if graphics and n_pred > 1 and not ancova:
|
|
967
|
+
# scatterplot matrix of the model variables, response
|
|
968
|
+
# first, with the least-squares fit; ~ .reg5Plot .plt.mat
|
|
969
|
+
mat_df = pd.DataFrame(
|
|
970
|
+
{y_name: yv}, index=Xd.index).join(Xd)
|
|
971
|
+
plots["scatter_matrix"] = scatter_matrix(
|
|
972
|
+
mat_df, fit="lm", digits_d=d)
|
|
973
|
+
if graphics and ancova:
|
|
974
|
+
# ANCOVA: grouped scatterplot with a parallel
|
|
975
|
+
# least-squares line per factor level; ~ .reg5ancova
|
|
976
|
+
plots["ancova"] = _reg_ancova_plot(
|
|
977
|
+
fit, Xd, yv, y_name, ancova_info, d)
|
|
978
|
+
if graphics and mod_info is not None:
|
|
979
|
+
plots["moderation"] = _moderation_plot(
|
|
980
|
+
fit, Xd, y_name, mod_info, d)
|
|
981
|
+
|
|
982
|
+
print("\n".join(lines))
|
|
983
|
+
|
|
984
|
+
out = RegressionResults(
|
|
985
|
+
formula=formula, n_obs=n_obs, n_keep=n_keep,
|
|
986
|
+
digits_d=d,
|
|
987
|
+
estimates=est, anova=anova,
|
|
988
|
+
fit={"se": se, "resid_range": res_range,
|
|
989
|
+
"Rsq": fit.rsquared if n_pred else np.nan,
|
|
990
|
+
"Rsq_adj": (fit.rsquared_adj if n_pred
|
|
991
|
+
else np.nan),
|
|
992
|
+
"PRESS": PRESS, "Rsq_PRESS": Rsq_press,
|
|
993
|
+
"sy": sy, "MSW": MSW},
|
|
994
|
+
tolerance=tol, vif=vif, subsets=sub_df,
|
|
995
|
+
residuals=res_tbl, predictions=pred_tbl,
|
|
996
|
+
plots=plots)
|
|
997
|
+
|
|
998
|
+
if Rmd is not None:
|
|
999
|
+
reg_rmd(out, y_name, pred_names, formula,
|
|
1000
|
+
list(data.columns), Rmd, Rmd_data, Rmd_format,
|
|
1001
|
+
Rmd_browser, results, explain, interpret, code,
|
|
1002
|
+
n_res_rows, n_pred_rows, res_sort, d)
|
|
1003
|
+
|
|
1004
|
+
return out
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
def _reg_kfold(yv, Xd, y_name, pred_names, n_keep, formula,
|
|
1008
|
+
kfold, seed, digits_d, smapi,
|
|
1009
|
+
new_scale="none", scale_response=False):
|
|
1010
|
+
"""K-fold cross-validation. Partition the complete cases into
|
|
1011
|
+
kfold folds; for each, fit on the other kfold-1 folds
|
|
1012
|
+
(training) and evaluate the fitted model on the held-out fold
|
|
1013
|
+
(testing). Report per-fold and mean se/MSE/Rsq for both, the
|
|
1014
|
+
training se as the residual sigma and the testing sp from the
|
|
1015
|
+
held-out prediction errors. ~ .regKfold
|
|
1016
|
+
|
|
1017
|
+
With new_scale, each fold's training and testing subsets are
|
|
1018
|
+
rescaled separately (predictors always, the response only
|
|
1019
|
+
when scale_response), the continuous variables only.
|
|
1020
|
+
|
|
1021
|
+
Folds are random; seed= makes them reproducible within Python
|
|
1022
|
+
but, because the RNG differs from R's, not identical to R for
|
|
1023
|
+
the same seed."""
|
|
1024
|
+
if kfold < 2:
|
|
1025
|
+
raise ValueError("kfold must be 2 or larger")
|
|
1026
|
+
d = 3 if digits_d is None else digits_d
|
|
1027
|
+
|
|
1028
|
+
# each row's fold, from a scrambled 1..n mod kfold, as R
|
|
1029
|
+
rng = np.random.default_rng(seed)
|
|
1030
|
+
nk = rng.permutation(np.arange(1, n_keep + 1)) % kfold
|
|
1031
|
+
|
|
1032
|
+
Xmat = Xd.to_numpy(dtype=float)
|
|
1033
|
+
# continuous columns (>2 distinct) are the ones rescaled;
|
|
1034
|
+
# the decision is global, the rescaling is per fold, as R
|
|
1035
|
+
do_scale = new_scale != "none"
|
|
1036
|
+
cont = ([c for c in range(Xmat.shape[1])
|
|
1037
|
+
if len(np.unique(Xmat[:, c])) > 2] if do_scale
|
|
1038
|
+
else [])
|
|
1039
|
+
scale_y = (do_scale and scale_response
|
|
1040
|
+
and len(np.unique(yv)) > 2)
|
|
1041
|
+
tr = {"n": [], "se": [], "MSE": [], "Rsq": []}
|
|
1042
|
+
te = {"n": [], "se": [], "MSE": [], "Rsq": []}
|
|
1043
|
+
for f in range(kfold):
|
|
1044
|
+
train, test = nk != f, nk == f
|
|
1045
|
+
ytr, yte = yv[train].copy(), yv[test].copy()
|
|
1046
|
+
Xtr_m, Xte_m = Xmat[train].copy(), Xmat[test].copy()
|
|
1047
|
+
if do_scale: # separate scaling per side
|
|
1048
|
+
for c in cont:
|
|
1049
|
+
Xtr_m[:, c] = _rescale(Xtr_m[:, c], new_scale, d)
|
|
1050
|
+
Xte_m[:, c] = _rescale(Xte_m[:, c], new_scale, d)
|
|
1051
|
+
if scale_y:
|
|
1052
|
+
ytr = _rescale(ytr, new_scale, d)
|
|
1053
|
+
yte = _rescale(yte, new_scale, d)
|
|
1054
|
+
|
|
1055
|
+
# training fit
|
|
1056
|
+
Xtr = smapi.add_constant(
|
|
1057
|
+
pd.DataFrame(Xtr_m), has_constant="add")
|
|
1058
|
+
fit = smapi.OLS(ytr, Xtr).fit()
|
|
1059
|
+
coefs = np.asarray(fit.params, dtype=float)
|
|
1060
|
+
tr["n"].append(int(train.sum()))
|
|
1061
|
+
tr["MSE"].append(float(fit.scale)) # residual mean sq
|
|
1062
|
+
tr["se"].append(math.sqrt(float(fit.scale)))
|
|
1063
|
+
tr["Rsq"].append(float(fit.rsquared))
|
|
1064
|
+
|
|
1065
|
+
# apply the training model to the held-out test fold
|
|
1066
|
+
n_te = int(test.sum())
|
|
1067
|
+
Xte = np.column_stack([np.ones(n_te), Xte_m])
|
|
1068
|
+
if Xte.shape[1] != coefs.shape[0]:
|
|
1069
|
+
raise ValueError(
|
|
1070
|
+
"a fold has more variables than estimated "
|
|
1071
|
+
"coefficients: at least one variable has no "
|
|
1072
|
+
"variation in a test fold")
|
|
1073
|
+
sse = float(((yte - Xte @ coefs) ** 2).sum())
|
|
1074
|
+
denom = n_te - coefs.shape[0]
|
|
1075
|
+
mse = sse / denom if denom > 0 else math.nan
|
|
1076
|
+
ssy = float(((yte - yte.mean()) ** 2).sum())
|
|
1077
|
+
te["n"].append(n_te)
|
|
1078
|
+
te["MSE"].append(mse)
|
|
1079
|
+
te["se"].append(math.sqrt(mse) if denom > 0 else math.nan)
|
|
1080
|
+
te["Rsq"].append(1 - sse / ssy if ssy > 0 else math.nan)
|
|
1081
|
+
|
|
1082
|
+
cv = pd.DataFrame({
|
|
1083
|
+
"fold": range(1, kfold + 1),
|
|
1084
|
+
"train_n": tr["n"], "train_se": tr["se"],
|
|
1085
|
+
"train_MSE": tr["MSE"], "train_Rsq": tr["Rsq"],
|
|
1086
|
+
"test_n": te["n"], "test_se": te["se"],
|
|
1087
|
+
"test_MSE": te["MSE"], "test_Rsq": te["Rsq"]})
|
|
1088
|
+
means = {"train_se": float(np.mean(tr["se"])),
|
|
1089
|
+
"train_MSE": float(np.mean(tr["MSE"])),
|
|
1090
|
+
"train_Rsq": float(np.mean(tr["Rsq"])),
|
|
1091
|
+
"test_se": float(np.mean(te["se"])),
|
|
1092
|
+
"test_MSE": float(np.mean(te["MSE"])),
|
|
1093
|
+
"test_Rsq": float(np.mean(te["Rsq"]))}
|
|
1094
|
+
|
|
1095
|
+
print("\n".join(_kfold_lines(cv, means, kfold, d)))
|
|
1096
|
+
return RegressionResults(
|
|
1097
|
+
formula=formula, n_keep=n_keep, kfold=kfold,
|
|
1098
|
+
digits_d=d, cv=cv, cv_means=means)
|
|
1099
|
+
|
|
1100
|
+
|
|
1101
|
+
def _kfold_lines(cv, means, kfold, d):
|
|
1102
|
+
"""The cross-validation table: a training block (se, MSE, Rsq)
|
|
1103
|
+
and a testing block (sp, MSE, Rsq), then the column means.
|
|
1104
|
+
~ .regKfold display."""
|
|
1105
|
+
def w(col): # width of a numeric column
|
|
1106
|
+
vals = [v for v in cv[col] if not math.isnan(v)]
|
|
1107
|
+
return max(6, max((len(fmt(v, d)) for v in vals),
|
|
1108
|
+
default=6))
|
|
1109
|
+
|
|
1110
|
+
def num(v):
|
|
1111
|
+
return "NA" if math.isnan(v) else fmt(v, d)
|
|
1112
|
+
|
|
1113
|
+
ws = {c: w(c) for c in ("train_se", "train_MSE", "train_Rsq",
|
|
1114
|
+
"test_se", "test_MSE", "test_Rsq")}
|
|
1115
|
+
wn = max(3, max(len(str(n)) for n in cv["train_n"]))
|
|
1116
|
+
wk = max(3, max(len(str(n)) for n in cv["test_n"]))
|
|
1117
|
+
gap = " "
|
|
1118
|
+
|
|
1119
|
+
def block_w(pre):
|
|
1120
|
+
return (ws[pre + "_se"] + ws[pre + "_MSE"]
|
|
1121
|
+
+ ws[pre + "_Rsq"] + 2 * len(gap))
|
|
1122
|
+
tw, kw = block_w("train"), block_w("test")
|
|
1123
|
+
|
|
1124
|
+
lines = [f"\n {kfold}-FOLD CROSS-VALIDATION", ""]
|
|
1125
|
+
# group titles, centered over each block (past the n column)
|
|
1126
|
+
tt, kt = "Model from Training Data", "Applied to Testing Data"
|
|
1127
|
+
lead = 5 + 1 + wn # "fold " + "|" area + n
|
|
1128
|
+
lines.append(" " * lead + tt.center(tw)
|
|
1129
|
+
+ " " * (3 + wk) + kt.center(kw))
|
|
1130
|
+
lines.append(" " * lead + "-" * tw
|
|
1131
|
+
+ " " * (3 + wk) + "-" * kw)
|
|
1132
|
+
lines.append(
|
|
1133
|
+
"fold" + " " + "n".rjust(wn)
|
|
1134
|
+
+ gap + "se".rjust(ws["train_se"])
|
|
1135
|
+
+ gap + "MSE".rjust(ws["train_MSE"])
|
|
1136
|
+
+ gap + "Rsq".rjust(ws["train_Rsq"])
|
|
1137
|
+
+ " " + "n".rjust(wk)
|
|
1138
|
+
+ gap + "sp".rjust(ws["test_se"])
|
|
1139
|
+
+ gap + "MSE".rjust(ws["test_MSE"])
|
|
1140
|
+
+ gap + "Rsq".rjust(ws["test_Rsq"]))
|
|
1141
|
+
for _, r in cv.iterrows():
|
|
1142
|
+
lines.append(
|
|
1143
|
+
f"{int(r['fold']):>3} |" + " "
|
|
1144
|
+
+ str(int(r["train_n"])).rjust(wn)
|
|
1145
|
+
+ gap + num(r["train_se"]).rjust(ws["train_se"])
|
|
1146
|
+
+ gap + num(r["train_MSE"]).rjust(ws["train_MSE"])
|
|
1147
|
+
+ gap + num(r["train_Rsq"]).rjust(ws["train_Rsq"])
|
|
1148
|
+
+ " " + str(int(r["test_n"])).rjust(wk)
|
|
1149
|
+
+ gap + num(r["test_se"]).rjust(ws["test_se"])
|
|
1150
|
+
+ gap + num(r["test_MSE"]).rjust(ws["test_MSE"])
|
|
1151
|
+
+ gap + num(r["test_Rsq"]).rjust(ws["test_Rsq"]))
|
|
1152
|
+
lines.append(" " * lead + "-" * tw
|
|
1153
|
+
+ " " * (3 + wk) + "-" * kw)
|
|
1154
|
+
lines.append(
|
|
1155
|
+
"Mean" + " " + " " * wn
|
|
1156
|
+
+ gap + num(means["train_se"]).rjust(ws["train_se"])
|
|
1157
|
+
+ gap + num(means["train_MSE"]).rjust(ws["train_MSE"])
|
|
1158
|
+
+ gap + num(means["train_Rsq"]).rjust(ws["train_Rsq"])
|
|
1159
|
+
+ " " + " " * wk
|
|
1160
|
+
+ gap + num(means["test_se"]).rjust(ws["test_se"])
|
|
1161
|
+
+ gap + num(means["test_MSE"]).rjust(ws["test_MSE"])
|
|
1162
|
+
+ gap + num(means["test_Rsq"]).rjust(ws["test_Rsq"]))
|
|
1163
|
+
return lines
|
|
1164
|
+
|
|
1165
|
+
|
|
1166
|
+
def _mod_slopes(fit, mod_info, wv):
|
|
1167
|
+
"""The simple-slope intercept and slope of the response on the
|
|
1168
|
+
focal predictor at the moderator W = wc, from the fitted
|
|
1169
|
+
Y = b0 + bx X + bw W + bxw X*W: at fixed W, intercept
|
|
1170
|
+
b0 + bw*wc and slope bx + bxw*wc. Returns the coefficients and
|
|
1171
|
+
the three (label, wc) levels, mean and +/-1 SD."""
|
|
1172
|
+
b0 = float(fit.params["const"])
|
|
1173
|
+
bx = float(fit.params[mod_info["x"]])
|
|
1174
|
+
bw = float(fit.params[mod_info["w"]])
|
|
1175
|
+
bxw = float(fit.params[mod_info["xw"]])
|
|
1176
|
+
m_w, s_w = float(wv.mean()), float(wv.std(ddof=1))
|
|
1177
|
+
levels = [("+1SD", m_w + s_w), ("Mean", m_w),
|
|
1178
|
+
("-1SD", m_w - s_w)]
|
|
1179
|
+
return b0, bx, bw, bxw, m_w, s_w, levels
|
|
1180
|
+
|
|
1181
|
+
|
|
1182
|
+
def _moderation_lines(fit, Xd, mod_info, d):
|
|
1183
|
+
"""The moderation text: the moderator's mean and SD, then the
|
|
1184
|
+
simple-slope intercept b0 and slope b1 at mean and +/-1 SD.
|
|
1185
|
+
~ .reg6mod out_mod"""
|
|
1186
|
+
w_name = mod_info["w"]
|
|
1187
|
+
b0, bx, bw, bxw, m_w, s_w, levels = _mod_slopes(
|
|
1188
|
+
fit, mod_info, Xd[w_name].to_numpy(dtype=float))
|
|
1189
|
+
out = ["", "", " MODERATION ANALYSIS", "",
|
|
1190
|
+
f"Mean of {w_name}: {fmt(m_w, d)}",
|
|
1191
|
+
f"SD of {w_name}: {fmt(s_w, d)}", ""]
|
|
1192
|
+
for tag, wc in (("mean+1SD", levels[0][1]),
|
|
1193
|
+
("mean ", levels[1][1]),
|
|
1194
|
+
("mean-1SD", levels[2][1])):
|
|
1195
|
+
out.append(f"{tag} for {w_name}: "
|
|
1196
|
+
f"b0={fmt(b0 + bw * wc, d)} "
|
|
1197
|
+
f"b1={fmt(bx + bxw * wc, d)}")
|
|
1198
|
+
return out
|
|
1199
|
+
|
|
1200
|
+
|
|
1201
|
+
def _moderation_plot(fit, Xd, y_name, mod_info, d):
|
|
1202
|
+
"""Interaction plot: the simple regression line of the
|
|
1203
|
+
response on the focal predictor at the moderator's mean and
|
|
1204
|
+
+/-1 SD, one line each. ~ .reg6mod plot"""
|
|
1205
|
+
x_name, w_name = mod_info["x"], mod_info["w"]
|
|
1206
|
+
b0, bx, bw, bxw, _, _, levels = _mod_slopes(
|
|
1207
|
+
fit, mod_info, Xd[w_name].to_numpy(dtype=float))
|
|
1208
|
+
xv = Xd[x_name].to_numpy(dtype=float)
|
|
1209
|
+
xs = np.array([float(xv.min()), float(xv.max())])
|
|
1210
|
+
style_opts = plotly_style()
|
|
1211
|
+
colors = {"+1SD": to_hex(BASE_COLORS[0]),
|
|
1212
|
+
"Mean": to_hex("gray20"),
|
|
1213
|
+
"-1SD": to_hex(BASE_COLORS[1])}
|
|
1214
|
+
widths = {"+1SD": 2, "Mean": 1.3, "-1SD": 2}
|
|
1215
|
+
fig = go.Figure()
|
|
1216
|
+
ys_all = []
|
|
1217
|
+
for lbl, wc in levels:
|
|
1218
|
+
ys = (b0 + bw * wc) + (bx + bxw * wc) * xs
|
|
1219
|
+
ys_all += list(ys)
|
|
1220
|
+
fig.add_trace(go.Scatter(
|
|
1221
|
+
x=xs, y=ys, mode="lines", name=lbl,
|
|
1222
|
+
line=dict(color=colors[lbl], width=widths[lbl]),
|
|
1223
|
+
hoverinfo="name+x+y"))
|
|
1224
|
+
axT1 = pretty(float(xs[0]), float(xs[1]))
|
|
1225
|
+
axT2 = pretty(min(ys_all), max(ys_all))
|
|
1226
|
+
ax_x = axis_num(x_name, axT1, axis_format(axT1, d))
|
|
1227
|
+
ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
|
|
1228
|
+
ax_y.update(showgrid=True,
|
|
1229
|
+
gridcolor=to_hex(style_opts["grid_col"]),
|
|
1230
|
+
gridwidth=1, griddash="dot")
|
|
1231
|
+
fig.update_layout(
|
|
1232
|
+
xaxis=ax_x, yaxis=ax_y,
|
|
1233
|
+
shapes=x_grid(axT1) + plot_border(), template=None,
|
|
1234
|
+
plot_bgcolor=to_hex(style_opts["panel_fill"]),
|
|
1235
|
+
paper_bgcolor=to_hex(style_opts["window_fill"]),
|
|
1236
|
+
legend=dict(title=dict(text=w_name)),
|
|
1237
|
+
title=dict(text="Moderator Variable Interaction Plot",
|
|
1238
|
+
x=0.5, xanchor="center",
|
|
1239
|
+
font=dict(size=round(
|
|
1240
|
+
16 * get_option("main_size", 1)))))
|
|
1241
|
+
return fig
|
|
1242
|
+
|
|
1243
|
+
|
|
1244
|
+
def _reg_ancova_models(fit, Xd, yv, y_name, info, d, smapi):
|
|
1245
|
+
"""The ANCOVA group models text: the test of the covariate x
|
|
1246
|
+
factor interaction (the parallel-lines assumption), the
|
|
1247
|
+
per-level equations of the no-interaction model, and the
|
|
1248
|
+
Plot suggestion. ~ .reg5ancova txmdl"""
|
|
1249
|
+
cov, fac = info["cov"], info["fac"]
|
|
1250
|
+
p1, p2 = info["preds"]
|
|
1251
|
+
ind_cols = [c for c in Xd.columns if c != cov]
|
|
1252
|
+
|
|
1253
|
+
# interaction model: add covariate x indicator columns, and
|
|
1254
|
+
# test them jointly against the no-interaction model (fit).
|
|
1255
|
+
# The sequential interaction SS = SSE(reduced) - SSE(full),
|
|
1256
|
+
# over the full model's residual mean square. ~ anova(lm(
|
|
1257
|
+
# y ~ cov * factor))[interaction row]
|
|
1258
|
+
inter = pd.DataFrame(
|
|
1259
|
+
{f"{cov}:{c}": Xd[cov].to_numpy() * Xd[c].to_numpy()
|
|
1260
|
+
for c in ind_cols}, index=Xd.index)
|
|
1261
|
+
Xf = pd.concat(
|
|
1262
|
+
[smapi.add_constant(Xd, has_constant="add"), inter],
|
|
1263
|
+
axis=1)
|
|
1264
|
+
full = smapi.OLS(yv, Xf).fit()
|
|
1265
|
+
df_int = len(ind_cols)
|
|
1266
|
+
df_res = int(full.df_resid)
|
|
1267
|
+
ss_int = float(fit.ssr) - float(full.ssr)
|
|
1268
|
+
F = (ss_int / df_int) / (float(full.ssr) / df_res)
|
|
1269
|
+
p = float(sps.f.sf(F, df_int, df_res))
|
|
1270
|
+
|
|
1271
|
+
b0 = float(fit.params["const"])
|
|
1272
|
+
b_slope = float(fit.params[cov])
|
|
1273
|
+
out = ["", "",
|
|
1274
|
+
f" MODELS OF {y_name} FOR LEVELS OF {fac}", "",
|
|
1275
|
+
"-- Test of Interaction", "",
|
|
1276
|
+
f"{p1}:{p2} df: {df_int} df resid: {df_res} "
|
|
1277
|
+
f"SS: {fmt(ss_int, 3)} F: {fmt(F, 3)} "
|
|
1278
|
+
f"p-value: {fmt(p, 3)}", "",
|
|
1279
|
+
"-- Assume parallel lines, no interaction of "
|
|
1280
|
+
f"{fac} with {cov}", ""]
|
|
1281
|
+
for i, lv in enumerate(info["levels"]):
|
|
1282
|
+
eff = 0.0 if i == 0 else float(
|
|
1283
|
+
fit.params.get(f"{fac}{lv}", 0.0))
|
|
1284
|
+
out.append(
|
|
1285
|
+
f"Level {lv}: y^_{y_name} = {fmt(b0 + eff, d)} + "
|
|
1286
|
+
f"{fmt(b_slope, d)}(x_{cov})")
|
|
1287
|
+
out += ["",
|
|
1288
|
+
"-- Visualize Separately Computed Regression Lines",
|
|
1289
|
+
"",
|
|
1290
|
+
f'XY("{cov}", "{y_name}", data=d, by="{fac}", '
|
|
1291
|
+
'fit="lm")']
|
|
1292
|
+
return out
|
|
1293
|
+
|
|
1294
|
+
|
|
1295
|
+
def _reg_ancova_plot(fit, Xd, yv, y_name, info, d):
|
|
1296
|
+
"""Grouped scatterplot with one parallel least-squares line
|
|
1297
|
+
per factor level: shared slope on the covariate, a separate
|
|
1298
|
+
intercept per level (the reference plus its group effect).
|
|
1299
|
+
~ .reg5ancova plot"""
|
|
1300
|
+
cov, fac, levels = info["cov"], info["fac"], info["levels"]
|
|
1301
|
+
fac_vals = info["fac_vals"]
|
|
1302
|
+
xv = Xd[cov].to_numpy(dtype=float)
|
|
1303
|
+
b0 = float(fit.params["const"])
|
|
1304
|
+
b_slope = float(fit.params[cov])
|
|
1305
|
+
xs = np.array([float(xv.min()), float(xv.max())])
|
|
1306
|
+
style_opts = plotly_style()
|
|
1307
|
+
fig = go.Figure()
|
|
1308
|
+
ys_all = list(yv)
|
|
1309
|
+
for i, lv in enumerate(levels):
|
|
1310
|
+
col = to_hex(BASE_COLORS[i % len(BASE_COLORS)])
|
|
1311
|
+
m = fac_vals == lv
|
|
1312
|
+
fig.add_trace(go.Scatter(
|
|
1313
|
+
x=xv[m], y=yv[m], mode="markers", name=lv,
|
|
1314
|
+
legendgroup=lv,
|
|
1315
|
+
marker=dict(symbol="circle", size=7,
|
|
1316
|
+
color=make_trans(col, 0.9), opacity=1,
|
|
1317
|
+
line=dict(color=col, width=1)),
|
|
1318
|
+
hoverinfo="x+y+name"))
|
|
1319
|
+
eff = 0.0 if i == 0 else float(
|
|
1320
|
+
fit.params.get(f"{fac}{lv}", 0.0))
|
|
1321
|
+
ys = (b0 + eff) + b_slope * xs
|
|
1322
|
+
ys_all += list(ys)
|
|
1323
|
+
fig.add_trace(go.Scatter(
|
|
1324
|
+
x=xs, y=ys, mode="lines", legendgroup=lv,
|
|
1325
|
+
line=dict(color=col, width=2),
|
|
1326
|
+
hoverinfo="skip", showlegend=False))
|
|
1327
|
+
axT1 = pretty(float(xs[0]), float(xs[1]))
|
|
1328
|
+
axT2 = pretty(min(ys_all), max(ys_all))
|
|
1329
|
+
ax_x = axis_num(cov, axT1, axis_format(axT1, d))
|
|
1330
|
+
ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
|
|
1331
|
+
ax_y.update(showgrid=True,
|
|
1332
|
+
gridcolor=to_hex(style_opts["grid_col"]),
|
|
1333
|
+
gridwidth=1, griddash="dot")
|
|
1334
|
+
fig.update_layout(
|
|
1335
|
+
xaxis=ax_x, yaxis=ax_y,
|
|
1336
|
+
shapes=x_grid(axT1) + plot_border(), template=None,
|
|
1337
|
+
plot_bgcolor=to_hex(style_opts["panel_fill"]),
|
|
1338
|
+
paper_bgcolor=to_hex(style_opts["window_fill"]),
|
|
1339
|
+
legend=dict(title=dict(text=fac)),
|
|
1340
|
+
title=dict(text="Scatterplot and Least-Squares Lines",
|
|
1341
|
+
x=0.5, xanchor="center",
|
|
1342
|
+
font=dict(size=round(
|
|
1343
|
+
16 * get_option("main_size", 1)))))
|
|
1344
|
+
return fig
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _reg_scatter(xv, yv, x_name, y_name, fit, smapi, digits_d,
|
|
1348
|
+
with_bands=True):
|
|
1349
|
+
"""Scatterplot with the least-squares line and, by default,
|
|
1350
|
+
the 95% confidence and prediction bands. ~ .reg5Plot"""
|
|
1351
|
+
od = np.argsort(xv, kind="stable")
|
|
1352
|
+
xs = xv[od]
|
|
1353
|
+
Xl = smapi.add_constant(
|
|
1354
|
+
pd.DataFrame({x_name: xs}), has_constant="add")
|
|
1355
|
+
pr = fit.get_prediction(Xl)
|
|
1356
|
+
sf = pr.summary_frame(alpha=0.05)
|
|
1357
|
+
style_opts = plotly_style()
|
|
1358
|
+
fig = go.Figure()
|
|
1359
|
+
if with_bands:
|
|
1360
|
+
fig.add_trace(go.Scatter( # prediction band
|
|
1361
|
+
x=np.concatenate([xs, xs[::-1]]),
|
|
1362
|
+
y=np.concatenate(
|
|
1363
|
+
[sf["obs_ci_upper"].to_numpy(),
|
|
1364
|
+
sf["obs_ci_lower"].to_numpy()[::-1]]),
|
|
1365
|
+
mode="none", fill="toself",
|
|
1366
|
+
fillcolor=as_plotly_color(
|
|
1367
|
+
get_option("se_fill", "#1A1A1A19")),
|
|
1368
|
+
hoverinfo="skip", showlegend=False))
|
|
1369
|
+
fig.add_trace(go.Scatter( # confidence band
|
|
1370
|
+
x=np.concatenate([xs, xs[::-1]]),
|
|
1371
|
+
y=np.concatenate(
|
|
1372
|
+
[sf["mean_ci_upper"].to_numpy(),
|
|
1373
|
+
sf["mean_ci_lower"].to_numpy()[::-1]]),
|
|
1374
|
+
mode="none", fill="toself",
|
|
1375
|
+
fillcolor=as_plotly_color(
|
|
1376
|
+
get_option("se_fill", "#1A1A1A19")),
|
|
1377
|
+
hoverinfo="skip", showlegend=False))
|
|
1378
|
+
pt_fill = get_option("pt_color", "#324E5C")
|
|
1379
|
+
fig.add_trace(go.Scatter(
|
|
1380
|
+
x=xv, y=yv, mode="markers",
|
|
1381
|
+
marker=dict(symbol="circle", size=7.25,
|
|
1382
|
+
sizemode="diameter",
|
|
1383
|
+
color=make_trans(pt_fill, 0.9),
|
|
1384
|
+
opacity=1,
|
|
1385
|
+
line=dict(color=to_hex(pt_fill), width=1)),
|
|
1386
|
+
hoverinfo="x+y", showlegend=False))
|
|
1387
|
+
fig.add_trace(go.Scatter(
|
|
1388
|
+
x=xs, y=sf["mean"].to_numpy(), mode="lines",
|
|
1389
|
+
line=dict(color=to_hex(get_option("fit_color",
|
|
1390
|
+
"#5C4032")),
|
|
1391
|
+
width=get_option("fit_lwd", 2)),
|
|
1392
|
+
hoverinfo="skip", showlegend=False))
|
|
1393
|
+
axT1 = pretty(float(xv.min()), float(xv.max()))
|
|
1394
|
+
ylo = min(float(yv.min()),
|
|
1395
|
+
float(sf["obs_ci_lower"].min())
|
|
1396
|
+
if with_bands else float(yv.min()))
|
|
1397
|
+
yhi = max(float(yv.max()),
|
|
1398
|
+
float(sf["obs_ci_upper"].max())
|
|
1399
|
+
if with_bands else float(yv.max()))
|
|
1400
|
+
axT2 = pretty(ylo, yhi)
|
|
1401
|
+
ax_x = axis_num(x_name, axT1,
|
|
1402
|
+
axis_format(axT1, digits_d))
|
|
1403
|
+
ax_y = axis_num(y_name, axT2,
|
|
1404
|
+
axis_format(axT2, digits_d))
|
|
1405
|
+
ax_y.update(showgrid=True,
|
|
1406
|
+
gridcolor=to_hex(style_opts["grid_col"]),
|
|
1407
|
+
gridwidth=1, griddash="dot")
|
|
1408
|
+
title = ("Reg Line, Confidence & Prediction Intervals"
|
|
1409
|
+
if with_bands
|
|
1410
|
+
else "Scatterplot and Least-Squares Line")
|
|
1411
|
+
fig.update_layout(
|
|
1412
|
+
xaxis=ax_x, yaxis=ax_y,
|
|
1413
|
+
shapes=x_grid(axT1) + plot_border(), template=None,
|
|
1414
|
+
plot_bgcolor=to_hex(style_opts["panel_fill"]),
|
|
1415
|
+
paper_bgcolor=to_hex(style_opts["window_fill"]),
|
|
1416
|
+
title=dict(text=title, x=0.5, xanchor="center",
|
|
1417
|
+
font=dict(size=round(
|
|
1418
|
+
16 * get_option("main_size", 1)))))
|
|
1419
|
+
return fig
|
|
1420
|
+
|
|
1421
|
+
|
|
1422
|
+
def _reg_resfit(fitted, resid, cooks, cooks_cut, labels,
|
|
1423
|
+
digits_d):
|
|
1424
|
+
"""Residuals vs fitted values, zero line, points at or
|
|
1425
|
+
above cooks_cut labeled (or the single largest when none
|
|
1426
|
+
reach it). ~ .reg3resfitResidual"""
|
|
1427
|
+
max_cook = float(np.nanmax(cooks))
|
|
1428
|
+
if max_cook < cooks_cut:
|
|
1429
|
+
cut = math.floor(max_cook * 100) / 100
|
|
1430
|
+
sub = ("Point with largest Cook's Distance of "
|
|
1431
|
+
f"{fmt(max_cook, 2)} is labeled")
|
|
1432
|
+
else:
|
|
1433
|
+
cut = cooks_cut
|
|
1434
|
+
sub = (f"Points with Cook's Distance > {cooks_cut} "
|
|
1435
|
+
"are labeled")
|
|
1436
|
+
style_opts = plotly_style()
|
|
1437
|
+
pt_fill = get_option("pt_color", "#324E5C")
|
|
1438
|
+
fig = go.Figure()
|
|
1439
|
+
fig.add_trace(go.Scatter(
|
|
1440
|
+
x=fitted, y=resid, mode="markers",
|
|
1441
|
+
marker=dict(symbol="circle", size=5,
|
|
1442
|
+
sizemode="diameter",
|
|
1443
|
+
color=make_trans(pt_fill, 0.9),
|
|
1444
|
+
opacity=1,
|
|
1445
|
+
line=dict(color=to_hex(pt_fill), width=0.5)),
|
|
1446
|
+
hoverinfo="x+y", showlegend=False))
|
|
1447
|
+
flag = np.where(cooks >= cut)[0]
|
|
1448
|
+
for i in flag:
|
|
1449
|
+
fig.add_annotation(
|
|
1450
|
+
x=float(fitted[i]), y=float(resid[i]),
|
|
1451
|
+
text=str(labels[i]), showarrow=False, yshift=12,
|
|
1452
|
+
font=dict(size=11, color=to_hex("gray30")))
|
|
1453
|
+
axT1 = pretty(float(fitted.min()), float(fitted.max()))
|
|
1454
|
+
axT2 = pretty(float(resid.min()), float(resid.max()))
|
|
1455
|
+
ax_x = axis_num("Fitted Values", axT1,
|
|
1456
|
+
axis_format(axT1, digits_d))
|
|
1457
|
+
ax_y = axis_num("Residuals", axT2,
|
|
1458
|
+
axis_format(axT2, digits_d))
|
|
1459
|
+
ax_y.update(showgrid=True,
|
|
1460
|
+
gridcolor=to_hex(style_opts["grid_col"]),
|
|
1461
|
+
gridwidth=1, griddash="dot")
|
|
1462
|
+
shapes = x_grid(axT1) + plot_border()
|
|
1463
|
+
shapes.append(dict( # zero residual line
|
|
1464
|
+
type="line", xref="paper", yref="y",
|
|
1465
|
+
x0=0, x1=1, y0=0, y1=0,
|
|
1466
|
+
line=dict(color=to_hex("gray50"), width=1,
|
|
1467
|
+
dash="dash")))
|
|
1468
|
+
fig.update_layout(
|
|
1469
|
+
xaxis=ax_x, yaxis=ax_y, shapes=shapes, template=None,
|
|
1470
|
+
plot_bgcolor=to_hex(style_opts["panel_fill"]),
|
|
1471
|
+
paper_bgcolor=to_hex(style_opts["window_fill"]),
|
|
1472
|
+
title=dict(
|
|
1473
|
+
text=("Residuals vs Fitted Values<br>"
|
|
1474
|
+
f"<sup>{sub}</sup>"),
|
|
1475
|
+
x=0.5, xanchor="center",
|
|
1476
|
+
font=dict(size=round(
|
|
1477
|
+
16 * get_option("main_size", 1)))))
|
|
1478
|
+
return fig
|
|
1479
|
+
|
|
1480
|
+
|
|
1481
|
+
def _reg_dn_residual(resid):
|
|
1482
|
+
"""Distribution of the residuals: the density display with
|
|
1483
|
+
its histogram backdrop. ~ .reg3dnResidual"""
|
|
1484
|
+
edges = _breaks_from_args(np.asarray(resid, dtype=float),
|
|
1485
|
+
None, None, None, "Sturges")
|
|
1486
|
+
fig = dn_plotly(np.asarray(resid, dtype=float),
|
|
1487
|
+
x_name="Residuals",
|
|
1488
|
+
x_lab="Residuals",
|
|
1489
|
+
show_histogram=True, hist_edges=edges,
|
|
1490
|
+
main="Distribution of Residuals")
|
|
1491
|
+
return fig
|