lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Regression.py ADDED
@@ -0,0 +1,1491 @@
1
+ # Regression.py — analog of Regression.R (core numeric OLS)
2
+ #
3
+ # Regression(): least-squares regression with the lessR output
4
+ # pipeline — background, estimated model with confidence
5
+ # intervals, model fit (with the PRESS R-squared), sequential
6
+ # ANOVA with the aggregate Model row, collinearity, the
7
+ # residuals-and-influence listing, and prediction intervals —
8
+ # plus the plotly graphics: the simple-regression scatterplot
9
+ # with confidence and prediction bands (~ .reg5Plot), the
10
+ # distribution of residuals (~ .reg3dnResidual), and residuals
11
+ # vs fitted values with Cook's-distance flagging
12
+ # (~ .reg3resfitResidual). As with the views, the pipeline is
13
+ # ported, not the lines.
14
+ #
15
+ # The model is a formula string, "Y ~ X1 + X2" ("Y ~ ." takes
16
+ # every other numeric column), the Python analog of the R
17
+ # formula. Expression terms such as log(Years) or I(Years^2)
18
+ # are materialized as data columns (~ .formula_expr); top-level
19
+ # interaction/crossing operators (: * ^) are not ported.
20
+ # Categorical predictors become treatment-coded
21
+ # indicator variables (VarLevel columns, first level the
22
+ # reference, ~ model.matrix); with exactly one covariate and
23
+ # one factor, the ANOVA reports term-level Type II sums of
24
+ # squares — each term adjusted for the other, matching R's
25
+ # .reg1ancova as corrected July 2026 (its earlier row
26
+ # replacement mislabeled the factor's SS). The
27
+ # the full Regression.R analysis is ported. Rmd= generates a
28
+ # Quarto (.qmd) report rather than R Markdown (see reg_rmd.py).
29
+ # quiet= does not exist, as in R: Regression() always prints.
30
+ #
31
+ # Numerics through statsmodels OLS (imported lazily, as
32
+ # plt_forecast does); influence measures from OLSInfluence.
33
+ # Returns a RegressionResults object holding the tables as
34
+ # DataFrames and the plotly figures in .plots — figures are
35
+ # not auto-shown (no R graphics device to open).
36
+
37
+ import math
38
+ import re
39
+
40
+ import numpy as np
41
+ import pandas as pd
42
+ import plotly.graph_objects as go
43
+ from scipy import stats as sps
44
+
45
+ from .dn_plotly import dn_plotly
46
+ from .plt_mat_plotly import scatter_matrix
47
+ from .reg_rmd import reg_rmd
48
+ from .plotly_utils import (
49
+ BASE_COLORS, as_plotly_color, axis_format, axis_num,
50
+ make_trans, plot_border, plotly_style, to_hex, x_grid)
51
+ from .utils import (
52
+ category_order, fmt, get_column, get_option, pretty)
53
+ from .X import _breaks_from_args
54
+
55
+
56
+ def _dash(n):
57
+ return "-" * n
58
+
59
+
60
+ def _getdigits(values, min_digits=3):
61
+ """Decimal digits for output: one more than the largest
62
+ number of decimal places in the response, at least
63
+ min_digits. R analog: .getdigits()"""
64
+ dmax = 0
65
+ for v in np.asarray(values, dtype=float)[:500]:
66
+ s = f"{v:.10f}".rstrip("0")
67
+ if "." in s:
68
+ dmax = max(dmax, len(s.split(".")[1]))
69
+ if dmax >= 8:
70
+ break
71
+ return max(min_digits, min(dmax + 1, 8))
72
+
73
+
74
+ def _rescalable(s):
75
+ """A variable is rescaled only if numeric with more than two
76
+ distinct values, so binary and indicator columns pass through
77
+ unchanged. ~ Regression.R unq.x > 2 && is.numeric"""
78
+ return (pd.api.types.is_numeric_dtype(s)
79
+ and s.dropna().nunique() > 2)
80
+
81
+
82
+ def _rescale(v, kind, digits_d):
83
+ """Rescale a numeric vector, missing values ignored, rounded
84
+ to digits_d. R analog: rescale()
85
+ z (x - mean) / sd (sd with n-1)
86
+ center x - mean
87
+ 0to1 (x - min) / (max - min)
88
+ robust (x - median) / IQR"""
89
+ if kind == "z":
90
+ out = (v - np.nanmean(v)) / np.nanstd(v, ddof=1)
91
+ elif kind == "center":
92
+ out = v - np.nanmean(v)
93
+ elif kind == "0to1":
94
+ lo, hi = np.nanmin(v), np.nanmax(v)
95
+ out = (v - lo) / (hi - lo)
96
+ else: # robust
97
+ q1, q3 = np.nanpercentile(v, [25, 75])
98
+ out = (v - np.nanmedian(v)) / (q3 - q1)
99
+ return np.round(out, digits_d)
100
+
101
+
102
+ def _split_top(s, sep):
103
+ """Split s on sep at parenthesis depth 0, so a separator
104
+ inside a function call is not a break."""
105
+ out, depth, start = [], 0, 0
106
+ for i, ch in enumerate(s):
107
+ if ch in "([":
108
+ depth += 1
109
+ elif ch in ")]":
110
+ depth -= 1
111
+ elif ch == sep and depth == 0:
112
+ out.append(s[start:i])
113
+ start = i + 1
114
+ out.append(s[start:])
115
+ return [t.strip() for t in out]
116
+
117
+
118
+ def _has_top(s, chars):
119
+ """True if any char in chars appears at parenthesis depth 0."""
120
+ depth = 0
121
+ for ch in s:
122
+ if ch in "([":
123
+ depth += 1
124
+ elif ch in ")]":
125
+ depth -= 1
126
+ elif ch in chars and depth == 0:
127
+ return True
128
+ return False
129
+
130
+
131
+ _BARE = re.compile(r"^[A-Za-z_.][\w.]*$")
132
+ _EXPR_FUN = {"log": np.log, "log10": np.log10, "log2": np.log2,
133
+ "sqrt": np.sqrt, "exp": np.exp, "abs": np.abs,
134
+ "sin": np.sin, "cos": np.cos, "tan": np.tan}
135
+
136
+
137
+ def _materialize(term, data):
138
+ """Evaluate a formula expression term (log(Years), sqrt(x),
139
+ I(Years^2), x/100 ...) into a numeric column named by the
140
+ term's text, as R's .formula_expr does with terms()
141
+ variables that are calls. Returns the (name, data) with the
142
+ new column added to a copy of data."""
143
+ py = re.sub(r"\bI\(", "(", term.replace("^", "**"))
144
+ ns = {c: data[c].to_numpy() for c in data.columns}
145
+ ns.update(_EXPR_FUN)
146
+ try:
147
+ val = eval(py, {"__builtins__": {}}, ns) # user's own model
148
+ except Exception as e:
149
+ raise ValueError(
150
+ f'cannot evaluate the formula term "{term}": {e}. '
151
+ "Name a data column, or compute the transformed "
152
+ "column first.")
153
+ data = data.copy()
154
+ data[term] = np.asarray(val, dtype=float)
155
+ return term, data
156
+
157
+
158
+ def _resolve_term(term, data):
159
+ """A bare column name stays; an expression is materialized
160
+ (~ .formula_expr). A top-level interaction/crossing operator
161
+ (: * ^) is not ported."""
162
+ if _BARE.match(term) and term in data.columns:
163
+ return term, data
164
+ if _has_top(term, ":*^"):
165
+ raise NotImplementedError(
166
+ f'formula operator in "{term}" is not ported: list '
167
+ "predictors with +, and compute any interaction "
168
+ "column first (I(a*b) materializes a product term)")
169
+ return _materialize(term, data)
170
+
171
+
172
+ def _parse_formula(my_formula, data):
173
+ """ "Y ~ X1 + X2" -> (response, [predictors], data). "Y ~ ."
174
+ takes every other numeric column; "Y ~ 1" is the null model.
175
+ Expression terms such as log(Years) or I(Years^2) are
176
+ materialized as data columns (~ .formula_expr); the returned
177
+ data carries those columns."""
178
+ if not isinstance(my_formula, str):
179
+ raise TypeError(
180
+ "the model is a formula string, such as "
181
+ '"Salary ~ Years + Pre"')
182
+ if my_formula.count("~") != 1:
183
+ raise ValueError(
184
+ 'the formula has one "~": "Y ~ X1 + X2"')
185
+ lhs, rhs = (s.strip() for s in my_formula.split("~"))
186
+ if not lhs:
187
+ raise ValueError("the formula names the response "
188
+ 'before the "~"')
189
+ y_name, data = _resolve_term(lhs, data)
190
+ if rhs == ".":
191
+ preds = [c for c in data.columns
192
+ if c != y_name
193
+ and pd.api.types.is_numeric_dtype(data[c])]
194
+ elif rhs in ("1", ""):
195
+ preds = []
196
+ else:
197
+ terms = _split_top(rhs, "+")
198
+ if any(not t for t in terms):
199
+ raise ValueError(f'cannot parse "{rhs}": list the '
200
+ "predictors separated by +")
201
+ preds = []
202
+ for t in terms:
203
+ nm, data = _resolve_term(t, data)
204
+ preds.append(nm)
205
+ return y_name, preds, data
206
+
207
+
208
+ def _expand_indicators(pred_names, pred_sers, note_fmt):
209
+ """Treatment-coded indicator variables for the categorical
210
+ predictors, in formula position: the first level (declared
211
+ Categorical order, else sorted) is the reference, and each
212
+ other level becomes a 0/1 column named VarLevel, the naming
213
+ of R's model.matrix(). Returns (names, series, notes,
214
+ cat_names). R analog: the "construct the indicator
215
+ variables" block of Regression.R / Logit.R"""
216
+ from .utils import category_order
217
+ names, sers, notes, cat_names = [], [], [], []
218
+ term_map = {} # original predictor -> its columns
219
+ for nm, s in zip(pred_names, pred_sers):
220
+ if pd.api.types.is_numeric_dtype(s):
221
+ names.append(nm)
222
+ sers.append(s)
223
+ term_map[nm] = [nm]
224
+ continue
225
+ cat_names.append(nm)
226
+ notes.append(note_fmt.format(nm))
227
+ term_map[nm] = []
228
+ for lv in category_order(
229
+ s if isinstance(s.dtype, pd.CategoricalDtype)
230
+ else s.astype(str))[1:]:
231
+ dnm = f"{nm}{lv}"
232
+ names.append(dnm)
233
+ term_map[nm].append(dnm)
234
+ sers.append((s.astype(str) == str(lv))
235
+ .astype(float).rename(dnm))
236
+ return names, sers, notes, cat_names, term_map
237
+
238
+
239
+ def _best_subsets(Xd, yv, tot_ss, MSW, best_sub, nbest=10):
240
+ """Best subset regressions over all predictor subsets,
241
+ keeping the nbest best of each size (the leaps() default),
242
+ scored by adjusted R-squared or Mallows' Cp on the full
243
+ model's error variance. Engine deviation: an exhaustive
244
+ all-subsets search replaces the leaps branch-and-bound —
245
+ same results, no size limit beyond practicality.
246
+ R analog: leaps::leaps() in .reg2Relations"""
247
+ from itertools import combinations
248
+ names = list(Xd.columns)
249
+ p = len(names)
250
+ if p > 15:
251
+ return None # 2^15 solves is the cap
252
+ n = len(yv)
253
+ Xc = Xd.to_numpy(dtype=float)
254
+ rows = []
255
+ for k in range(1, p + 1):
256
+ size_rows = []
257
+ for cols in combinations(range(p), k):
258
+ Xs = np.column_stack(
259
+ [np.ones(n), Xc[:, list(cols)]])
260
+ rss = float(((yv - Xs @ np.linalg.lstsq(
261
+ Xs, yv, rcond=None)[0]) ** 2).sum())
262
+ if best_sub == "adjr2":
263
+ crit = 1 - (rss / (n - k - 1)) \
264
+ / (tot_ss / (n - 1))
265
+ else: # Mallows' Cp
266
+ crit = rss / MSW - (n - 2 * (k + 1))
267
+ size_rows.append((cols, crit))
268
+ size_rows.sort(key=lambda r: r[1],
269
+ reverse=best_sub == "adjr2")
270
+ rows += size_rows[:nbest]
271
+ lbl = "R2adj" if best_sub == "adjr2" else "Cp"
272
+ out = pd.DataFrame(
273
+ [[int(j in cols) for j in range(p)] + [crit, len(cols)]
274
+ for cols, crit in rows],
275
+ columns=names + [lbl, "X's"])
276
+ return out.sort_values(
277
+ lbl, ascending=best_sub != "adjr2",
278
+ kind="stable").reset_index(drop=True)
279
+
280
+
281
+ def _prntbl(df, digits_d, int_cols=()):
282
+ """Aligned text table with row labels, floats at digits_d.
283
+ Columns named in int_cols print as integers. R analog:
284
+ .prntbl()"""
285
+ show = pd.DataFrame(index=df.index.astype(str))
286
+ for c in df.columns:
287
+ col = df[c]
288
+ if c in int_cols:
289
+ show[c] = ["" if pd.isna(v) else str(int(round(v)))
290
+ for v in col]
291
+ elif pd.api.types.is_float_dtype(col):
292
+ show[c] = [fmt(v, digits_d) if pd.notna(v) else ""
293
+ for v in col]
294
+ else:
295
+ show[c] = col.astype(str)
296
+ return show.to_string()
297
+
298
+
299
+ def _int_vars(df, names):
300
+ """Of the given data columns, those whose values are all whole
301
+ numbers -- shown without decimals, as the integer variables
302
+ they are (an integer column with a missing value elsewhere is
303
+ stored as float, so this is decided by value, not dtype)."""
304
+ out = []
305
+ for nm in names:
306
+ if nm not in df.columns:
307
+ continue
308
+ col = df[nm]
309
+ if not pd.api.types.is_numeric_dtype(col):
310
+ continue
311
+ v = col.to_numpy(dtype=float)
312
+ v = v[np.isfinite(v)]
313
+ if len(v) and np.all(v == np.floor(v)):
314
+ out.append(nm)
315
+ return out
316
+
317
+
318
+ class RegressionResults:
319
+ """Numeric results and figures of Regression(): estimates,
320
+ fit, anova (DataFrames/dict), residuals and predictions
321
+ listings, and the plotly figures in .plots."""
322
+
323
+ def __init__(self, **kw):
324
+ self.__dict__.update(kw)
325
+
326
+ def __repr__(self):
327
+ return (f"<lessPy Regression: {self.formula}, "
328
+ f"n={self.n_keep}>")
329
+
330
+
331
+ def Regression(my_formula, data=None, filter=None, digits_d=None,
332
+ brief=False,
333
+ n_res_rows=None, res_sort="cooks",
334
+ n_pred_rows=None, pred_sort="predint",
335
+ subsets=None, best_sub="adjr2",
336
+ cooks_cut=1,
337
+ X1_new=None, X2_new=None, X3_new=None,
338
+ X4_new=None, X5_new=None, X6_new=None,
339
+ kfold=0, seed=None,
340
+ new_scale="none", scale_response=False,
341
+ mod=None, mod_transf="center",
342
+ Rmd=None, Rmd_data=None, Rmd_format="html",
343
+ Rmd_browser=True,
344
+ results=True, explain=True, interpret=True,
345
+ code=True,
346
+ graphics=True):
347
+ """Least-squares regression of a formula string,
348
+ "Y ~ X1 + X2", with the lessR analysis pipeline: estimates,
349
+ fit, ANOVA, collinearity, residuals and influence,
350
+ prediction intervals, and the regression graphics. Always
351
+ prints, as in R; returns a RegressionResults object with
352
+ the figures in .plots."""
353
+ import statsmodels.api as smapi
354
+
355
+ if data is None:
356
+ raise ValueError(
357
+ "data= is required: a pandas DataFrame containing "
358
+ "the model's variables")
359
+ if res_sort not in ("cooks", "rstudent", "dffits", "off"):
360
+ raise ValueError(
361
+ 'res_sort: "cooks", "rstudent", "dffits", or "off"')
362
+ if pred_sort not in ("predint", "off"):
363
+ raise ValueError('pred_sort: "predint" or "off"')
364
+ if best_sub not in ("adjr2", "Cp"):
365
+ raise ValueError('best_sub: "adjr2" or "Cp"')
366
+ if new_scale not in ("none", "z", "center", "0to1", "robust"):
367
+ raise ValueError('new_scale: "none", "z", "center", '
368
+ '"0to1", or "robust"')
369
+ if mod_transf not in ("center", "z", "none"):
370
+ raise ValueError('mod_transf: "center", "z", or "none"')
371
+ if Rmd is not None:
372
+ if Rmd_format not in ("html", "pdf", "docx", "word",
373
+ "none"):
374
+ raise ValueError('Rmd_format: "html", "pdf", '
375
+ '"docx", or "none"')
376
+ if brief:
377
+ raise ValueError(
378
+ "a Quarto report needs the full analysis, so "
379
+ "Rmd= is not available with brief=True")
380
+ if filter is not None:
381
+ data = data.query(filter)
382
+
383
+ y_name, pred_names, data = _parse_formula(my_formula, data)
384
+ formula = (f"{y_name} ~ "
385
+ + (" + ".join(pred_names) if pred_names else "1"))
386
+ n_pred = len(pred_names)
387
+
388
+ y_ser = get_column(data, y_name, "response")
389
+ if not pd.api.types.is_numeric_dtype(y_ser):
390
+ raise TypeError(
391
+ f"'{y_name}' is {y_ser.dtype}: the response of "
392
+ "Regression() is numeric. For a two-level "
393
+ "categorical response use Logit().")
394
+ pred_sers = [get_column(data, nm, "predictor")
395
+ for nm in pred_names]
396
+
397
+ # digits from the response, before any rescaling (~ R sets
398
+ # digits_d here, ahead of the new_scale block)
399
+ if digits_d is None:
400
+ digits_d = _getdigits(
401
+ y_ser.dropna().to_numpy(dtype=float))
402
+ d = digits_d
403
+
404
+ # new_scale: rescale the numeric model variables in place --
405
+ # predictors always, the response only when scale_response;
406
+ # binary and non-numeric variables pass through unchanged
407
+ # (~ Regression.R rescale block, rescale.R). For kfold the
408
+ # rescaling is per fold, done in _reg_kfold.
409
+ transf = None
410
+ rescale_lines = []
411
+ if new_scale != "none":
412
+ transf = {"z": "Standardized", "center": "Centered",
413
+ "0to1": "Min-Max (0 to 1)",
414
+ "robust": "Robust Version of Standardized"
415
+ }[new_scale]
416
+ if kfold == 0:
417
+ if scale_response and _rescalable(y_ser):
418
+ y_ser = pd.Series(
419
+ _rescale(y_ser.to_numpy(dtype=float),
420
+ new_scale, d),
421
+ index=y_ser.index, name=y_name)
422
+ pred_sers = [
423
+ pd.Series(_rescale(s.to_numpy(dtype=float),
424
+ new_scale, d),
425
+ index=s.index, name=s.name)
426
+ if _rescalable(s) else s
427
+ for s in pred_sers]
428
+ head = pd.concat(
429
+ [y_ser] + pred_sers, axis=1).head(6)
430
+ rescale_lines = (
431
+ ["Rescaled Data, First Six Rows", ""]
432
+ + head.to_string().split("\n") + [""])
433
+
434
+ # mod: moderation. Add the X*W interaction and refit
435
+ # Y ~ X + W + X*W. The two predictors are first centered (or
436
+ # standardized) per mod_transf, which curbs the collinearity
437
+ # the product term would otherwise introduce. ~ Regression.R
438
+ # mod block, .reg6mod
439
+ mod_info = None
440
+ if mod is not None:
441
+ if n_pred != 2:
442
+ raise ValueError(
443
+ "mod moderation currently needs exactly 2 "
444
+ "predictors")
445
+ if mod not in pred_names:
446
+ raise ValueError(
447
+ f"mod variable '{mod}' must be one of the two "
448
+ "predictors")
449
+ if any(not pd.api.types.is_numeric_dtype(s)
450
+ for s in pred_sers):
451
+ raise TypeError(
452
+ "both predictors of a moderation must be numeric")
453
+ cc = pd.concat([y_ser] + pred_sers,
454
+ axis=1).notna().all(axis=1)
455
+ if mod_transf != "none":
456
+ is_z = mod_transf == "z"
457
+ scaled = []
458
+ for s in pred_sers:
459
+ v = s.astype(float)
460
+ mu = v[cc].mean()
461
+ v = ((v - mu) / v[cc].std(ddof=1) if is_z
462
+ else v - mu)
463
+ scaled.append(pd.Series(v, index=s.index,
464
+ name=s.name))
465
+ pred_sers = scaled
466
+ w_name = mod
467
+ x_name = next(p for p in pred_names if p != mod)
468
+ xw_name = f"{w_name}.{x_name}"
469
+ inter = pd.Series(
470
+ pred_sers[0].to_numpy(dtype=float)
471
+ * pred_sers[1].to_numpy(dtype=float),
472
+ index=pred_sers[0].index, name=xw_name)
473
+ pred_names = pred_names + [xw_name]
474
+ pred_sers = pred_sers + [inter]
475
+ mod_info = {"x": x_name, "w": w_name, "xw": xw_name}
476
+
477
+ used = pd.concat([y_ser] + pred_sers, axis=1)
478
+ keep = ~used.isna().any(axis=1)
479
+ n_obs = len(data)
480
+ n_keep = int(keep.sum())
481
+
482
+ # categorical predictors become indicator variables; with
483
+ # exactly one covariate and one factor, the ANOVA reports
484
+ # Type II sums of squares (the ANCOVA table)
485
+ (pred_names, pred_sers_x, ind_notes, cat_names,
486
+ term_map) = _expand_indicators(
487
+ pred_names, [s[keep] for s in pred_sers],
488
+ ">>> {0} is not numeric. "
489
+ "Converted to indicator variables.")
490
+ ancova = (n_pred == 2 and len(cat_names) == 1)
491
+ ancova_terms = ([nm for nm in term_map
492
+ if nm not in cat_names] + cat_names
493
+ if ancova else None)
494
+ ancova_info = None
495
+ if ancova:
496
+ # the covariate is the numeric term (maps to itself), the
497
+ # factor is the one categorical predictor; keep its
498
+ # original values, in level order, for the group plot
499
+ cov_name = next(k for k, v in term_map.items()
500
+ if v == [k])
501
+ fac_name = cat_names[0]
502
+ fac_series = next(s for s in pred_sers
503
+ if s.name == fac_name)[keep]
504
+ fac_series = (fac_series
505
+ if isinstance(fac_series.dtype,
506
+ pd.CategoricalDtype)
507
+ else fac_series.astype(str))
508
+ ancova_info = {
509
+ "cov": cov_name, "fac": fac_name,
510
+ "preds": list(term_map.keys()),
511
+ "levels": [str(lv) for lv in
512
+ category_order(fac_series)],
513
+ "fac_vals": fac_series.astype(str).to_numpy()}
514
+ # collinearity and subsets follow the count of listed
515
+ # predictors, before indicator expansion, as R
516
+ n_pred_orig = n_pred
517
+ n_pred = len(pred_names)
518
+
519
+ if n_keep < n_pred + 2:
520
+ raise ValueError(
521
+ f"only {n_keep} complete rows: too few to estimate "
522
+ f"{n_pred + 1} coefficients")
523
+ yv = y_ser[keep].to_numpy(dtype=float)
524
+ Xd = pd.DataFrame(
525
+ {nm: s.to_numpy(dtype=float)
526
+ for nm, s in zip(pred_names, pred_sers_x)},
527
+ index=y_ser[keep].index)
528
+ row_labels = Xd.index.astype(str)
529
+
530
+ if kfold and kfold > 0:
531
+ # cross-validation replaces the single-model analysis;
532
+ # graphics and the residual/prediction listings are off,
533
+ # exactly as R turns them off for kfold > 0
534
+ if n_pred == 0:
535
+ raise ValueError(
536
+ "kfold cross-validation needs at least one "
537
+ "predictor")
538
+ return _reg_kfold(yv, Xd, y_name, pred_names, n_keep,
539
+ formula, kfold, seed, d, smapi,
540
+ new_scale, scale_response)
541
+
542
+ X = smapi.add_constant(Xd, has_constant="add") \
543
+ if n_pred > 0 else pd.DataFrame(
544
+ {"const": np.ones(n_keep)}, index=Xd.index)
545
+ fit = smapi.OLS(yv, X).fit()
546
+ infl = fit.get_influence()
547
+ hat = infl.hat_matrix_diag
548
+ rstudent = infl.resid_studentized_external
549
+ dffits = infl.dffits[0]
550
+ cooks = infl.cooks_distance[0]
551
+ resid = np.asarray(fit.resid, dtype=float)
552
+ fitted = np.asarray(fit.fittedvalues, dtype=float)
553
+ df_res = int(fit.df_resid)
554
+
555
+ lines = list(rescale_lines) # console output
556
+ for note in ind_notes:
557
+ lines += [note, ""]
558
+
559
+ # ---------- BACKGROUND -------------------------------------
560
+ lines += ["", " BACKGROUND", ""]
561
+ for i, nm in enumerate([y_name] + pred_names):
562
+ if i == 0:
563
+ lbl = "Response Variable: "
564
+ elif n_pred > 1:
565
+ lbl = f"Predictor Variable {i}: "
566
+ else:
567
+ lbl = "Predictor Variable: "
568
+ lines.append(lbl + nm)
569
+ if transf is not None:
570
+ lines += ["", f"Data are {transf}"]
571
+ lines += ["",
572
+ f"Number of cases (rows) of data: {n_obs}",
573
+ f"Number of cases retained for analysis: "
574
+ f"{n_keep}"]
575
+
576
+ # ---------- BASIC ANALYSIS ---------------------------------
577
+ lines += ["", "", " BASIC ANALYSIS", ""]
578
+
579
+ # estimates with 95% confidence intervals, ~ .reg1modelBasic
580
+ ci = fit.conf_int(alpha=0.05)
581
+ est = pd.DataFrame({
582
+ "Estimate": fit.params,
583
+ "Std Err": fit.bse,
584
+ "t-value": fit.tvalues,
585
+ "p-value": fit.pvalues,
586
+ "Lower 95%": ci[0],
587
+ "Upper 95%": ci[1],
588
+ })
589
+ est.index = ["(Intercept)"] + pred_names
590
+ lines += [f"-- Estimated Model for {y_name}", ""]
591
+ buf = max(len(s) for s in est.index)
592
+ w = [max(9, max(len(fmt(v, d)) for v in est[c]) + 1)
593
+ for c in est.columns]
594
+ lines.append(" " * buf
595
+ + f"{'Estimate':>{w[0] + 1}}"
596
+ + f"{'Std Err':>{w[1] + 2}}"
597
+ + f"{'t-value':>9}{'p-value':>9}"
598
+ + f"{'Lower 95%':>{w[4] + 3}}"
599
+ + f"{'Upper 95%':>{w[5] + 3}}")
600
+ for lbl, r in est.iterrows():
601
+ lines.append(
602
+ f"{lbl:<{buf}}"
603
+ + f"{fmt(r['Estimate'], d):>{w[0] + 1}}"
604
+ + f"{fmt(r['Std Err'], d):>{w[1] + 2}}"
605
+ + f"{fmt(r['t-value'], 3):>9}"
606
+ + f"{fmt(r['p-value'], 3):>9}"
607
+ + f"{fmt(r['Lower 95%'], d):>{w[4] + 3}}"
608
+ + f"{fmt(r['Upper 95%'], d):>{w[5] + 3}}")
609
+
610
+ # model fit, ~ .reg1fitBasic
611
+ tot_ss = float(((yv - yv.mean()) ** 2).sum())
612
+ sy = math.sqrt(tot_ss / (n_keep - 1))
613
+ se = math.sqrt(fit.scale)
614
+ tcut = -sps.t.ppf(0.025, df=df_res)
615
+ res_range = 2 * tcut * se
616
+ prs = resid / (1 - hat)
617
+ prs = np.where(hat >= 1 - 1e-13, np.nan, prs)
618
+ PRESS = float(np.nansum(prs ** 2))
619
+ Rsq_press = (np.nan if tot_ss == 0
620
+ else 1 - PRESS / tot_ss)
621
+ lines += ["", "-- Model Fit", "",
622
+ f"Standard deviation of {y_name}: {fmt(sy, d)}",
623
+ "",
624
+ f"Standard deviation of residuals: {fmt(se, d)}"
625
+ f" for df={df_res}",
626
+ f"95% range of residuals: {fmt(res_range, d)}"
627
+ f" = 2 * ({fmt(tcut, 3)} * {fmt(se, d)})"]
628
+ if n_pred > 0:
629
+ lines += ["",
630
+ f"R-squared: {fmt(fit.rsquared, 3)} "
631
+ f"Adjusted R-squared: "
632
+ f"{fmt(fit.rsquared_adj, 3)} "
633
+ f"PRESS R-squared: {fmt(Rsq_press, 3)}",
634
+ "",
635
+ "Null hypothesis of all 0 population slope "
636
+ "coefficients:",
637
+ f" F-statistic: {fmt(fit.fvalue, 3)} "
638
+ f"df: {int(fit.df_model)} and {df_res} "
639
+ f"p-value: {fmt(fit.f_pvalue, 3)}"]
640
+
641
+ # ANOVA: sequential (Type I) with the Model row,
642
+ # ~ .reg1anvBasic; for the ANCOVA case (one covariate, one
643
+ # factor) term-level Type II sums of squares, each term
644
+ # adjusted for the other
645
+ anova = None
646
+ MSW = fit.scale
647
+ res_ss = float(fit.ssr)
648
+ if n_pred > 0 and ancova:
649
+ rows = []
650
+ for t in ancova_terms:
651
+ others = [c for c in pred_names
652
+ if c not in term_map[t]]
653
+ Xr = smapi.add_constant(Xd[others],
654
+ has_constant="add")
655
+ ss = float(smapi.OLS(yv, Xr).fit().ssr) - res_ss
656
+ dft = len(term_map[t])
657
+ f_v = (ss / dft) / MSW
658
+ rows.append([t, dft, ss, ss / dft, f_v,
659
+ float(sps.f.sf(f_v, dft, df_res))])
660
+ anova = pd.DataFrame(
661
+ rows + [["Residuals", df_res, res_ss, MSW,
662
+ np.nan, np.nan]],
663
+ columns=["term", "df", "Sum Sq", "Mean Sq",
664
+ "F-value", "p-value"]).set_index("term")
665
+ lines += ["", "-- Analysis of Variance from Type II "
666
+ "Sums of Squares", ""]
667
+ anv_names = ancova_terms
668
+ elif n_pred > 0:
669
+ seq_ss = []
670
+ rss_prev = tot_ss
671
+ for i in range(1, n_pred + 1):
672
+ Xi = smapi.add_constant(Xd.iloc[:, :i],
673
+ has_constant="add")
674
+ rss_i = float(smapi.OLS(yv, Xi).fit().ssr)
675
+ seq_ss.append(rss_prev - rss_i)
676
+ rss_prev = rss_i
677
+ rows = []
678
+ for nm, ss in zip(pred_names, seq_ss):
679
+ f_v = ss / MSW
680
+ rows.append([nm, 1, ss, ss, f_v,
681
+ float(sps.f.sf(f_v, 1, df_res))])
682
+ mod_df = n_pred
683
+ mod_ss = sum(seq_ss)
684
+ mod_ms = mod_ss / mod_df
685
+ mod_f = mod_ms / MSW
686
+ mod_p = float(sps.f.sf(mod_f, mod_df, df_res))
687
+ anova = pd.DataFrame(
688
+ rows + [["Model", mod_df, mod_ss, mod_ms, mod_f,
689
+ mod_p],
690
+ ["Residuals", df_res, res_ss, MSW,
691
+ np.nan, np.nan],
692
+ [y_name, n_keep - 1, tot_ss,
693
+ tot_ss / (n_keep - 1), np.nan, np.nan]],
694
+ columns=["term", "df", "Sum Sq", "Mean Sq",
695
+ "F-value", "p-value"]).set_index("term")
696
+ lines += ["", "-- Analysis of Variance", ""]
697
+ anv_names = pred_names
698
+ if n_pred > 0:
699
+ c1 = max(len(s) for s in anova.index)
700
+ wn = [max(9, max(len(fmt(v, d))
701
+ for v in anova[c].dropna()) + 1)
702
+ for c in ("Sum Sq", "Mean Sq", "F-value")]
703
+ lines.append(" " * c1 + f"{'df':>7}"
704
+ + f"{'Sum Sq':>{wn[0]}}"
705
+ + f"{'Mean Sq':>{wn[1]}}"
706
+ + f"{'F-value':>{wn[2]}}"
707
+ + f"{'p-value':>9}")
708
+
709
+ def anv_line(lbl, r, with_test=True):
710
+ t = (f"{lbl:<{c1}}{int(r['df']):>7}"
711
+ + f"{fmt(r['Sum Sq'], d):>{wn[0]}}"
712
+ + f"{fmt(r['Mean Sq'], d):>{wn[1]}}")
713
+ if with_test:
714
+ t += (f"{fmt(r['F-value'], d):>{wn[2]}}"
715
+ + f"{fmt(r['p-value'], 3):>9}")
716
+ return t
717
+
718
+ for nm in anv_names:
719
+ lines.append(anv_line(nm, anova.loc[nm]))
720
+ if not ancova: # Type II has no total
721
+ lines.append("")
722
+ lines.append(anv_line("Model",
723
+ anova.loc["Model"]))
724
+ lines.append(anv_line("Residuals",
725
+ anova.loc["Residuals"],
726
+ with_test=False))
727
+ if not ancova:
728
+ lines.append(anv_line(y_name, anova.loc[y_name],
729
+ with_test=False))
730
+
731
+ # ---------- MODERATION ANALYSIS ----------------------------
732
+ # the simple slopes at the moderator's mean and +/-1 SD; as R,
733
+ # this section is generated with the graphics (~ .reg6mod)
734
+ if mod_info is not None and graphics:
735
+ lines += _moderation_lines(fit, Xd, mod_info, d)
736
+
737
+ # ---------- ANCOVA GROUP MODELS ----------------------------
738
+ # the interaction test and the per-level parallel-line
739
+ # equations; as R, generated with the graphics (~ .reg5ancova)
740
+ if ancova and graphics:
741
+ lines += _reg_ancova_models(fit, Xd, yv, y_name,
742
+ ancova_info, d, smapi)
743
+
744
+ # ---------- RELATIONS AMONG THE VARIABLES ------------------
745
+ # collinearity, ~ .reg2Relations (tolerance and VIF from the
746
+ # coefficient standard errors, the R computation), and the
747
+ # best-subset models
748
+ tol = vif = None
749
+ sub_df = None
750
+ if not brief and n_pred_orig > 1:
751
+ vif = np.array([
752
+ (Xd[nm].var(ddof=1) * (n_keep - 1)
753
+ * est.loc[nm, "Std Err"] ** 2) / MSW
754
+ for nm in pred_names])
755
+ tol = 1 / vif
756
+ lines += ["", "", " RELATIONS AMONG THE VARIABLES",
757
+ "", "-- Collinearity", ""]
758
+ c1 = max(len(s) for s in pred_names)
759
+ lines.append(" " * c1 + f"{'Tolerance':>11}"
760
+ + f"{'VIF':>9}")
761
+ for nm, t_i, v_i in zip(pred_names, tol, vif):
762
+ lines.append(f"{nm:<{c1}}{fmt(t_i, 3):>11}"
763
+ + f"{fmt(v_i, 3):>9}")
764
+
765
+ # best subsets: default on, subsets=n caps the listing
766
+ max_sublns = 50
767
+ do_subsets = True if subsets is None else subsets
768
+ if not isinstance(do_subsets, bool) \
769
+ and isinstance(do_subsets, (int, float)) \
770
+ and do_subsets > 1:
771
+ max_sublns = int(do_subsets)
772
+ do_subsets = True
773
+ if do_subsets:
774
+ sub_df = _best_subsets(Xd, yv, tot_ss, MSW,
775
+ best_sub)
776
+ if sub_df is not None:
777
+ lines += ["", "-- Best Subset Regression Models"]
778
+ if n_pred > 5:
779
+ lines.append("up to 10 subsets of each "
780
+ "number of predictors")
781
+ crit_lbl = sub_df.columns[-2]
782
+ xs_lbl = "X's"
783
+ wids = [max(4, len(nm) + 1) for nm in pred_names]
784
+ hdr = ("".join(f"{nm:>{w}}" for nm, w in
785
+ zip(pred_names, wids))
786
+ + f"{crit_lbl:>9}{xs_lbl:>7}")
787
+ lines += ["", hdr]
788
+ shown = min(max_sublns, len(sub_df))
789
+ for i in range(shown):
790
+ if shown > 40 and (i + 1) % 30 == 0:
791
+ lines.append(hdr)
792
+ r = sub_df.iloc[i]
793
+ lines.append(
794
+ "".join(f"{int(r[nm]):>{w}}"
795
+ for nm, w in zip(pred_names,
796
+ wids))
797
+ + f" {r[crit_lbl]:>8.3f}"
798
+ + f" {int(r[xs_lbl]):>6}")
799
+ if len(sub_df) > max_sublns:
800
+ lines += ["",
801
+ f">>> Only first {shown} of "
802
+ f"{len(sub_df)} rows printed",
803
+ " To indicate more, add "
804
+ "subsets=n, where n is the number "
805
+ "of lines"]
806
+ lines += ["",
807
+ "[exhaustive search of all predictor "
808
+ "subsets]"]
809
+
810
+ # ---------- RESIDUALS AND INFLUENCE ------------------------
811
+ res_tbl = None
812
+ if brief and n_res_rows is None:
813
+ n_res_rows = 0
814
+ if n_res_rows is None:
815
+ n_res_rows = n_keep if n_keep < 20 else 20
816
+ if n_res_rows == "all":
817
+ n_res_rows = n_keep
818
+ n_res_rows = min(int(n_res_rows), n_keep)
819
+
820
+ if n_res_rows > 0:
821
+ res_tbl = pd.DataFrame(index=row_labels)
822
+ for nm in pred_names:
823
+ res_tbl[nm] = Xd[nm].to_numpy()
824
+ res_tbl[y_name] = yv
825
+ res_tbl["fitted"] = fitted
826
+ res_tbl["resid"] = resid
827
+ res_tbl["rstdnt"] = rstudent
828
+ res_tbl["dffits"] = dffits
829
+ res_tbl["cooks"] = np.round(cooks, 5)
830
+ if res_sort == "cooks":
831
+ res_tbl = res_tbl.sort_values(
832
+ "cooks", ascending=False)
833
+ elif res_sort == "rstudent":
834
+ res_tbl = res_tbl.reindex(
835
+ res_tbl["rstdnt"].abs().sort_values(
836
+ ascending=False).index)
837
+ elif res_sort == "dffits":
838
+ res_tbl = res_tbl.reindex(
839
+ res_tbl["dffits"].abs().sort_values(
840
+ ascending=False).index)
841
+ lines += ["", "", " RESIDUALS AND INFLUENCE", "",
842
+ "-- Data, Fitted, Residual, Studentized "
843
+ "Residual, Dffits, Cook's Distance"]
844
+ if res_sort == "cooks":
845
+ lines.append(" [sorted by Cook's Distance]")
846
+ elif res_sort == "rstudent":
847
+ lines.append(" [sorted by Studentized Residual,"
848
+ " ignoring + or - sign]")
849
+ elif res_sort == "dffits":
850
+ lines.append(" [sorted by dffits, ignoring + or"
851
+ " - sign]")
852
+ more = (" rows of data, or do n_res_rows=\"all\"]"
853
+ if n_res_rows < n_keep else "]")
854
+ lines.append(f" [n_res_rows = {n_res_rows}, out of "
855
+ f"{n_keep}{more}")
856
+ res_ints = _int_vars(res_tbl, list(pred_names) + [y_name])
857
+ lines += _prntbl(res_tbl.head(n_res_rows), d,
858
+ int_cols=res_ints).split("\n")
859
+
860
+ # ---------- PREDICTION ERROR -------------------------------
861
+ pred_tbl = None
862
+ new_data = X1_new is not None
863
+ if brief and n_pred_rows is None and not new_data:
864
+ n_pred_rows = 0
865
+ if n_pred_rows is None:
866
+ n_pred_rows = n_keep if n_keep < 25 else 10
867
+ if n_pred_rows == "all":
868
+ n_pred_rows = n_keep
869
+
870
+ if n_pred_rows > 0 or new_data:
871
+ if new_data: # X1_new..X6_new grid
872
+ grids = [np.atleast_1d(g) for g in
873
+ (X1_new, X2_new, X3_new, X4_new,
874
+ X5_new, X6_new)[:n_pred]
875
+ if g is not None]
876
+ if len(grids) != n_pred:
877
+ raise ValueError(
878
+ "specify new values for every predictor: "
879
+ f"X1_new ... X{n_pred}_new")
880
+ mesh = np.meshgrid(*grids, indexing="ij")
881
+ Xnew = pd.DataFrame(
882
+ {nm: m.ravel() for nm, m in
883
+ zip(pred_names, mesh)})
884
+ Xn = smapi.add_constant(Xnew, has_constant="add")
885
+ pr = fit.get_prediction(Xn)
886
+ base = Xnew.copy()
887
+ base[y_name] = ""
888
+ base.index = [""] * len(base)
889
+ else:
890
+ pr = fit.get_prediction(X)
891
+ base = pd.DataFrame(index=row_labels)
892
+ for nm in pred_names:
893
+ base[nm] = Xd[nm].to_numpy()
894
+ base[y_name] = yv
895
+ sf = pr.summary_frame(alpha=0.05)
896
+ s_pred = np.sqrt(fit.scale + pr.se_mean ** 2)
897
+ pred_tbl = base
898
+ pred_tbl["pred"] = sf["mean"].to_numpy()
899
+ pred_tbl["s_pred"] = s_pred
900
+ pred_tbl["pi.lwr"] = sf["obs_ci_lower"].to_numpy()
901
+ pred_tbl["pi.upr"] = sf["obs_ci_upper"].to_numpy()
902
+ pred_tbl["width"] = (pred_tbl["pi.upr"]
903
+ - pred_tbl["pi.lwr"])
904
+ if pred_sort == "predint":
905
+ pred_tbl = pred_tbl.sort_values("pi.lwr")
906
+
907
+ hdr = ["", "", " PREDICTION ERROR", "",
908
+ "-- Data, Predicted, Standard Error of "
909
+ "Prediction, 95% Prediction Intervals",
910
+ " [sorted by lower bound of prediction "
911
+ "interval]"]
912
+ if n_pred_rows < n_keep and not new_data:
913
+ hdr.append(' [to see all intervals add '
914
+ 'n_pred_rows="all"]')
915
+ hdr.append(_dash(46))
916
+ lines += hdr
917
+ pred_ints = _int_vars(pred_tbl, list(pred_names)
918
+ + [y_name])
919
+ if new_data or n_pred_rows >= len(pred_tbl):
920
+ lines += _prntbl(pred_tbl, d,
921
+ int_cols=pred_ints).split("\n")
922
+ else:
923
+ # three pieces, as R: around the widest interval
924
+ # when it falls in that half (else the start/end),
925
+ # and around the narrowest interval in the middle
926
+ widths = pred_tbl["width"].to_numpy()
927
+ nr = len(pred_tbl)
928
+ min_row = int(widths.argmin())
929
+ max_row = int(widths.argmax())
930
+ max_side = max_row < n_keep / 2
931
+ piece = max(1, round(n_pred_rows / 3))
932
+ pr2 = piece // 2
933
+ if max_side:
934
+ r1 = list(range(max(max_row - pr2, 0),
935
+ min(max_row + pr2 + 1, nr)))
936
+ r3 = list(range(max(nr - piece, 0), nr))
937
+ else:
938
+ r1 = list(range(0, min(piece, nr)))
939
+ r3 = list(range(max(max_row - pr2, 0),
940
+ min(max_row + pr2 + 1, nr)))
941
+ r2 = list(range(max(min_row - pr2, 0),
942
+ min(min_row + pr2 + 1, nr)))
943
+ body = _prntbl(pred_tbl, d,
944
+ int_cols=pred_ints).split("\n")
945
+ head_ln, rows_ln = body[0], body[1:]
946
+ lines.append(head_ln)
947
+ for k, rr in enumerate((r1, r2, r3)):
948
+ if k > 0:
949
+ lines.append("...")
950
+ for i in rr:
951
+ lines.append(rows_ln[i])
952
+
953
+ # ---------- graphics ---------------------------------------
954
+ plots = {}
955
+ if graphics and n_pred > 0 and n_res_rows > 0:
956
+ plots["residuals_density"] = _reg_dn_residual(resid)
957
+ plots["residuals_fitted"] = _reg_resfit(
958
+ fitted, resid, cooks, cooks_cut, row_labels, d)
959
+ if graphics and n_pred == 1:
960
+ pi_df = (pred_tbl if (pred_tbl is not None
961
+ and not new_data) else None)
962
+ plots["scatter"] = _reg_scatter(
963
+ Xd.iloc[:, 0].to_numpy(), yv, pred_names[0],
964
+ y_name, fit, smapi, d,
965
+ with_bands=pi_df is not None)
966
+ if graphics and n_pred > 1 and not ancova:
967
+ # scatterplot matrix of the model variables, response
968
+ # first, with the least-squares fit; ~ .reg5Plot .plt.mat
969
+ mat_df = pd.DataFrame(
970
+ {y_name: yv}, index=Xd.index).join(Xd)
971
+ plots["scatter_matrix"] = scatter_matrix(
972
+ mat_df, fit="lm", digits_d=d)
973
+ if graphics and ancova:
974
+ # ANCOVA: grouped scatterplot with a parallel
975
+ # least-squares line per factor level; ~ .reg5ancova
976
+ plots["ancova"] = _reg_ancova_plot(
977
+ fit, Xd, yv, y_name, ancova_info, d)
978
+ if graphics and mod_info is not None:
979
+ plots["moderation"] = _moderation_plot(
980
+ fit, Xd, y_name, mod_info, d)
981
+
982
+ print("\n".join(lines))
983
+
984
+ out = RegressionResults(
985
+ formula=formula, n_obs=n_obs, n_keep=n_keep,
986
+ digits_d=d,
987
+ estimates=est, anova=anova,
988
+ fit={"se": se, "resid_range": res_range,
989
+ "Rsq": fit.rsquared if n_pred else np.nan,
990
+ "Rsq_adj": (fit.rsquared_adj if n_pred
991
+ else np.nan),
992
+ "PRESS": PRESS, "Rsq_PRESS": Rsq_press,
993
+ "sy": sy, "MSW": MSW},
994
+ tolerance=tol, vif=vif, subsets=sub_df,
995
+ residuals=res_tbl, predictions=pred_tbl,
996
+ plots=plots)
997
+
998
+ if Rmd is not None:
999
+ reg_rmd(out, y_name, pred_names, formula,
1000
+ list(data.columns), Rmd, Rmd_data, Rmd_format,
1001
+ Rmd_browser, results, explain, interpret, code,
1002
+ n_res_rows, n_pred_rows, res_sort, d)
1003
+
1004
+ return out
1005
+
1006
+
1007
+ def _reg_kfold(yv, Xd, y_name, pred_names, n_keep, formula,
1008
+ kfold, seed, digits_d, smapi,
1009
+ new_scale="none", scale_response=False):
1010
+ """K-fold cross-validation. Partition the complete cases into
1011
+ kfold folds; for each, fit on the other kfold-1 folds
1012
+ (training) and evaluate the fitted model on the held-out fold
1013
+ (testing). Report per-fold and mean se/MSE/Rsq for both, the
1014
+ training se as the residual sigma and the testing sp from the
1015
+ held-out prediction errors. ~ .regKfold
1016
+
1017
+ With new_scale, each fold's training and testing subsets are
1018
+ rescaled separately (predictors always, the response only
1019
+ when scale_response), the continuous variables only.
1020
+
1021
+ Folds are random; seed= makes them reproducible within Python
1022
+ but, because the RNG differs from R's, not identical to R for
1023
+ the same seed."""
1024
+ if kfold < 2:
1025
+ raise ValueError("kfold must be 2 or larger")
1026
+ d = 3 if digits_d is None else digits_d
1027
+
1028
+ # each row's fold, from a scrambled 1..n mod kfold, as R
1029
+ rng = np.random.default_rng(seed)
1030
+ nk = rng.permutation(np.arange(1, n_keep + 1)) % kfold
1031
+
1032
+ Xmat = Xd.to_numpy(dtype=float)
1033
+ # continuous columns (>2 distinct) are the ones rescaled;
1034
+ # the decision is global, the rescaling is per fold, as R
1035
+ do_scale = new_scale != "none"
1036
+ cont = ([c for c in range(Xmat.shape[1])
1037
+ if len(np.unique(Xmat[:, c])) > 2] if do_scale
1038
+ else [])
1039
+ scale_y = (do_scale and scale_response
1040
+ and len(np.unique(yv)) > 2)
1041
+ tr = {"n": [], "se": [], "MSE": [], "Rsq": []}
1042
+ te = {"n": [], "se": [], "MSE": [], "Rsq": []}
1043
+ for f in range(kfold):
1044
+ train, test = nk != f, nk == f
1045
+ ytr, yte = yv[train].copy(), yv[test].copy()
1046
+ Xtr_m, Xte_m = Xmat[train].copy(), Xmat[test].copy()
1047
+ if do_scale: # separate scaling per side
1048
+ for c in cont:
1049
+ Xtr_m[:, c] = _rescale(Xtr_m[:, c], new_scale, d)
1050
+ Xte_m[:, c] = _rescale(Xte_m[:, c], new_scale, d)
1051
+ if scale_y:
1052
+ ytr = _rescale(ytr, new_scale, d)
1053
+ yte = _rescale(yte, new_scale, d)
1054
+
1055
+ # training fit
1056
+ Xtr = smapi.add_constant(
1057
+ pd.DataFrame(Xtr_m), has_constant="add")
1058
+ fit = smapi.OLS(ytr, Xtr).fit()
1059
+ coefs = np.asarray(fit.params, dtype=float)
1060
+ tr["n"].append(int(train.sum()))
1061
+ tr["MSE"].append(float(fit.scale)) # residual mean sq
1062
+ tr["se"].append(math.sqrt(float(fit.scale)))
1063
+ tr["Rsq"].append(float(fit.rsquared))
1064
+
1065
+ # apply the training model to the held-out test fold
1066
+ n_te = int(test.sum())
1067
+ Xte = np.column_stack([np.ones(n_te), Xte_m])
1068
+ if Xte.shape[1] != coefs.shape[0]:
1069
+ raise ValueError(
1070
+ "a fold has more variables than estimated "
1071
+ "coefficients: at least one variable has no "
1072
+ "variation in a test fold")
1073
+ sse = float(((yte - Xte @ coefs) ** 2).sum())
1074
+ denom = n_te - coefs.shape[0]
1075
+ mse = sse / denom if denom > 0 else math.nan
1076
+ ssy = float(((yte - yte.mean()) ** 2).sum())
1077
+ te["n"].append(n_te)
1078
+ te["MSE"].append(mse)
1079
+ te["se"].append(math.sqrt(mse) if denom > 0 else math.nan)
1080
+ te["Rsq"].append(1 - sse / ssy if ssy > 0 else math.nan)
1081
+
1082
+ cv = pd.DataFrame({
1083
+ "fold": range(1, kfold + 1),
1084
+ "train_n": tr["n"], "train_se": tr["se"],
1085
+ "train_MSE": tr["MSE"], "train_Rsq": tr["Rsq"],
1086
+ "test_n": te["n"], "test_se": te["se"],
1087
+ "test_MSE": te["MSE"], "test_Rsq": te["Rsq"]})
1088
+ means = {"train_se": float(np.mean(tr["se"])),
1089
+ "train_MSE": float(np.mean(tr["MSE"])),
1090
+ "train_Rsq": float(np.mean(tr["Rsq"])),
1091
+ "test_se": float(np.mean(te["se"])),
1092
+ "test_MSE": float(np.mean(te["MSE"])),
1093
+ "test_Rsq": float(np.mean(te["Rsq"]))}
1094
+
1095
+ print("\n".join(_kfold_lines(cv, means, kfold, d)))
1096
+ return RegressionResults(
1097
+ formula=formula, n_keep=n_keep, kfold=kfold,
1098
+ digits_d=d, cv=cv, cv_means=means)
1099
+
1100
+
1101
+ def _kfold_lines(cv, means, kfold, d):
1102
+ """The cross-validation table: a training block (se, MSE, Rsq)
1103
+ and a testing block (sp, MSE, Rsq), then the column means.
1104
+ ~ .regKfold display."""
1105
+ def w(col): # width of a numeric column
1106
+ vals = [v for v in cv[col] if not math.isnan(v)]
1107
+ return max(6, max((len(fmt(v, d)) for v in vals),
1108
+ default=6))
1109
+
1110
+ def num(v):
1111
+ return "NA" if math.isnan(v) else fmt(v, d)
1112
+
1113
+ ws = {c: w(c) for c in ("train_se", "train_MSE", "train_Rsq",
1114
+ "test_se", "test_MSE", "test_Rsq")}
1115
+ wn = max(3, max(len(str(n)) for n in cv["train_n"]))
1116
+ wk = max(3, max(len(str(n)) for n in cv["test_n"]))
1117
+ gap = " "
1118
+
1119
+ def block_w(pre):
1120
+ return (ws[pre + "_se"] + ws[pre + "_MSE"]
1121
+ + ws[pre + "_Rsq"] + 2 * len(gap))
1122
+ tw, kw = block_w("train"), block_w("test")
1123
+
1124
+ lines = [f"\n {kfold}-FOLD CROSS-VALIDATION", ""]
1125
+ # group titles, centered over each block (past the n column)
1126
+ tt, kt = "Model from Training Data", "Applied to Testing Data"
1127
+ lead = 5 + 1 + wn # "fold " + "|" area + n
1128
+ lines.append(" " * lead + tt.center(tw)
1129
+ + " " * (3 + wk) + kt.center(kw))
1130
+ lines.append(" " * lead + "-" * tw
1131
+ + " " * (3 + wk) + "-" * kw)
1132
+ lines.append(
1133
+ "fold" + " " + "n".rjust(wn)
1134
+ + gap + "se".rjust(ws["train_se"])
1135
+ + gap + "MSE".rjust(ws["train_MSE"])
1136
+ + gap + "Rsq".rjust(ws["train_Rsq"])
1137
+ + " " + "n".rjust(wk)
1138
+ + gap + "sp".rjust(ws["test_se"])
1139
+ + gap + "MSE".rjust(ws["test_MSE"])
1140
+ + gap + "Rsq".rjust(ws["test_Rsq"]))
1141
+ for _, r in cv.iterrows():
1142
+ lines.append(
1143
+ f"{int(r['fold']):>3} |" + " "
1144
+ + str(int(r["train_n"])).rjust(wn)
1145
+ + gap + num(r["train_se"]).rjust(ws["train_se"])
1146
+ + gap + num(r["train_MSE"]).rjust(ws["train_MSE"])
1147
+ + gap + num(r["train_Rsq"]).rjust(ws["train_Rsq"])
1148
+ + " " + str(int(r["test_n"])).rjust(wk)
1149
+ + gap + num(r["test_se"]).rjust(ws["test_se"])
1150
+ + gap + num(r["test_MSE"]).rjust(ws["test_MSE"])
1151
+ + gap + num(r["test_Rsq"]).rjust(ws["test_Rsq"]))
1152
+ lines.append(" " * lead + "-" * tw
1153
+ + " " * (3 + wk) + "-" * kw)
1154
+ lines.append(
1155
+ "Mean" + " " + " " * wn
1156
+ + gap + num(means["train_se"]).rjust(ws["train_se"])
1157
+ + gap + num(means["train_MSE"]).rjust(ws["train_MSE"])
1158
+ + gap + num(means["train_Rsq"]).rjust(ws["train_Rsq"])
1159
+ + " " + " " * wk
1160
+ + gap + num(means["test_se"]).rjust(ws["test_se"])
1161
+ + gap + num(means["test_MSE"]).rjust(ws["test_MSE"])
1162
+ + gap + num(means["test_Rsq"]).rjust(ws["test_Rsq"]))
1163
+ return lines
1164
+
1165
+
1166
+ def _mod_slopes(fit, mod_info, wv):
1167
+ """The simple-slope intercept and slope of the response on the
1168
+ focal predictor at the moderator W = wc, from the fitted
1169
+ Y = b0 + bx X + bw W + bxw X*W: at fixed W, intercept
1170
+ b0 + bw*wc and slope bx + bxw*wc. Returns the coefficients and
1171
+ the three (label, wc) levels, mean and +/-1 SD."""
1172
+ b0 = float(fit.params["const"])
1173
+ bx = float(fit.params[mod_info["x"]])
1174
+ bw = float(fit.params[mod_info["w"]])
1175
+ bxw = float(fit.params[mod_info["xw"]])
1176
+ m_w, s_w = float(wv.mean()), float(wv.std(ddof=1))
1177
+ levels = [("+1SD", m_w + s_w), ("Mean", m_w),
1178
+ ("-1SD", m_w - s_w)]
1179
+ return b0, bx, bw, bxw, m_w, s_w, levels
1180
+
1181
+
1182
+ def _moderation_lines(fit, Xd, mod_info, d):
1183
+ """The moderation text: the moderator's mean and SD, then the
1184
+ simple-slope intercept b0 and slope b1 at mean and +/-1 SD.
1185
+ ~ .reg6mod out_mod"""
1186
+ w_name = mod_info["w"]
1187
+ b0, bx, bw, bxw, m_w, s_w, levels = _mod_slopes(
1188
+ fit, mod_info, Xd[w_name].to_numpy(dtype=float))
1189
+ out = ["", "", " MODERATION ANALYSIS", "",
1190
+ f"Mean of {w_name}: {fmt(m_w, d)}",
1191
+ f"SD of {w_name}: {fmt(s_w, d)}", ""]
1192
+ for tag, wc in (("mean+1SD", levels[0][1]),
1193
+ ("mean ", levels[1][1]),
1194
+ ("mean-1SD", levels[2][1])):
1195
+ out.append(f"{tag} for {w_name}: "
1196
+ f"b0={fmt(b0 + bw * wc, d)} "
1197
+ f"b1={fmt(bx + bxw * wc, d)}")
1198
+ return out
1199
+
1200
+
1201
+ def _moderation_plot(fit, Xd, y_name, mod_info, d):
1202
+ """Interaction plot: the simple regression line of the
1203
+ response on the focal predictor at the moderator's mean and
1204
+ +/-1 SD, one line each. ~ .reg6mod plot"""
1205
+ x_name, w_name = mod_info["x"], mod_info["w"]
1206
+ b0, bx, bw, bxw, _, _, levels = _mod_slopes(
1207
+ fit, mod_info, Xd[w_name].to_numpy(dtype=float))
1208
+ xv = Xd[x_name].to_numpy(dtype=float)
1209
+ xs = np.array([float(xv.min()), float(xv.max())])
1210
+ style_opts = plotly_style()
1211
+ colors = {"+1SD": to_hex(BASE_COLORS[0]),
1212
+ "Mean": to_hex("gray20"),
1213
+ "-1SD": to_hex(BASE_COLORS[1])}
1214
+ widths = {"+1SD": 2, "Mean": 1.3, "-1SD": 2}
1215
+ fig = go.Figure()
1216
+ ys_all = []
1217
+ for lbl, wc in levels:
1218
+ ys = (b0 + bw * wc) + (bx + bxw * wc) * xs
1219
+ ys_all += list(ys)
1220
+ fig.add_trace(go.Scatter(
1221
+ x=xs, y=ys, mode="lines", name=lbl,
1222
+ line=dict(color=colors[lbl], width=widths[lbl]),
1223
+ hoverinfo="name+x+y"))
1224
+ axT1 = pretty(float(xs[0]), float(xs[1]))
1225
+ axT2 = pretty(min(ys_all), max(ys_all))
1226
+ ax_x = axis_num(x_name, axT1, axis_format(axT1, d))
1227
+ ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
1228
+ ax_y.update(showgrid=True,
1229
+ gridcolor=to_hex(style_opts["grid_col"]),
1230
+ gridwidth=1, griddash="dot")
1231
+ fig.update_layout(
1232
+ xaxis=ax_x, yaxis=ax_y,
1233
+ shapes=x_grid(axT1) + plot_border(), template=None,
1234
+ plot_bgcolor=to_hex(style_opts["panel_fill"]),
1235
+ paper_bgcolor=to_hex(style_opts["window_fill"]),
1236
+ legend=dict(title=dict(text=w_name)),
1237
+ title=dict(text="Moderator Variable Interaction Plot",
1238
+ x=0.5, xanchor="center",
1239
+ font=dict(size=round(
1240
+ 16 * get_option("main_size", 1)))))
1241
+ return fig
1242
+
1243
+
1244
+ def _reg_ancova_models(fit, Xd, yv, y_name, info, d, smapi):
1245
+ """The ANCOVA group models text: the test of the covariate x
1246
+ factor interaction (the parallel-lines assumption), the
1247
+ per-level equations of the no-interaction model, and the
1248
+ Plot suggestion. ~ .reg5ancova txmdl"""
1249
+ cov, fac = info["cov"], info["fac"]
1250
+ p1, p2 = info["preds"]
1251
+ ind_cols = [c for c in Xd.columns if c != cov]
1252
+
1253
+ # interaction model: add covariate x indicator columns, and
1254
+ # test them jointly against the no-interaction model (fit).
1255
+ # The sequential interaction SS = SSE(reduced) - SSE(full),
1256
+ # over the full model's residual mean square. ~ anova(lm(
1257
+ # y ~ cov * factor))[interaction row]
1258
+ inter = pd.DataFrame(
1259
+ {f"{cov}:{c}": Xd[cov].to_numpy() * Xd[c].to_numpy()
1260
+ for c in ind_cols}, index=Xd.index)
1261
+ Xf = pd.concat(
1262
+ [smapi.add_constant(Xd, has_constant="add"), inter],
1263
+ axis=1)
1264
+ full = smapi.OLS(yv, Xf).fit()
1265
+ df_int = len(ind_cols)
1266
+ df_res = int(full.df_resid)
1267
+ ss_int = float(fit.ssr) - float(full.ssr)
1268
+ F = (ss_int / df_int) / (float(full.ssr) / df_res)
1269
+ p = float(sps.f.sf(F, df_int, df_res))
1270
+
1271
+ b0 = float(fit.params["const"])
1272
+ b_slope = float(fit.params[cov])
1273
+ out = ["", "",
1274
+ f" MODELS OF {y_name} FOR LEVELS OF {fac}", "",
1275
+ "-- Test of Interaction", "",
1276
+ f"{p1}:{p2} df: {df_int} df resid: {df_res} "
1277
+ f"SS: {fmt(ss_int, 3)} F: {fmt(F, 3)} "
1278
+ f"p-value: {fmt(p, 3)}", "",
1279
+ "-- Assume parallel lines, no interaction of "
1280
+ f"{fac} with {cov}", ""]
1281
+ for i, lv in enumerate(info["levels"]):
1282
+ eff = 0.0 if i == 0 else float(
1283
+ fit.params.get(f"{fac}{lv}", 0.0))
1284
+ out.append(
1285
+ f"Level {lv}: y^_{y_name} = {fmt(b0 + eff, d)} + "
1286
+ f"{fmt(b_slope, d)}(x_{cov})")
1287
+ out += ["",
1288
+ "-- Visualize Separately Computed Regression Lines",
1289
+ "",
1290
+ f'XY("{cov}", "{y_name}", data=d, by="{fac}", '
1291
+ 'fit="lm")']
1292
+ return out
1293
+
1294
+
1295
+ def _reg_ancova_plot(fit, Xd, yv, y_name, info, d):
1296
+ """Grouped scatterplot with one parallel least-squares line
1297
+ per factor level: shared slope on the covariate, a separate
1298
+ intercept per level (the reference plus its group effect).
1299
+ ~ .reg5ancova plot"""
1300
+ cov, fac, levels = info["cov"], info["fac"], info["levels"]
1301
+ fac_vals = info["fac_vals"]
1302
+ xv = Xd[cov].to_numpy(dtype=float)
1303
+ b0 = float(fit.params["const"])
1304
+ b_slope = float(fit.params[cov])
1305
+ xs = np.array([float(xv.min()), float(xv.max())])
1306
+ style_opts = plotly_style()
1307
+ fig = go.Figure()
1308
+ ys_all = list(yv)
1309
+ for i, lv in enumerate(levels):
1310
+ col = to_hex(BASE_COLORS[i % len(BASE_COLORS)])
1311
+ m = fac_vals == lv
1312
+ fig.add_trace(go.Scatter(
1313
+ x=xv[m], y=yv[m], mode="markers", name=lv,
1314
+ legendgroup=lv,
1315
+ marker=dict(symbol="circle", size=7,
1316
+ color=make_trans(col, 0.9), opacity=1,
1317
+ line=dict(color=col, width=1)),
1318
+ hoverinfo="x+y+name"))
1319
+ eff = 0.0 if i == 0 else float(
1320
+ fit.params.get(f"{fac}{lv}", 0.0))
1321
+ ys = (b0 + eff) + b_slope * xs
1322
+ ys_all += list(ys)
1323
+ fig.add_trace(go.Scatter(
1324
+ x=xs, y=ys, mode="lines", legendgroup=lv,
1325
+ line=dict(color=col, width=2),
1326
+ hoverinfo="skip", showlegend=False))
1327
+ axT1 = pretty(float(xs[0]), float(xs[1]))
1328
+ axT2 = pretty(min(ys_all), max(ys_all))
1329
+ ax_x = axis_num(cov, axT1, axis_format(axT1, d))
1330
+ ax_y = axis_num(y_name, axT2, axis_format(axT2, d))
1331
+ ax_y.update(showgrid=True,
1332
+ gridcolor=to_hex(style_opts["grid_col"]),
1333
+ gridwidth=1, griddash="dot")
1334
+ fig.update_layout(
1335
+ xaxis=ax_x, yaxis=ax_y,
1336
+ shapes=x_grid(axT1) + plot_border(), template=None,
1337
+ plot_bgcolor=to_hex(style_opts["panel_fill"]),
1338
+ paper_bgcolor=to_hex(style_opts["window_fill"]),
1339
+ legend=dict(title=dict(text=fac)),
1340
+ title=dict(text="Scatterplot and Least-Squares Lines",
1341
+ x=0.5, xanchor="center",
1342
+ font=dict(size=round(
1343
+ 16 * get_option("main_size", 1)))))
1344
+ return fig
1345
+
1346
+
1347
+ def _reg_scatter(xv, yv, x_name, y_name, fit, smapi, digits_d,
1348
+ with_bands=True):
1349
+ """Scatterplot with the least-squares line and, by default,
1350
+ the 95% confidence and prediction bands. ~ .reg5Plot"""
1351
+ od = np.argsort(xv, kind="stable")
1352
+ xs = xv[od]
1353
+ Xl = smapi.add_constant(
1354
+ pd.DataFrame({x_name: xs}), has_constant="add")
1355
+ pr = fit.get_prediction(Xl)
1356
+ sf = pr.summary_frame(alpha=0.05)
1357
+ style_opts = plotly_style()
1358
+ fig = go.Figure()
1359
+ if with_bands:
1360
+ fig.add_trace(go.Scatter( # prediction band
1361
+ x=np.concatenate([xs, xs[::-1]]),
1362
+ y=np.concatenate(
1363
+ [sf["obs_ci_upper"].to_numpy(),
1364
+ sf["obs_ci_lower"].to_numpy()[::-1]]),
1365
+ mode="none", fill="toself",
1366
+ fillcolor=as_plotly_color(
1367
+ get_option("se_fill", "#1A1A1A19")),
1368
+ hoverinfo="skip", showlegend=False))
1369
+ fig.add_trace(go.Scatter( # confidence band
1370
+ x=np.concatenate([xs, xs[::-1]]),
1371
+ y=np.concatenate(
1372
+ [sf["mean_ci_upper"].to_numpy(),
1373
+ sf["mean_ci_lower"].to_numpy()[::-1]]),
1374
+ mode="none", fill="toself",
1375
+ fillcolor=as_plotly_color(
1376
+ get_option("se_fill", "#1A1A1A19")),
1377
+ hoverinfo="skip", showlegend=False))
1378
+ pt_fill = get_option("pt_color", "#324E5C")
1379
+ fig.add_trace(go.Scatter(
1380
+ x=xv, y=yv, mode="markers",
1381
+ marker=dict(symbol="circle", size=7.25,
1382
+ sizemode="diameter",
1383
+ color=make_trans(pt_fill, 0.9),
1384
+ opacity=1,
1385
+ line=dict(color=to_hex(pt_fill), width=1)),
1386
+ hoverinfo="x+y", showlegend=False))
1387
+ fig.add_trace(go.Scatter(
1388
+ x=xs, y=sf["mean"].to_numpy(), mode="lines",
1389
+ line=dict(color=to_hex(get_option("fit_color",
1390
+ "#5C4032")),
1391
+ width=get_option("fit_lwd", 2)),
1392
+ hoverinfo="skip", showlegend=False))
1393
+ axT1 = pretty(float(xv.min()), float(xv.max()))
1394
+ ylo = min(float(yv.min()),
1395
+ float(sf["obs_ci_lower"].min())
1396
+ if with_bands else float(yv.min()))
1397
+ yhi = max(float(yv.max()),
1398
+ float(sf["obs_ci_upper"].max())
1399
+ if with_bands else float(yv.max()))
1400
+ axT2 = pretty(ylo, yhi)
1401
+ ax_x = axis_num(x_name, axT1,
1402
+ axis_format(axT1, digits_d))
1403
+ ax_y = axis_num(y_name, axT2,
1404
+ axis_format(axT2, digits_d))
1405
+ ax_y.update(showgrid=True,
1406
+ gridcolor=to_hex(style_opts["grid_col"]),
1407
+ gridwidth=1, griddash="dot")
1408
+ title = ("Reg Line, Confidence & Prediction Intervals"
1409
+ if with_bands
1410
+ else "Scatterplot and Least-Squares Line")
1411
+ fig.update_layout(
1412
+ xaxis=ax_x, yaxis=ax_y,
1413
+ shapes=x_grid(axT1) + plot_border(), template=None,
1414
+ plot_bgcolor=to_hex(style_opts["panel_fill"]),
1415
+ paper_bgcolor=to_hex(style_opts["window_fill"]),
1416
+ title=dict(text=title, x=0.5, xanchor="center",
1417
+ font=dict(size=round(
1418
+ 16 * get_option("main_size", 1)))))
1419
+ return fig
1420
+
1421
+
1422
+ def _reg_resfit(fitted, resid, cooks, cooks_cut, labels,
1423
+ digits_d):
1424
+ """Residuals vs fitted values, zero line, points at or
1425
+ above cooks_cut labeled (or the single largest when none
1426
+ reach it). ~ .reg3resfitResidual"""
1427
+ max_cook = float(np.nanmax(cooks))
1428
+ if max_cook < cooks_cut:
1429
+ cut = math.floor(max_cook * 100) / 100
1430
+ sub = ("Point with largest Cook's Distance of "
1431
+ f"{fmt(max_cook, 2)} is labeled")
1432
+ else:
1433
+ cut = cooks_cut
1434
+ sub = (f"Points with Cook's Distance > {cooks_cut} "
1435
+ "are labeled")
1436
+ style_opts = plotly_style()
1437
+ pt_fill = get_option("pt_color", "#324E5C")
1438
+ fig = go.Figure()
1439
+ fig.add_trace(go.Scatter(
1440
+ x=fitted, y=resid, mode="markers",
1441
+ marker=dict(symbol="circle", size=5,
1442
+ sizemode="diameter",
1443
+ color=make_trans(pt_fill, 0.9),
1444
+ opacity=1,
1445
+ line=dict(color=to_hex(pt_fill), width=0.5)),
1446
+ hoverinfo="x+y", showlegend=False))
1447
+ flag = np.where(cooks >= cut)[0]
1448
+ for i in flag:
1449
+ fig.add_annotation(
1450
+ x=float(fitted[i]), y=float(resid[i]),
1451
+ text=str(labels[i]), showarrow=False, yshift=12,
1452
+ font=dict(size=11, color=to_hex("gray30")))
1453
+ axT1 = pretty(float(fitted.min()), float(fitted.max()))
1454
+ axT2 = pretty(float(resid.min()), float(resid.max()))
1455
+ ax_x = axis_num("Fitted Values", axT1,
1456
+ axis_format(axT1, digits_d))
1457
+ ax_y = axis_num("Residuals", axT2,
1458
+ axis_format(axT2, digits_d))
1459
+ ax_y.update(showgrid=True,
1460
+ gridcolor=to_hex(style_opts["grid_col"]),
1461
+ gridwidth=1, griddash="dot")
1462
+ shapes = x_grid(axT1) + plot_border()
1463
+ shapes.append(dict( # zero residual line
1464
+ type="line", xref="paper", yref="y",
1465
+ x0=0, x1=1, y0=0, y1=0,
1466
+ line=dict(color=to_hex("gray50"), width=1,
1467
+ dash="dash")))
1468
+ fig.update_layout(
1469
+ xaxis=ax_x, yaxis=ax_y, shapes=shapes, template=None,
1470
+ plot_bgcolor=to_hex(style_opts["panel_fill"]),
1471
+ paper_bgcolor=to_hex(style_opts["window_fill"]),
1472
+ title=dict(
1473
+ text=("Residuals vs Fitted Values<br>"
1474
+ f"<sup>{sub}</sup>"),
1475
+ x=0.5, xanchor="center",
1476
+ font=dict(size=round(
1477
+ 16 * get_option("main_size", 1)))))
1478
+ return fig
1479
+
1480
+
1481
+ def _reg_dn_residual(resid):
1482
+ """Distribution of the residuals: the density display with
1483
+ its histogram backdrop. ~ .reg3dnResidual"""
1484
+ edges = _breaks_from_args(np.asarray(resid, dtype=float),
1485
+ None, None, None, "Sturges")
1486
+ fig = dn_plotly(np.asarray(resid, dtype=float),
1487
+ x_name="Residuals",
1488
+ x_lab="Residuals",
1489
+ show_histogram=True, hist_edges=edges,
1490
+ main="Distribution of Residuals")
1491
+ return fig