lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Logit.py ADDED
@@ -0,0 +1,615 @@
1
+ # Logit.py — analog of Logit.R (core numeric predictors)
2
+ #
3
+ # Logit(): logistic regression with the lessR output pipeline —
4
+ # the estimated model on the logit scale with Wald confidence
5
+ # intervals, odds ratios, model fit (deviances, AIC),
6
+ # collinearity via the auxiliary linear model, the residuals and
7
+ # influence listing (with R's glm formulas for the studentized
8
+ # residual, dffits, and Cook's distance), the classification
9
+ # table sorted by fitted probability, confusion matrices per
10
+ # prob_cut threshold with accuracy/sensitivity/precision, and
11
+ # the fitted-sigmoid plot for a single predictor. As elsewhere,
12
+ # the pipeline is ported, not the lines.
13
+ #
14
+ # The model is a formula string, "Y ~ X1 + X2", as Regression().
15
+ # The response is numeric 0/1 or a two-level categorical
16
+ # (second level = the reference group predicted as 1, re-ordered
17
+ # by ref_group=). Categorical predictors become treatment-coded
18
+ # indicator variables (VarLevel columns, ~ model.matrix), with
19
+ # R's ">>> Note" announcement. Expression terms such as
20
+ # log(Years) or I(Years^2) are materialized as columns (shared
21
+ # _parse_formula, ~ .formula_expr). A multiple logit model draws
22
+ # the symmetric scatterplot matrix with loess smooths
23
+ # (plt_mat_plotly, ~ logit.4Pred). Rmd= writes a Quarto (.qmd)
24
+ # classification report (logit_rmd.py) — designed on the
25
+ # Regression report, as R's Logit has no Rmd. quiet= does not
26
+ # exist; brief= trims residuals and prediction, as R.
27
+ #
28
+ # Numerics through statsmodels GLM (Binomial), imported lazily.
29
+ # Returns a LogitResults object; figures in .plots are not
30
+ # auto-shown.
31
+
32
+ import math
33
+
34
+ import numpy as np
35
+ import pandas as pd
36
+ import plotly.graph_objects as go
37
+
38
+ from .plotly_utils import (
39
+ axis_format, axis_num, make_trans, plot_border,
40
+ plotly_style, to_hex, x_grid)
41
+ from .plt_mat_plotly import scatter_matrix
42
+ from .logit_rmd import logit_rmd
43
+ from .Regression import (
44
+ _expand_indicators, _parse_formula, _prntbl)
45
+ from .utils import fmt, get_column, get_option, pretty
46
+
47
+
48
+ class LogitResults:
49
+ """Numeric results and figures of Logit(): estimates,
50
+ odds_ratios, fit, residuals and predictions listings,
51
+ confusion matrices, and the plotly figures in .plots."""
52
+
53
+ def __init__(self, **kw):
54
+ self.__dict__.update(kw)
55
+
56
+ def __repr__(self):
57
+ return (f"<lessPy Logit: {self.formula}, "
58
+ f"n={self.n_keep}>")
59
+
60
+
61
+ def _glm_influence(y01, mu, hat, k_params):
62
+ """R's glm influence measures from the hat values and the
63
+ Pearson residuals, dispersion 1 (binomial): rstudent, as R
64
+ returns it for a glm (the standardized Pearson residual,
65
+ pearson / sqrt(1 - hat) — verified against rstudent() on
66
+ this model), dffits = rstudent * sqrt(hat / (1 - hat)),
67
+ and Cook's distance. R analogs: rstudent(), dffits(),
68
+ cooks.distance.glm()"""
69
+ with np.errstate(divide="ignore", invalid="ignore"):
70
+ pear = (y01 - mu) / np.sqrt(mu * (1 - mu))
71
+ rstud = pear / np.sqrt(1 - hat)
72
+ dffits = rstud * np.sqrt(hat / (1 - hat))
73
+ cooks = (pear / (1 - hat)) ** 2 * hat / k_params
74
+ return rstud, dffits, cooks
75
+
76
+
77
+ def Logit(my_formula, data=None, filter=None, ref_group=None,
78
+ digits_d=4, brief=False,
79
+ res_rows=None, res_sort="cooks",
80
+ pred=True, pred_all=False, prob_cut=0.5, cooks_cut=1,
81
+ X1_new=None, X2_new=None, X3_new=None,
82
+ X4_new=None, X5_new=None, X6_new=None,
83
+ pt_size=0.9, transparency=0.8,
84
+ Rmd=None, Rmd_data=None, Rmd_format="html",
85
+ Rmd_browser=True,
86
+ results=True, explain=True, interpret=True, code=True,
87
+ xlab=None, ylab=None, graphics=True):
88
+ """Logistic regression of a formula string, "Y ~ X1 + X2",
89
+ with the lessR analysis pipeline: estimates and odds ratios,
90
+ fit, collinearity, residuals and influence, classification
91
+ with confusion matrices, and the fitted-sigmoid plot. Always
92
+ prints, as in R; returns a LogitResults object with the
93
+ figures in .plots."""
94
+ import statsmodels.api as smapi
95
+
96
+ if data is None:
97
+ raise ValueError(
98
+ "data= is required: a pandas DataFrame containing "
99
+ "the model's variables")
100
+ if res_sort not in ("cooks", "rstudent", "dffits", "off"):
101
+ raise ValueError(
102
+ 'res_sort: "cooks", "rstudent", "dffits", or "off"')
103
+ if Rmd is not None:
104
+ if Rmd_format not in ("html", "pdf", "docx", "word",
105
+ "none"):
106
+ raise ValueError('Rmd_format: "html", "pdf", '
107
+ '"docx", or "none"')
108
+ if brief:
109
+ raise ValueError(
110
+ "a Quarto report needs the full analysis, so "
111
+ "Rmd= is not available with brief=True")
112
+ if filter is not None:
113
+ data = data.query(filter)
114
+ if brief:
115
+ if res_rows is None:
116
+ res_rows = 0
117
+ pred = False
118
+
119
+ y_name, pred_names, data = _parse_formula(my_formula, data)
120
+ formula = (f"{y_name} ~ "
121
+ + (" + ".join(pred_names) if pred_names else "1"))
122
+ n_pred = len(pred_names)
123
+ if n_pred == 0:
124
+ raise ValueError("Logit() requires at least one "
125
+ "predictor")
126
+
127
+ y_ser = get_column(data, y_name, "response")
128
+ pred_sers = [get_column(data, nm, "predictor")
129
+ for nm in pred_names]
130
+
131
+ used = pd.concat([y_ser] + pred_sers, axis=1)
132
+ keep = ~used.isna().any(axis=1)
133
+ n_obs = len(data)
134
+ n_keep = int(keep.sum())
135
+ yk = y_ser[keep]
136
+
137
+ # categorical predictors become indicator variables, as R
138
+ (pred_names, pred_sers_x, ind_notes, _cat_names,
139
+ _term_map) = _expand_indicators(
140
+ pred_names, [s[keep] for s in pred_sers],
141
+ ">>> Note: {0} is not a numeric variable.\n"
142
+ " Indicator variables are created and "
143
+ "analyzed.")
144
+ n_pred = len(pred_names)
145
+ Xd = pd.DataFrame(
146
+ {nm: s.to_numpy(dtype=float)
147
+ for nm, s in zip(pred_names, pred_sers_x)},
148
+ index=yk.index)
149
+
150
+ # response: numeric 0/1, or two-level categorical with the
151
+ # second level the reference group (predicted as 1)
152
+ y_is_factor = not pd.api.types.is_numeric_dtype(yk)
153
+ if y_is_factor:
154
+ if isinstance(yk.dtype, pd.CategoricalDtype):
155
+ levels = [lv for lv in yk.cat.categories
156
+ if lv in set(yk)]
157
+ else:
158
+ levels = sorted(yk.astype(str).unique())
159
+ if len(levels) != 2:
160
+ raise ValueError(
161
+ f"Response variable: {y_name}\n"
162
+ "If numeric, can only have values of 0 or 1.\n"
163
+ "If a factor, can only have two levels.")
164
+ if ref_group is not None:
165
+ if ref_group not in levels:
166
+ raise ValueError(
167
+ f"Values of response {y_name}: "
168
+ f"{levels[0]} {levels[1]}\n"
169
+ "You specified a non-existent value, "
170
+ f"ref_group = {ref_group}")
171
+ if levels[1] != ref_group:
172
+ levels = [levels[1], levels[0]]
173
+ # single numeric predictor: order the levels so the
174
+ # slope is positive, as R
175
+ if n_pred == 1:
176
+ avg = Xd.iloc[:, 0].groupby(
177
+ yk.astype(str).to_numpy()).mean()
178
+ if avg[str(levels[0])] > avg[str(levels[1])]:
179
+ levels = [levels[1], levels[0]]
180
+ y01 = (yk.astype(str) ==
181
+ str(levels[1])).to_numpy(dtype=float)
182
+ else:
183
+ if ref_group is not None:
184
+ raise ValueError(
185
+ "Parameter ref_group only applies when the "
186
+ "response is a factor.")
187
+ vals = set(yk.unique())
188
+ if not vals <= {0, 1}:
189
+ raise ValueError(
190
+ f"Response variable: {y_name}\n"
191
+ "If numeric, can only have values of 0 or 1.\n"
192
+ "If a factor, can only have two levels.")
193
+ levels = None
194
+ y01 = yk.to_numpy(dtype=float)
195
+
196
+ new_data = X1_new is not None
197
+ row_labels = Xd.index.astype(str)
198
+ d = digits_d
199
+
200
+ X = smapi.add_constant(Xd, has_constant="add")
201
+ glm = smapi.GLM(y01, X,
202
+ family=smapi.families.Binomial()).fit(
203
+ tol=1e-12) # match R glm precision
204
+ if glm.params.isna().any():
205
+ bad = ", ".join(glm.params.index[glm.params.isna()])
206
+ raise ValueError(
207
+ "Variable redundant with a prior predictor in the "
208
+ f"model: {bad}")
209
+ mu = np.asarray(glm.fittedvalues, dtype=float)
210
+ hat = glm.get_influence().hat_matrix_diag
211
+ rstud, dffits, cooks = _glm_influence(
212
+ y01, mu, hat, len(glm.params))
213
+
214
+ lines = []
215
+ for note in ind_notes:
216
+ lines += [note, ""]
217
+
218
+ # ---------- variables and cases ----------------------------
219
+ for i, nm in enumerate([y_name] + pred_names):
220
+ if i == 0:
221
+ lbl = "Response Variable: "
222
+ elif n_pred > 1:
223
+ lbl = f"Predictor Variable {i}: "
224
+ else:
225
+ lbl = "Predictor Variable: "
226
+ lines.append(lbl + nm)
227
+ lines += ["",
228
+ f"Number of cases (rows) of data: {n_obs}",
229
+ f"Number of cases retained for analysis: "
230
+ f"{n_keep}"]
231
+
232
+ # ---------- BASIC ANALYSIS ---------------------------------
233
+ lines += ["", "", " BASIC ANALYSIS", "",
234
+ f"-- Estimated Model of {y_name} for the Logit "
235
+ "of Reference Group Membership", ""]
236
+ ci = glm.conf_int(alpha=0.05) # Wald, ~ confint.default
237
+ est = pd.DataFrame({
238
+ "Estimate": glm.params,
239
+ "Std Err": glm.bse,
240
+ "z-value": glm.tvalues,
241
+ "p-value": glm.pvalues,
242
+ "Lower 95%": ci[0],
243
+ "Upper 95%": ci[1],
244
+ })
245
+ est.index = ["(Intercept)"] + pred_names
246
+ buf = max(len(s) for s in est.index)
247
+ w = [max(9, max(len(fmt(v, d)) for v in est[c]) + 1)
248
+ for c in est.columns]
249
+ lines.append(" " * buf
250
+ + f"{'Estimate':>{w[0] + 1}}"
251
+ + f"{'Std Err':>{w[1] + 2}}"
252
+ + f"{'z-value':>9}{'p-value':>9}"
253
+ + f"{'Lower 95%':>{w[4] + 3}}"
254
+ + f"{'Upper 95%':>{w[5] + 3}}")
255
+ for lbl, r in est.iterrows():
256
+ lines.append(
257
+ f"{lbl:<{buf}}"
258
+ + f"{fmt(r['Estimate'], d):>{w[0] + 1}}"
259
+ + f"{fmt(r['Std Err'], d):>{w[1] + 2}}"
260
+ + f"{fmt(r['z-value'], 3):>9}"
261
+ + f"{fmt(r['p-value'], 3):>9}"
262
+ + f"{fmt(r['Lower 95%'], d):>{w[4] + 3}}"
263
+ + f"{fmt(r['Upper 95%'], d):>{w[5] + 3}}")
264
+
265
+ # odds ratios and 95% CI
266
+ orci = pd.DataFrame({
267
+ "Odds Ratio": np.exp(est["Estimate"]),
268
+ "Lower 95%": np.exp(est["Lower 95%"]),
269
+ "Upper 95%": np.exp(est["Upper 95%"]),
270
+ }, index=est.index)
271
+ lines += ["", "",
272
+ "-- Odds Ratios and Confidence Intervals", ""]
273
+ wo = [max(10, max(len(fmt(v, d)) for v in orci[c]) + 1)
274
+ for c in orci.columns]
275
+ lines.append(" " * (buf + 2)
276
+ + f"{'Odds Ratio':>{wo[0] + 1}}"
277
+ + f"{'Lower 95%':>{wo[1] + 3}}"
278
+ + f"{'Upper 95%':>{wo[2] + 3}}")
279
+ for lbl, r in orci.iterrows():
280
+ lines.append(
281
+ f"{lbl:<{buf}} "
282
+ + f"{fmt(r['Odds Ratio'], d):>{wo[0] + 1}}"
283
+ + f"{fmt(r['Lower 95%'], d):>{wo[1] + 3}}"
284
+ + f"{fmt(r['Upper 95%'], d):>{wo[2] + 3}}")
285
+
286
+ # model fit
287
+ n_iter = len(glm.fit_history["deviance"]) - 1
288
+ lines += ["", "", "-- Model Fit", "",
289
+ f" Null deviance: {fmt(glm.null_deviance, 3)}"
290
+ f" on {int(glm.df_resid + n_pred)} degrees of "
291
+ "freedom",
292
+ f"Residual deviance: {fmt(glm.deviance, 3)} on "
293
+ f"{int(glm.df_resid)} degrees of freedom", "",
294
+ f"AIC: {fmt(glm.aic, 3)}", "",
295
+ f"Number of iterations to convergence: {n_iter}"]
296
+
297
+ # collinearity: R computes tolerance/VIF from the auxiliary
298
+ # linear model of the (numeric) response on the predictors
299
+ tol = vif = None
300
+ if n_pred > 1:
301
+ ols = smapi.OLS(y01, X).fit()
302
+ MSW = ols.scale
303
+ vif = np.array([
304
+ (Xd[nm].var(ddof=1) * (n_keep - 1)
305
+ * ols.bse[nm] ** 2) / MSW
306
+ for nm in pred_names])
307
+ tol = 1 / vif
308
+ lines += ["", "", "Collinearity", ""]
309
+ c1 = max(len(s) for s in pred_names)
310
+ lines.append(" " * c1 + f"{'Tolerance':>11}"
311
+ + f"{'VIF':>9}")
312
+ for nm, t_i, v_i in zip(pred_names, tol, vif):
313
+ lines.append(f"{nm:<{c1}}{fmt(t_i, 3):>11}"
314
+ + f"{fmt(v_i, 3):>9}")
315
+
316
+ # ---------- ANALYSIS OF RESIDUALS AND INFLUENCE ------------
317
+ res_tbl = None
318
+ if res_rows is None:
319
+ res_rows = n_keep if n_keep < 20 else 20
320
+ if res_rows == "all":
321
+ res_rows = n_keep
322
+ res_rows = min(int(res_rows), n_keep)
323
+ if res_rows > 0:
324
+ res_tbl = pd.DataFrame(index=row_labels)
325
+ for nm in pred_names:
326
+ res_tbl[nm] = Xd[nm].to_numpy()
327
+ res_tbl[y_name] = yk.to_numpy()
328
+ res_tbl["P(Y=1)"] = mu
329
+ res_tbl["residual"] = y01 - mu
330
+ res_tbl["rstudent"] = rstud
331
+ res_tbl["dffits"] = dffits
332
+ res_tbl["cooks"] = cooks
333
+ if res_sort == "cooks":
334
+ res_tbl = res_tbl.sort_values("cooks",
335
+ ascending=False)
336
+ elif res_sort == "rstudent":
337
+ res_tbl = res_tbl.reindex(
338
+ res_tbl["rstudent"].abs().sort_values(
339
+ ascending=False).index)
340
+ elif res_sort == "dffits":
341
+ res_tbl = res_tbl.reindex(
342
+ res_tbl["dffits"].abs().sort_values(
343
+ ascending=False).index)
344
+ lines += ["", "",
345
+ " ANALYSIS OF RESIDUALS AND INFLUENCE", "",
346
+ "Data, Fitted, Residual, Standardized "
347
+ "Pearson Residual, Dffits, Cook's Distance"]
348
+ if res_sort == "cooks":
349
+ lines.append(" [sorted by Cook's Distance]")
350
+ elif res_sort == "rstudent":
351
+ lines.append(" [sorted by Standardized Pearson "
352
+ "Residual, ignoring + or - sign]")
353
+ elif res_sort == "dffits":
354
+ lines.append(" [sorted by dffits, ignoring + or"
355
+ " - sign]")
356
+ lines.append(f" [res_rows = {res_rows} out of "
357
+ f"{n_keep} cases (rows) of data]")
358
+ lines.append("-" * 68)
359
+ lines += _prntbl(res_tbl.head(res_rows), d).split("\n")
360
+ lines.append("-" * 68)
361
+
362
+ # ---------- PREDICTION -------------------------------------
363
+ pred_tbl = None
364
+ confusion = []
365
+ plots = {}
366
+ prob_cuts = [float(p) for p in np.atleast_1d(prob_cut)]
367
+ p_cut = prob_cuts[0] if len(prob_cuts) == 1 else 0.5
368
+ lv2_txt = str(levels[1]) if levels is not None else ""
369
+
370
+ if pred:
371
+ if new_data:
372
+ grids = [np.atleast_1d(g) for g in
373
+ (X1_new, X2_new, X3_new, X4_new,
374
+ X5_new, X6_new)[:n_pred]
375
+ if g is not None]
376
+ if len(grids) != n_pred:
377
+ raise ValueError(
378
+ "Specified new data values for one "
379
+ "predictor variable, so do for all.")
380
+ mesh = np.meshgrid(*grids, indexing="ij")
381
+ Xnew = pd.DataFrame(
382
+ {nm: m.ravel() for nm, m in
383
+ zip(pred_names, mesh)})
384
+ Xn = smapi.add_constant(Xnew, has_constant="add")
385
+ prd = glm.get_prediction(Xn)
386
+ fitv = np.asarray(prd.predicted)
387
+ sev = np.asarray(prd.se)
388
+ pred_tbl = Xnew.copy()
389
+ pred_tbl[y_name] = ""
390
+ pred_tbl.index = [""] * len(pred_tbl)
391
+ else:
392
+ prd = glm.get_prediction(X)
393
+ fitv = np.asarray(prd.predicted)
394
+ sev = np.asarray(prd.se)
395
+ pred_tbl = pd.DataFrame(index=row_labels)
396
+ for nm in pred_names:
397
+ pred_tbl[nm] = Xd[nm].to_numpy()
398
+ pred_tbl[y_name] = yk.to_numpy()
399
+ pred_tbl["label"] = (fitv >= p_cut).astype(int)
400
+ pred_tbl["fitted"] = fitv
401
+ pred_tbl["std.err"] = sev
402
+ pred_tbl = pred_tbl.sort_values("fitted")
403
+
404
+ lines += ["", "", " PREDICTION", "",
405
+ "Probability threshold for classification "
406
+ f"{lv2_txt}: {p_cut}", ""]
407
+ if levels is not None:
408
+ lines += [f" 0: {levels[0]}",
409
+ f" 1: {levels[1]}", ""]
410
+ lines += ["Data, Fitted Values, Standard Errors",
411
+ " [sorted by fitted value]"]
412
+ if n_keep > 50 and not pred_all and not new_data:
413
+ lines.append(" [pred_all=TRUE to see all "
414
+ "intervals displayed]")
415
+ lines.append("-" * 68)
416
+ body = _prntbl(pred_tbl, d).split("\n")
417
+ head_ln, rows_ln = body[0], body[1:]
418
+ if n_keep < 25 or pred_all or new_data:
419
+ lines += body
420
+ else:
421
+ fits_sorted = pred_tbl["fitted"].to_numpy()
422
+ i_mid = int(np.abs(0.5 - fits_sorted).argmin())
423
+ i_mid = min(max(i_mid, 2), n_keep - 3)
424
+ lines.append(head_ln)
425
+ lines += rows_ln[:4]
426
+ lines += ["", "... for the rows of data where "
427
+ "fitted is close to 0.5 ...", ""]
428
+ lines += rows_ln[i_mid - 2:i_mid + 3]
429
+ lines += ["", "... for the last 4 rows of sorted "
430
+ "data ...", ""]
431
+ lines += rows_ln[n_keep - 4:]
432
+ lines.append("-" * 68)
433
+
434
+ # confusion matrix per prob_cut threshold
435
+ if not new_data:
436
+ lines += ["", "", "-" * 28,
437
+ "Specified confusion matrices",
438
+ "-" * 28, ""]
439
+ for pc in prob_cuts:
440
+ confusion.append(_logit_confuse(
441
+ lines, y01, fitv, pc, y_name, lv2_txt,
442
+ glm.params, pred_names, d))
443
+ else:
444
+ lines += ["", "",
445
+ "With X1_new, etc., no confusion "
446
+ "matrix."]
447
+
448
+ # sigmoid plot for a single numeric predictor
449
+ if graphics and n_pred == 1 and not new_data:
450
+ plots["logit_fit"] = _logit_plot(
451
+ Xd.iloc[:, 0].to_numpy(), y01, mu, levels,
452
+ y_name, pred_names[0],
453
+ prob_cuts[0] if len(prob_cuts) == 1 else None,
454
+ glm.params, xlab, ylab, pt_size, transparency,
455
+ d)
456
+
457
+ # scatterplot matrix of the model variables for a
458
+ # multiple logit model, symmetric with loess smooths and
459
+ # no correlations, ~ logit.4Pred pairs(panel=smooth)
460
+ if graphics and n_pred > 1 and not new_data:
461
+ mat_df = pd.DataFrame(
462
+ {y_name: y01}, index=Xd.index).join(Xd)
463
+ plots["scatter_matrix"] = scatter_matrix(
464
+ mat_df, fit="loess", cor_coef=False, band=False,
465
+ digits_d=d)
466
+
467
+ print("\n".join(lines))
468
+
469
+ out = LogitResults(
470
+ formula=formula, n_obs=n_obs, n_keep=n_keep,
471
+ digits_d=d, levels=levels,
472
+ estimates=est, odds_ratios=orci,
473
+ fit={"null_deviance": glm.null_deviance,
474
+ "deviance": glm.deviance,
475
+ "df_null": int(glm.df_resid + n_pred),
476
+ "df_residual": int(glm.df_resid),
477
+ "aic": glm.aic, "iterations": n_iter},
478
+ tolerance=tol, vif=vif,
479
+ residuals=res_tbl, predictions=pred_tbl,
480
+ confusion=confusion, plots=plots)
481
+
482
+ if Rmd is not None:
483
+ logit_rmd(out, y_name, pred_names, formula,
484
+ list(data.columns), Rmd, Rmd_data, Rmd_format,
485
+ Rmd_browser, results, explain, interpret, code)
486
+
487
+ return out
488
+
489
+
490
+ def _logit_confuse(lines, y01, fitv, pc, y_name, lv2_txt,
491
+ params, pred_names, digits_d):
492
+ """One confusion matrix at threshold pc, with the accuracy,
493
+ sensitivity, and precision. Appends the printed block to
494
+ lines and returns the counts. R analog: .logit5Confuse()"""
495
+ label = (fitv >= pc).astype(int)
496
+ hit0 = int(((y01 == 0) & (label == 0)).sum())
497
+ mis0 = int(((y01 == 0) & (label == 1)).sum())
498
+ hit1 = int(((y01 == 1) & (label == 1)).sum())
499
+ mis1 = int(((y01 == 1) & (label == 0)).sum())
500
+ tot0, tot1 = hit0 + mis0, hit1 + mis1
501
+ totG = tot0 + tot1
502
+ per0 = hit0 / tot0 if tot0 else np.nan
503
+ per1 = hit1 / tot1 if tot1 else np.nan
504
+ perT = (hit0 + hit1) / totG
505
+
506
+ lines.append("Probability threshold for predicting "
507
+ f"{lv2_txt}: {pc}")
508
+ if len(pred_names) == 1:
509
+ x_cut = ((math.log(pc / (1 - pc)) - params.iloc[0])
510
+ / params.iloc[1])
511
+ lines.append("Corresponding cutoff threshold for "
512
+ f"{pred_names[0]}: {round(x_cut, 3)}")
513
+ lines.append("")
514
+ ln = len(y_name)
515
+ pad = " " * ln
516
+ lines.append(f"{pad} Baseline Predicted")
517
+ lines.append("-" * 51)
518
+ lines.append(f"{pad} Total %Tot 0 1"
519
+ " %Correct")
520
+ lines.append("-" * 51)
521
+ lines.append(f"{pad} 1 {tot1:6d} {100 * tot1 / totG:5.1f}"
522
+ f" {mis1:6d} {hit1:6d} "
523
+ f"{100 * per1:.1f}")
524
+ lines.append(f"{y_name} 0 {tot0:6d} "
525
+ f"{100 * tot0 / totG:5.1f} {hit0:6d} "
526
+ f"{mis0:6d} {100 * per0:.1f}")
527
+ lines.append("-" * 51)
528
+ lines.append(f"{pad} Total {totG:6d}" + " " * 26
529
+ + f"{100 * perT:.1f}")
530
+ lines.append("")
531
+ accuracy = 100 * (hit0 + hit1) / totG
532
+ recall = 100 * hit1 / (hit1 + mis1) if tot1 else np.nan
533
+ precision = (100 * hit1 / (hit1 + mis0)
534
+ if (hit1 + mis0) else np.nan)
535
+ lines += [f"Accuracy: {fmt(accuracy, 2)}",
536
+ f"Sensitivity: {fmt(recall, 2)}",
537
+ f"Precision: {fmt(precision, 2)}", ""]
538
+ return {"prob_cut": pc, "hit0": hit0, "mis0": mis0,
539
+ "hit1": hit1, "mis1": mis1,
540
+ "accuracy": accuracy, "sensitivity": recall,
541
+ "precision": precision}
542
+
543
+
544
+ def _logit_plot(xv, y01, mu, levels, y_name, x_name, pc,
545
+ params, xlab, ylab, pt_size, transparency,
546
+ digits_d):
547
+ """Scatterplot of the 0/1 response on the predictor with
548
+ the fitted logistic curve; dashed crosshairs at the
549
+ probability threshold and its predictor cutoff; the two
550
+ response levels labeled on the right axis.
551
+ R analog: the sigmoid plot of .logit4Pred()"""
552
+ style_opts = plotly_style()
553
+ x_lab = x_name if xlab is None else xlab
554
+ lv2 = str(levels[1]) if levels is not None else "1"
555
+ y_lab = (f"Probability {y_name} = {lv2}"
556
+ if ylab is None else ylab)
557
+ pt_fill = get_option("pt_color", "#324E5C")
558
+ fig = go.Figure()
559
+ fig.add_trace(go.Scatter(
560
+ x=xv, y=y01, mode="markers",
561
+ marker=dict(symbol="circle",
562
+ size=max(1.0, pt_size * 7.25),
563
+ sizemode="diameter",
564
+ color=make_trans(pt_fill,
565
+ 1 - transparency),
566
+ opacity=1,
567
+ line=dict(color=to_hex(pt_fill),
568
+ width=1)),
569
+ hoverinfo="x+y", showlegend=False))
570
+ od = np.argsort(xv, kind="stable")
571
+ fig.add_trace(go.Scatter(
572
+ x=xv[od], y=mu[od], mode="lines",
573
+ line=dict(color=to_hex(pt_fill), width=2),
574
+ hoverinfo="skip", showlegend=False))
575
+ axT1 = pretty(float(xv.min()), float(xv.max()))
576
+ axT2 = [0, 0.2, 0.4, 0.6, 0.8, 1]
577
+ ax_x = axis_num(x_lab, axT1,
578
+ axis_format(axT1, digits_d))
579
+ ax_y = axis_num(y_lab, axT2,
580
+ [f"{v:g}" for v in axT2])
581
+ ax_y.update(range=[-0.10, 1.10], showgrid=True,
582
+ gridcolor=to_hex(style_opts["grid_col"]),
583
+ gridwidth=1, griddash="dot")
584
+ shapes = x_grid(axT1) + plot_border()
585
+ if pc is not None: # threshold crosshairs
586
+ x_cut = ((math.log(pc / (1 - pc)) - params.iloc[0])
587
+ / params.iloc[1])
588
+ shapes.append(dict(
589
+ type="line", xref="paper", yref="y",
590
+ x0=0, x1=1, y0=pc, y1=pc,
591
+ line=dict(color=to_hex("gray35"), width=0.75,
592
+ dash="dash")))
593
+ if float(xv.min()) <= x_cut <= float(xv.max()):
594
+ shapes.append(dict(
595
+ type="line", xref="x", yref="paper",
596
+ x0=x_cut, x1=x_cut, y0=0, y1=1,
597
+ line=dict(color=to_hex("gray35"), width=0.75,
598
+ dash="dash")))
599
+ anns = []
600
+ if levels is not None: # right-axis level labels
601
+ for yv_i, lv in ((0, levels[0]), (1, levels[1])):
602
+ anns.append(dict(
603
+ xref="paper", yref="y", x=1.01, y=yv_i,
604
+ text=str(lv), showarrow=False,
605
+ xanchor="left",
606
+ font=dict(size=round(
607
+ 14 * get_option("axis_size", 0.9)),
608
+ color=to_hex(get_option(
609
+ "axis_color", "black")))))
610
+ fig.update_layout(
611
+ xaxis=ax_x, yaxis=ax_y, shapes=shapes,
612
+ annotations=anns, template=None,
613
+ plot_bgcolor=to_hex(style_opts["panel_fill"]),
614
+ paper_bgcolor=to_hex(style_opts["window_fill"]))
615
+ return fig