lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Logit.py
ADDED
|
@@ -0,0 +1,615 @@
|
|
|
1
|
+
# Logit.py — analog of Logit.R (core numeric predictors)
|
|
2
|
+
#
|
|
3
|
+
# Logit(): logistic regression with the lessR output pipeline —
|
|
4
|
+
# the estimated model on the logit scale with Wald confidence
|
|
5
|
+
# intervals, odds ratios, model fit (deviances, AIC),
|
|
6
|
+
# collinearity via the auxiliary linear model, the residuals and
|
|
7
|
+
# influence listing (with R's glm formulas for the studentized
|
|
8
|
+
# residual, dffits, and Cook's distance), the classification
|
|
9
|
+
# table sorted by fitted probability, confusion matrices per
|
|
10
|
+
# prob_cut threshold with accuracy/sensitivity/precision, and
|
|
11
|
+
# the fitted-sigmoid plot for a single predictor. As elsewhere,
|
|
12
|
+
# the pipeline is ported, not the lines.
|
|
13
|
+
#
|
|
14
|
+
# The model is a formula string, "Y ~ X1 + X2", as Regression().
|
|
15
|
+
# The response is numeric 0/1 or a two-level categorical
|
|
16
|
+
# (second level = the reference group predicted as 1, re-ordered
|
|
17
|
+
# by ref_group=). Categorical predictors become treatment-coded
|
|
18
|
+
# indicator variables (VarLevel columns, ~ model.matrix), with
|
|
19
|
+
# R's ">>> Note" announcement. Expression terms such as
|
|
20
|
+
# log(Years) or I(Years^2) are materialized as columns (shared
|
|
21
|
+
# _parse_formula, ~ .formula_expr). A multiple logit model draws
|
|
22
|
+
# the symmetric scatterplot matrix with loess smooths
|
|
23
|
+
# (plt_mat_plotly, ~ logit.4Pred). Rmd= writes a Quarto (.qmd)
|
|
24
|
+
# classification report (logit_rmd.py) — designed on the
|
|
25
|
+
# Regression report, as R's Logit has no Rmd. quiet= does not
|
|
26
|
+
# exist; brief= trims residuals and prediction, as R.
|
|
27
|
+
#
|
|
28
|
+
# Numerics through statsmodels GLM (Binomial), imported lazily.
|
|
29
|
+
# Returns a LogitResults object; figures in .plots are not
|
|
30
|
+
# auto-shown.
|
|
31
|
+
|
|
32
|
+
import math
|
|
33
|
+
|
|
34
|
+
import numpy as np
|
|
35
|
+
import pandas as pd
|
|
36
|
+
import plotly.graph_objects as go
|
|
37
|
+
|
|
38
|
+
from .plotly_utils import (
|
|
39
|
+
axis_format, axis_num, make_trans, plot_border,
|
|
40
|
+
plotly_style, to_hex, x_grid)
|
|
41
|
+
from .plt_mat_plotly import scatter_matrix
|
|
42
|
+
from .logit_rmd import logit_rmd
|
|
43
|
+
from .Regression import (
|
|
44
|
+
_expand_indicators, _parse_formula, _prntbl)
|
|
45
|
+
from .utils import fmt, get_column, get_option, pretty
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class LogitResults:
|
|
49
|
+
"""Numeric results and figures of Logit(): estimates,
|
|
50
|
+
odds_ratios, fit, residuals and predictions listings,
|
|
51
|
+
confusion matrices, and the plotly figures in .plots."""
|
|
52
|
+
|
|
53
|
+
def __init__(self, **kw):
|
|
54
|
+
self.__dict__.update(kw)
|
|
55
|
+
|
|
56
|
+
def __repr__(self):
|
|
57
|
+
return (f"<lessPy Logit: {self.formula}, "
|
|
58
|
+
f"n={self.n_keep}>")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _glm_influence(y01, mu, hat, k_params):
|
|
62
|
+
"""R's glm influence measures from the hat values and the
|
|
63
|
+
Pearson residuals, dispersion 1 (binomial): rstudent, as R
|
|
64
|
+
returns it for a glm (the standardized Pearson residual,
|
|
65
|
+
pearson / sqrt(1 - hat) — verified against rstudent() on
|
|
66
|
+
this model), dffits = rstudent * sqrt(hat / (1 - hat)),
|
|
67
|
+
and Cook's distance. R analogs: rstudent(), dffits(),
|
|
68
|
+
cooks.distance.glm()"""
|
|
69
|
+
with np.errstate(divide="ignore", invalid="ignore"):
|
|
70
|
+
pear = (y01 - mu) / np.sqrt(mu * (1 - mu))
|
|
71
|
+
rstud = pear / np.sqrt(1 - hat)
|
|
72
|
+
dffits = rstud * np.sqrt(hat / (1 - hat))
|
|
73
|
+
cooks = (pear / (1 - hat)) ** 2 * hat / k_params
|
|
74
|
+
return rstud, dffits, cooks
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def Logit(my_formula, data=None, filter=None, ref_group=None,
|
|
78
|
+
digits_d=4, brief=False,
|
|
79
|
+
res_rows=None, res_sort="cooks",
|
|
80
|
+
pred=True, pred_all=False, prob_cut=0.5, cooks_cut=1,
|
|
81
|
+
X1_new=None, X2_new=None, X3_new=None,
|
|
82
|
+
X4_new=None, X5_new=None, X6_new=None,
|
|
83
|
+
pt_size=0.9, transparency=0.8,
|
|
84
|
+
Rmd=None, Rmd_data=None, Rmd_format="html",
|
|
85
|
+
Rmd_browser=True,
|
|
86
|
+
results=True, explain=True, interpret=True, code=True,
|
|
87
|
+
xlab=None, ylab=None, graphics=True):
|
|
88
|
+
"""Logistic regression of a formula string, "Y ~ X1 + X2",
|
|
89
|
+
with the lessR analysis pipeline: estimates and odds ratios,
|
|
90
|
+
fit, collinearity, residuals and influence, classification
|
|
91
|
+
with confusion matrices, and the fitted-sigmoid plot. Always
|
|
92
|
+
prints, as in R; returns a LogitResults object with the
|
|
93
|
+
figures in .plots."""
|
|
94
|
+
import statsmodels.api as smapi
|
|
95
|
+
|
|
96
|
+
if data is None:
|
|
97
|
+
raise ValueError(
|
|
98
|
+
"data= is required: a pandas DataFrame containing "
|
|
99
|
+
"the model's variables")
|
|
100
|
+
if res_sort not in ("cooks", "rstudent", "dffits", "off"):
|
|
101
|
+
raise ValueError(
|
|
102
|
+
'res_sort: "cooks", "rstudent", "dffits", or "off"')
|
|
103
|
+
if Rmd is not None:
|
|
104
|
+
if Rmd_format not in ("html", "pdf", "docx", "word",
|
|
105
|
+
"none"):
|
|
106
|
+
raise ValueError('Rmd_format: "html", "pdf", '
|
|
107
|
+
'"docx", or "none"')
|
|
108
|
+
if brief:
|
|
109
|
+
raise ValueError(
|
|
110
|
+
"a Quarto report needs the full analysis, so "
|
|
111
|
+
"Rmd= is not available with brief=True")
|
|
112
|
+
if filter is not None:
|
|
113
|
+
data = data.query(filter)
|
|
114
|
+
if brief:
|
|
115
|
+
if res_rows is None:
|
|
116
|
+
res_rows = 0
|
|
117
|
+
pred = False
|
|
118
|
+
|
|
119
|
+
y_name, pred_names, data = _parse_formula(my_formula, data)
|
|
120
|
+
formula = (f"{y_name} ~ "
|
|
121
|
+
+ (" + ".join(pred_names) if pred_names else "1"))
|
|
122
|
+
n_pred = len(pred_names)
|
|
123
|
+
if n_pred == 0:
|
|
124
|
+
raise ValueError("Logit() requires at least one "
|
|
125
|
+
"predictor")
|
|
126
|
+
|
|
127
|
+
y_ser = get_column(data, y_name, "response")
|
|
128
|
+
pred_sers = [get_column(data, nm, "predictor")
|
|
129
|
+
for nm in pred_names]
|
|
130
|
+
|
|
131
|
+
used = pd.concat([y_ser] + pred_sers, axis=1)
|
|
132
|
+
keep = ~used.isna().any(axis=1)
|
|
133
|
+
n_obs = len(data)
|
|
134
|
+
n_keep = int(keep.sum())
|
|
135
|
+
yk = y_ser[keep]
|
|
136
|
+
|
|
137
|
+
# categorical predictors become indicator variables, as R
|
|
138
|
+
(pred_names, pred_sers_x, ind_notes, _cat_names,
|
|
139
|
+
_term_map) = _expand_indicators(
|
|
140
|
+
pred_names, [s[keep] for s in pred_sers],
|
|
141
|
+
">>> Note: {0} is not a numeric variable.\n"
|
|
142
|
+
" Indicator variables are created and "
|
|
143
|
+
"analyzed.")
|
|
144
|
+
n_pred = len(pred_names)
|
|
145
|
+
Xd = pd.DataFrame(
|
|
146
|
+
{nm: s.to_numpy(dtype=float)
|
|
147
|
+
for nm, s in zip(pred_names, pred_sers_x)},
|
|
148
|
+
index=yk.index)
|
|
149
|
+
|
|
150
|
+
# response: numeric 0/1, or two-level categorical with the
|
|
151
|
+
# second level the reference group (predicted as 1)
|
|
152
|
+
y_is_factor = not pd.api.types.is_numeric_dtype(yk)
|
|
153
|
+
if y_is_factor:
|
|
154
|
+
if isinstance(yk.dtype, pd.CategoricalDtype):
|
|
155
|
+
levels = [lv for lv in yk.cat.categories
|
|
156
|
+
if lv in set(yk)]
|
|
157
|
+
else:
|
|
158
|
+
levels = sorted(yk.astype(str).unique())
|
|
159
|
+
if len(levels) != 2:
|
|
160
|
+
raise ValueError(
|
|
161
|
+
f"Response variable: {y_name}\n"
|
|
162
|
+
"If numeric, can only have values of 0 or 1.\n"
|
|
163
|
+
"If a factor, can only have two levels.")
|
|
164
|
+
if ref_group is not None:
|
|
165
|
+
if ref_group not in levels:
|
|
166
|
+
raise ValueError(
|
|
167
|
+
f"Values of response {y_name}: "
|
|
168
|
+
f"{levels[0]} {levels[1]}\n"
|
|
169
|
+
"You specified a non-existent value, "
|
|
170
|
+
f"ref_group = {ref_group}")
|
|
171
|
+
if levels[1] != ref_group:
|
|
172
|
+
levels = [levels[1], levels[0]]
|
|
173
|
+
# single numeric predictor: order the levels so the
|
|
174
|
+
# slope is positive, as R
|
|
175
|
+
if n_pred == 1:
|
|
176
|
+
avg = Xd.iloc[:, 0].groupby(
|
|
177
|
+
yk.astype(str).to_numpy()).mean()
|
|
178
|
+
if avg[str(levels[0])] > avg[str(levels[1])]:
|
|
179
|
+
levels = [levels[1], levels[0]]
|
|
180
|
+
y01 = (yk.astype(str) ==
|
|
181
|
+
str(levels[1])).to_numpy(dtype=float)
|
|
182
|
+
else:
|
|
183
|
+
if ref_group is not None:
|
|
184
|
+
raise ValueError(
|
|
185
|
+
"Parameter ref_group only applies when the "
|
|
186
|
+
"response is a factor.")
|
|
187
|
+
vals = set(yk.unique())
|
|
188
|
+
if not vals <= {0, 1}:
|
|
189
|
+
raise ValueError(
|
|
190
|
+
f"Response variable: {y_name}\n"
|
|
191
|
+
"If numeric, can only have values of 0 or 1.\n"
|
|
192
|
+
"If a factor, can only have two levels.")
|
|
193
|
+
levels = None
|
|
194
|
+
y01 = yk.to_numpy(dtype=float)
|
|
195
|
+
|
|
196
|
+
new_data = X1_new is not None
|
|
197
|
+
row_labels = Xd.index.astype(str)
|
|
198
|
+
d = digits_d
|
|
199
|
+
|
|
200
|
+
X = smapi.add_constant(Xd, has_constant="add")
|
|
201
|
+
glm = smapi.GLM(y01, X,
|
|
202
|
+
family=smapi.families.Binomial()).fit(
|
|
203
|
+
tol=1e-12) # match R glm precision
|
|
204
|
+
if glm.params.isna().any():
|
|
205
|
+
bad = ", ".join(glm.params.index[glm.params.isna()])
|
|
206
|
+
raise ValueError(
|
|
207
|
+
"Variable redundant with a prior predictor in the "
|
|
208
|
+
f"model: {bad}")
|
|
209
|
+
mu = np.asarray(glm.fittedvalues, dtype=float)
|
|
210
|
+
hat = glm.get_influence().hat_matrix_diag
|
|
211
|
+
rstud, dffits, cooks = _glm_influence(
|
|
212
|
+
y01, mu, hat, len(glm.params))
|
|
213
|
+
|
|
214
|
+
lines = []
|
|
215
|
+
for note in ind_notes:
|
|
216
|
+
lines += [note, ""]
|
|
217
|
+
|
|
218
|
+
# ---------- variables and cases ----------------------------
|
|
219
|
+
for i, nm in enumerate([y_name] + pred_names):
|
|
220
|
+
if i == 0:
|
|
221
|
+
lbl = "Response Variable: "
|
|
222
|
+
elif n_pred > 1:
|
|
223
|
+
lbl = f"Predictor Variable {i}: "
|
|
224
|
+
else:
|
|
225
|
+
lbl = "Predictor Variable: "
|
|
226
|
+
lines.append(lbl + nm)
|
|
227
|
+
lines += ["",
|
|
228
|
+
f"Number of cases (rows) of data: {n_obs}",
|
|
229
|
+
f"Number of cases retained for analysis: "
|
|
230
|
+
f"{n_keep}"]
|
|
231
|
+
|
|
232
|
+
# ---------- BASIC ANALYSIS ---------------------------------
|
|
233
|
+
lines += ["", "", " BASIC ANALYSIS", "",
|
|
234
|
+
f"-- Estimated Model of {y_name} for the Logit "
|
|
235
|
+
"of Reference Group Membership", ""]
|
|
236
|
+
ci = glm.conf_int(alpha=0.05) # Wald, ~ confint.default
|
|
237
|
+
est = pd.DataFrame({
|
|
238
|
+
"Estimate": glm.params,
|
|
239
|
+
"Std Err": glm.bse,
|
|
240
|
+
"z-value": glm.tvalues,
|
|
241
|
+
"p-value": glm.pvalues,
|
|
242
|
+
"Lower 95%": ci[0],
|
|
243
|
+
"Upper 95%": ci[1],
|
|
244
|
+
})
|
|
245
|
+
est.index = ["(Intercept)"] + pred_names
|
|
246
|
+
buf = max(len(s) for s in est.index)
|
|
247
|
+
w = [max(9, max(len(fmt(v, d)) for v in est[c]) + 1)
|
|
248
|
+
for c in est.columns]
|
|
249
|
+
lines.append(" " * buf
|
|
250
|
+
+ f"{'Estimate':>{w[0] + 1}}"
|
|
251
|
+
+ f"{'Std Err':>{w[1] + 2}}"
|
|
252
|
+
+ f"{'z-value':>9}{'p-value':>9}"
|
|
253
|
+
+ f"{'Lower 95%':>{w[4] + 3}}"
|
|
254
|
+
+ f"{'Upper 95%':>{w[5] + 3}}")
|
|
255
|
+
for lbl, r in est.iterrows():
|
|
256
|
+
lines.append(
|
|
257
|
+
f"{lbl:<{buf}}"
|
|
258
|
+
+ f"{fmt(r['Estimate'], d):>{w[0] + 1}}"
|
|
259
|
+
+ f"{fmt(r['Std Err'], d):>{w[1] + 2}}"
|
|
260
|
+
+ f"{fmt(r['z-value'], 3):>9}"
|
|
261
|
+
+ f"{fmt(r['p-value'], 3):>9}"
|
|
262
|
+
+ f"{fmt(r['Lower 95%'], d):>{w[4] + 3}}"
|
|
263
|
+
+ f"{fmt(r['Upper 95%'], d):>{w[5] + 3}}")
|
|
264
|
+
|
|
265
|
+
# odds ratios and 95% CI
|
|
266
|
+
orci = pd.DataFrame({
|
|
267
|
+
"Odds Ratio": np.exp(est["Estimate"]),
|
|
268
|
+
"Lower 95%": np.exp(est["Lower 95%"]),
|
|
269
|
+
"Upper 95%": np.exp(est["Upper 95%"]),
|
|
270
|
+
}, index=est.index)
|
|
271
|
+
lines += ["", "",
|
|
272
|
+
"-- Odds Ratios and Confidence Intervals", ""]
|
|
273
|
+
wo = [max(10, max(len(fmt(v, d)) for v in orci[c]) + 1)
|
|
274
|
+
for c in orci.columns]
|
|
275
|
+
lines.append(" " * (buf + 2)
|
|
276
|
+
+ f"{'Odds Ratio':>{wo[0] + 1}}"
|
|
277
|
+
+ f"{'Lower 95%':>{wo[1] + 3}}"
|
|
278
|
+
+ f"{'Upper 95%':>{wo[2] + 3}}")
|
|
279
|
+
for lbl, r in orci.iterrows():
|
|
280
|
+
lines.append(
|
|
281
|
+
f"{lbl:<{buf}} "
|
|
282
|
+
+ f"{fmt(r['Odds Ratio'], d):>{wo[0] + 1}}"
|
|
283
|
+
+ f"{fmt(r['Lower 95%'], d):>{wo[1] + 3}}"
|
|
284
|
+
+ f"{fmt(r['Upper 95%'], d):>{wo[2] + 3}}")
|
|
285
|
+
|
|
286
|
+
# model fit
|
|
287
|
+
n_iter = len(glm.fit_history["deviance"]) - 1
|
|
288
|
+
lines += ["", "", "-- Model Fit", "",
|
|
289
|
+
f" Null deviance: {fmt(glm.null_deviance, 3)}"
|
|
290
|
+
f" on {int(glm.df_resid + n_pred)} degrees of "
|
|
291
|
+
"freedom",
|
|
292
|
+
f"Residual deviance: {fmt(glm.deviance, 3)} on "
|
|
293
|
+
f"{int(glm.df_resid)} degrees of freedom", "",
|
|
294
|
+
f"AIC: {fmt(glm.aic, 3)}", "",
|
|
295
|
+
f"Number of iterations to convergence: {n_iter}"]
|
|
296
|
+
|
|
297
|
+
# collinearity: R computes tolerance/VIF from the auxiliary
|
|
298
|
+
# linear model of the (numeric) response on the predictors
|
|
299
|
+
tol = vif = None
|
|
300
|
+
if n_pred > 1:
|
|
301
|
+
ols = smapi.OLS(y01, X).fit()
|
|
302
|
+
MSW = ols.scale
|
|
303
|
+
vif = np.array([
|
|
304
|
+
(Xd[nm].var(ddof=1) * (n_keep - 1)
|
|
305
|
+
* ols.bse[nm] ** 2) / MSW
|
|
306
|
+
for nm in pred_names])
|
|
307
|
+
tol = 1 / vif
|
|
308
|
+
lines += ["", "", "Collinearity", ""]
|
|
309
|
+
c1 = max(len(s) for s in pred_names)
|
|
310
|
+
lines.append(" " * c1 + f"{'Tolerance':>11}"
|
|
311
|
+
+ f"{'VIF':>9}")
|
|
312
|
+
for nm, t_i, v_i in zip(pred_names, tol, vif):
|
|
313
|
+
lines.append(f"{nm:<{c1}}{fmt(t_i, 3):>11}"
|
|
314
|
+
+ f"{fmt(v_i, 3):>9}")
|
|
315
|
+
|
|
316
|
+
# ---------- ANALYSIS OF RESIDUALS AND INFLUENCE ------------
|
|
317
|
+
res_tbl = None
|
|
318
|
+
if res_rows is None:
|
|
319
|
+
res_rows = n_keep if n_keep < 20 else 20
|
|
320
|
+
if res_rows == "all":
|
|
321
|
+
res_rows = n_keep
|
|
322
|
+
res_rows = min(int(res_rows), n_keep)
|
|
323
|
+
if res_rows > 0:
|
|
324
|
+
res_tbl = pd.DataFrame(index=row_labels)
|
|
325
|
+
for nm in pred_names:
|
|
326
|
+
res_tbl[nm] = Xd[nm].to_numpy()
|
|
327
|
+
res_tbl[y_name] = yk.to_numpy()
|
|
328
|
+
res_tbl["P(Y=1)"] = mu
|
|
329
|
+
res_tbl["residual"] = y01 - mu
|
|
330
|
+
res_tbl["rstudent"] = rstud
|
|
331
|
+
res_tbl["dffits"] = dffits
|
|
332
|
+
res_tbl["cooks"] = cooks
|
|
333
|
+
if res_sort == "cooks":
|
|
334
|
+
res_tbl = res_tbl.sort_values("cooks",
|
|
335
|
+
ascending=False)
|
|
336
|
+
elif res_sort == "rstudent":
|
|
337
|
+
res_tbl = res_tbl.reindex(
|
|
338
|
+
res_tbl["rstudent"].abs().sort_values(
|
|
339
|
+
ascending=False).index)
|
|
340
|
+
elif res_sort == "dffits":
|
|
341
|
+
res_tbl = res_tbl.reindex(
|
|
342
|
+
res_tbl["dffits"].abs().sort_values(
|
|
343
|
+
ascending=False).index)
|
|
344
|
+
lines += ["", "",
|
|
345
|
+
" ANALYSIS OF RESIDUALS AND INFLUENCE", "",
|
|
346
|
+
"Data, Fitted, Residual, Standardized "
|
|
347
|
+
"Pearson Residual, Dffits, Cook's Distance"]
|
|
348
|
+
if res_sort == "cooks":
|
|
349
|
+
lines.append(" [sorted by Cook's Distance]")
|
|
350
|
+
elif res_sort == "rstudent":
|
|
351
|
+
lines.append(" [sorted by Standardized Pearson "
|
|
352
|
+
"Residual, ignoring + or - sign]")
|
|
353
|
+
elif res_sort == "dffits":
|
|
354
|
+
lines.append(" [sorted by dffits, ignoring + or"
|
|
355
|
+
" - sign]")
|
|
356
|
+
lines.append(f" [res_rows = {res_rows} out of "
|
|
357
|
+
f"{n_keep} cases (rows) of data]")
|
|
358
|
+
lines.append("-" * 68)
|
|
359
|
+
lines += _prntbl(res_tbl.head(res_rows), d).split("\n")
|
|
360
|
+
lines.append("-" * 68)
|
|
361
|
+
|
|
362
|
+
# ---------- PREDICTION -------------------------------------
|
|
363
|
+
pred_tbl = None
|
|
364
|
+
confusion = []
|
|
365
|
+
plots = {}
|
|
366
|
+
prob_cuts = [float(p) for p in np.atleast_1d(prob_cut)]
|
|
367
|
+
p_cut = prob_cuts[0] if len(prob_cuts) == 1 else 0.5
|
|
368
|
+
lv2_txt = str(levels[1]) if levels is not None else ""
|
|
369
|
+
|
|
370
|
+
if pred:
|
|
371
|
+
if new_data:
|
|
372
|
+
grids = [np.atleast_1d(g) for g in
|
|
373
|
+
(X1_new, X2_new, X3_new, X4_new,
|
|
374
|
+
X5_new, X6_new)[:n_pred]
|
|
375
|
+
if g is not None]
|
|
376
|
+
if len(grids) != n_pred:
|
|
377
|
+
raise ValueError(
|
|
378
|
+
"Specified new data values for one "
|
|
379
|
+
"predictor variable, so do for all.")
|
|
380
|
+
mesh = np.meshgrid(*grids, indexing="ij")
|
|
381
|
+
Xnew = pd.DataFrame(
|
|
382
|
+
{nm: m.ravel() for nm, m in
|
|
383
|
+
zip(pred_names, mesh)})
|
|
384
|
+
Xn = smapi.add_constant(Xnew, has_constant="add")
|
|
385
|
+
prd = glm.get_prediction(Xn)
|
|
386
|
+
fitv = np.asarray(prd.predicted)
|
|
387
|
+
sev = np.asarray(prd.se)
|
|
388
|
+
pred_tbl = Xnew.copy()
|
|
389
|
+
pred_tbl[y_name] = ""
|
|
390
|
+
pred_tbl.index = [""] * len(pred_tbl)
|
|
391
|
+
else:
|
|
392
|
+
prd = glm.get_prediction(X)
|
|
393
|
+
fitv = np.asarray(prd.predicted)
|
|
394
|
+
sev = np.asarray(prd.se)
|
|
395
|
+
pred_tbl = pd.DataFrame(index=row_labels)
|
|
396
|
+
for nm in pred_names:
|
|
397
|
+
pred_tbl[nm] = Xd[nm].to_numpy()
|
|
398
|
+
pred_tbl[y_name] = yk.to_numpy()
|
|
399
|
+
pred_tbl["label"] = (fitv >= p_cut).astype(int)
|
|
400
|
+
pred_tbl["fitted"] = fitv
|
|
401
|
+
pred_tbl["std.err"] = sev
|
|
402
|
+
pred_tbl = pred_tbl.sort_values("fitted")
|
|
403
|
+
|
|
404
|
+
lines += ["", "", " PREDICTION", "",
|
|
405
|
+
"Probability threshold for classification "
|
|
406
|
+
f"{lv2_txt}: {p_cut}", ""]
|
|
407
|
+
if levels is not None:
|
|
408
|
+
lines += [f" 0: {levels[0]}",
|
|
409
|
+
f" 1: {levels[1]}", ""]
|
|
410
|
+
lines += ["Data, Fitted Values, Standard Errors",
|
|
411
|
+
" [sorted by fitted value]"]
|
|
412
|
+
if n_keep > 50 and not pred_all and not new_data:
|
|
413
|
+
lines.append(" [pred_all=TRUE to see all "
|
|
414
|
+
"intervals displayed]")
|
|
415
|
+
lines.append("-" * 68)
|
|
416
|
+
body = _prntbl(pred_tbl, d).split("\n")
|
|
417
|
+
head_ln, rows_ln = body[0], body[1:]
|
|
418
|
+
if n_keep < 25 or pred_all or new_data:
|
|
419
|
+
lines += body
|
|
420
|
+
else:
|
|
421
|
+
fits_sorted = pred_tbl["fitted"].to_numpy()
|
|
422
|
+
i_mid = int(np.abs(0.5 - fits_sorted).argmin())
|
|
423
|
+
i_mid = min(max(i_mid, 2), n_keep - 3)
|
|
424
|
+
lines.append(head_ln)
|
|
425
|
+
lines += rows_ln[:4]
|
|
426
|
+
lines += ["", "... for the rows of data where "
|
|
427
|
+
"fitted is close to 0.5 ...", ""]
|
|
428
|
+
lines += rows_ln[i_mid - 2:i_mid + 3]
|
|
429
|
+
lines += ["", "... for the last 4 rows of sorted "
|
|
430
|
+
"data ...", ""]
|
|
431
|
+
lines += rows_ln[n_keep - 4:]
|
|
432
|
+
lines.append("-" * 68)
|
|
433
|
+
|
|
434
|
+
# confusion matrix per prob_cut threshold
|
|
435
|
+
if not new_data:
|
|
436
|
+
lines += ["", "", "-" * 28,
|
|
437
|
+
"Specified confusion matrices",
|
|
438
|
+
"-" * 28, ""]
|
|
439
|
+
for pc in prob_cuts:
|
|
440
|
+
confusion.append(_logit_confuse(
|
|
441
|
+
lines, y01, fitv, pc, y_name, lv2_txt,
|
|
442
|
+
glm.params, pred_names, d))
|
|
443
|
+
else:
|
|
444
|
+
lines += ["", "",
|
|
445
|
+
"With X1_new, etc., no confusion "
|
|
446
|
+
"matrix."]
|
|
447
|
+
|
|
448
|
+
# sigmoid plot for a single numeric predictor
|
|
449
|
+
if graphics and n_pred == 1 and not new_data:
|
|
450
|
+
plots["logit_fit"] = _logit_plot(
|
|
451
|
+
Xd.iloc[:, 0].to_numpy(), y01, mu, levels,
|
|
452
|
+
y_name, pred_names[0],
|
|
453
|
+
prob_cuts[0] if len(prob_cuts) == 1 else None,
|
|
454
|
+
glm.params, xlab, ylab, pt_size, transparency,
|
|
455
|
+
d)
|
|
456
|
+
|
|
457
|
+
# scatterplot matrix of the model variables for a
|
|
458
|
+
# multiple logit model, symmetric with loess smooths and
|
|
459
|
+
# no correlations, ~ logit.4Pred pairs(panel=smooth)
|
|
460
|
+
if graphics and n_pred > 1 and not new_data:
|
|
461
|
+
mat_df = pd.DataFrame(
|
|
462
|
+
{y_name: y01}, index=Xd.index).join(Xd)
|
|
463
|
+
plots["scatter_matrix"] = scatter_matrix(
|
|
464
|
+
mat_df, fit="loess", cor_coef=False, band=False,
|
|
465
|
+
digits_d=d)
|
|
466
|
+
|
|
467
|
+
print("\n".join(lines))
|
|
468
|
+
|
|
469
|
+
out = LogitResults(
|
|
470
|
+
formula=formula, n_obs=n_obs, n_keep=n_keep,
|
|
471
|
+
digits_d=d, levels=levels,
|
|
472
|
+
estimates=est, odds_ratios=orci,
|
|
473
|
+
fit={"null_deviance": glm.null_deviance,
|
|
474
|
+
"deviance": glm.deviance,
|
|
475
|
+
"df_null": int(glm.df_resid + n_pred),
|
|
476
|
+
"df_residual": int(glm.df_resid),
|
|
477
|
+
"aic": glm.aic, "iterations": n_iter},
|
|
478
|
+
tolerance=tol, vif=vif,
|
|
479
|
+
residuals=res_tbl, predictions=pred_tbl,
|
|
480
|
+
confusion=confusion, plots=plots)
|
|
481
|
+
|
|
482
|
+
if Rmd is not None:
|
|
483
|
+
logit_rmd(out, y_name, pred_names, formula,
|
|
484
|
+
list(data.columns), Rmd, Rmd_data, Rmd_format,
|
|
485
|
+
Rmd_browser, results, explain, interpret, code)
|
|
486
|
+
|
|
487
|
+
return out
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def _logit_confuse(lines, y01, fitv, pc, y_name, lv2_txt,
|
|
491
|
+
params, pred_names, digits_d):
|
|
492
|
+
"""One confusion matrix at threshold pc, with the accuracy,
|
|
493
|
+
sensitivity, and precision. Appends the printed block to
|
|
494
|
+
lines and returns the counts. R analog: .logit5Confuse()"""
|
|
495
|
+
label = (fitv >= pc).astype(int)
|
|
496
|
+
hit0 = int(((y01 == 0) & (label == 0)).sum())
|
|
497
|
+
mis0 = int(((y01 == 0) & (label == 1)).sum())
|
|
498
|
+
hit1 = int(((y01 == 1) & (label == 1)).sum())
|
|
499
|
+
mis1 = int(((y01 == 1) & (label == 0)).sum())
|
|
500
|
+
tot0, tot1 = hit0 + mis0, hit1 + mis1
|
|
501
|
+
totG = tot0 + tot1
|
|
502
|
+
per0 = hit0 / tot0 if tot0 else np.nan
|
|
503
|
+
per1 = hit1 / tot1 if tot1 else np.nan
|
|
504
|
+
perT = (hit0 + hit1) / totG
|
|
505
|
+
|
|
506
|
+
lines.append("Probability threshold for predicting "
|
|
507
|
+
f"{lv2_txt}: {pc}")
|
|
508
|
+
if len(pred_names) == 1:
|
|
509
|
+
x_cut = ((math.log(pc / (1 - pc)) - params.iloc[0])
|
|
510
|
+
/ params.iloc[1])
|
|
511
|
+
lines.append("Corresponding cutoff threshold for "
|
|
512
|
+
f"{pred_names[0]}: {round(x_cut, 3)}")
|
|
513
|
+
lines.append("")
|
|
514
|
+
ln = len(y_name)
|
|
515
|
+
pad = " " * ln
|
|
516
|
+
lines.append(f"{pad} Baseline Predicted")
|
|
517
|
+
lines.append("-" * 51)
|
|
518
|
+
lines.append(f"{pad} Total %Tot 0 1"
|
|
519
|
+
" %Correct")
|
|
520
|
+
lines.append("-" * 51)
|
|
521
|
+
lines.append(f"{pad} 1 {tot1:6d} {100 * tot1 / totG:5.1f}"
|
|
522
|
+
f" {mis1:6d} {hit1:6d} "
|
|
523
|
+
f"{100 * per1:.1f}")
|
|
524
|
+
lines.append(f"{y_name} 0 {tot0:6d} "
|
|
525
|
+
f"{100 * tot0 / totG:5.1f} {hit0:6d} "
|
|
526
|
+
f"{mis0:6d} {100 * per0:.1f}")
|
|
527
|
+
lines.append("-" * 51)
|
|
528
|
+
lines.append(f"{pad} Total {totG:6d}" + " " * 26
|
|
529
|
+
+ f"{100 * perT:.1f}")
|
|
530
|
+
lines.append("")
|
|
531
|
+
accuracy = 100 * (hit0 + hit1) / totG
|
|
532
|
+
recall = 100 * hit1 / (hit1 + mis1) if tot1 else np.nan
|
|
533
|
+
precision = (100 * hit1 / (hit1 + mis0)
|
|
534
|
+
if (hit1 + mis0) else np.nan)
|
|
535
|
+
lines += [f"Accuracy: {fmt(accuracy, 2)}",
|
|
536
|
+
f"Sensitivity: {fmt(recall, 2)}",
|
|
537
|
+
f"Precision: {fmt(precision, 2)}", ""]
|
|
538
|
+
return {"prob_cut": pc, "hit0": hit0, "mis0": mis0,
|
|
539
|
+
"hit1": hit1, "mis1": mis1,
|
|
540
|
+
"accuracy": accuracy, "sensitivity": recall,
|
|
541
|
+
"precision": precision}
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _logit_plot(xv, y01, mu, levels, y_name, x_name, pc,
|
|
545
|
+
params, xlab, ylab, pt_size, transparency,
|
|
546
|
+
digits_d):
|
|
547
|
+
"""Scatterplot of the 0/1 response on the predictor with
|
|
548
|
+
the fitted logistic curve; dashed crosshairs at the
|
|
549
|
+
probability threshold and its predictor cutoff; the two
|
|
550
|
+
response levels labeled on the right axis.
|
|
551
|
+
R analog: the sigmoid plot of .logit4Pred()"""
|
|
552
|
+
style_opts = plotly_style()
|
|
553
|
+
x_lab = x_name if xlab is None else xlab
|
|
554
|
+
lv2 = str(levels[1]) if levels is not None else "1"
|
|
555
|
+
y_lab = (f"Probability {y_name} = {lv2}"
|
|
556
|
+
if ylab is None else ylab)
|
|
557
|
+
pt_fill = get_option("pt_color", "#324E5C")
|
|
558
|
+
fig = go.Figure()
|
|
559
|
+
fig.add_trace(go.Scatter(
|
|
560
|
+
x=xv, y=y01, mode="markers",
|
|
561
|
+
marker=dict(symbol="circle",
|
|
562
|
+
size=max(1.0, pt_size * 7.25),
|
|
563
|
+
sizemode="diameter",
|
|
564
|
+
color=make_trans(pt_fill,
|
|
565
|
+
1 - transparency),
|
|
566
|
+
opacity=1,
|
|
567
|
+
line=dict(color=to_hex(pt_fill),
|
|
568
|
+
width=1)),
|
|
569
|
+
hoverinfo="x+y", showlegend=False))
|
|
570
|
+
od = np.argsort(xv, kind="stable")
|
|
571
|
+
fig.add_trace(go.Scatter(
|
|
572
|
+
x=xv[od], y=mu[od], mode="lines",
|
|
573
|
+
line=dict(color=to_hex(pt_fill), width=2),
|
|
574
|
+
hoverinfo="skip", showlegend=False))
|
|
575
|
+
axT1 = pretty(float(xv.min()), float(xv.max()))
|
|
576
|
+
axT2 = [0, 0.2, 0.4, 0.6, 0.8, 1]
|
|
577
|
+
ax_x = axis_num(x_lab, axT1,
|
|
578
|
+
axis_format(axT1, digits_d))
|
|
579
|
+
ax_y = axis_num(y_lab, axT2,
|
|
580
|
+
[f"{v:g}" for v in axT2])
|
|
581
|
+
ax_y.update(range=[-0.10, 1.10], showgrid=True,
|
|
582
|
+
gridcolor=to_hex(style_opts["grid_col"]),
|
|
583
|
+
gridwidth=1, griddash="dot")
|
|
584
|
+
shapes = x_grid(axT1) + plot_border()
|
|
585
|
+
if pc is not None: # threshold crosshairs
|
|
586
|
+
x_cut = ((math.log(pc / (1 - pc)) - params.iloc[0])
|
|
587
|
+
/ params.iloc[1])
|
|
588
|
+
shapes.append(dict(
|
|
589
|
+
type="line", xref="paper", yref="y",
|
|
590
|
+
x0=0, x1=1, y0=pc, y1=pc,
|
|
591
|
+
line=dict(color=to_hex("gray35"), width=0.75,
|
|
592
|
+
dash="dash")))
|
|
593
|
+
if float(xv.min()) <= x_cut <= float(xv.max()):
|
|
594
|
+
shapes.append(dict(
|
|
595
|
+
type="line", xref="x", yref="paper",
|
|
596
|
+
x0=x_cut, x1=x_cut, y0=0, y1=1,
|
|
597
|
+
line=dict(color=to_hex("gray35"), width=0.75,
|
|
598
|
+
dash="dash")))
|
|
599
|
+
anns = []
|
|
600
|
+
if levels is not None: # right-axis level labels
|
|
601
|
+
for yv_i, lv in ((0, levels[0]), (1, levels[1])):
|
|
602
|
+
anns.append(dict(
|
|
603
|
+
xref="paper", yref="y", x=1.01, y=yv_i,
|
|
604
|
+
text=str(lv), showarrow=False,
|
|
605
|
+
xanchor="left",
|
|
606
|
+
font=dict(size=round(
|
|
607
|
+
14 * get_option("axis_size", 0.9)),
|
|
608
|
+
color=to_hex(get_option(
|
|
609
|
+
"axis_color", "black")))))
|
|
610
|
+
fig.update_layout(
|
|
611
|
+
xaxis=ax_x, yaxis=ax_y, shapes=shapes,
|
|
612
|
+
annotations=anns, template=None,
|
|
613
|
+
plot_bgcolor=to_hex(style_opts["panel_fill"]),
|
|
614
|
+
paper_bgcolor=to_hex(style_opts["window_fill"]))
|
|
615
|
+
return fig
|