lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/logit_rmd.py
ADDED
|
@@ -0,0 +1,410 @@
|
|
|
1
|
+
# logit_rmd.py — Quarto (.qmd) report for Logit(), the
|
|
2
|
+
# classification analog of reg_rmd.py.
|
|
3
|
+
#
|
|
4
|
+
# R's Logit() has NO Rmd= feature (no logit.Rmd, inst/Rmd holds
|
|
5
|
+
# only reg/), so this report is not a port: it is designed on the
|
|
6
|
+
# reg_rmd pattern, adapted to logistic regression — the log-odds
|
|
7
|
+
# model and odds ratios, the deviance fit with the likelihood-
|
|
8
|
+
# ratio test and McFadden pseudo-R^2, and the classification
|
|
9
|
+
# (confusion) table. Same design choices as the Regression
|
|
10
|
+
# report: Quarto target, direct per-section emitters, live code
|
|
11
|
+
# chunks for tables/plots, narrative numbers baked at generation
|
|
12
|
+
# time, and the explain/interpret/results/code toggles.
|
|
13
|
+
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
from scipy import stats as sps
|
|
18
|
+
|
|
19
|
+
from .reg_rmd import _chunk, _maybe_render, _xAnd, _xNum, _xP, _xU
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def logit_rmd(result, y_name, pred_names, formula_str, data_cols,
|
|
23
|
+
rmd, rmd_data, rmd_format, rmd_browser,
|
|
24
|
+
results, explain, interpret, code):
|
|
25
|
+
"""Write a Quarto report reproducing the logistic-regression
|
|
26
|
+
analysis. Returns the path of the .qmd file written."""
|
|
27
|
+
n_pred = len(pred_names)
|
|
28
|
+
d = result.digits_d
|
|
29
|
+
Y = y_name
|
|
30
|
+
X = _xAnd(pred_names)
|
|
31
|
+
pl = "s" if n_pred > 1 else ""
|
|
32
|
+
event = (f"{Y} = {result.levels[1]}" if result.levels
|
|
33
|
+
else f"{Y} = 1")
|
|
34
|
+
|
|
35
|
+
data_name = "d"
|
|
36
|
+
read_expr = (f'pd.read_csv("{rmd_data}")' if rmd_data
|
|
37
|
+
else 'pd.read_csv("your_data.csv")')
|
|
38
|
+
call = f'Logit("{formula_str}", data={data_name})'
|
|
39
|
+
|
|
40
|
+
tx = []
|
|
41
|
+
tx += _front_matter(Y, X, n_pred, results, explain,
|
|
42
|
+
interpret, code)
|
|
43
|
+
tx += _setup_chunk(read_expr, data_name, call, code)
|
|
44
|
+
tx += _sec_intro(Y, X, pl, event)
|
|
45
|
+
tx += _sec_data(read_expr, data_name, data_cols, code)
|
|
46
|
+
tx += _sec_model(Y, X, pred_names, pl, event, result, d,
|
|
47
|
+
n_pred, results, explain, interpret, code)
|
|
48
|
+
tx += _sec_fit(Y, result, d, n_pred, results, explain)
|
|
49
|
+
tx += _sec_relations(Y, X, pl, result, n_pred, code)
|
|
50
|
+
tx += _sec_classification(Y, event, result, d, results,
|
|
51
|
+
explain, code)
|
|
52
|
+
if result.residuals is not None:
|
|
53
|
+
tx += _sec_influence(Y, result, code)
|
|
54
|
+
if result.predictions is not None:
|
|
55
|
+
tx += _sec_prediction(Y, event, result, code)
|
|
56
|
+
|
|
57
|
+
path = Path(rmd)
|
|
58
|
+
if path.suffix.lower() != ".qmd":
|
|
59
|
+
path = path.with_suffix(".qmd")
|
|
60
|
+
path.write_text("\n".join(tx))
|
|
61
|
+
print(f"\nQuarto report written: {path}")
|
|
62
|
+
_maybe_render(path, rmd_format, rmd_browser)
|
|
63
|
+
return path
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# ------------------------------------------------------------
|
|
67
|
+
# front matter and setup
|
|
68
|
+
# ------------------------------------------------------------
|
|
69
|
+
|
|
70
|
+
def _front_matter(Y, X, n_pred, results, explain, interpret, code):
|
|
71
|
+
title = (f"Logistic Regression of {Y}" if n_pred > 1
|
|
72
|
+
else f"Logistic Regression of {Y} on {X}")
|
|
73
|
+
return [
|
|
74
|
+
"---",
|
|
75
|
+
f'title: "{title}"',
|
|
76
|
+
"format:",
|
|
77
|
+
" html:",
|
|
78
|
+
" toc: true",
|
|
79
|
+
" toc-depth: 4",
|
|
80
|
+
" embed-resources: true",
|
|
81
|
+
"engine: jupyter",
|
|
82
|
+
"---",
|
|
83
|
+
"",
|
|
84
|
+
"------",
|
|
85
|
+
"",
|
|
86
|
+
f"_Output Options: explain={explain}, "
|
|
87
|
+
f"interpret={interpret}, results={results}, "
|
|
88
|
+
f"code={code}_",
|
|
89
|
+
"",
|
|
90
|
+
"------",
|
|
91
|
+
""]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _setup_chunk(read_expr, data_name, call, code):
|
|
95
|
+
return _chunk(
|
|
96
|
+
["import pandas as pd",
|
|
97
|
+
"from lessPy import Logit",
|
|
98
|
+
f"{data_name} = {read_expr}",
|
|
99
|
+
f"r = {call}"],
|
|
100
|
+
echo=code, output=False)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# ------------------------------------------------------------
|
|
104
|
+
# 1 - Introduction
|
|
105
|
+
# ------------------------------------------------------------
|
|
106
|
+
|
|
107
|
+
def _sec_intro(Y, X, pl, event):
|
|
108
|
+
return [
|
|
109
|
+
"## Introduction",
|
|
110
|
+
"",
|
|
111
|
+
f"The _response variable_ {Y} is binary. This logistic "
|
|
112
|
+
"regression models the probability of the outcome "
|
|
113
|
+
f"_{event}_ from the _predictor variable{pl}_ {X}. It is "
|
|
114
|
+
"an example of _supervised machine learning_: a "
|
|
115
|
+
"classifier that learns, from the training data, how the "
|
|
116
|
+
f"feature{pl} {X} relate to the target {Y}.",
|
|
117
|
+
"",
|
|
118
|
+
"Rather than the value of the response directly, the "
|
|
119
|
+
"model estimates its probability, and from a probability "
|
|
120
|
+
"threshold assigns each case to a predicted class.",
|
|
121
|
+
""]
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ------------------------------------------------------------
|
|
125
|
+
# 2 - Data
|
|
126
|
+
# ------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
def _sec_data(read_expr, data_name, data_cols, code):
|
|
129
|
+
out = ["## Data", "",
|
|
130
|
+
"Read the data into a pandas data frame.", ""]
|
|
131
|
+
out += _chunk([f"{data_name} = {read_expr}", data_name],
|
|
132
|
+
echo=code)
|
|
133
|
+
out += [
|
|
134
|
+
"The rows are the _training data_, from which the model "
|
|
135
|
+
"is estimated.",
|
|
136
|
+
"",
|
|
137
|
+
"Data from the following variables are available for "
|
|
138
|
+
f"analysis: {_xAnd(data_cols)}.",
|
|
139
|
+
""]
|
|
140
|
+
return out
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# ------------------------------------------------------------
|
|
144
|
+
# 3 - Model
|
|
145
|
+
# ------------------------------------------------------------
|
|
146
|
+
|
|
147
|
+
def _sec_model(Y, X, pred_names, pl, event, result, d, n_pred,
|
|
148
|
+
results, explain, interpret, code):
|
|
149
|
+
coefs = result.estimates["Estimate"]
|
|
150
|
+
out = ["## Model", "", "### Specified Model", "",
|
|
151
|
+
f"The response {Y} is not modeled directly. Instead "
|
|
152
|
+
"the model is linear in the _log odds_ (the logit) of "
|
|
153
|
+
f"the probability $\\hat\\pi$ of the outcome "
|
|
154
|
+
f"_{event}_:",
|
|
155
|
+
"",
|
|
156
|
+
_eq_logit_b(pred_names),
|
|
157
|
+
""]
|
|
158
|
+
if explain:
|
|
159
|
+
out += [
|
|
160
|
+
"The _odds_ are $\\hat\\pi / (1 - \\hat\\pi)$, and "
|
|
161
|
+
"the logit is their natural logarithm. Solving for "
|
|
162
|
+
"the probability gives the S-shaped logistic curve:",
|
|
163
|
+
"",
|
|
164
|
+
"$$\\hat\\pi = \\frac{1}{1 + e^{-(b_0 + b_1 X_1 + "
|
|
165
|
+
"\\cdots)}}$$",
|
|
166
|
+
"",
|
|
167
|
+
"A slope coefficient $b_j$ is the change in the log "
|
|
168
|
+
"odds for a one-unit increase in its predictor; "
|
|
169
|
+
f"exponentiated, $e^{{b_j}}$ is the _odds ratio_, "
|
|
170
|
+
"the multiplicative change in the odds.",
|
|
171
|
+
""]
|
|
172
|
+
out += ["Estimate the coefficients by maximum likelihood "
|
|
173
|
+
"with the _lessPy_ function _Logit()_. The analysis "
|
|
174
|
+
"ran in the setup chunk above.",
|
|
175
|
+
"", "### Estimated Model", ""]
|
|
176
|
+
if results:
|
|
177
|
+
out += _chunk(["r.estimates"], echo=code)
|
|
178
|
+
out += [
|
|
179
|
+
"The estimated logit model, with maximum-likelihood "
|
|
180
|
+
"coefficients:",
|
|
181
|
+
"",
|
|
182
|
+
_eq_logit_est(pred_names, coefs, d),
|
|
183
|
+
"",
|
|
184
|
+
"### Odds Ratios", "",
|
|
185
|
+
"Exponentiating each coefficient gives its odds ratio "
|
|
186
|
+
"and 95% confidence interval, the more interpretable "
|
|
187
|
+
"scale for the size of an effect.",
|
|
188
|
+
""]
|
|
189
|
+
if results:
|
|
190
|
+
out += _chunk(["r.odds_ratios"], echo=code)
|
|
191
|
+
if interpret:
|
|
192
|
+
out += _intr_or(Y, event, pred_names, result, d, n_pred)
|
|
193
|
+
return out
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _eq_logit_b(pred_names):
|
|
197
|
+
s = "$$\\ln\\!\\left(\\frac{\\hat\\pi}{1-\\hat\\pi}\\right) " \
|
|
198
|
+
"= b_0 + b_1 X_{" + pred_names[0] + "}"
|
|
199
|
+
for i, nm in enumerate(pred_names[1:], start=2):
|
|
200
|
+
s += f" + b_{i} X_{{{nm}}}"
|
|
201
|
+
return s + "$$"
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _eq_logit_est(pred_names, coefs, d):
|
|
205
|
+
s = ("$$\\ln\\!\\left(\\frac{\\hat\\pi}{1-\\hat\\pi}\\right) "
|
|
206
|
+
f"= {_xP(coefs.iloc[0], d)}")
|
|
207
|
+
for i, nm in enumerate(pred_names, start=1):
|
|
208
|
+
b = coefs.iloc[i]
|
|
209
|
+
sign = "+" if b >= 0 else "-"
|
|
210
|
+
s += f" {sign} {_xP(abs(b), d)} X_{{{nm}}}"
|
|
211
|
+
return s + "$$"
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _intr_or(Y, event, pred_names, result, d, n_pred):
|
|
215
|
+
pvals = result.estimates["p-value"]
|
|
216
|
+
orr = result.odds_ratios
|
|
217
|
+
sig = [pred_names[i] for i in range(n_pred)
|
|
218
|
+
if pvals.iloc[i + 1] <= 0.05]
|
|
219
|
+
ns = [pred_names[i] for i in range(n_pred)
|
|
220
|
+
if pvals.iloc[i + 1] > 0.05]
|
|
221
|
+
out = []
|
|
222
|
+
if sig:
|
|
223
|
+
out.append(
|
|
224
|
+
f"At $\\alpha$ = 0.05, {_xAnd([f'_{s}_' for s in sig])} "
|
|
225
|
+
f"significantly predict{'s' if len(sig) == 1 else ''} "
|
|
226
|
+
f"the odds of {event}.")
|
|
227
|
+
for nm in sig:
|
|
228
|
+
o = float(orr.loc[nm, "Odds Ratio"])
|
|
229
|
+
lo = float(orr.loc[nm, "Lower 95%"])
|
|
230
|
+
hi = float(orr.loc[nm, "Upper 95%"])
|
|
231
|
+
direction = ("multiplies" if o >= 1
|
|
232
|
+
else "multiplies (reduces)")
|
|
233
|
+
bullet = f"- _{nm}_: " if n_pred > 1 else ""
|
|
234
|
+
out.append(
|
|
235
|
+
f"{bullet}each one-unit increase {direction} the "
|
|
236
|
+
f"odds of {event} by {_xP(o, d)} "
|
|
237
|
+
f"(95% CI {_xP(lo, d)} to {_xP(hi, d)}), the "
|
|
238
|
+
"other predictors held constant."
|
|
239
|
+
if n_pred > 1 else
|
|
240
|
+
f"{bullet}each one-unit increase in {nm} "
|
|
241
|
+
f"{direction} the odds of {event} by {_xP(o, d)} "
|
|
242
|
+
f"(95% CI {_xP(lo, d)} to {_xP(hi, d)}).")
|
|
243
|
+
out.append("")
|
|
244
|
+
if ns:
|
|
245
|
+
pl2 = "s" if len(ns) > 1 else ""
|
|
246
|
+
out += [
|
|
247
|
+
f"{_xU(_xNum(len(ns)))} predictor{pl2} "
|
|
248
|
+
f"({_xAnd(ns)}) did not reach significance, so "
|
|
249
|
+
"there is a reasonable possibility of no effect on "
|
|
250
|
+
f"the odds of {event}.",
|
|
251
|
+
""]
|
|
252
|
+
return out
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# ------------------------------------------------------------
|
|
256
|
+
# 4 - Fit
|
|
257
|
+
# ------------------------------------------------------------
|
|
258
|
+
|
|
259
|
+
def _sec_fit(Y, result, d, n_pred, results, explain):
|
|
260
|
+
f = result.fit
|
|
261
|
+
g2 = f["null_deviance"] - f["deviance"]
|
|
262
|
+
df_lr = f["df_null"] - f["df_residual"]
|
|
263
|
+
p = float(sps.chi2.sf(g2, df_lr))
|
|
264
|
+
mcf = 1 - f["deviance"] / f["null_deviance"]
|
|
265
|
+
out = ["## Fit", "", "### Deviance", ""]
|
|
266
|
+
if explain:
|
|
267
|
+
out += [
|
|
268
|
+
"Least squares does not apply to a binary outcome; "
|
|
269
|
+
"fit is assessed with _deviance_, minus twice the "
|
|
270
|
+
"maximized log-likelihood. The _null deviance_ is "
|
|
271
|
+
"the deviance of the intercept-only model (no "
|
|
272
|
+
"predictors); the _residual deviance_ is that of the "
|
|
273
|
+
"fitted model. The drop from null to residual "
|
|
274
|
+
"deviance is the variation the predictors explain.",
|
|
275
|
+
""]
|
|
276
|
+
if results:
|
|
277
|
+
out += _chunk(["r.fit"], echo=False)
|
|
278
|
+
out += [
|
|
279
|
+
f"Null deviance {_xP(f['null_deviance'], d)} on "
|
|
280
|
+
f"{f['df_null']} df; residual deviance "
|
|
281
|
+
f"{_xP(f['deviance'], d)} on {f['df_residual']} df; "
|
|
282
|
+
f"AIC {_xP(f['aic'], d)}.",
|
|
283
|
+
"", "### Likelihood-Ratio Test", "",
|
|
284
|
+
"The overall test of whether the predictors as a set "
|
|
285
|
+
"relate to the response compares the two deviances. "
|
|
286
|
+
"The likelihood-ratio statistic is their difference, "
|
|
287
|
+
"a chi-square on the difference in degrees of "
|
|
288
|
+
"freedom.",
|
|
289
|
+
"",
|
|
290
|
+
f"$$G^2 = D_{{null}} - D_{{residual}} = "
|
|
291
|
+
f"{_xP(f['null_deviance'], d)} - "
|
|
292
|
+
f"{_xP(f['deviance'], d)} = {_xP(g2, d)}$$",
|
|
293
|
+
"",
|
|
294
|
+
f"With {df_lr} degrees of freedom, $G^2$ = "
|
|
295
|
+
f"{_xP(g2, d)} has a _p_-value of {_xP(p, 4)}, so "
|
|
296
|
+
+ ("the predictors are related to the response."
|
|
297
|
+
if p < 0.05 else
|
|
298
|
+
"the predictors are not detectably related to "
|
|
299
|
+
"the response."),
|
|
300
|
+
"", "### Pseudo $R^2$", "",
|
|
301
|
+
"McFadden's pseudo-$R^2$ rescales the deviance drop "
|
|
302
|
+
"to a 0-to-1 fit index (not the proportion of "
|
|
303
|
+
"variance of an OLS $R^2$).",
|
|
304
|
+
"",
|
|
305
|
+
f"$$R^2_{{McF}} = 1 - \\frac{{D_{{residual}}}}"
|
|
306
|
+
f"{{D_{{null}}}} = 1 - "
|
|
307
|
+
f"\\frac{{{_xP(f['deviance'], d)}}}"
|
|
308
|
+
f"{{{_xP(f['null_deviance'], d)}}} = {_xP(mcf, 3)}$$",
|
|
309
|
+
""]
|
|
310
|
+
return out
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
# ------------------------------------------------------------
|
|
314
|
+
# 5 - Relations
|
|
315
|
+
# ------------------------------------------------------------
|
|
316
|
+
|
|
317
|
+
def _sec_relations(Y, X, pl, result, n_pred, code):
|
|
318
|
+
if n_pred == 1:
|
|
319
|
+
out = ["## Relations", "", "### Fitted Probabilities", "",
|
|
320
|
+
f"The fitted logistic curve shows how the "
|
|
321
|
+
f"probability of the modeled outcome varies with "
|
|
322
|
+
f"{X}, with the 0/1 data points.",
|
|
323
|
+
""]
|
|
324
|
+
if "logit_fit" in result.plots:
|
|
325
|
+
out += _chunk(['r.plots["logit_fit"]'], echo=code)
|
|
326
|
+
return out
|
|
327
|
+
out = ["## Relations", "", "### Scatter Plot Matrix", "",
|
|
328
|
+
f"The pairwise relations among the response {Y} and "
|
|
329
|
+
f"the predictors {X} appear in a _scatterplot "
|
|
330
|
+
"matrix_, each off-diagonal cell with a loess smooth. "
|
|
331
|
+
f"The {Y} row and column, being 0/1, show how the "
|
|
332
|
+
"outcome rate shifts across each predictor.",
|
|
333
|
+
""]
|
|
334
|
+
if "scatter_matrix" in result.plots:
|
|
335
|
+
out += _chunk(['r.plots["scatter_matrix"]'], echo=code)
|
|
336
|
+
return out
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
# ------------------------------------------------------------
|
|
340
|
+
# 6 - Classification
|
|
341
|
+
# ------------------------------------------------------------
|
|
342
|
+
|
|
343
|
+
def _sec_classification(Y, event, result, d, results, explain,
|
|
344
|
+
code):
|
|
345
|
+
if not result.confusion:
|
|
346
|
+
return []
|
|
347
|
+
cm = result.confusion[0]
|
|
348
|
+
out = ["## Classification", ""]
|
|
349
|
+
if explain:
|
|
350
|
+
out += [
|
|
351
|
+
"Applying a probability threshold to the fitted "
|
|
352
|
+
"probabilities assigns each case a predicted class. "
|
|
353
|
+
"The _confusion matrix_ cross-tabulates the "
|
|
354
|
+
"predicted class against the actual class.",
|
|
355
|
+
""]
|
|
356
|
+
out += [
|
|
357
|
+
"At the probability threshold "
|
|
358
|
+
f"{_xP(cm['prob_cut'], 2)} for predicting _{event}_:",
|
|
359
|
+
""]
|
|
360
|
+
if results:
|
|
361
|
+
out += _chunk(
|
|
362
|
+
["import pandas as pd",
|
|
363
|
+
"cm = r.confusion[0]",
|
|
364
|
+
'pd.DataFrame('
|
|
365
|
+
'{"predicted 0": [cm["hit0"], cm["mis1"]], '
|
|
366
|
+
'"predicted 1": [cm["mis0"], cm["hit1"]]}, '
|
|
367
|
+
'index=["actual 0", "actual 1"])'],
|
|
368
|
+
echo=code)
|
|
369
|
+
out += [
|
|
370
|
+
f"Accuracy is {_xP(cm['accuracy'], 2)}% of cases "
|
|
371
|
+
f"classified correctly. Sensitivity (recall) is "
|
|
372
|
+
f"{_xP(cm['sensitivity'], 2)}%, the share of actual "
|
|
373
|
+
f"_{event}_ cases caught; precision is "
|
|
374
|
+
f"{_xP(cm['precision'], 2)}%, the share of predicted "
|
|
375
|
+
f"_{event}_ cases that were correct.",
|
|
376
|
+
""]
|
|
377
|
+
return out
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
# ------------------------------------------------------------
|
|
381
|
+
# 7 - Influence
|
|
382
|
+
# ------------------------------------------------------------
|
|
383
|
+
|
|
384
|
+
def _sec_influence(Y, result, code):
|
|
385
|
+
out = ["## Influence", "",
|
|
386
|
+
"Which cases contribute most to the lack of fit? For "
|
|
387
|
+
"each case the analysis reports the fitted "
|
|
388
|
+
"probability, the residual, the externally "
|
|
389
|
+
"Studentized residual (_rstudent_), the standardized "
|
|
390
|
+
"change in fit (_dffits_), and Cook's Distance "
|
|
391
|
+
"(_cooks_).",
|
|
392
|
+
""]
|
|
393
|
+
out += _chunk(["r.residuals"], echo=code)
|
|
394
|
+
return out
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# ------------------------------------------------------------
|
|
398
|
+
# 8 - Prediction
|
|
399
|
+
# ------------------------------------------------------------
|
|
400
|
+
|
|
401
|
+
def _sec_prediction(Y, event, result, code):
|
|
402
|
+
out = ["## Prediction", "",
|
|
403
|
+
"For each case the model gives the predicted "
|
|
404
|
+
f"probability of _{event}_ and its standard error. "
|
|
405
|
+
"Applied to new predictor values, the same model "
|
|
406
|
+
"predicts the probability of the outcome for cases "
|
|
407
|
+
"not in the training data.",
|
|
408
|
+
""]
|
|
409
|
+
out += _chunk(["r.predictions"], echo=code)
|
|
410
|
+
return out
|
lessPy/order_by.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# order_by.py — analog of order_by.R.
|
|
2
|
+
#
|
|
3
|
+
# order_by(): return a copy of a data frame with its rows sorted
|
|
4
|
+
# by one or more columns, each ascending ("+") or descending
|
|
5
|
+
# ("-"), or by the special keyword "row.names" (the index) or
|
|
6
|
+
# "random" (a shuffle). Prints the sort specification unless quiet.
|
|
7
|
+
#
|
|
8
|
+
# Column names are passed as strings, as elsewhere in lessPy.
|
|
9
|
+
# pandas sorts categoricals by their category order natively, so
|
|
10
|
+
# R's xtfrm() handling of factors needs no special case. The sort
|
|
11
|
+
# is stable (kind="stable"), matching R's stable order().
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def order_by(data, by, direction=None, seed=None, quiet=False):
|
|
18
|
+
"""Sort the rows of data and return the sorted copy. by is a
|
|
19
|
+
column name or list of names, or the keyword "row.names" (sort
|
|
20
|
+
by the index) or "random" (shuffle). direction is "+"
|
|
21
|
+
(ascending, the default) or "-" (descending), one per sort
|
|
22
|
+
column. seed makes a "random" sort reproducible.
|
|
23
|
+
R analog: order_by()"""
|
|
24
|
+
by_list = [by] if isinstance(by, str) else list(by)
|
|
25
|
+
|
|
26
|
+
if not quiet:
|
|
27
|
+
print("\nSort Specification")
|
|
28
|
+
|
|
29
|
+
if by_list == ["row.names"]:
|
|
30
|
+
d = _by_row_names(data, direction, quiet)
|
|
31
|
+
elif by_list == ["random"]:
|
|
32
|
+
d = _by_random(data, seed, quiet)
|
|
33
|
+
else:
|
|
34
|
+
d = _by_columns(data, by_list, direction, quiet)
|
|
35
|
+
|
|
36
|
+
if not quiet:
|
|
37
|
+
print()
|
|
38
|
+
return d
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _dir_word(sign):
|
|
42
|
+
return "ascending" if sign == "+" else "descending"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _check_dirs(direction, n):
|
|
46
|
+
if direction is None:
|
|
47
|
+
return ["+"] * n
|
|
48
|
+
dirs = [direction] if isinstance(direction, str) \
|
|
49
|
+
else list(direction)
|
|
50
|
+
if len(dirs) != n:
|
|
51
|
+
raise ValueError(
|
|
52
|
+
f"number of sort columns ({n}) must equal the number "
|
|
53
|
+
f"of direction values ({len(dirs)})")
|
|
54
|
+
for s in dirs:
|
|
55
|
+
if s not in ("+", "-"):
|
|
56
|
+
raise ValueError(
|
|
57
|
+
f"direction value {s!r}: use '+' (ascending) or "
|
|
58
|
+
"'-' (descending)")
|
|
59
|
+
return dirs
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _by_row_names(data, direction, quiet):
|
|
63
|
+
if direction is not None and (
|
|
64
|
+
not isinstance(direction, str)
|
|
65
|
+
and len(direction) != 1):
|
|
66
|
+
raise ValueError(
|
|
67
|
+
"sorting by row.names takes exactly one direction "
|
|
68
|
+
"value ('+' or '-')")
|
|
69
|
+
sign = _check_dirs(direction, 1)[0]
|
|
70
|
+
if not quiet:
|
|
71
|
+
print(f" row.names --> {_dir_word(sign)}")
|
|
72
|
+
return data.sort_index(ascending=(sign == "+"),
|
|
73
|
+
kind="stable")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _by_random(data, seed, quiet):
|
|
77
|
+
if not quiet:
|
|
78
|
+
print(" random")
|
|
79
|
+
rng = np.random.default_rng(seed)
|
|
80
|
+
order = rng.permutation(len(data))
|
|
81
|
+
return data.iloc[order]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _by_columns(data, by_list, direction, quiet):
|
|
85
|
+
missing = [c for c in by_list if c not in data.columns]
|
|
86
|
+
if missing:
|
|
87
|
+
raise ValueError(f"column(s) not found: {missing}")
|
|
88
|
+
dirs = _check_dirs(direction, len(by_list))
|
|
89
|
+
if not quiet:
|
|
90
|
+
for c, s in zip(by_list, dirs):
|
|
91
|
+
print(f" {c} --> {_dir_word(s)}")
|
|
92
|
+
ascending = [s == "+" for s in dirs]
|
|
93
|
+
return data.sort_values(by=by_list, ascending=ascending,
|
|
94
|
+
kind="stable")
|