lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/simCLT.py
ADDED
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
# simCLT.py — analog of simCLT.R.
|
|
2
|
+
#
|
|
3
|
+
# simCLT(): a Central Limit Theorem simulation. Draw ns samples,
|
|
4
|
+
# each of size n, from a chosen population (normal, uniform,
|
|
5
|
+
# lognormal, or a U-shaped "antinormal"), then show the sampling
|
|
6
|
+
# distribution of the sample mean converging toward normality.
|
|
7
|
+
# Prints the population and sample statistics as in R and returns
|
|
8
|
+
# a results object with the two plotly figures in .plots.
|
|
9
|
+
#
|
|
10
|
+
# The random draws use numpy's RNG, so they do not reproduce R's
|
|
11
|
+
# stream value-for-value (a simulation, like Flows). The population
|
|
12
|
+
# constants (mu, sigma, lognormal skew/median) and the CLT
|
|
13
|
+
# standardization are the exact R formulas. R's "antinormal" needs
|
|
14
|
+
# the external "triangle" package; here the triangular components
|
|
15
|
+
# come from scipy.stats.triang, so no extra dependency is added.
|
|
16
|
+
#
|
|
17
|
+
# NOTE: R's uniform branch samples runif(ns*n, 0, 4) — it ignores
|
|
18
|
+
# p1/p2 while still reporting mu/sigma from p1/p2. That is a latent
|
|
19
|
+
# bug; this port samples uniform(p1, p2) so the data match the
|
|
20
|
+
# reported population.
|
|
21
|
+
|
|
22
|
+
import numpy as np
|
|
23
|
+
|
|
24
|
+
from .utils import fmt
|
|
25
|
+
|
|
26
|
+
_GHOST = "#F8F8FF"
|
|
27
|
+
_STEEL = "#A2B5CD" # lightsteelblue3
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class SimCLTResults:
|
|
31
|
+
"""Numeric results of simCLT(): the population parameters, the
|
|
32
|
+
data and sample-mean summaries, the observed 95% ranges, the
|
|
33
|
+
per-sample means/sds, and the two plotly figures in .plots
|
|
34
|
+
("population" and "sampling")."""
|
|
35
|
+
|
|
36
|
+
def __init__(self, **kw):
|
|
37
|
+
self.__dict__.update(kw)
|
|
38
|
+
|
|
39
|
+
def __repr__(self):
|
|
40
|
+
return (f"<lessPy simCLT: {self.dist}, "
|
|
41
|
+
f"ns={self.n_samples}, n={self.n}>")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def simCLT(ns=None, n=None, p1=0, p2=1, seed=None, dist="normal",
|
|
45
|
+
fill=_STEEL, n_display=0, digits_d=3, subtitle=True,
|
|
46
|
+
pop=True):
|
|
47
|
+
"""Central Limit Theorem simulation: ns samples each of size n
|
|
48
|
+
from a population (dist = "normal", "uniform", "lognormal", or
|
|
49
|
+
"antinormal"). p1/p2 are the population parameters (normal:
|
|
50
|
+
mean/sd; uniform: min/max; lognormal: meanlog/sdlog;
|
|
51
|
+
antinormal: 0/max). Prints the statistics and returns a
|
|
52
|
+
SimCLTResults with .plots. R analog: simCLT()"""
|
|
53
|
+
if dist not in ("normal", "uniform", "lognormal", "antinormal"):
|
|
54
|
+
raise ValueError('dist: "normal", "uniform", '
|
|
55
|
+
'"lognormal", "antinormal"')
|
|
56
|
+
if ns is None:
|
|
57
|
+
raise ValueError("specify the number of samples: ns")
|
|
58
|
+
if n is None:
|
|
59
|
+
raise ValueError("specify the size of each sample: n")
|
|
60
|
+
|
|
61
|
+
rng = np.random.default_rng(seed)
|
|
62
|
+
skew = med = None
|
|
63
|
+
sigma = None
|
|
64
|
+
|
|
65
|
+
if dist == "normal":
|
|
66
|
+
mu, sigma = p1, p2
|
|
67
|
+
data = rng.normal(mu, sigma, ns * n)
|
|
68
|
+
fig_pop = _pop_normal(mu, sigma, fill, subtitle)
|
|
69
|
+
|
|
70
|
+
elif dist == "uniform":
|
|
71
|
+
lo, hi = p1, p2
|
|
72
|
+
mu = (lo + hi) / 2
|
|
73
|
+
sigma = (hi - lo) / np.sqrt(12)
|
|
74
|
+
data = rng.uniform(lo, hi, ns * n)
|
|
75
|
+
fig_pop = _pop_uniform(lo, hi, fill, subtitle)
|
|
76
|
+
|
|
77
|
+
elif dist == "lognormal":
|
|
78
|
+
meanlog, sdlog = p1, p2
|
|
79
|
+
vlog = sdlog ** 2
|
|
80
|
+
mu = np.exp(meanlog + vlog / 2)
|
|
81
|
+
sigma = mu * np.sqrt(np.exp(vlog) - 1)
|
|
82
|
+
skew = (np.exp(vlog) + 2) * np.sqrt(np.exp(vlog) - 1)
|
|
83
|
+
med = np.exp(meanlog)
|
|
84
|
+
data = rng.lognormal(meanlog, sdlog, ns * n)
|
|
85
|
+
fig_pop = _pop_lognormal(meanlog, sdlog, mu, sigma, fill,
|
|
86
|
+
subtitle)
|
|
87
|
+
|
|
88
|
+
else: # antinormal (U-shaped)
|
|
89
|
+
if p1 != 0:
|
|
90
|
+
raise ValueError("minimum of the anti-normal "
|
|
91
|
+
"distribution must be 0")
|
|
92
|
+
xmax = p2
|
|
93
|
+
mu = xmax / 2
|
|
94
|
+
data = _rantinormal(rng, ns * n, xmax)
|
|
95
|
+
fig_pop = _pop_antinormal(xmax, fill, subtitle)
|
|
96
|
+
|
|
97
|
+
mx = data.mean()
|
|
98
|
+
sx = data.std(ddof=1)
|
|
99
|
+
by_rep = data.reshape(ns, n)
|
|
100
|
+
ymean = by_rep.mean(axis=1)
|
|
101
|
+
ysd = by_rep.std(axis=1, ddof=1)
|
|
102
|
+
|
|
103
|
+
fig_samp = _samp_plot(ymean, fill, ns, n, dist, subtitle)
|
|
104
|
+
|
|
105
|
+
q_units = np.quantile(ymean, [0.025, 0.975])
|
|
106
|
+
se = (sx if dist == "antinormal" else sigma) / np.sqrt(n)
|
|
107
|
+
z = (ymean - mu) / se
|
|
108
|
+
q_se = np.quantile(z, [0.025, 0.975])
|
|
109
|
+
|
|
110
|
+
_report(dist, mu, sigma, skew, med, ns, n, mx, sx, ymean,
|
|
111
|
+
q_units, q_se, digits_d, n_display, by_rep)
|
|
112
|
+
|
|
113
|
+
return SimCLTResults(
|
|
114
|
+
dist=dist, mu=mu, sigma=sigma, skew=skew, median=med,
|
|
115
|
+
n_samples=ns, n=n, data_mean=mx, data_sd=sx,
|
|
116
|
+
mean_of_means=ymean.mean(), sd_of_means=ymean.std(ddof=1),
|
|
117
|
+
range_units=q_units, range_se=q_se, ymean=ymean, ysd=ysd,
|
|
118
|
+
plots={"population": fig_pop, "sampling": fig_samp})
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# --- population sampling -------------------------------------------
|
|
122
|
+
|
|
123
|
+
def _rantinormal(rng, size, xmax):
|
|
124
|
+
from scipy.stats import triang
|
|
125
|
+
half = xmax / 2
|
|
126
|
+
u = rng.random(size)
|
|
127
|
+
lo = u < 0.5
|
|
128
|
+
out = np.empty(size)
|
|
129
|
+
# lower half: triangle a=0, b=half, mode ~0 (decreasing)
|
|
130
|
+
lo_c = 0.01 / half
|
|
131
|
+
out[lo] = triang.ppf(rng.random(lo.sum()), lo_c, loc=0,
|
|
132
|
+
scale=half)
|
|
133
|
+
# upper half: triangle a=half, b=xmax, mode ~xmax (increasing)
|
|
134
|
+
hi_c = (half - 0.01) / half
|
|
135
|
+
out[~lo] = triang.ppf(rng.random((~lo).sum()), hi_c,
|
|
136
|
+
loc=half, scale=half)
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# --- population figures --------------------------------------------
|
|
141
|
+
|
|
142
|
+
def _pop_fig(x, y, xlab, sub, fill):
|
|
143
|
+
import plotly.graph_objects as go
|
|
144
|
+
fig = go.Figure()
|
|
145
|
+
fig.add_trace(go.Scatter(
|
|
146
|
+
x=np.concatenate(([x[0]], x, [x[-1]])),
|
|
147
|
+
y=np.concatenate(([0.0], y, [0.0])),
|
|
148
|
+
mode="lines", fill="toself",
|
|
149
|
+
fillcolor=fill, line=dict(color="black"),
|
|
150
|
+
hoverinfo="skip", showlegend=False))
|
|
151
|
+
fig.update_layout(
|
|
152
|
+
plot_bgcolor=_GHOST, paper_bgcolor="white",
|
|
153
|
+
xaxis=dict(title=xlab, showgrid=False, zeroline=False,
|
|
154
|
+
linecolor="black", mirror=True),
|
|
155
|
+
yaxis=dict(showticklabels=False, showgrid=False,
|
|
156
|
+
zeroline=False, linecolor="black", mirror=True),
|
|
157
|
+
margin=dict(t=30, r=20, b=50, l=30),
|
|
158
|
+
title=dict(text=sub or "", x=0.5, xanchor="center",
|
|
159
|
+
font=dict(size=11)))
|
|
160
|
+
return fig
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _pop_normal(mu, sigma, fill, subtitle):
|
|
164
|
+
from scipy.stats import norm
|
|
165
|
+
x = np.linspace(mu - 4 * sigma, mu + 4 * sigma, 500)
|
|
166
|
+
sub = (f"mu={fmt(mu, 3)} sigma={fmt(sigma, 3)}"
|
|
167
|
+
if subtitle else "")
|
|
168
|
+
return _pop_fig(x, norm.pdf(x, mu, sigma),
|
|
169
|
+
"Normal Population", sub, fill)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _pop_uniform(lo, hi, fill, subtitle):
|
|
173
|
+
x = np.linspace(lo, hi, 500)
|
|
174
|
+
y = np.full_like(x, 1.0 / (hi - lo))
|
|
175
|
+
sub = f"min={lo} max={hi}" if subtitle else ""
|
|
176
|
+
return _pop_fig(x, y, "Uniform Population", sub, fill)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _pop_lognormal(meanlog, sdlog, mu, sigma, fill, subtitle):
|
|
180
|
+
from scipy.stats import lognorm
|
|
181
|
+
xmax = np.ceil(mu + 4 * sigma)
|
|
182
|
+
x = np.linspace(1e-6, xmax, 500)
|
|
183
|
+
y = lognorm.pdf(x, sdlog, scale=np.exp(meanlog))
|
|
184
|
+
sub = (f"meanlog={meanlog} sdlog={sdlog}"
|
|
185
|
+
if subtitle else "")
|
|
186
|
+
return _pop_fig(x, y, "Lognormal Population", sub, fill)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _pop_antinormal(xmax, fill, subtitle):
|
|
190
|
+
from scipy.stats import triang
|
|
191
|
+
half = xmax / 2
|
|
192
|
+
x1 = np.linspace(0, half, 250)
|
|
193
|
+
y1 = triang.pdf(x1, 0.01 / half, loc=0, scale=half)
|
|
194
|
+
x2 = np.linspace(half, xmax, 250)
|
|
195
|
+
y2 = triang.pdf(x2, (half - 0.01) / half, loc=half, scale=half)
|
|
196
|
+
x = np.concatenate((x1, x2))
|
|
197
|
+
y = np.concatenate((y1, y2))
|
|
198
|
+
sub = f"min=0 max={xmax}" if subtitle else ""
|
|
199
|
+
return _pop_fig(x, y, "Anti-Normal Population", sub, fill)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
# --- sampling-distribution figure ----------------------------------
|
|
203
|
+
|
|
204
|
+
def _samp_plot(ymean, fill, ns, n, dist, subtitle):
|
|
205
|
+
import plotly.graph_objects as go
|
|
206
|
+
from scipy.stats import norm
|
|
207
|
+
m, s = ymean.mean(), ymean.std(ddof=1)
|
|
208
|
+
fig = go.Figure()
|
|
209
|
+
fig.add_trace(go.Histogram(
|
|
210
|
+
x=ymean, histnorm="probability density",
|
|
211
|
+
marker=dict(color=fill, line=dict(color="black", width=1)),
|
|
212
|
+
showlegend=False, hoverinfo="skip"))
|
|
213
|
+
x = np.linspace(ymean.min(), ymean.max(), 300)
|
|
214
|
+
fig.add_trace(go.Scatter(
|
|
215
|
+
x=x, y=norm.pdf(x, m, s), mode="lines",
|
|
216
|
+
line=dict(color="black"), showlegend=False,
|
|
217
|
+
hoverinfo="skip"))
|
|
218
|
+
sub = (f"{ns} samples, each of size {n} from {dist}"
|
|
219
|
+
if subtitle else "")
|
|
220
|
+
fig.update_layout(
|
|
221
|
+
bargap=0, plot_bgcolor="white", paper_bgcolor="white",
|
|
222
|
+
xaxis=dict(title="Sample Mean", showgrid=False,
|
|
223
|
+
linecolor="black", mirror=True),
|
|
224
|
+
yaxis=dict(showticklabels=False, showgrid=False,
|
|
225
|
+
linecolor="black", mirror=True),
|
|
226
|
+
margin=dict(t=30, r=20, b=50, l=30),
|
|
227
|
+
title=dict(text=sub, x=0.5, xanchor="center",
|
|
228
|
+
font=dict(size=11)))
|
|
229
|
+
return fig
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
# --- text report ----------------------------------------------------
|
|
233
|
+
|
|
234
|
+
def _report(dist, mu, sigma, skew, med, ns, n, mx, sx, ymean,
|
|
235
|
+
q_units, q_se, digits_d, n_display, by_rep):
|
|
236
|
+
print()
|
|
237
|
+
print(f"Population mean, mu : {mu}")
|
|
238
|
+
if sigma is not None:
|
|
239
|
+
print(f"Pop std dev, sigma : {sigma}")
|
|
240
|
+
if dist == "lognormal":
|
|
241
|
+
print(f"Population skew: {skew}")
|
|
242
|
+
print(f"Population median: {med}")
|
|
243
|
+
print()
|
|
244
|
+
print(f"Number of samples : {ns}")
|
|
245
|
+
print(f"Size of each sample : {n}")
|
|
246
|
+
print()
|
|
247
|
+
print(f"Mean of the data: {mx}")
|
|
248
|
+
print(f"Std Dev of the data: {sx}")
|
|
249
|
+
print()
|
|
250
|
+
print("Analysis of Sample Means")
|
|
251
|
+
print(f" Mean: {fmt(ymean.mean(), digits_d)}")
|
|
252
|
+
print(f"Std Dev: {fmt(ymean.std(ddof=1), digits_d)}")
|
|
253
|
+
print()
|
|
254
|
+
print("Observed 95% range of Sample Mean")
|
|
255
|
+
print(f" Original Units: {q_units[0]} {q_units[1]}")
|
|
256
|
+
lead = (" Estimated from Data, "
|
|
257
|
+
if dist == "antinormal" else " ")
|
|
258
|
+
print(f"{lead}Standard Errors: {q_se[0]} {q_se[1]}")
|
|
259
|
+
if n_display > 0:
|
|
260
|
+
for i in range(n_display):
|
|
261
|
+
vals = " ".join(fmt(v, digits_d) for v in by_rep[i])
|
|
262
|
+
print(f"\nSample {i + 1}")
|
|
263
|
+
print(f" Mean: {fmt(ymean[i], digits_d)}")
|
|
264
|
+
print(f" Rounded Data Values: {vals}")
|
|
265
|
+
print()
|
lessPy/simFlips.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# simFlips.py — analog of simFlips.R.
|
|
2
|
+
#
|
|
3
|
+
# simFlips(): flip a coin n times and plot the running proportion
|
|
4
|
+
# of heads converging toward the true probability prob — a law of
|
|
5
|
+
# large numbers demonstration. Optionally shows each individual
|
|
6
|
+
# 0/1 flip. Prints a short summary and returns a results object
|
|
7
|
+
# with the plotly figure in .plots.
|
|
8
|
+
#
|
|
9
|
+
# The flips use numpy's RNG (seed=), so they do not reproduce R's
|
|
10
|
+
# stream value-for-value (a simulation, like simCLT/simMeans).
|
|
11
|
+
# R's interactive "pause" mode is not ported (no batch equivalent).
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from .utils import fmt
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class SimFlipsResults:
|
|
19
|
+
"""Numeric results of simFlips(): the 0/1 flips, the running
|
|
20
|
+
proportion of heads, the head/tail counts, the target prob,
|
|
21
|
+
the final running mean, and the plotly figure in .plots
|
|
22
|
+
("flips")."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, **kw):
|
|
25
|
+
self.__dict__.update(kw)
|
|
26
|
+
|
|
27
|
+
def __repr__(self):
|
|
28
|
+
return (f"<lessPy simFlips: n={self.n}, "
|
|
29
|
+
f"heads={self.n_heads}>")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def simFlips(n=None, prob=0.5, seed=None, show_title=True,
|
|
33
|
+
show_flips=True, grid="#E5E5E5"):
|
|
34
|
+
"""Flip a coin n times (P(heads) = prob) and plot the running
|
|
35
|
+
proportion of heads against the number of flips, with a
|
|
36
|
+
reference line at prob. show_flips overlays each 0/1 flip.
|
|
37
|
+
Prints a summary and returns a SimFlipsResults with .plots.
|
|
38
|
+
R analog: simFlips()"""
|
|
39
|
+
if n is None:
|
|
40
|
+
raise ValueError("specify the number of flips: n")
|
|
41
|
+
if not 0 <= prob <= 1:
|
|
42
|
+
raise ValueError("prob must be between 0 and 1")
|
|
43
|
+
|
|
44
|
+
rng = np.random.default_rng(seed)
|
|
45
|
+
flips = rng.binomial(1, prob, n)
|
|
46
|
+
ybar = np.cumsum(flips) / np.arange(1, n + 1)
|
|
47
|
+
n_heads = int(flips.sum())
|
|
48
|
+
n_tails = n - n_heads
|
|
49
|
+
|
|
50
|
+
fig = _plot(n, flips, ybar, prob, grid, show_flips,
|
|
51
|
+
show_title, n_heads, n_tails)
|
|
52
|
+
|
|
53
|
+
print(f"\nNumber of flips: {n}")
|
|
54
|
+
print(f"P(heads) : {prob}")
|
|
55
|
+
print(f"Heads : {n_heads}")
|
|
56
|
+
print(f"Tails : {n_tails}")
|
|
57
|
+
print(f"Sample mean after {n} flips: {fmt(ybar[-1], 3)}")
|
|
58
|
+
print()
|
|
59
|
+
|
|
60
|
+
return SimFlipsResults(
|
|
61
|
+
n=n, prob=prob, flips=flips, running_mean=ybar,
|
|
62
|
+
n_heads=n_heads, n_tails=n_tails, final_mean=ybar[-1],
|
|
63
|
+
plots={"flips": fig})
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _plot(n, flips, ybar, prob, grid, show_flips, show_title,
|
|
67
|
+
n_heads, n_tails):
|
|
68
|
+
import plotly.graph_objects as go
|
|
69
|
+
fig = go.Figure()
|
|
70
|
+
# reference line at the true probability
|
|
71
|
+
fig.add_hline(y=prob, line=dict(color="lightsteelblue",
|
|
72
|
+
width=2))
|
|
73
|
+
if show_flips:
|
|
74
|
+
fig.add_trace(go.Scatter(
|
|
75
|
+
x=list(range(1, n + 1)), y=flips, mode="markers",
|
|
76
|
+
marker=dict(symbol="diamond", size=6, color="darkblue",
|
|
77
|
+
line=dict(color="lightsteelblue", width=1)),
|
|
78
|
+
showlegend=False,
|
|
79
|
+
hovertemplate="flip %{x}: %{y}<extra></extra>"))
|
|
80
|
+
# running proportion of heads
|
|
81
|
+
fig.add_trace(go.Scatter(
|
|
82
|
+
x=list(range(1, n + 1)), y=ybar, mode="lines",
|
|
83
|
+
line=dict(color="gray", width=3), showlegend=False,
|
|
84
|
+
hovertemplate="after %{x}: %{y:.3f}<extra></extra>"))
|
|
85
|
+
|
|
86
|
+
for y, txt in ((prob, f"μ={prob}"), (1, f"{n_heads} Heads"),
|
|
87
|
+
(0, f"{n_tails} Tails")):
|
|
88
|
+
fig.add_annotation(
|
|
89
|
+
xref="paper", x=1.005, y=y, xanchor="left",
|
|
90
|
+
text=txt, showarrow=False, font=dict(size=10))
|
|
91
|
+
|
|
92
|
+
title = (f"Sample Mean after {n} Coin Flips: "
|
|
93
|
+
f"{fmt(ybar[-1], 3)}") if show_title else ""
|
|
94
|
+
fig.update_layout(
|
|
95
|
+
plot_bgcolor="#F8F8FF", paper_bgcolor="white",
|
|
96
|
+
margin=dict(t=36, r=70, b=40, l=50),
|
|
97
|
+
xaxis=dict(title="Number of Flips", range=[1, n],
|
|
98
|
+
showgrid=False, linecolor="black", mirror=True),
|
|
99
|
+
yaxis=dict(title="Estimate", range=[0, 1], showgrid=True,
|
|
100
|
+
gridcolor=grid, gridwidth=0.5,
|
|
101
|
+
linecolor="black", mirror=True),
|
|
102
|
+
title=dict(text=title, x=0.5, xanchor="center",
|
|
103
|
+
font=dict(size=12)))
|
|
104
|
+
return fig
|
lessPy/simMeans.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# simMeans.py — analog of simMeans.R.
|
|
2
|
+
#
|
|
3
|
+
# simMeans(): draw ns samples, each of size n, from a normal
|
|
4
|
+
# population and plot every sample mean against its sample index,
|
|
5
|
+
# with a horizontal centerline at the population mean mu. Shows
|
|
6
|
+
# how the sample means scatter about mu and how their spread (the
|
|
7
|
+
# empirical standard error) shrinks as n grows. Prints the per-
|
|
8
|
+
# sample means, SDs and raw values, then the analysis of the
|
|
9
|
+
# sample means, and returns a results object with the plotly
|
|
10
|
+
# figure in .plots.
|
|
11
|
+
#
|
|
12
|
+
# The draws use numpy's RNG (seed=), so they do not reproduce R's
|
|
13
|
+
# stream value-for-value (a simulation, like simCLT). R's
|
|
14
|
+
# interactive "pause" mode (one sample per Enter press) has no
|
|
15
|
+
# batch equivalent and is not ported.
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
|
|
19
|
+
from .utils import fmt
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class SimMeansResults:
|
|
23
|
+
"""Numeric results of simMeans(): the population mu/sigma, the
|
|
24
|
+
per-sample means and SDs (in display order), the display order
|
|
25
|
+
index, the mean of the sample means and their SD (the direct
|
|
26
|
+
standard-error estimate), and the plotly figure in .plots
|
|
27
|
+
("means")."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, **kw):
|
|
30
|
+
self.__dict__.update(kw)
|
|
31
|
+
|
|
32
|
+
def __repr__(self):
|
|
33
|
+
return (f"<lessPy simMeans: ns={self.n_samples}, "
|
|
34
|
+
f"n={self.n}>")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def simMeans(ns=None, n=None, mu=0, sigma=1, seed=None,
|
|
38
|
+
show_title=True, show_data=True, max_data=10,
|
|
39
|
+
grid="#E5E5E5", ylim_bound=None, sort=True,
|
|
40
|
+
set_mu=False, digits_d=2):
|
|
41
|
+
"""Simulate ns samples of size n from a normal population
|
|
42
|
+
(mean mu, sd sigma) and plot each sample mean vs its index,
|
|
43
|
+
centered on mu. sort orders the samples by their mean;
|
|
44
|
+
set_mu hides a randomly chosen mu/sigma (a guess-the-mean
|
|
45
|
+
exercise); show_data lists up to max_data raw values per
|
|
46
|
+
sample. Prints the tables and returns a SimMeansResults with
|
|
47
|
+
.plots. R analog: simMeans()"""
|
|
48
|
+
if ns is None:
|
|
49
|
+
raise ValueError("specify the number of samples: ns")
|
|
50
|
+
if n is None:
|
|
51
|
+
raise ValueError("specify the size of each sample: n")
|
|
52
|
+
if sigma < 0:
|
|
53
|
+
raise ValueError("sigma cannot be negative")
|
|
54
|
+
|
|
55
|
+
rng = np.random.default_rng(seed)
|
|
56
|
+
if set_mu:
|
|
57
|
+
mu = int(rng.integers(0, 101))
|
|
58
|
+
sigma = int(rng.integers(1, 26))
|
|
59
|
+
|
|
60
|
+
if not set_mu:
|
|
61
|
+
print(f"\nPopulation mean, mu: {mu}")
|
|
62
|
+
print(f"Pop std dev, sigma : {sigma}")
|
|
63
|
+
print(f"\nNumber of samples : {ns}")
|
|
64
|
+
print(f"Size of each sample: {n}")
|
|
65
|
+
|
|
66
|
+
data = rng.normal(mu, sigma, ns * n).reshape(ns, n)
|
|
67
|
+
ymean = data.mean(axis=1)
|
|
68
|
+
ysd = data.std(axis=1, ddof=1)
|
|
69
|
+
|
|
70
|
+
if sort:
|
|
71
|
+
o = np.argsort(ymean, kind="stable")
|
|
72
|
+
else:
|
|
73
|
+
o = np.arange(ns)
|
|
74
|
+
ymean, ysd, data = ymean[o], ysd[o], data[o]
|
|
75
|
+
|
|
76
|
+
if ylim_bound is None:
|
|
77
|
+
dev = max(abs(mu - ymean.min()), abs(ymean.max() - mu))
|
|
78
|
+
lo, hi = mu - dev, mu + dev
|
|
79
|
+
else:
|
|
80
|
+
lo, hi = mu - ylim_bound, mu + ylim_bound
|
|
81
|
+
|
|
82
|
+
fig = _plot(ns, ymean, mu, sigma, n, lo, hi, grid,
|
|
83
|
+
show_title and not set_mu)
|
|
84
|
+
|
|
85
|
+
_report(ns, n, o, ymean, ysd, data, mu, sigma, max_data,
|
|
86
|
+
show_data, digits_d)
|
|
87
|
+
|
|
88
|
+
return SimMeansResults(
|
|
89
|
+
mu=mu, sigma=sigma, n_samples=ns, n=n, order=o,
|
|
90
|
+
ymean=ymean, ysd=ysd, mean_of_means=ymean.mean(),
|
|
91
|
+
se=ymean.std(ddof=1), plots={"means": fig})
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _plot(ns, ymean, mu, sigma, n, lo, hi, grid, show_title):
|
|
95
|
+
import plotly.graph_objects as go
|
|
96
|
+
fig = go.Figure()
|
|
97
|
+
fig.add_hline(y=mu, line=dict(color="darkslateblue", width=1.5))
|
|
98
|
+
fig.add_trace(go.Scatter(
|
|
99
|
+
x=list(range(1, ns + 1)), y=ymean, mode="markers",
|
|
100
|
+
marker=dict(symbol="circle", size=7, color="#4398D0",
|
|
101
|
+
line=dict(color="black", width=1)),
|
|
102
|
+
showlegend=False,
|
|
103
|
+
hovertemplate="sample %{x}<br>mean %{y:.3f}"
|
|
104
|
+
"<extra></extra>"))
|
|
105
|
+
title = (f"μ={mu} σ={sigma} n={n}"
|
|
106
|
+
if show_title else "")
|
|
107
|
+
fig.update_layout(
|
|
108
|
+
plot_bgcolor="white", paper_bgcolor="white",
|
|
109
|
+
margin=dict(t=36, r=30, b=30, l=50),
|
|
110
|
+
xaxis=dict(title="", range=[1, ns], showgrid=True,
|
|
111
|
+
gridcolor=grid, gridwidth=0.5,
|
|
112
|
+
linecolor="black", mirror=True),
|
|
113
|
+
yaxis=dict(title="Sample Mean", range=[lo, hi],
|
|
114
|
+
showgrid=True, gridcolor=grid, gridwidth=0.5,
|
|
115
|
+
linecolor="black", mirror=True),
|
|
116
|
+
title=dict(text=title, x=0.5, xanchor="center",
|
|
117
|
+
font=dict(size=12)))
|
|
118
|
+
return fig
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _report(ns, n, o, ymean, ysd, data, mu, sigma, max_data,
|
|
122
|
+
show_data, digits_d):
|
|
123
|
+
maxd = min(n, max_data)
|
|
124
|
+
w = max(8, digits_d + 6)
|
|
125
|
+
head = f"\n{'Sample':>6}{'Mean':>{w}}{'SD':>{w}}"
|
|
126
|
+
if show_data:
|
|
127
|
+
head += " " + "".join(f"{j:>{w}}"
|
|
128
|
+
for j in range(1, maxd + 1))
|
|
129
|
+
print(head)
|
|
130
|
+
for i in range(ns):
|
|
131
|
+
row = (f"{int(o[i]) + 1:>6}"
|
|
132
|
+
f"{fmt(ymean[i], digits_d):>{w}}"
|
|
133
|
+
f"{fmt(ysd[i], digits_d):>{w}} ")
|
|
134
|
+
if show_data:
|
|
135
|
+
row += "".join(f"{fmt(data[i, j], digits_d):>{w}}"
|
|
136
|
+
for j in range(maxd))
|
|
137
|
+
if n > max_data:
|
|
138
|
+
row += " ..."
|
|
139
|
+
print(row)
|
|
140
|
+
|
|
141
|
+
print("\nAnalysis of Sample Means")
|
|
142
|
+
print(f" Mean: {fmt(ymean.mean(), digits_d):>{w}}")
|
|
143
|
+
print(f"Std Dev: {fmt(ymean.std(ddof=1), digits_d):>{w}}"
|
|
144
|
+
" Direct estimate of the standard error of the "
|
|
145
|
+
"sample mean")
|
|
146
|
+
print()
|