mfe-toolbox 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mfe/__init__.py +18 -0
- mfe/bootstrap/__init__.py +18 -0
- mfe/bootstrap/spa.py +217 -0
- mfe/bootstrap/stepM.py +210 -0
- mfe/bootstrap/wild.py +169 -0
- mfe/crosssection/__init__.py +15 -0
- mfe/crosssection/fm.py +171 -0
- mfe/crosssection/ols.py +176 -0
- mfe/crosssection/pca.py +143 -0
- mfe/distributions/__init__.py +21 -0
- mfe/distributions/ged.py +85 -0
- mfe/distributions/mvnorm.py +166 -0
- mfe/distributions/skewt.py +140 -0
- mfe/multivariate/__init__.py +22 -0
- mfe/multivariate/_core.html +2944 -0
- mfe/multivariate/_core.pyx +303 -0
- mfe/multivariate/base.py +193 -0
- mfe/multivariate/bekk.py +415 -0
- mfe/multivariate/ccc.py +84 -0
- mfe/multivariate/dcc.py +244 -0
- mfe/multivariate/gogarch.py +581 -0
- mfe/multivariate/rcc.py +330 -0
- mfe/realized/__init__.py +39 -0
- mfe/realized/_core.html +3075 -0
- mfe/realized/_core.pyx +225 -0
- mfe/realized/_core_fallback.py +26 -0
- mfe/realized/_types.py +73 -0
- mfe/realized/covariance.py +217 -0
- mfe/realized/jumps.py +76 -0
- mfe/realized/kernel.py +208 -0
- mfe/realized/multivariate_kernel.py +133 -0
- mfe/realized/noise.py +67 -0
- mfe/realized/quantile_var.py +123 -0
- mfe/realized/quarticity.py +44 -0
- mfe/realized/range_.py +131 -0
- mfe/realized/sampling.py +205 -0
- mfe/realized/tsrv.py +178 -0
- mfe/realized/variance.py +256 -0
- mfe/tests_stat/__init__.py +24 -0
- mfe/tests_stat/arch_lm.py +95 -0
- mfe/tests_stat/forecast_eval.py +186 -0
- mfe/tests_stat/serial.py +150 -0
- mfe/timeseries/__init__.py +18 -0
- mfe/timeseries/beveridge_nelson.py +289 -0
- mfe/timeseries/var.py +599 -0
- mfe/univariate/__init__.py +15 -0
- mfe/univariate/har.py +225 -0
- mfe/univariate/heavy.py +332 -0
- mfe/utils/__init__.py +16 -0
- mfe/utils/lags.py +96 -0
- mfe/utils/typing.py +27 -0
- mfe/utils/vcv.py +73 -0
- mfe_toolbox-0.1.0.dist-info/METADATA +84 -0
- mfe_toolbox-0.1.0.dist-info/RECORD +55 -0
- mfe_toolbox-0.1.0.dist-info/WHEEL +4 -0
mfe/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
mfe — Python port of the MFE Toolbox (Kevin Sheppard / Oxford MFE).
|
|
3
|
+
|
|
4
|
+
Submodules
|
|
5
|
+
----------
|
|
6
|
+
realized Realized volatility measures (kernel, BPV, preaveraged, etc.)
|
|
7
|
+
univariate HAR-RV, HEAVY, and thin wrappers around arch package
|
|
8
|
+
multivariate DCC, BEKK, CCC, GO-GARCH
|
|
9
|
+
timeseries ARMA, VAR, HAR mean models
|
|
10
|
+
bootstrap Wild bootstrap, stationary bootstrap extensions
|
|
11
|
+
crosssection Fama-MacBeth, Newey-West, PCA
|
|
12
|
+
distributions GED, Skew-t, stable distributions
|
|
13
|
+
tests_stat Jump tests, ARCH-LM, forecast evaluation (MZ, DM, SPA)
|
|
14
|
+
utils Lag matrices, robust VCV, typing aliases
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
__version__ = "0.1.0"
|
|
18
|
+
__author__ = "Gabin Taibi"
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
mfe.bootstrap — Dependent-data bootstrap methods.
|
|
3
|
+
|
|
4
|
+
wild_bootstrap_rv Wild bootstrap CI for realized volatility statistics
|
|
5
|
+
wild_bootstrap_test Wild bootstrap p-value for hypothesis tests
|
|
6
|
+
spa_test Hansen (2005) Superior Predictive Ability test
|
|
7
|
+
step_m Romano-Wolf (2005) Stepdown Multiple Hypothesis Test (FWER control)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from mfe.bootstrap.wild import wild_bootstrap_rv, wild_bootstrap_test, WildBootstrapResult
|
|
11
|
+
from mfe.bootstrap.spa import spa_test, SPAResult
|
|
12
|
+
from mfe.bootstrap.stepM import step_m, StepMResult
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"wild_bootstrap_rv", "wild_bootstrap_test", "WildBootstrapResult",
|
|
16
|
+
"spa_test", "SPAResult",
|
|
17
|
+
"step_m", "StepMResult",
|
|
18
|
+
]
|
mfe/bootstrap/spa.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Superior Predictive Ability (SPA) test — Hansen (2005).
|
|
3
|
+
|
|
4
|
+
Hansen, P.R. (2005): "A Test for Superior Predictive Ability",
|
|
5
|
+
Journal of Business & Economic Statistics, 23(4), 365-380.
|
|
6
|
+
|
|
7
|
+
Also implements White (2000) Reality Check as a special case.
|
|
8
|
+
|
|
9
|
+
Setup
|
|
10
|
+
-----
|
|
11
|
+
Given M models and a benchmark, let d_{k,t} = L(y_t, f_{0,t}) - L(y_t, f_{k,t})
|
|
12
|
+
be the loss differential at time t for model k vs. benchmark (model 0).
|
|
13
|
+
d_{k,t} > 0 means model k is better than benchmark at time t.
|
|
14
|
+
|
|
15
|
+
H0: max_k E[d_{k,t}] <= 0 (no model beats the benchmark on average)
|
|
16
|
+
H1: max_k E[d_{k,t}] > 0 (at least one model is strictly better)
|
|
17
|
+
|
|
18
|
+
Test statistic
|
|
19
|
+
--------------
|
|
20
|
+
T_SPA = max_k ( sqrt(T) * d_bar_k / sigma_k )
|
|
21
|
+
where d_bar_k = mean(d_{k,t}) and sigma_k^2 is the long-run variance of d_{k,t}.
|
|
22
|
+
|
|
23
|
+
Under H0, T_SPA has a distribution that depends on the correlation structure of
|
|
24
|
+
{d_{k,t}} across k. P-values are computed by the stationary bootstrap.
|
|
25
|
+
|
|
26
|
+
Hansen's SPA uses a "studentized" version with three variants of the null:
|
|
27
|
+
- "consistent" (default): removes irrelevant models from the null (d_bar_k << 0)
|
|
28
|
+
- "upper": retains all models (equivalent to White's Reality Check)
|
|
29
|
+
- "lower": most conservative, all models treated as tied with benchmark
|
|
30
|
+
|
|
31
|
+
References
|
|
32
|
+
----------
|
|
33
|
+
White, H. (2000): "A Reality Check for Data Snooping", Econometrica.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
|
|
40
|
+
import numpy as np
|
|
41
|
+
from scipy import stats
|
|
42
|
+
|
|
43
|
+
from mfe.utils.typing import FloatArray
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class SPAResult:
|
|
48
|
+
statistic: float # max_k t_k = max_k sqrt(T)*d_bar_k/sigma_k
|
|
49
|
+
p_value_consistent: float # Hansen consistent p-value (default to report)
|
|
50
|
+
p_value_upper: float # White Reality Check p-value
|
|
51
|
+
p_value_lower: float # lower bound p-value
|
|
52
|
+
d_bar: FloatArray # (M,) mean loss differentials
|
|
53
|
+
t_stats: FloatArray # (M,) studentized stats per model
|
|
54
|
+
n_models: int
|
|
55
|
+
n_obs: int
|
|
56
|
+
n_bootstrap: int
|
|
57
|
+
bootstrap_distribution: FloatArray = field(default_factory=lambda: np.array([]))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _stationary_bootstrap_indices(
|
|
61
|
+
T: int,
|
|
62
|
+
n_boot: int,
|
|
63
|
+
avg_block_len: float,
|
|
64
|
+
rng: np.random.Generator,
|
|
65
|
+
) -> np.ndarray:
|
|
66
|
+
"""
|
|
67
|
+
Stationary bootstrap of Politis & Romano (1994).
|
|
68
|
+
Returns (n_boot, T) array of resample indices.
|
|
69
|
+
p = 1/avg_block_len is the geometric block-end probability.
|
|
70
|
+
"""
|
|
71
|
+
p = 1.0 / avg_block_len
|
|
72
|
+
idx = np.empty((n_boot, T), dtype=np.int64)
|
|
73
|
+
for b in range(n_boot):
|
|
74
|
+
i = rng.integers(0, T)
|
|
75
|
+
for t in range(T):
|
|
76
|
+
idx[b, t] = i
|
|
77
|
+
if rng.random() < p:
|
|
78
|
+
i = rng.integers(0, T)
|
|
79
|
+
else:
|
|
80
|
+
i = (i + 1) % T
|
|
81
|
+
return idx
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _long_run_variance(
|
|
85
|
+
d: FloatArray,
|
|
86
|
+
bandwidth: int | None = None,
|
|
87
|
+
) -> FloatArray:
|
|
88
|
+
"""
|
|
89
|
+
Newey-West long-run variance for each column of d.
|
|
90
|
+
Returns (M,) array of sigma_k^2.
|
|
91
|
+
"""
|
|
92
|
+
T, M = d.shape
|
|
93
|
+
if bandwidth is None:
|
|
94
|
+
bandwidth = int(np.ceil(1.2 * T ** (1 / 3)))
|
|
95
|
+
|
|
96
|
+
sigma2 = np.empty(M, dtype=np.float64)
|
|
97
|
+
for k in range(M):
|
|
98
|
+
dk = d[:, k] - d[:, k].mean()
|
|
99
|
+
var = float(dk @ dk) / T
|
|
100
|
+
for lag in range(1, bandwidth + 1):
|
|
101
|
+
w = 1.0 - lag / (bandwidth + 1)
|
|
102
|
+
gamma = float(dk[lag:] @ dk[:T - lag]) / T
|
|
103
|
+
var += 2 * w * gamma
|
|
104
|
+
sigma2[k] = max(var, 1e-30)
|
|
105
|
+
|
|
106
|
+
return sigma2
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def spa_test(
|
|
110
|
+
loss_benchmark: FloatArray,
|
|
111
|
+
loss_models: FloatArray,
|
|
112
|
+
n_bootstrap: int = 999,
|
|
113
|
+
avg_block_len: float | None = None,
|
|
114
|
+
bandwidth: int | None = None,
|
|
115
|
+
rng: np.random.Generator | None = None,
|
|
116
|
+
) -> SPAResult:
|
|
117
|
+
"""
|
|
118
|
+
Hansen (2005) Superior Predictive Ability test.
|
|
119
|
+
|
|
120
|
+
Parameters
|
|
121
|
+
----------
|
|
122
|
+
loss_benchmark : (T,) loss series for the benchmark model (lower = better)
|
|
123
|
+
loss_models : (T, M) loss series for M alternative models
|
|
124
|
+
n_bootstrap : stationary bootstrap replications (default 999)
|
|
125
|
+
avg_block_len : average block length for stationary bootstrap;
|
|
126
|
+
if None, uses T^{1/3}
|
|
127
|
+
bandwidth : Newey-West bandwidth for long-run variance;
|
|
128
|
+
if None, uses 1.2 * T^{1/3}
|
|
129
|
+
rng : numpy Generator; if None uses default_rng()
|
|
130
|
+
|
|
131
|
+
Returns
|
|
132
|
+
-------
|
|
133
|
+
SPAResult
|
|
134
|
+
Report .p_value_consistent for the standard SPA p-value.
|
|
135
|
+
Report .p_value_upper for White's Reality Check p-value.
|
|
136
|
+
|
|
137
|
+
Notes
|
|
138
|
+
-----
|
|
139
|
+
Loss convention: LOWER is BETTER (e.g. MSE, MAE, negative log-lik).
|
|
140
|
+
Loss differential d_{k,t} = L_benchmark_t - L_model_k_t.
|
|
141
|
+
Positive d_bar_k means model k beats benchmark on average.
|
|
142
|
+
"""
|
|
143
|
+
lb = np.asarray(loss_benchmark, dtype=np.float64)
|
|
144
|
+
lm = np.asarray(loss_models, dtype=np.float64)
|
|
145
|
+
if lm.ndim == 1:
|
|
146
|
+
lm = lm[:, None]
|
|
147
|
+
|
|
148
|
+
T, M = lm.shape
|
|
149
|
+
if rng is None:
|
|
150
|
+
rng = np.random.default_rng()
|
|
151
|
+
if avg_block_len is None:
|
|
152
|
+
avg_block_len = max(2.0, float(T ** (1 / 3)))
|
|
153
|
+
|
|
154
|
+
# Loss differentials: d_{k,t} = L_bench_t - L_model_k_t
|
|
155
|
+
# d_bar_k > 0 means model k is better than benchmark
|
|
156
|
+
d = lb[:, None] - lm # (T, M)
|
|
157
|
+
|
|
158
|
+
d_bar = d.mean(axis=0) # (M,)
|
|
159
|
+
|
|
160
|
+
# Long-run variance
|
|
161
|
+
sigma2 = _long_run_variance(d, bandwidth=bandwidth)
|
|
162
|
+
sigma = np.sqrt(sigma2)
|
|
163
|
+
|
|
164
|
+
# Studentized statistics: t_k = sqrt(T) * d_bar_k / sigma_k
|
|
165
|
+
t_stats = np.sqrt(T) * d_bar / sigma # (M,)
|
|
166
|
+
T_spa = float(np.max(t_stats))
|
|
167
|
+
|
|
168
|
+
# Stationary bootstrap to get null distribution of max t_k
|
|
169
|
+
idx = _stationary_bootstrap_indices(T, n_bootstrap, avg_block_len, rng)
|
|
170
|
+
|
|
171
|
+
# Three null variants of Hansen (2005)
|
|
172
|
+
# "consistent": zero out models with strongly negative d_bar (irrelevant)
|
|
173
|
+
# "upper": keep all models (White RC)
|
|
174
|
+
# "lower": only models with d_bar > 0 (conservative)
|
|
175
|
+
|
|
176
|
+
# Threshold for "consistent": c_k = max(d_bar_k, -sqrt(sigma2_k * log(log(T)) / T))
|
|
177
|
+
c_consistent = np.maximum(d_bar, -np.sqrt(sigma2 * np.log(np.log(T)) / T))
|
|
178
|
+
c_upper = d_bar.copy() # White Reality Check: mean-center on d_bar
|
|
179
|
+
c_lower = np.maximum(d_bar, 0.0)
|
|
180
|
+
|
|
181
|
+
boot_max_consistent = np.empty(n_bootstrap, dtype=np.float64)
|
|
182
|
+
boot_max_upper = np.empty(n_bootstrap, dtype=np.float64)
|
|
183
|
+
boot_max_lower = np.empty(n_bootstrap, dtype=np.float64)
|
|
184
|
+
|
|
185
|
+
d_centered_consistent = d - c_consistent[None, :]
|
|
186
|
+
d_centered_upper = d - c_upper[None, :]
|
|
187
|
+
d_centered_lower = d - c_lower[None, :]
|
|
188
|
+
|
|
189
|
+
for b in range(n_bootstrap):
|
|
190
|
+
d_boot_c = d_centered_consistent[idx[b]].mean(axis=0)
|
|
191
|
+
d_boot_u = d_centered_upper[idx[b]].mean(axis=0)
|
|
192
|
+
d_boot_l = d_centered_lower[idx[b]].mean(axis=0)
|
|
193
|
+
|
|
194
|
+
t_boot_c = np.sqrt(T) * d_boot_c / sigma
|
|
195
|
+
t_boot_u = np.sqrt(T) * d_boot_u / sigma
|
|
196
|
+
t_boot_l = np.sqrt(T) * d_boot_l / sigma
|
|
197
|
+
|
|
198
|
+
boot_max_consistent[b] = float(np.max(t_boot_c))
|
|
199
|
+
boot_max_upper[b] = float(np.max(t_boot_u))
|
|
200
|
+
boot_max_lower[b] = float(np.max(t_boot_l))
|
|
201
|
+
|
|
202
|
+
pval_consistent = float(np.mean(boot_max_consistent >= T_spa))
|
|
203
|
+
pval_upper = float(np.mean(boot_max_upper >= T_spa))
|
|
204
|
+
pval_lower = float(np.mean(boot_max_lower >= T_spa))
|
|
205
|
+
|
|
206
|
+
return SPAResult(
|
|
207
|
+
statistic=T_spa,
|
|
208
|
+
p_value_consistent=pval_consistent,
|
|
209
|
+
p_value_upper=pval_upper,
|
|
210
|
+
p_value_lower=pval_lower,
|
|
211
|
+
d_bar=d_bar,
|
|
212
|
+
t_stats=t_stats,
|
|
213
|
+
n_models=M,
|
|
214
|
+
n_obs=T,
|
|
215
|
+
n_bootstrap=n_bootstrap,
|
|
216
|
+
bootstrap_distribution=boot_max_consistent,
|
|
217
|
+
)
|
mfe/bootstrap/stepM.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""
|
|
2
|
+
StepM: Stepdown Multiple Hypothesis Testing with FWER control.
|
|
3
|
+
|
|
4
|
+
Romano, J.P. & Wolf, M. (2005): "Stepwise Multiple Testing as Formalized Data
|
|
5
|
+
Snooping", Econometrica, 73(4), 1237-1282.
|
|
6
|
+
|
|
7
|
+
Setup
|
|
8
|
+
-----
|
|
9
|
+
M null hypotheses H_k: mu_k <= 0 for k = 1..M, where mu_k = E[d_{k,t}] is the
|
|
10
|
+
mean performance differential of model k vs. the benchmark.
|
|
11
|
+
|
|
12
|
+
The algorithm controls the familywise error rate (FWER):
|
|
13
|
+
FWER = P(reject at least one true H_k) <= alpha
|
|
14
|
+
|
|
15
|
+
This is more powerful than Bonferroni and more interpretable than SPA:
|
|
16
|
+
it returns which models are significantly better, not just whether any is.
|
|
17
|
+
|
|
18
|
+
Algorithm (Algorithm 4.1 of Romano & Wolf 2005)
|
|
19
|
+
------------------------------------------------
|
|
20
|
+
1. Start with all M models.
|
|
21
|
+
2. Compute test statistics t_k = sqrt(T) * d_bar_k / sigma_k.
|
|
22
|
+
3. Use the stationary bootstrap to get the joint null distribution of
|
|
23
|
+
max_k t_k (over the remaining models).
|
|
24
|
+
4. Reject the model with the largest t_k if it exceeds the bootstrap
|
|
25
|
+
critical value at level alpha.
|
|
26
|
+
5. Remove rejected models from the set and repeat.
|
|
27
|
+
6. Stop when no more rejections occur.
|
|
28
|
+
|
|
29
|
+
The result is a set of models significantly better than the benchmark.
|
|
30
|
+
|
|
31
|
+
This matches the MFE MATLAB implementation under bootstrap/stepm.m.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
from dataclasses import dataclass, field
|
|
37
|
+
|
|
38
|
+
import numpy as np
|
|
39
|
+
|
|
40
|
+
from mfe.utils.typing import FloatArray
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class StepMResult:
|
|
45
|
+
"""StepM multiple hypothesis testing result."""
|
|
46
|
+
rejected: list[int] # indices (0-based) of models that significantly beat benchmark
|
|
47
|
+
accepted: list[int] # indices that could not be rejected (H0 not rejected)
|
|
48
|
+
t_stats: FloatArray # (M,) studentized statistics for all models
|
|
49
|
+
p_values_raw: FloatArray # (M,) unadjusted p-values
|
|
50
|
+
p_values_adjusted: FloatArray # (M,) FWER-adjusted p-values (stepdown)
|
|
51
|
+
n_models: int
|
|
52
|
+
n_obs: int
|
|
53
|
+
n_bootstrap: int
|
|
54
|
+
alpha: float
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _stationary_bootstrap(
|
|
58
|
+
d: FloatArray,
|
|
59
|
+
n_boot: int,
|
|
60
|
+
avg_block_len: float,
|
|
61
|
+
rng: np.random.Generator,
|
|
62
|
+
) -> np.ndarray:
|
|
63
|
+
"""(n_boot, T, M) bootstrap resamples of the centered loss differentials."""
|
|
64
|
+
T, M = d.shape
|
|
65
|
+
p = 1.0 / avg_block_len
|
|
66
|
+
out = np.empty((n_boot, T, M), dtype=np.float64)
|
|
67
|
+
for b in range(n_boot):
|
|
68
|
+
i = rng.integers(0, T)
|
|
69
|
+
for t in range(T):
|
|
70
|
+
out[b, t] = d[i]
|
|
71
|
+
if rng.random() < p:
|
|
72
|
+
i = rng.integers(0, T)
|
|
73
|
+
else:
|
|
74
|
+
i = (i + 1) % T
|
|
75
|
+
return out
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _long_run_std(d: FloatArray, bandwidth: int) -> FloatArray:
|
|
79
|
+
"""(M,) Newey-West long-run standard deviations."""
|
|
80
|
+
T, M = d.shape
|
|
81
|
+
sigma = np.empty(M, dtype=np.float64)
|
|
82
|
+
for k in range(M):
|
|
83
|
+
dk = d[:, k] - d[:, k].mean()
|
|
84
|
+
var = float(dk @ dk) / T
|
|
85
|
+
for lag in range(1, bandwidth + 1):
|
|
86
|
+
w = 1.0 - lag / (bandwidth + 1)
|
|
87
|
+
var += 2 * w * float(dk[lag:] @ dk[:T - lag]) / T
|
|
88
|
+
sigma[k] = max(var, 1e-30) ** 0.5
|
|
89
|
+
return sigma
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def step_m(
|
|
93
|
+
loss_benchmark: FloatArray,
|
|
94
|
+
loss_models: FloatArray,
|
|
95
|
+
alpha: float = 0.05,
|
|
96
|
+
n_bootstrap: int = 999,
|
|
97
|
+
avg_block_len: float | None = None,
|
|
98
|
+
bandwidth: int | None = None,
|
|
99
|
+
rng: np.random.Generator | None = None,
|
|
100
|
+
) -> StepMResult:
|
|
101
|
+
"""
|
|
102
|
+
Romano-Wolf StepM stepdown multiple hypothesis test.
|
|
103
|
+
|
|
104
|
+
Tests H_k: E[L_bench - L_model_k] <= 0 for k = 1..M.
|
|
105
|
+
Rejects H_k (model k beats benchmark) for k in result.rejected.
|
|
106
|
+
Controls FWER <= alpha across all M tests.
|
|
107
|
+
|
|
108
|
+
Parameters
|
|
109
|
+
----------
|
|
110
|
+
loss_benchmark : (T,) benchmark loss series (lower = better)
|
|
111
|
+
loss_models : (T, M) alternative model loss series
|
|
112
|
+
alpha : familywise error rate (default 0.05)
|
|
113
|
+
n_bootstrap : stationary bootstrap replications
|
|
114
|
+
avg_block_len : average block length; if None uses T^{1/3}
|
|
115
|
+
bandwidth : Newey-West bandwidth; if None uses 1.2 * T^{1/3}
|
|
116
|
+
rng : random generator
|
|
117
|
+
|
|
118
|
+
Returns
|
|
119
|
+
-------
|
|
120
|
+
StepMResult
|
|
121
|
+
.rejected — 0-based indices of models significantly beating benchmark
|
|
122
|
+
.accepted — the rest
|
|
123
|
+
"""
|
|
124
|
+
lb = np.asarray(loss_benchmark, dtype=np.float64)
|
|
125
|
+
lm = np.asarray(loss_models, dtype=np.float64)
|
|
126
|
+
if lm.ndim == 1:
|
|
127
|
+
lm = lm[:, None]
|
|
128
|
+
|
|
129
|
+
T, M = lm.shape
|
|
130
|
+
if rng is None:
|
|
131
|
+
rng = np.random.default_rng()
|
|
132
|
+
if avg_block_len is None:
|
|
133
|
+
avg_block_len = max(2.0, T ** (1 / 3))
|
|
134
|
+
if bandwidth is None:
|
|
135
|
+
bandwidth = max(1, int(1.2 * T ** (1 / 3)))
|
|
136
|
+
|
|
137
|
+
# Loss differentials: d_{k,t} = L_bench_t - L_model_k_t
|
|
138
|
+
d = lb[:, None] - lm # (T, M)
|
|
139
|
+
d_bar = d.mean(axis=0) # (M,)
|
|
140
|
+
sigma = _long_run_std(d, bandwidth) # (M,)
|
|
141
|
+
t_stats = np.sqrt(T) * d_bar / sigma # (M,)
|
|
142
|
+
|
|
143
|
+
# Unadjusted p-values (individual, no FWER control)
|
|
144
|
+
# Use bootstrap max distribution over all M models
|
|
145
|
+
boot = _stationary_bootstrap(d - d_bar[None, :], n_bootstrap, avg_block_len, rng)
|
|
146
|
+
# boot: (n_boot, T, M) — resampled centered loss diffs
|
|
147
|
+
|
|
148
|
+
p_raw = np.empty(M, dtype=np.float64)
|
|
149
|
+
boot_max_all = (np.sqrt(T) * boot.mean(axis=1) / sigma[None, :]).max(axis=1)
|
|
150
|
+
for k in range(M):
|
|
151
|
+
boot_k = np.sqrt(T) * boot[:, :, k].mean(axis=1) / sigma[k]
|
|
152
|
+
p_raw[k] = float(np.mean(boot_k >= t_stats[k]))
|
|
153
|
+
|
|
154
|
+
# Stepdown procedure
|
|
155
|
+
remaining = list(range(M))
|
|
156
|
+
rejected = []
|
|
157
|
+
p_adjusted = np.ones(M, dtype=np.float64)
|
|
158
|
+
|
|
159
|
+
step = 0
|
|
160
|
+
while remaining:
|
|
161
|
+
# Bootstrap max over remaining models
|
|
162
|
+
boot_max = np.empty(n_bootstrap, dtype=np.float64)
|
|
163
|
+
for b in range(n_bootstrap):
|
|
164
|
+
t_boot_remaining = np.sqrt(T) * boot[b, :, :][:, remaining].mean(axis=0) / sigma[remaining]
|
|
165
|
+
boot_max[b] = float(np.max(t_boot_remaining))
|
|
166
|
+
|
|
167
|
+
# Critical value at level alpha
|
|
168
|
+
cv = float(np.quantile(boot_max, 1 - alpha))
|
|
169
|
+
|
|
170
|
+
# Find the model with max t_stat among remaining
|
|
171
|
+
t_remaining = t_stats[remaining]
|
|
172
|
+
max_idx_in_remaining = int(np.argmax(t_remaining))
|
|
173
|
+
max_k = remaining[max_idx_in_remaining]
|
|
174
|
+
max_t = float(t_stats[max_k])
|
|
175
|
+
|
|
176
|
+
if max_t > cv:
|
|
177
|
+
# Reject this model
|
|
178
|
+
p_adjusted[max_k] = float(np.mean(boot_max >= max_t))
|
|
179
|
+
rejected.append(max_k)
|
|
180
|
+
remaining.remove(max_k)
|
|
181
|
+
step += 1
|
|
182
|
+
else:
|
|
183
|
+
# No more rejections possible
|
|
184
|
+
break
|
|
185
|
+
|
|
186
|
+
# Adjusted p-values for accepted: use the last step's distribution
|
|
187
|
+
# (conservative: bound by the p-value from the last step)
|
|
188
|
+
for k in remaining:
|
|
189
|
+
p_adjusted[k] = float(np.mean(boot_max_all >= t_stats[k]))
|
|
190
|
+
|
|
191
|
+
# Monotonise: stepdown p-values must be non-decreasing when sorted by t_stat descending
|
|
192
|
+
order = np.argsort(-t_stats)
|
|
193
|
+
p_mono = p_adjusted[order].copy()
|
|
194
|
+
for i in range(1, M):
|
|
195
|
+
p_mono[i] = max(p_mono[i], p_mono[i - 1])
|
|
196
|
+
p_adjusted[order] = p_mono
|
|
197
|
+
|
|
198
|
+
accepted = [k for k in range(M) if k not in rejected]
|
|
199
|
+
|
|
200
|
+
return StepMResult(
|
|
201
|
+
rejected=sorted(rejected),
|
|
202
|
+
accepted=sorted(accepted),
|
|
203
|
+
t_stats=t_stats,
|
|
204
|
+
p_values_raw=p_raw,
|
|
205
|
+
p_values_adjusted=p_adjusted,
|
|
206
|
+
n_models=M,
|
|
207
|
+
n_obs=T,
|
|
208
|
+
n_bootstrap=n_bootstrap,
|
|
209
|
+
alpha=alpha,
|
|
210
|
+
)
|
mfe/bootstrap/wild.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Wild bootstrap for realized volatility and related statistics.
|
|
3
|
+
|
|
4
|
+
Gonçalves, S. & Meddahi, N. (2009): "Bootstrapping Realized Volatility",
|
|
5
|
+
Econometrica, 77(1), 283-306.
|
|
6
|
+
|
|
7
|
+
The wild bootstrap resamples by multiplying each squared return by an i.i.d.
|
|
8
|
+
multiplier w_t drawn from a two-point distribution that matches the first
|
|
9
|
+
two moments of the standard normal.
|
|
10
|
+
|
|
11
|
+
This is appropriate for realized volatility statistics because:
|
|
12
|
+
1. The squared-return sequence has heterogeneous conditional variance.
|
|
13
|
+
2. Block resampling destroys the i.i.d.-ness of squared returns under the null.
|
|
14
|
+
3. The wild bootstrap is consistent for RV-based test statistics even in the
|
|
15
|
+
presence of microstructure noise (with appropriate pre-averaging).
|
|
16
|
+
|
|
17
|
+
Two-point Rademacher multiplier: w_t = +1 or -1 with prob 1/2.
|
|
18
|
+
Mammen (1993) multiplier: w_t = -(sqrt(5)-1)/2 or (sqrt(5)+1)/2.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from typing import Callable
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
|
|
28
|
+
from mfe.utils.typing import FloatArray
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
MultiplierType = type # "rademacher" | "mammen" | "normal"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _rademacher(n: int, rng: np.random.Generator) -> FloatArray:
|
|
35
|
+
"""w_t in {-1, +1} with prob 1/2."""
|
|
36
|
+
return rng.choice([-1.0, 1.0], size=n).astype(np.float64)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _mammen(n: int, rng: np.random.Generator) -> FloatArray:
|
|
40
|
+
"""Mammen (1993) two-point: matches first three moments of N(0,1)."""
|
|
41
|
+
sqrt5 = np.sqrt(5.0)
|
|
42
|
+
p = (sqrt5 + 1) / (2 * sqrt5)
|
|
43
|
+
a = -(sqrt5 - 1) / 2
|
|
44
|
+
b = (sqrt5 + 1) / 2
|
|
45
|
+
u = rng.random(n)
|
|
46
|
+
return np.where(u < p, a, b).astype(np.float64)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _normal(n: int, rng: np.random.Generator) -> FloatArray:
|
|
50
|
+
return rng.standard_normal(n)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_MULTIPLIER_FUNCS = {
|
|
54
|
+
"rademacher": _rademacher,
|
|
55
|
+
"mammen": _mammen,
|
|
56
|
+
"normal": _normal,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class WildBootstrapResult:
|
|
62
|
+
"""Result from a wild bootstrap confidence interval computation."""
|
|
63
|
+
statistic: float
|
|
64
|
+
ci_lower: float
|
|
65
|
+
ci_upper: float
|
|
66
|
+
ci_level: float
|
|
67
|
+
bootstrap_distribution: FloatArray
|
|
68
|
+
n_replications: int
|
|
69
|
+
multiplier: str
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def wild_bootstrap_rv(
|
|
73
|
+
returns: FloatArray,
|
|
74
|
+
statistic_fn: Callable[[FloatArray], float] | None = None,
|
|
75
|
+
n_replications: int = 999,
|
|
76
|
+
ci_level: float = 0.95,
|
|
77
|
+
multiplier: str = "rademacher",
|
|
78
|
+
rng: np.random.Generator | None = None,
|
|
79
|
+
) -> WildBootstrapResult:
|
|
80
|
+
"""
|
|
81
|
+
Wild bootstrap confidence interval for a realized volatility statistic.
|
|
82
|
+
|
|
83
|
+
The bootstrap DGP is:
|
|
84
|
+
r_t^* = w_t * r_t
|
|
85
|
+
|
|
86
|
+
where w_t is i.i.d. from the specified multiplier distribution.
|
|
87
|
+
The statistic is re-evaluated on {r_t^*}.
|
|
88
|
+
|
|
89
|
+
Parameters
|
|
90
|
+
----------
|
|
91
|
+
returns : (T,) log-return array
|
|
92
|
+
statistic_fn : function mapping returns -> float; default is sum(r^2) (RV)
|
|
93
|
+
n_replications : number of bootstrap replications
|
|
94
|
+
ci_level : confidence level (e.g. 0.95 for 95% CI)
|
|
95
|
+
multiplier : "rademacher" | "mammen" | "normal"
|
|
96
|
+
rng : numpy random generator; if None, uses default_rng()
|
|
97
|
+
|
|
98
|
+
Returns
|
|
99
|
+
-------
|
|
100
|
+
WildBootstrapResult
|
|
101
|
+
"""
|
|
102
|
+
r = np.asarray(returns, dtype=np.float64)
|
|
103
|
+
T = len(r)
|
|
104
|
+
|
|
105
|
+
if rng is None:
|
|
106
|
+
rng = np.random.default_rng()
|
|
107
|
+
|
|
108
|
+
if statistic_fn is None:
|
|
109
|
+
def statistic_fn(x: FloatArray) -> float:
|
|
110
|
+
return float(np.sum(x ** 2))
|
|
111
|
+
|
|
112
|
+
mult_func = _MULTIPLIER_FUNCS.get(multiplier)
|
|
113
|
+
if mult_func is None:
|
|
114
|
+
raise ValueError(f"multiplier must be one of {list(_MULTIPLIER_FUNCS)}, got '{multiplier}'")
|
|
115
|
+
|
|
116
|
+
# Point estimate
|
|
117
|
+
stat0 = statistic_fn(r)
|
|
118
|
+
|
|
119
|
+
# Bootstrap distribution
|
|
120
|
+
boot_stats = np.empty(n_replications, dtype=np.float64)
|
|
121
|
+
for b in range(n_replications):
|
|
122
|
+
w = mult_func(T, rng)
|
|
123
|
+
r_star = w * r
|
|
124
|
+
boot_stats[b] = statistic_fn(r_star)
|
|
125
|
+
|
|
126
|
+
alpha = 1.0 - ci_level
|
|
127
|
+
ci_lo = float(np.percentile(boot_stats, 100 * alpha / 2))
|
|
128
|
+
ci_hi = float(np.percentile(boot_stats, 100 * (1 - alpha / 2)))
|
|
129
|
+
|
|
130
|
+
return WildBootstrapResult(
|
|
131
|
+
statistic=stat0,
|
|
132
|
+
ci_lower=ci_lo,
|
|
133
|
+
ci_upper=ci_hi,
|
|
134
|
+
ci_level=ci_level,
|
|
135
|
+
bootstrap_distribution=boot_stats,
|
|
136
|
+
n_replications=n_replications,
|
|
137
|
+
multiplier=multiplier,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def wild_bootstrap_test(
|
|
142
|
+
returns: FloatArray,
|
|
143
|
+
null_statistic: float,
|
|
144
|
+
statistic_fn: Callable[[FloatArray], float] | None = None,
|
|
145
|
+
n_replications: int = 999,
|
|
146
|
+
multiplier: str = "rademacher",
|
|
147
|
+
rng: np.random.Generator | None = None,
|
|
148
|
+
) -> tuple[float, float]:
|
|
149
|
+
"""
|
|
150
|
+
Wild bootstrap p-value for a two-sided hypothesis test.
|
|
151
|
+
|
|
152
|
+
Parameters
|
|
153
|
+
----------
|
|
154
|
+
null_statistic : the value of the statistic under the null hypothesis
|
|
155
|
+
|
|
156
|
+
Returns
|
|
157
|
+
-------
|
|
158
|
+
(observed_statistic, bootstrap_p_value)
|
|
159
|
+
"""
|
|
160
|
+
result = wild_bootstrap_rv(
|
|
161
|
+
returns,
|
|
162
|
+
statistic_fn=statistic_fn,
|
|
163
|
+
n_replications=n_replications,
|
|
164
|
+
multiplier=multiplier,
|
|
165
|
+
rng=rng,
|
|
166
|
+
)
|
|
167
|
+
p_val = float(np.mean(np.abs(result.bootstrap_distribution - null_statistic) >=
|
|
168
|
+
abs(result.statistic - null_statistic)))
|
|
169
|
+
return result.statistic, p_val
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""
|
|
2
|
+
mfe.crosssection — Cross-sectional econometrics.
|
|
3
|
+
|
|
4
|
+
ols OLS with White heteroskedastic SEs
|
|
5
|
+
olsnw OLS with Newey-West HAC SEs
|
|
6
|
+
fama_macbeth Two-pass FM regression with Shanken correction
|
|
7
|
+
rolling_betas Rolling time-series betas for FM pass 1
|
|
8
|
+
pca Principal component analysis with factor interpretation
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from mfe.crosssection.ols import ols, olsnw, OLSResult
|
|
12
|
+
from mfe.crosssection.fm import fama_macbeth, rolling_betas, FMResult
|
|
13
|
+
from mfe.crosssection.pca import pca, PCAResult
|
|
14
|
+
|
|
15
|
+
__all__ = ["ols", "olsnw", "OLSResult", "fama_macbeth", "rolling_betas", "FMResult", "pca", "PCAResult"]
|