corrscore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corrscore/__init__.py +21 -0
- corrscore/backtest.py +148 -0
- corrscore/bootstrap.py +81 -0
- corrscore/diebold_mariano.py +129 -0
- corrscore/mcs.py +143 -0
- corrscore/py.typed +0 -0
- corrscore/scoring.py +373 -0
- corrscore/utils.py +37 -0
- corrscore-0.1.0.dist-info/METADATA +129 -0
- corrscore-0.1.0.dist-info/RECORD +12 -0
- corrscore-0.1.0.dist-info/WHEEL +4 -0
- corrscore-0.1.0.dist-info/licenses/LICENSE +21 -0
corrscore/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
from .backtest import BacktestResult, backtest_zero_overlap
|
|
2
|
+
from .bootstrap import BootstrapResult, circular_block_bootstrap
|
|
3
|
+
from .diebold_mariano import DieboldMarianoResult, diebold_mariano
|
|
4
|
+
from .mcs import MCSResult, model_confidence_set
|
|
5
|
+
from .scoring import matrix_energy_score, matrix_geodesic_variogram_score, matrix_variogram_score
|
|
6
|
+
from .utils import asymmetric_weighted_mean
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"matrix_energy_score",
|
|
10
|
+
"matrix_variogram_score",
|
|
11
|
+
"matrix_geodesic_variogram_score",
|
|
12
|
+
"backtest_zero_overlap",
|
|
13
|
+
"BacktestResult",
|
|
14
|
+
"circular_block_bootstrap",
|
|
15
|
+
"BootstrapResult",
|
|
16
|
+
"diebold_mariano",
|
|
17
|
+
"DieboldMarianoResult",
|
|
18
|
+
"model_confidence_set",
|
|
19
|
+
"MCSResult",
|
|
20
|
+
"asymmetric_weighted_mean",
|
|
21
|
+
]
|
corrscore/backtest.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""The zero-overlap walk-forward backtest driver.
|
|
2
|
+
|
|
3
|
+
Design note: an earlier sketch of this API had a `window` parameter
|
|
4
|
+
defaulting to `horizon`. Working through the actual mechanics precisely
|
|
5
|
+
during implementation surfaced a cleaner, more honestly-scoped contract,
|
|
6
|
+
documented here rather than silently substituted. This harness does not,
|
|
7
|
+
and cannot, police how much history a caller's `forecast_fn` consults
|
|
8
|
+
internally -- that model is opaque to the harness (it might be a
|
|
9
|
+
full-history discounted filter, a short trailing window, or anything
|
|
10
|
+
else), exactly the same responsibility boundary scikit-learn's
|
|
11
|
+
`TimeSeriesSplit` leaves to the caller. What the harness genuinely *can*
|
|
12
|
+
and does enforce, unconditionally, is that `ground_truth_fn` is always
|
|
13
|
+
called with a start point strictly after the origin
|
|
14
|
+
(`origin + 1 + purge_gap`), never a window that reaches back before or
|
|
15
|
+
across it. That guards against a real class of bug this package exists
|
|
16
|
+
to prevent: computing "ground truth" as a window *ending* near the
|
|
17
|
+
origin rather than a window *starting* strictly after it, which silently
|
|
18
|
+
leaks estimation-window information into the evaluation. This API makes
|
|
19
|
+
that specific mistake structurally impossible to reproduce, since the
|
|
20
|
+
caller never controls the start point passed to `ground_truth_fn`.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from typing import Any, Callable, Mapping, NamedTuple, Sequence
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
import numpy.typing as npt
|
|
28
|
+
|
|
29
|
+
from .scoring import matrix_energy_score
|
|
30
|
+
|
|
31
|
+
ForecastFn = Callable[[int], Mapping[str, Any]]
|
|
32
|
+
GroundTruthFn = Callable[[int, int], npt.ArrayLike]
|
|
33
|
+
ScoreFn = Callable[[Mapping[str, Any], npt.ArrayLike], float]
|
|
34
|
+
SeverityFn = Callable[[npt.NDArray[np.float64]], float]
|
|
35
|
+
|
|
36
|
+
__all__ = ["BacktestResult", "backtest_zero_overlap"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class BacktestResult(NamedTuple):
|
|
40
|
+
"""Result of `backtest_zero_overlap`.
|
|
41
|
+
|
|
42
|
+
Attributes
|
|
43
|
+
----------
|
|
44
|
+
origins : list of int
|
|
45
|
+
The origins actually scored, in order.
|
|
46
|
+
scores : dict of str -> ndarray
|
|
47
|
+
Per-model, per-origin scores, one array per key of the
|
|
48
|
+
`forecast_fns` mapping passed in, aligned with `origins`.
|
|
49
|
+
severity : ndarray or None
|
|
50
|
+
Per-origin realized-severity values (`severity_fn(y)` at each
|
|
51
|
+
origin), or None if `severity_fn` was not supplied. Intended for
|
|
52
|
+
`corrscore.asymmetric_weighted_mean`.
|
|
53
|
+
horizon : int
|
|
54
|
+
purge_gap : int
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
origins: list[int]
|
|
58
|
+
scores: dict[str, npt.NDArray[np.float64]]
|
|
59
|
+
severity: npt.NDArray[np.float64] | None
|
|
60
|
+
horizon: int
|
|
61
|
+
purge_gap: int
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def backtest_zero_overlap(
|
|
65
|
+
forecast_fns: Mapping[str, ForecastFn] | ForecastFn,
|
|
66
|
+
ground_truth_fn: GroundTruthFn,
|
|
67
|
+
origins: Sequence[int],
|
|
68
|
+
horizon: int,
|
|
69
|
+
purge_gap: int = 0,
|
|
70
|
+
score_fn: ScoreFn = matrix_energy_score,
|
|
71
|
+
severity_fn: SeverityFn | None = None,
|
|
72
|
+
) -> BacktestResult:
|
|
73
|
+
"""Score one or more forecasting methods against a proper,
|
|
74
|
+
zero-overlap-by-construction ground truth.
|
|
75
|
+
|
|
76
|
+
For each `origin` in `origins`: computes
|
|
77
|
+
`y = ground_truth_fn(origin + 1 + purge_gap, origin + 1 + purge_gap
|
|
78
|
+
+ horizon)`, then scores each `forecast_fns[name](origin)` against
|
|
79
|
+
`y` via `score_fn`. `purge_gap=0` (the default) gives zero shared
|
|
80
|
+
days between the forecast origin and the ground-truth window by
|
|
81
|
+
construction; a larger `purge_gap` is a deliberate relaxation the
|
|
82
|
+
caller must opt into explicitly -- never a silent default.
|
|
83
|
+
|
|
84
|
+
Parameters
|
|
85
|
+
----------
|
|
86
|
+
forecast_fns : callable, or dict of str -> callable
|
|
87
|
+
Each callable maps an origin (int) to a forecast dict in the
|
|
88
|
+
shape `matrix_energy_score`/`matrix_variogram_score` expect
|
|
89
|
+
(see `corrscore.scoring`). A single callable is treated as
|
|
90
|
+
`{"model": forecast_fns}`. Sharing one `ground_truth_fn` call
|
|
91
|
+
per origin across every model is deliberate: ground truth is
|
|
92
|
+
model-independent and often the more expensive computation, and
|
|
93
|
+
this shape is exactly what `corrscore.circular_block_bootstrap`,
|
|
94
|
+
`corrscore.diebold_mariano`, and `corrscore.model_confidence_set`
|
|
95
|
+
expect as input (`result.scores[name]`).
|
|
96
|
+
ground_truth_fn : callable
|
|
97
|
+
`(start, end) -> K x K matrix`. Called only with
|
|
98
|
+
`start = origin + 1 + purge_gap`, `end = start + horizon` --
|
|
99
|
+
never anything else. It is the caller's responsibility that
|
|
100
|
+
this function computes a genuinely forward-looking realized
|
|
101
|
+
estimate from `[start, end)`, not a trailing one -- the timing
|
|
102
|
+
guard above prevents overlap, but not a `ground_truth_fn` that
|
|
103
|
+
is itself defined as a trailing window.
|
|
104
|
+
origins : sequence of int
|
|
105
|
+
horizon : int
|
|
106
|
+
purge_gap : int, default=0
|
|
107
|
+
score_fn : callable, default=matrix_energy_score
|
|
108
|
+
severity_fn : callable, optional
|
|
109
|
+
`y -> float`, a per-origin realized-severity summary (e.g. mean
|
|
110
|
+
absolute off-diagonal correlation) for later use with
|
|
111
|
+
`corrscore.asymmetric_weighted_mean`.
|
|
112
|
+
|
|
113
|
+
Returns
|
|
114
|
+
-------
|
|
115
|
+
BacktestResult
|
|
116
|
+
"""
|
|
117
|
+
if callable(forecast_fns):
|
|
118
|
+
forecast_fns = {"model": forecast_fns}
|
|
119
|
+
if purge_gap < 0:
|
|
120
|
+
raise ValueError(f"purge_gap must be >= 0, got {purge_gap}")
|
|
121
|
+
if horizon < 1:
|
|
122
|
+
raise ValueError(f"horizon must be >= 1, got {horizon}")
|
|
123
|
+
|
|
124
|
+
names = list(forecast_fns.keys())
|
|
125
|
+
scores: dict[str, list[float]] = {name: [] for name in names}
|
|
126
|
+
severities: list[float] = []
|
|
127
|
+
used_origins: list[int] = []
|
|
128
|
+
|
|
129
|
+
for origin in origins:
|
|
130
|
+
start = origin + 1 + purge_gap
|
|
131
|
+
end = start + horizon
|
|
132
|
+
y = np.asarray(ground_truth_fn(start, end), dtype=float)
|
|
133
|
+
for name in names:
|
|
134
|
+
forecast = forecast_fns[name](origin)
|
|
135
|
+
scores[name].append(score_fn(forecast, y))
|
|
136
|
+
if severity_fn is not None:
|
|
137
|
+
severities.append(severity_fn(y))
|
|
138
|
+
used_origins.append(origin)
|
|
139
|
+
|
|
140
|
+
scores_arr = {name: np.array(values) for name, values in scores.items()}
|
|
141
|
+
severity_arr = np.array(severities) if severity_fn is not None else None
|
|
142
|
+
return BacktestResult(
|
|
143
|
+
origins=used_origins,
|
|
144
|
+
scores=scores_arr,
|
|
145
|
+
severity=severity_arr,
|
|
146
|
+
horizon=horizon,
|
|
147
|
+
purge_gap=purge_gap,
|
|
148
|
+
)
|
corrscore/bootstrap.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Circular block-bootstrap significance testing on paired per-origin
|
|
2
|
+
score differentials, via `arch.bootstrap.CircularBlockBootstrap` -- kept
|
|
3
|
+
as a real dependency rather than vendored, since `arch` clears this
|
|
4
|
+
package's "widely used, actively maintained, field-standard" bar for
|
|
5
|
+
reuse, and block-bootstrap correctness (edge handling, unbiased block
|
|
6
|
+
placement) carries real reimplementation risk for a "just wrap it
|
|
7
|
+
correctly" component.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import NamedTuple
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import numpy.typing as npt
|
|
15
|
+
from arch.bootstrap import CircularBlockBootstrap
|
|
16
|
+
|
|
17
|
+
__all__ = ["BootstrapResult", "circular_block_bootstrap"]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BootstrapResult(NamedTuple):
|
|
21
|
+
"""Result of `circular_block_bootstrap` for one block length.
|
|
22
|
+
|
|
23
|
+
Attributes
|
|
24
|
+
----------
|
|
25
|
+
obs : float
|
|
26
|
+
The observed mean of `scores_a - scores_b`.
|
|
27
|
+
ci_lo, ci_hi : float
|
|
28
|
+
95% percentile confidence interval from the bootstrap
|
|
29
|
+
distribution of that mean.
|
|
30
|
+
p_value : float
|
|
31
|
+
Two-sided bootstrap p-value for H0: true mean difference = 0.
|
|
32
|
+
significant : bool
|
|
33
|
+
`ci_lo > 0 or ci_hi < 0` -- whether zero falls outside the 95%
|
|
34
|
+
interval.
|
|
35
|
+
block_len : int
|
|
36
|
+
n_boot : int
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
obs: float
|
|
40
|
+
ci_lo: float
|
|
41
|
+
ci_hi: float
|
|
42
|
+
p_value: float
|
|
43
|
+
significant: bool
|
|
44
|
+
block_len: int
|
|
45
|
+
n_boot: int
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def circular_block_bootstrap(
|
|
49
|
+
scores_a: npt.ArrayLike,
|
|
50
|
+
scores_b: npt.ArrayLike,
|
|
51
|
+
block_lengths: list[int],
|
|
52
|
+
n_boot: int = 2000,
|
|
53
|
+
seed: int | np.random.Generator | None = None,
|
|
54
|
+
) -> dict[int, BootstrapResult]:
|
|
55
|
+
"""Percentile circular block bootstrap on the paired differential
|
|
56
|
+
`d_i = scores_a[i] - scores_b[i]`, swept across `block_lengths` as a
|
|
57
|
+
sensitivity check rather than relying on one automatically "optimal"
|
|
58
|
+
block length.
|
|
59
|
+
|
|
60
|
+
Returns a dict keyed by each entry of `block_lengths`. The most
|
|
61
|
+
conservative (largest-p) entry is the one worth reporting as the
|
|
62
|
+
headline result.
|
|
63
|
+
"""
|
|
64
|
+
diffs = np.asarray(scores_a, dtype=float) - np.asarray(scores_b, dtype=float)
|
|
65
|
+
results: dict[int, BootstrapResult] = {}
|
|
66
|
+
for block_len in block_lengths:
|
|
67
|
+
bs = CircularBlockBootstrap(block_len, diffs, seed=seed)
|
|
68
|
+
boot_means = bs.apply(lambda z: np.mean(z), n_boot).ravel()
|
|
69
|
+
ci_lo, ci_hi = np.percentile(boot_means, [2.5, 97.5])
|
|
70
|
+
tail_frac = min(float(np.mean(boot_means <= 0)), float(np.mean(boot_means >= 0)))
|
|
71
|
+
p_value = min(1.0, 2.0 * tail_frac)
|
|
72
|
+
results[block_len] = BootstrapResult(
|
|
73
|
+
obs=float(diffs.mean()),
|
|
74
|
+
ci_lo=float(ci_lo),
|
|
75
|
+
ci_hi=float(ci_hi),
|
|
76
|
+
p_value=p_value,
|
|
77
|
+
significant=bool(ci_lo > 0 or ci_hi < 0),
|
|
78
|
+
block_len=block_len,
|
|
79
|
+
n_boot=n_boot,
|
|
80
|
+
)
|
|
81
|
+
return results
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""The Diebold-Mariano test (Diebold & Mariano, 1995), vendored rather
|
|
2
|
+
than depending on the small, thinly-adopted `dieboldmariano` PyPI
|
|
3
|
+
package.
|
|
4
|
+
|
|
5
|
+
Deliberately mirrors R's `forecast::dm.test` (Hyndman et al.) formula
|
|
6
|
+
by formula -- not because that package is depended on, but because it
|
|
7
|
+
is the field's de facto reference implementation and this module's own
|
|
8
|
+
test suite cross-checks against its output on fixed synthetic data as a
|
|
9
|
+
development-time oracle (`tests/_reference/dm_test_oracle_values.py`),
|
|
10
|
+
per this package's "oracle, not dependency" policy. In particular: the
|
|
11
|
+
Harvey, Leybourne & Newbold (1997) small-sample correction is always
|
|
12
|
+
applied (R's `dm.test` has no toggle for it either) and the p-value
|
|
13
|
+
comes from a Student-t distribution with `n - 1` degrees of freedom, not
|
|
14
|
+
the plain asymptotic normal.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import NamedTuple
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import numpy.typing as npt
|
|
22
|
+
from scipy.stats import t as student_t
|
|
23
|
+
|
|
24
|
+
__all__ = ["DieboldMarianoResult", "diebold_mariano"]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class DieboldMarianoResult(NamedTuple):
|
|
28
|
+
"""Result of `diebold_mariano`.
|
|
29
|
+
|
|
30
|
+
Attributes
|
|
31
|
+
----------
|
|
32
|
+
statistic : float
|
|
33
|
+
The Harvey-Leybourne-Newbold-corrected DM statistic.
|
|
34
|
+
p_value : float
|
|
35
|
+
Two-sided p-value, Student-t with `n - 1` degrees of freedom.
|
|
36
|
+
mean_diff : float
|
|
37
|
+
Mean of `loss_a - loss_b` (uncorrected).
|
|
38
|
+
n : int
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
statistic: float
|
|
42
|
+
p_value: float
|
|
43
|
+
mean_diff: float
|
|
44
|
+
n: int
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _autocovariance(d: npt.NDArray[np.float64], lag: int) -> float:
|
|
48
|
+
"""Sample autocovariance at `lag`, normalized by the FULL sample
|
|
49
|
+
size `n` (not `n - lag`) at every lag -- the convention R's `acf()`
|
|
50
|
+
uses (and that `forecast::dm.test` relies on for its long-run-
|
|
51
|
+
variance estimate), not the `n - lag`-normalized "unbiased" variant.
|
|
52
|
+
Confirmed to matter here: with `n - lag` normalization this
|
|
53
|
+
function's h=1 oracle cases (lag=0 only, where the two conventions
|
|
54
|
+
coincide) passed, while every h>1 case (which needs lag>0
|
|
55
|
+
autocovariances) silently disagreed with R's own output -- exactly
|
|
56
|
+
how the discrepancy was actually caught."""
|
|
57
|
+
n = d.size
|
|
58
|
+
dbar = d.mean()
|
|
59
|
+
if lag == 0:
|
|
60
|
+
return float(np.mean((d - dbar) ** 2))
|
|
61
|
+
return float(np.sum((d[lag:] - dbar) * (d[:-lag] - dbar)) / n)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def diebold_mariano(
|
|
65
|
+
loss_a: npt.ArrayLike,
|
|
66
|
+
loss_b: npt.ArrayLike,
|
|
67
|
+
h: int = 1,
|
|
68
|
+
varestimator: str = "acf",
|
|
69
|
+
) -> DieboldMarianoResult:
|
|
70
|
+
"""Diebold-Mariano test on the paired loss differential
|
|
71
|
+
`d_t = loss_a[t] - loss_b[t]`.
|
|
72
|
+
|
|
73
|
+
Parameters
|
|
74
|
+
----------
|
|
75
|
+
loss_a, loss_b : array-like
|
|
76
|
+
Per-origin losses (e.g. `BacktestResult.scores[name]`) -- NOT
|
|
77
|
+
raw forecast errors; unlike R's `dm.test(e1, e2, power=...)`,
|
|
78
|
+
this function does not apply a power transform, since the
|
|
79
|
+
objects being compared here are already non-negative scores
|
|
80
|
+
(energy score, variogram score). Passing already-nonnegative
|
|
81
|
+
losses with R's `power=1` reproduces this function's `d`
|
|
82
|
+
exactly (see the reference-oracle test suite).
|
|
83
|
+
h : int, default=1
|
|
84
|
+
Forecast horizon. Controls the long-run-variance truncation lag
|
|
85
|
+
(`h - 1`) and the Harvey-Leybourne-Newbold correction factor.
|
|
86
|
+
varestimator : {"acf", "bartlett"}, default="acf"
|
|
87
|
+
"acf": unweighted sum of sample autocovariances up to lag
|
|
88
|
+
`h - 1` (R's own default, and the only option when `h == 1`,
|
|
89
|
+
where the two coincide). "bartlett": Bartlett-kernel-weighted
|
|
90
|
+
sum (`1 - k/h` weights), matching R's `varestimator="bartlett"`.
|
|
91
|
+
|
|
92
|
+
Returns
|
|
93
|
+
-------
|
|
94
|
+
DieboldMarianoResult
|
|
95
|
+
"""
|
|
96
|
+
d = np.asarray(loss_a, dtype=float) - np.asarray(loss_b, dtype=float)
|
|
97
|
+
n = d.size
|
|
98
|
+
if h < 1:
|
|
99
|
+
raise ValueError(f"h must be >= 1, got {h}")
|
|
100
|
+
if h > n:
|
|
101
|
+
raise ValueError(f"h ({h}) cannot exceed the number of observations ({n})")
|
|
102
|
+
|
|
103
|
+
gamma0 = _autocovariance(d, 0)
|
|
104
|
+
if varestimator == "acf" or h == 1:
|
|
105
|
+
long_run_var = gamma0 + 2.0 * sum(_autocovariance(d, k) for k in range(1, h))
|
|
106
|
+
elif varestimator == "bartlett":
|
|
107
|
+
long_run_var = gamma0 + 2.0 * sum(
|
|
108
|
+
(1.0 - k / h) * _autocovariance(d, k) for k in range(1, h)
|
|
109
|
+
)
|
|
110
|
+
else:
|
|
111
|
+
raise ValueError(f"Unknown varestimator: {varestimator!r}")
|
|
112
|
+
long_run_var /= n
|
|
113
|
+
|
|
114
|
+
if long_run_var <= 0:
|
|
115
|
+
raise ValueError(
|
|
116
|
+
"Estimated long-run variance of the loss differential is <= 0 "
|
|
117
|
+
"(a degenerate or perfectly-periodic differential); the DM "
|
|
118
|
+
"statistic is undefined. R's dm.test falls back to h=1 with a "
|
|
119
|
+
"warning in this situation -- retry with h=1 or varestimator="
|
|
120
|
+
"'bartlett' explicitly rather than silently doing so here."
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
dbar = float(d.mean())
|
|
124
|
+
raw_statistic = dbar / np.sqrt(long_run_var)
|
|
125
|
+
correction = np.sqrt((n + 1 - 2 * h + (h / n) * (h - 1)) / n)
|
|
126
|
+
statistic = float(raw_statistic * correction)
|
|
127
|
+
p_value = float(2.0 * student_t.cdf(-abs(statistic), df=n - 1))
|
|
128
|
+
|
|
129
|
+
return DieboldMarianoResult(statistic=statistic, p_value=p_value, mean_diff=dbar, n=n)
|
corrscore/mcs.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""The Model Confidence Set (Hansen, Lunde & Nason, 2011), vendored
|
|
2
|
+
rather than depending on `model-confidence-set` (JLDC's GitHub-only,
|
|
3
|
+
no-PyPI-release Python port): reproduced directly from the paper, with
|
|
4
|
+
R's actively-maintained CRAN `MCS` package (Catania & Bernardi) and the
|
|
5
|
+
JLDC port used only as development-time correctness references, never
|
|
6
|
+
imported at runtime.
|
|
7
|
+
|
|
8
|
+
Implements the range statistic (T_R) elimination algorithm: repeatedly
|
|
9
|
+
test whether the current candidate set is statistically distinguishable
|
|
10
|
+
from its own best member; if so, drop the single worst-performing model
|
|
11
|
+
and repeat, until the surviving set cannot be rejected. The null
|
|
12
|
+
distribution and each round's studentizing standard errors both come
|
|
13
|
+
from the SAME joint circular block bootstrap of the loss matrix
|
|
14
|
+
(reusing `arch.bootstrap.CircularBlockBootstrap`, matching this
|
|
15
|
+
package's dependency policy).
|
|
16
|
+
|
|
17
|
+
Honest scope note: this is this package's own reasonable
|
|
18
|
+
operationalization of Hansen et al.'s range statistic and elimination
|
|
19
|
+
rule, not a literal line-by-line transcription of any specific existing
|
|
20
|
+
implementation's internal choices (e.g. R's `MCS` package's automatic
|
|
21
|
+
block-length selection is not reproduced -- `block_len` is a required,
|
|
22
|
+
caller-chosen parameter here, swept manually if robustness to it
|
|
23
|
+
matters, the same discipline `circular_block_bootstrap` already uses).
|
|
24
|
+
The test suite cross-checks this implementation's *verdict* (which
|
|
25
|
+
models survive) against R's `MCS::MCSprocedure` on fixed synthetic
|
|
26
|
+
data where the correct verdict is unambiguous by construction, not
|
|
27
|
+
byte-exact statistic/p-value agreement -- see
|
|
28
|
+
`tests/test_mcs.py` for why that is the honest bar for this
|
|
29
|
+
specific dependency.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from typing import Mapping, NamedTuple
|
|
34
|
+
|
|
35
|
+
import numpy as np
|
|
36
|
+
import numpy.typing as npt
|
|
37
|
+
from arch.bootstrap import CircularBlockBootstrap
|
|
38
|
+
|
|
39
|
+
__all__ = ["MCSResult", "model_confidence_set"]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class MCSResult(NamedTuple):
|
|
43
|
+
"""Result of `model_confidence_set`.
|
|
44
|
+
|
|
45
|
+
Attributes
|
|
46
|
+
----------
|
|
47
|
+
survivors : list of str
|
|
48
|
+
Models statistically indistinguishable from the best one at
|
|
49
|
+
`alpha`, in no particular order.
|
|
50
|
+
alpha : float
|
|
51
|
+
eliminated : list of (str, float)
|
|
52
|
+
Models removed, in elimination order, paired with the round's
|
|
53
|
+
p-value at the moment of that model's removal.
|
|
54
|
+
final_p_value : float
|
|
55
|
+
The surviving set's own p-value (>= alpha; or 1.0 if only one
|
|
56
|
+
model ever remained, a case in which rejection is trivially
|
|
57
|
+
impossible).
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
survivors: list[str]
|
|
61
|
+
alpha: float
|
|
62
|
+
eliminated: list[tuple[str, float]]
|
|
63
|
+
final_p_value: float
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _round_statistics(
|
|
67
|
+
sub_loss: npt.NDArray[np.float64], block_len: int, n_boot: int, seed: int | np.random.Generator | None
|
|
68
|
+
) -> tuple[npt.NDArray[np.float64], float]:
|
|
69
|
+
"""One elimination round: pairwise studentized statistics for the
|
|
70
|
+
current model set, and the bootstrap p-value for the joint range
|
|
71
|
+
null. Returns `(t_obs, p_value)`, `t_obs` shape (m, m)."""
|
|
72
|
+
m = sub_loss.shape[1]
|
|
73
|
+
dbar = np.array([[np.mean(sub_loss[:, i] - sub_loss[:, j]) for j in range(m)] for i in range(m)])
|
|
74
|
+
|
|
75
|
+
bs = CircularBlockBootstrap(block_len, sub_loss, seed=seed)
|
|
76
|
+
boot_dbar = np.empty((n_boot, m, m))
|
|
77
|
+
for rep, (pos, _kw) in enumerate(bs.bootstrap(n_boot)):
|
|
78
|
+
resampled = pos[0]
|
|
79
|
+
boot_dbar[rep] = np.array(
|
|
80
|
+
[[np.mean(resampled[:, i] - resampled[:, j]) for j in range(m)] for i in range(m)]
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
se = boot_dbar.std(axis=0, ddof=1)
|
|
84
|
+
se[se == 0] = np.inf # identical-loss pairs (including the diagonal) -> t_ij = 0, never inf/nan
|
|
85
|
+
t_obs = dbar / se
|
|
86
|
+
t_range_obs = float(np.abs(t_obs).max())
|
|
87
|
+
|
|
88
|
+
t_boot = (boot_dbar - dbar[None, :, :]) / se[None, :, :]
|
|
89
|
+
t_range_boot = np.abs(t_boot).max(axis=(1, 2))
|
|
90
|
+
p_value = float(np.mean(t_range_boot >= t_range_obs))
|
|
91
|
+
return t_obs, p_value
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def model_confidence_set(
|
|
95
|
+
scores: Mapping[str, npt.ArrayLike],
|
|
96
|
+
alpha: float = 0.10,
|
|
97
|
+
block_len: int = 5,
|
|
98
|
+
n_boot: int = 1000,
|
|
99
|
+
seed: int | np.random.Generator | None = None,
|
|
100
|
+
) -> MCSResult:
|
|
101
|
+
"""The Model Confidence Set via the range statistic.
|
|
102
|
+
|
|
103
|
+
Parameters
|
|
104
|
+
----------
|
|
105
|
+
scores : dict of str -> array-like
|
|
106
|
+
Per-model, per-origin losses, all the same length (e.g.
|
|
107
|
+
`BacktestResult.scores`). At least two models required.
|
|
108
|
+
alpha : float, default=0.10
|
|
109
|
+
block_len : int, default=5
|
|
110
|
+
Circular block-bootstrap block length. Required and
|
|
111
|
+
caller-chosen (see module docstring) -- rerun with different
|
|
112
|
+
values to check robustness, matching `circular_block_bootstrap`.
|
|
113
|
+
n_boot : int, default=1000
|
|
114
|
+
seed : int, np.random.Generator, or None
|
|
115
|
+
|
|
116
|
+
Returns
|
|
117
|
+
-------
|
|
118
|
+
MCSResult
|
|
119
|
+
"""
|
|
120
|
+
names = list(scores.keys())
|
|
121
|
+
if len(names) < 2:
|
|
122
|
+
raise ValueError("model_confidence_set needs at least two models")
|
|
123
|
+
loss = np.column_stack([np.asarray(scores[name], dtype=float) for name in names])
|
|
124
|
+
|
|
125
|
+
current = list(range(len(names)))
|
|
126
|
+
eliminated: list[tuple[str, float]] = []
|
|
127
|
+
p_value = 1.0
|
|
128
|
+
|
|
129
|
+
while True:
|
|
130
|
+
if len(current) == 1:
|
|
131
|
+
p_value = 1.0
|
|
132
|
+
break
|
|
133
|
+
sub_names = [names[i] for i in current]
|
|
134
|
+
t_obs, p_value = _round_statistics(loss[:, current], block_len, n_boot, seed)
|
|
135
|
+
if p_value >= alpha:
|
|
136
|
+
break
|
|
137
|
+
avg_t = t_obs.mean(axis=1)
|
|
138
|
+
worst_local = int(np.argmax(avg_t))
|
|
139
|
+
eliminated.append((sub_names[worst_local], p_value))
|
|
140
|
+
current.pop(worst_local)
|
|
141
|
+
|
|
142
|
+
survivors = [names[i] for i in current]
|
|
143
|
+
return MCSResult(survivors=survivors, alpha=alpha, eliminated=eliminated, final_p_value=p_value)
|
corrscore/py.typed
ADDED
|
File without changes
|
corrscore/scoring.py
ADDED
|
@@ -0,0 +1,373 @@
|
|
|
1
|
+
"""Matrix-aware proper scoring rules for correlation/covariance-matrix
|
|
2
|
+
forecasts: the energy score (Gneiting & Raftery, 2007) and the variogram
|
|
3
|
+
score (Scheuerer & Hamill, 2015), both adapted to score a full K x K
|
|
4
|
+
matrix rather than the vector-valued observation either was originally
|
|
5
|
+
published for.
|
|
6
|
+
|
|
7
|
+
Forecast representation (see `matrix_energy_score` and
|
|
8
|
+
`matrix_variogram_score` docstrings for the exact shape of each): a
|
|
9
|
+
plain dict tagged by `"kind"`, dispatching across a closed-form
|
|
10
|
+
tractability spectrum --
|
|
11
|
+
|
|
12
|
+
- "point": a single deterministic matrix (M=1).
|
|
13
|
+
- "mixture": any number of discrete atoms -- exact,
|
|
14
|
+
O(K_atoms^2) cost, no simulation, for
|
|
15
|
+
any number of atoms (not hard-coded to
|
|
16
|
+
two regimes).
|
|
17
|
+
- "isotropic_gaussian_mixture": discrete atoms each with an isotropic
|
|
18
|
+
Gaussian scatter -- exact for the
|
|
19
|
+
ENERGY score via the confluent
|
|
20
|
+
hypergeometric mean-norm formula; only
|
|
21
|
+
ever reachable through this explicit
|
|
22
|
+
kind, never inferred from a plain
|
|
23
|
+
ensemble automatically, because the
|
|
24
|
+
isotropy assumption is a real one.
|
|
25
|
+
- "ensemble": a general Monte Carlo draw set -- the
|
|
26
|
+
only kind with no closed form; used for
|
|
27
|
+
anything else, including simulated
|
|
28
|
+
paths from a regime-switching
|
|
29
|
+
correlation model.
|
|
30
|
+
|
|
31
|
+
`matrix_variogram_score` has a narrower closed form than the energy
|
|
32
|
+
score: exact for "point" and "mixture" (both are fully determined by a
|
|
33
|
+
finite set of deterministic atoms), Monte-Carlo-estimated for "ensemble"
|
|
34
|
+
and "isotropic_gaussian_mixture" -- no closed form for a fractional
|
|
35
|
+
absolute moment of a Gaussian difference has been derived for this
|
|
36
|
+
package; sampling is the honest v1 answer for those two kinds.
|
|
37
|
+
|
|
38
|
+
`matrix_geodesic_variogram_score` is a geometric (Fisher-Rao-aware)
|
|
39
|
+
variant of `matrix_variogram_score`: every free entry is passed through
|
|
40
|
+
`phi`, the signed Fisher-Rao arc length from rho=0 (a correlation entry's
|
|
41
|
+
distance-to-independence, treating the entry as its own isolated 2x2
|
|
42
|
+
correlation matrix under the affine-invariant metric), before the
|
|
43
|
+
ordinary variogram-score machinery runs on the transformed values. This
|
|
44
|
+
is `matrix_variogram_score` itself, unmodified, called on
|
|
45
|
+
phi-transformed entries -- not a new formula. It needs no new propriety
|
|
46
|
+
argument: |a-b|^p is already conditionally negative definite (of
|
|
47
|
+
negative type) on all of R for any real a, b (the same classical fact
|
|
48
|
+
that already licenses the flat variogram score), and composing a fixed
|
|
49
|
+
measurable transform with an already-valid kernel changes nothing about
|
|
50
|
+
that -- this construction is in fact proper for *any* fixed measurable
|
|
51
|
+
per-entry transform, not only ones with a metric interpretation.
|
|
52
|
+
Validated on synthetic matrices, on a real 16-factor equity-factor
|
|
53
|
+
panel, and -- the one comparison that matters most -- on real,
|
|
54
|
+
walk-forward regime-switching correlation forecast ensembles, where it
|
|
55
|
+
reproduced a statistically significant discrimination advantage over
|
|
56
|
+
the flat variogram score's own blind spot near the boundary of valid
|
|
57
|
+
correlation matrices. A whole-matrix geometric energy-score analogue was
|
|
58
|
+
prototyped alongside this one but did not reproduce the same advantage
|
|
59
|
+
on that comparison, so it is not (yet) shipped in this package; this is,
|
|
60
|
+
for now, the one geometric scoring rule promoted from prototype to
|
|
61
|
+
package code.
|
|
62
|
+
"""
|
|
63
|
+
from __future__ import annotations
|
|
64
|
+
|
|
65
|
+
from typing import Any, Callable, Mapping, Sequence
|
|
66
|
+
|
|
67
|
+
import numpy as np
|
|
68
|
+
import numpy.typing as npt
|
|
69
|
+
from scipy.spatial.distance import pdist
|
|
70
|
+
from scipy.special import gammaln, hyp1f1
|
|
71
|
+
|
|
72
|
+
Matrix = npt.NDArray[np.float64]
|
|
73
|
+
Forecast = Mapping[str, Any]
|
|
74
|
+
|
|
75
|
+
__all__ = ["matrix_energy_score", "matrix_variogram_score", "matrix_geodesic_variogram_score"]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _frobenius(a: npt.ArrayLike, b: npt.ArrayLike) -> float:
|
|
79
|
+
"""Full K x K Frobenius distance -- the standard convention for a
|
|
80
|
+
matrix-valued energy score, not the upper-triangle-only convention
|
|
81
|
+
`matrix_variogram_score` uses (see that function's own docstring for
|
|
82
|
+
why the two conventions deliberately differ)."""
|
|
83
|
+
return float(np.linalg.norm(np.asarray(a, dtype=float) - np.asarray(b, dtype=float), ord="fro"))
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _upper(m: npt.ArrayLike) -> npt.NDArray[np.float64]:
|
|
87
|
+
"""The K(K-1)/2 free upper-triangle entries (diagonal, always 1 for
|
|
88
|
+
a correlation matrix, and the mirrored lower triangle both excluded,
|
|
89
|
+
since they carry no independent information)."""
|
|
90
|
+
arr = np.asarray(m, dtype=float)
|
|
91
|
+
k = arr.shape[0]
|
|
92
|
+
return arr[np.triu_indices(k, k=1)]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _expected_norm_isotropic_gaussian(norm_mu_sq: float, sigma: float, n: int) -> float:
|
|
96
|
+
"""E[||Z||] for Z ~ N(mu, sigma^2 I_n), via the noncentral-chi mean
|
|
97
|
+
formula (Johnson, Kotz & Balakrishnan, 1994), evaluated through the
|
|
98
|
+
confluent hypergeometric function `scipy.special.hyp1f1` rather than
|
|
99
|
+
simulated.
|
|
100
|
+
|
|
101
|
+
`norm_mu_sq` is ||mu||^2 -- the formula only depends on mu through
|
|
102
|
+
this scalar, by spherical symmetry. `n == 0` (a degenerate,
|
|
103
|
+
parameter-free space, e.g. K=1) returns 0 directly.
|
|
104
|
+
"""
|
|
105
|
+
if n <= 0:
|
|
106
|
+
return 0.0
|
|
107
|
+
if sigma <= 0.0:
|
|
108
|
+
return float(np.sqrt(norm_mu_sq))
|
|
109
|
+
log_ratio = gammaln((n + 1) / 2.0) - gammaln(n / 2.0)
|
|
110
|
+
coef = sigma * np.sqrt(2.0) * np.exp(log_ratio)
|
|
111
|
+
return float(coef * hyp1f1(-0.5, n / 2.0, -norm_mu_sq / (2.0 * sigma**2)))
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _expected_matrix_norm_isotropic(diff_free_entries: npt.NDArray[np.float64], sigma: float) -> float:
|
|
115
|
+
"""E[||A||_F] where A is the (implicit) symmetric, zero-diagonal K x
|
|
116
|
+
K matrix built by placing an isotropic Gaussian free-entry vector u
|
|
117
|
+
~ N(diff_free_entries, sigma^2 I_n) (n = K(K-1)/2) into both the
|
|
118
|
+
upper and lower triangle. Each free entry appears twice in A (once
|
|
119
|
+
per triangle), so ||A||_F = sqrt(2) * ||u||_2 -- this reuses
|
|
120
|
+
`_expected_norm_isotropic_gaussian` above with that bookkeeping
|
|
121
|
+
factor folded in, not a separate derivation."""
|
|
122
|
+
n = diff_free_entries.size
|
|
123
|
+
norm_mu_sq = float(diff_free_entries @ diff_free_entries)
|
|
124
|
+
return np.sqrt(2.0) * _expected_norm_isotropic_gaussian(norm_mu_sq, sigma, n)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _energy_score_point(q: npt.ArrayLike, y: npt.ArrayLike) -> float:
|
|
128
|
+
return _frobenius(q, y)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _energy_score_mixture(components: Sequence[tuple[float, npt.ArrayLike]], y: npt.ArrayLike) -> float:
|
|
132
|
+
weights = np.array([float(p) for p, _ in components])
|
|
133
|
+
mats = [np.asarray(q, dtype=float) for _, q in components]
|
|
134
|
+
k = len(mats)
|
|
135
|
+
term1 = sum(weights[i] * _frobenius(mats[i], y) for i in range(k))
|
|
136
|
+
term2 = 0.0
|
|
137
|
+
for i in range(k):
|
|
138
|
+
for j in range(k):
|
|
139
|
+
term2 += weights[i] * weights[j] * _frobenius(mats[i], mats[j])
|
|
140
|
+
return float(term1 - 0.5 * term2)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _energy_score_isotropic_gaussian_mixture(
|
|
144
|
+
components: Sequence[tuple[float, npt.ArrayLike, float]], y: npt.ArrayLike
|
|
145
|
+
) -> float:
|
|
146
|
+
weights = np.array([float(p) for p, _, _ in components])
|
|
147
|
+
mats_u = [_upper(q) for _, q, _ in components]
|
|
148
|
+
sigmas = [float(s) for _, _, s in components]
|
|
149
|
+
y_u = _upper(y)
|
|
150
|
+
k = len(components)
|
|
151
|
+
|
|
152
|
+
term1 = 0.0
|
|
153
|
+
for i in range(k):
|
|
154
|
+
term1 += weights[i] * _expected_matrix_norm_isotropic(mats_u[i] - y_u, sigmas[i])
|
|
155
|
+
|
|
156
|
+
term2 = 0.0
|
|
157
|
+
for i in range(k):
|
|
158
|
+
for j in range(k):
|
|
159
|
+
combined_sigma = float(np.sqrt(sigmas[i] ** 2 + sigmas[j] ** 2))
|
|
160
|
+
term2 += weights[i] * weights[j] * _expected_matrix_norm_isotropic(mats_u[i] - mats_u[j], combined_sigma)
|
|
161
|
+
|
|
162
|
+
return float(term1 - 0.5 * term2)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _energy_score_ensemble(draws: Sequence[npt.ArrayLike], y: npt.ArrayLike) -> float:
|
|
166
|
+
flat = np.stack([np.asarray(d, dtype=float).ravel() for d in draws])
|
|
167
|
+
y_flat = np.asarray(y, dtype=float).ravel()
|
|
168
|
+
m = flat.shape[0]
|
|
169
|
+
term1 = float(np.mean(np.linalg.norm(flat - y_flat, axis=1)))
|
|
170
|
+
if m < 2:
|
|
171
|
+
return term1
|
|
172
|
+
pairwise = pdist(flat, metric="euclidean")
|
|
173
|
+
term2 = float(pairwise.sum() / (m**2))
|
|
174
|
+
return term1 - term2
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def matrix_energy_score(forecast: Forecast, y: npt.ArrayLike) -> float:
|
|
178
|
+
"""The energy score (Gneiting & Raftery, 2007), dispatched across the
|
|
179
|
+
closed-form spectrum (module docstring) by `forecast["kind"]`:
|
|
180
|
+
|
|
181
|
+
- {"kind": "point", "Q": Q} -> the M=1 special case, plain
|
|
182
|
+
Frobenius distance.
|
|
183
|
+
- {"kind": "mixture", "components": [(p_1, Q_1), ..., (p_K, Q_K)]}
|
|
184
|
+
-> Tier 1, exact. Any number of atoms, not just two.
|
|
185
|
+
- {"kind": "isotropic_gaussian_mixture",
|
|
186
|
+
"components": [(p_1, Q_1, sigma_1), ...]} -> Tier 2, exact.
|
|
187
|
+
`sigma_k` is the per-component scatter in the K(K-1)/2-dimensional
|
|
188
|
+
free-entry space, not in the full K x K ambient space.
|
|
189
|
+
- {"kind": "ensemble", "draws": [Q_1, ..., Q_M]} -> Tier 3, the
|
|
190
|
+
general Monte Carlo form, O(M^2) pairwise distances.
|
|
191
|
+
|
|
192
|
+
`y` is the realized K x K correlation (or covariance) matrix.
|
|
193
|
+
"""
|
|
194
|
+
kind = forecast["kind"]
|
|
195
|
+
if kind == "point":
|
|
196
|
+
return _energy_score_point(forecast["Q"], y)
|
|
197
|
+
if kind == "mixture":
|
|
198
|
+
return _energy_score_mixture(forecast["components"], y)
|
|
199
|
+
if kind == "isotropic_gaussian_mixture":
|
|
200
|
+
return _energy_score_isotropic_gaussian_mixture(forecast["components"], y)
|
|
201
|
+
if kind == "ensemble":
|
|
202
|
+
return _energy_score_ensemble(forecast["draws"], y)
|
|
203
|
+
raise ValueError(f"Unknown forecast kind: {kind!r}")
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _variogram_diff_p_point(q: npt.ArrayLike, p: float) -> npt.NDArray[np.float64]:
|
|
207
|
+
q_u = _upper(q)
|
|
208
|
+
return np.abs(q_u[:, None] - q_u[None, :]) ** p
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _variogram_diff_p_mixture(components: Sequence[tuple[float, npt.ArrayLike]], p: float) -> npt.NDArray[np.float64]:
|
|
212
|
+
"""Exact: a discrete mixture's entries are fully determined once the
|
|
213
|
+
atom is known, so E|X_i - X_j|^p is just the atom-weighted average
|
|
214
|
+
of the deterministic per-atom values -- no sampling needed."""
|
|
215
|
+
weights = [float(w) for w, _ in components]
|
|
216
|
+
total = None
|
|
217
|
+
for w, q in zip(weights, (c[1] for c in components)):
|
|
218
|
+
term = w * _variogram_diff_p_point(q, p)
|
|
219
|
+
total = term if total is None else total + term
|
|
220
|
+
assert total is not None
|
|
221
|
+
return total
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _sample_upper_triangle(forecast: Forecast, n_samples: int, rng: np.random.Generator) -> npt.NDArray[np.float64]:
|
|
225
|
+
"""Draws used only by `matrix_variogram_score`'s Monte Carlo path
|
|
226
|
+
(see module docstring for why "ensemble" and
|
|
227
|
+
"isotropic_gaussian_mixture" have no closed form here). "ensemble"
|
|
228
|
+
reuses its existing draws directly (deterministic, no extra
|
|
229
|
+
sampling); "isotropic_gaussian_mixture" is genuinely sampled."""
|
|
230
|
+
kind = forecast["kind"]
|
|
231
|
+
if kind == "ensemble":
|
|
232
|
+
return np.stack([_upper(d) for d in forecast["draws"]])
|
|
233
|
+
if kind == "isotropic_gaussian_mixture":
|
|
234
|
+
components = forecast["components"]
|
|
235
|
+
weights = np.array([float(w) for w, _, _ in components])
|
|
236
|
+
weights = weights / weights.sum()
|
|
237
|
+
atom_idx = rng.choice(len(components), size=n_samples, p=weights)
|
|
238
|
+
n_free = _upper(components[0][1]).size
|
|
239
|
+
out = np.empty((n_samples, n_free))
|
|
240
|
+
for k, (_, q, sigma) in enumerate(components):
|
|
241
|
+
mask = atom_idx == k
|
|
242
|
+
count = int(mask.sum())
|
|
243
|
+
if count == 0:
|
|
244
|
+
continue
|
|
245
|
+
noise = rng.normal(0.0, float(sigma), size=(count, n_free)) if sigma > 0 else 0.0
|
|
246
|
+
out[mask] = _upper(q) + noise
|
|
247
|
+
return out
|
|
248
|
+
raise ValueError(f"Forecast kind {kind!r} has no Monte Carlo sampling path")
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def matrix_variogram_score(
|
|
252
|
+
forecast: Forecast,
|
|
253
|
+
y: npt.ArrayLike,
|
|
254
|
+
p: float = 0.5,
|
|
255
|
+
weights: npt.ArrayLike | None = None,
|
|
256
|
+
n_samples: int = 500,
|
|
257
|
+
random_state: int | np.random.Generator | None = None,
|
|
258
|
+
) -> float:
|
|
259
|
+
"""The variogram score (Scheuerer & Hamill, 2015), adapted to index
|
|
260
|
+
i, j over the K(K-1)/2 free upper-triangle entries of the
|
|
261
|
+
correlation matrix, rather than the K original series the published
|
|
262
|
+
formula was defined for.
|
|
263
|
+
|
|
264
|
+
Exact (no sampling) for "point" and "mixture" forecasts. For
|
|
265
|
+
"ensemble" and "isotropic_gaussian_mixture", `E|X_i - X_j|^p` is
|
|
266
|
+
estimated by Monte Carlo (`n_samples` draws) -- no closed form for
|
|
267
|
+
this fractional absolute moment has been derived for this package
|
|
268
|
+
(unlike `matrix_energy_score`'s Tier 2 path); see the module
|
|
269
|
+
docstring. Cost and memory scale as O(n_samples * n_entries^2),
|
|
270
|
+
where n_entries = K(K-1)/2 -- the default `n_samples=500` keeps this
|
|
271
|
+
modest through K=16 (n_entries=120); pass a smaller value for larger
|
|
272
|
+
K if memory becomes a concern.
|
|
273
|
+
|
|
274
|
+
`weights` (if given) must be an (n_entries, n_entries) array;
|
|
275
|
+
defaults to uniform (all-ones), matching Scheuerer & Hamill's own
|
|
276
|
+
default and the published formula literally.
|
|
277
|
+
"""
|
|
278
|
+
y_u = _upper(y)
|
|
279
|
+
n = y_u.size
|
|
280
|
+
y_diff_p = np.abs(y_u[:, None] - y_u[None, :]) ** p
|
|
281
|
+
w = np.ones((n, n)) if weights is None else np.asarray(weights, dtype=float)
|
|
282
|
+
|
|
283
|
+
kind = forecast["kind"]
|
|
284
|
+
if kind == "point":
|
|
285
|
+
exp_diff_p = _variogram_diff_p_point(forecast["Q"], p)
|
|
286
|
+
elif kind == "mixture":
|
|
287
|
+
exp_diff_p = _variogram_diff_p_mixture(forecast["components"], p)
|
|
288
|
+
elif kind in ("ensemble", "isotropic_gaussian_mixture"):
|
|
289
|
+
rng = random_state if isinstance(random_state, np.random.Generator) else np.random.default_rng(random_state)
|
|
290
|
+
samples = _sample_upper_triangle(forecast, n_samples, rng)
|
|
291
|
+
exp_diff_p = (np.abs(samples[:, :, None] - samples[:, None, :]) ** p).mean(axis=0)
|
|
292
|
+
else:
|
|
293
|
+
raise ValueError(f"Unknown forecast kind: {kind!r}")
|
|
294
|
+
|
|
295
|
+
return float(np.sum(w * (y_diff_p - exp_diff_p) ** 2))
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _phi(rho: npt.ArrayLike) -> npt.NDArray[np.float64]:
|
|
299
|
+
"""Signed Fisher-Rao arc length from rho=0: treats a single
|
|
300
|
+
correlation entry as its own isolated 2x2 correlation matrix
|
|
301
|
+
C(rho) = [[1,rho],[rho,1]] under the affine-invariant metric, and
|
|
302
|
+
returns sign(rho) * d_FR(I, C(rho)). Closed form (derived and
|
|
303
|
+
verified to machine precision against the general eigenvalue-based
|
|
304
|
+
Fisher-Rao formula):
|
|
305
|
+
|
|
306
|
+
phi(rho) = sign(rho) * sqrt(0.5 * (log(1+|rho|)^2 + log(1-|rho|)^2))
|
|
307
|
+
|
|
308
|
+
Smooth, odd, and strictly increasing on (-1, 1) -> R (verified
|
|
309
|
+
numerically); a close cousin of the century-old Fisher z-transform
|
|
310
|
+
arctanh(rho), diverging slightly faster as rho -> +-1.
|
|
311
|
+
"""
|
|
312
|
+
rho = np.asarray(rho, dtype=float)
|
|
313
|
+
a = np.abs(rho)
|
|
314
|
+
d = np.sqrt(0.5 * (np.log1p(a) ** 2 + np.log1p(-a) ** 2))
|
|
315
|
+
return np.sign(rho) * d
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _phi_transform_matrix(q: npt.ArrayLike) -> npt.NDArray[np.float64]:
|
|
319
|
+
"""`_phi` applied entrywise off-diagonal. The diagonal is zeroed
|
|
320
|
+
before transforming, not left at the input's own (always exactly
|
|
321
|
+
1.0) diagonal: `_upper` never reads it, but `_phi(1.0)` is -inf
|
|
322
|
+
(log1p(-1)), which is harmless numerically (never read) but raises a
|
|
323
|
+
spurious divide-by-zero warning on every call otherwise."""
|
|
324
|
+
q = np.asarray(q, dtype=float).copy()
|
|
325
|
+
np.fill_diagonal(q, 0.0)
|
|
326
|
+
return _phi(q)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _phi_transform_forecast(forecast: Forecast) -> Forecast:
|
|
330
|
+
kind = forecast["kind"]
|
|
331
|
+
if kind == "point":
|
|
332
|
+
return {"kind": "point", "Q": _phi_transform_matrix(forecast["Q"])}
|
|
333
|
+
if kind == "mixture":
|
|
334
|
+
return {"kind": "mixture", "components": [(w, _phi_transform_matrix(q)) for w, q in forecast["components"]]}
|
|
335
|
+
if kind == "ensemble":
|
|
336
|
+
return {"kind": "ensemble", "draws": [_phi_transform_matrix(d) for d in forecast["draws"]]}
|
|
337
|
+
raise ValueError(f"Unknown forecast kind: {kind!r}")
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def matrix_geodesic_variogram_score(
|
|
341
|
+
forecast: Forecast,
|
|
342
|
+
y: npt.ArrayLike,
|
|
343
|
+
p: float = 0.5,
|
|
344
|
+
weights: npt.ArrayLike | None = None,
|
|
345
|
+
n_samples: int = 500,
|
|
346
|
+
random_state: int | np.random.Generator | None = None,
|
|
347
|
+
) -> float:
|
|
348
|
+
"""The geometric (Fisher-Rao-aware) variogram score: `matrix_
|
|
349
|
+
variogram_score` itself, unmodified, called on every free entry
|
|
350
|
+
passed through `_phi` first (see module docstring). Same forecast
|
|
351
|
+
dict shapes, same `p`/`weights`/`n_samples`/`random_state` semantics,
|
|
352
|
+
same closed-form-vs-Monte-Carlo split by `forecast["kind"]` as
|
|
353
|
+
`matrix_variogram_score` -- this function only changes what happens
|
|
354
|
+
to the raw entries before that machinery runs.
|
|
355
|
+
|
|
356
|
+
Proper for the same reason `matrix_variogram_score` is (it elicits a
|
|
357
|
+
pairwise-moment vector for any fixed measurable per-entry transform)
|
|
358
|
+
-- no new argument specific to `_phi` is needed. Not strictly proper,
|
|
359
|
+
for the identical reason the flat variogram score is not, though the
|
|
360
|
+
two scores' blind spots are provably different (a common shift in
|
|
361
|
+
phi-space vs. a common shift in raw rho-space) -- running both
|
|
362
|
+
alongside each other is a real, not just pragmatic, recommendation.
|
|
363
|
+
|
|
364
|
+
Validated on real, walk-forward regime-switching correlation forecast
|
|
365
|
+
ensembles: a statistically significant discrimination advantage over
|
|
366
|
+
the flat variogram score on that comparison, unlike a separate
|
|
367
|
+
whole-matrix geodesic energy score prototype, which did not reproduce
|
|
368
|
+
an advantage there and has accordingly not (yet) been promoted into
|
|
369
|
+
this package.
|
|
370
|
+
"""
|
|
371
|
+
y_t = _phi_transform_matrix(y)
|
|
372
|
+
forecast_t = _phi_transform_forecast(forecast)
|
|
373
|
+
return matrix_variogram_score(forecast_t, y_t, p=p, weights=weights, n_samples=n_samples, random_state=random_state)
|
corrscore/utils.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Outcome-severity-weighted aggregation -- a small, general-purpose
|
|
2
|
+
utility for weighting per-origin scores by how severe the realized
|
|
3
|
+
outcome was, surfaced while dogfooding this package against a real
|
|
4
|
+
regime-detection backtest.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import numpy.typing as npt
|
|
10
|
+
|
|
11
|
+
__all__ = ["asymmetric_weighted_mean"]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def asymmetric_weighted_mean(values: npt.ArrayLike, severity: npt.ArrayLike, kappa: float = 1.0) -> float:
|
|
15
|
+
"""Weight per-origin `values` (typically per-origin scores from a
|
|
16
|
+
`BacktestResult`) by realized outcome severity: origins at or above
|
|
17
|
+
the median `severity` are weighted `kappa` times as heavily as
|
|
18
|
+
origins below it.
|
|
19
|
+
|
|
20
|
+
`severity` should be a non-circular, outcome-derived measure -- e.g.
|
|
21
|
+
the mean absolute off-diagonal correlation of the realized
|
|
22
|
+
ground-truth window (`BacktestResult.severity`, via
|
|
23
|
+
`backtest_zero_overlap`'s `severity_fn`) -- not any model's own
|
|
24
|
+
regime call, to avoid biasing which origins get up-weighted toward
|
|
25
|
+
whichever model is being evaluated.
|
|
26
|
+
|
|
27
|
+
`kappa=1.0` (the default) reduces to a plain mean.
|
|
28
|
+
"""
|
|
29
|
+
values_arr = np.asarray(values, dtype=float)
|
|
30
|
+
severity_arr = np.asarray(severity, dtype=float)
|
|
31
|
+
if values_arr.shape != severity_arr.shape:
|
|
32
|
+
raise ValueError(
|
|
33
|
+
f"values and severity must have the same shape, got {values_arr.shape} and {severity_arr.shape}"
|
|
34
|
+
)
|
|
35
|
+
median = np.median(severity_arr)
|
|
36
|
+
weights = np.where(severity_arr >= median, kappa, 1.0)
|
|
37
|
+
return float(np.sum(values_arr * weights) / np.sum(weights))
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: corrscore
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts: zero-overlap walk-forward evaluation, energy/variogram scoring, and significance testing (block bootstrap, Diebold-Mariano, Model Confidence Set).
|
|
5
|
+
Author-email: Vinh Nguyen <vinhnguyen3455@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Classifier: Intended Audience :: Financial and Insurance Industry
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Office/Business :: Financial :: Investment
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Requires-Dist: arch>=6.3
|
|
19
|
+
Requires-Dist: numpy>=1.24
|
|
20
|
+
Requires-Dist: scipy>=1.10
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
23
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
24
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# corrscore
|
|
28
|
+
|
|
29
|
+
Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts, in Python.
|
|
30
|
+
|
|
31
|
+
[](LICENSE)
|
|
32
|
+
|
|
33
|
+
## Why this exists
|
|
34
|
+
|
|
35
|
+
Forecasting a correlation or covariance matrix is common in risk management and portfolio
|
|
36
|
+
construction — but evaluating that forecast correctly is not routine. Two mistakes are easy to
|
|
37
|
+
make and hard to notice:
|
|
38
|
+
|
|
39
|
+
1. **Naive matrix-comparison metrics aren't proper scoring rules.** A metric that isn't a proper
|
|
40
|
+
scoring rule can reward a forecaster for hedging toward a "safe" answer instead of reporting
|
|
41
|
+
their honest best guess — the evaluation itself creates bad incentives.
|
|
42
|
+
2. **Walk-forward evaluation windows are easy to overlap with the estimation window**, silently
|
|
43
|
+
leaking future information into a backtest and inflating apparent skill.
|
|
44
|
+
|
|
45
|
+
`corrscore` evaluates a forecast; it never produces one. It doesn't fit a model, doesn't implement
|
|
46
|
+
any particular correlation-dynamics model, and doesn't fetch or clean data — it's a focused
|
|
47
|
+
evaluation layer you drop on top of whatever you're already forecasting with.
|
|
48
|
+
|
|
49
|
+
## What it offers
|
|
50
|
+
|
|
51
|
+
- **Matrix-aware energy and variogram scores.** The two standard proper scoring rules from the
|
|
52
|
+
forecast-verification literature, generalized from their usual vector-valued form to score a
|
|
53
|
+
full `K x K` correlation/covariance matrix directly.
|
|
54
|
+
- **A geometric variant of the variogram score**, aware of the fact that correlation matrices live
|
|
55
|
+
on a curved space rather than flat Euclidean space — sharper at detecting forecast danger as a
|
|
56
|
+
matrix approaches the boundary of validity (near-singular, highly correlated regimes).
|
|
57
|
+
- **Four forecast representations**, not just point forecasts: a single deterministic matrix, a
|
|
58
|
+
discrete mixture of any number of atoms, an isotropic-Gaussian mixture, or a general Monte Carlo
|
|
59
|
+
ensemble — with closed-form scoring wherever one exists, Monte Carlo estimation only where it
|
|
60
|
+
doesn't.
|
|
61
|
+
- **A backtesting harness (`backtest_zero_overlap`)** that makes the specific, easy-to-make
|
|
62
|
+
lookahead bug — ground truth computed from a window that overlaps the forecast origin —
|
|
63
|
+
structurally impossible to reproduce, rather than something you have to remember to get right.
|
|
64
|
+
- **Significance testing**, not just point comparisons: a circular block bootstrap for
|
|
65
|
+
serially-dependent score differentials, the Diebold-Mariano test, and the Model Confidence Set
|
|
66
|
+
— so "is model A really better than model B" has an actual answer.
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from corrscore import matrix_energy_score, backtest_zero_overlap
|
|
70
|
+
from corrscore import circular_block_bootstrap, diebold_mariano, model_confidence_set
|
|
71
|
+
|
|
72
|
+
result = backtest_zero_overlap(
|
|
73
|
+
forecast_fns={"naive": naive_forecast, "filter": my_model.forecast},
|
|
74
|
+
ground_truth_fn=realized_correlation, # (start, end) -> K x K matrix
|
|
75
|
+
origins=origins,
|
|
76
|
+
horizon=10,
|
|
77
|
+
)
|
|
78
|
+
sig = circular_block_bootstrap(result.scores["filter"], result.scores["naive"], block_lengths=[1, 3, 6, 10])
|
|
79
|
+
mcs = model_confidence_set(result.scores, alpha=0.10)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Install
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install corrscore
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Requires Python 3.10-3.12. Runtime dependencies are `numpy`, `scipy`, and `arch` (for the circular
|
|
89
|
+
block bootstrap) — nothing else.
|
|
90
|
+
|
|
91
|
+
## Validation
|
|
92
|
+
|
|
93
|
+
Every scoring rule and test in this package is checked against an independent source of truth, not
|
|
94
|
+
just its own self-consistency: `diebold_mariano` and `model_confidence_set` are cross-checked
|
|
95
|
+
against live-generated R oracles (`forecast::dm.test` byte-exact, `MCS::MCSprocedure`
|
|
96
|
+
verdict-matched), and every closed-form scoring formula is checked against brute-force Monte Carlo
|
|
97
|
+
simulation of the object it claims to score. Fully type-hinted and `mypy`-clean. See `tests/` for
|
|
98
|
+
the full suite.
|
|
99
|
+
|
|
100
|
+
## Related work
|
|
101
|
+
|
|
102
|
+
No existing package (Python or R) treats a correlation/covariance matrix as the forecast object
|
|
103
|
+
with purge-aware walk-forward splitting and a proper scoring rule built in — see
|
|
104
|
+
[`docs/survey/`](docs/survey/) for the full landscape survey and the specific reuse-vs.-vendor
|
|
105
|
+
decision behind each dependency.
|
|
106
|
+
|
|
107
|
+
## Citation
|
|
108
|
+
|
|
109
|
+
A companion paper describing the package's design and methodology is available as a working
|
|
110
|
+
paper on SSRN: [Nguyen (2026), "corrscore: Matrix-Aware Proper Scoring Rules and Significance
|
|
111
|
+
Testing for Correlation and Covariance Forecasts in
|
|
112
|
+
Python"](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=7358478). See
|
|
113
|
+
[`CITATION.cff`](CITATION.cff) for a machine-readable citation.
|
|
114
|
+
|
|
115
|
+
## Development
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
git clone https://github.com/vinhnguyen3455/corrscore
|
|
119
|
+
cd corrscore
|
|
120
|
+
pip install -e ".[dev]"
|
|
121
|
+
pytest -q
|
|
122
|
+
mypy src/corrscore
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Issues and pull requests welcome.
|
|
126
|
+
|
|
127
|
+
## License
|
|
128
|
+
|
|
129
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
corrscore/__init__.py,sha256=xjKePWKScFs2h7EjlYW9HPJIPoNTZU5evw7M4TFGC-E,728
|
|
2
|
+
corrscore/backtest.py,sha256=QfYoZNOFNAB-8vQ_ukS07ZzdWaAk9VXJGhsE9m20Wfk,6143
|
|
3
|
+
corrscore/bootstrap.py,sha256=qJkLDh95JucUQ5NyCsO4zDDPy_Xm3K7X0uaj4KWFbZU,2801
|
|
4
|
+
corrscore/diebold_mariano.py,sha256=Oi3aZDFhZAG5HY3Q5CpjOnHk7-kbejUPwQWIKu-RrJU,5093
|
|
5
|
+
corrscore/mcs.py,sha256=ki6KH47dpFgCaotrCyKa_B9nR2f98cRJ46SD-MZmEsw,5636
|
|
6
|
+
corrscore/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
corrscore/scoring.py,sha256=0QJrtjylgaGovnoq7Mdz-oe8J1aDFeVhGPNsRsEVVuc,17546
|
|
8
|
+
corrscore/utils.py,sha256=_CV08KOct3AWVcLV36CmcTkXuiC6qv3ZQu5osJs1xwo,1587
|
|
9
|
+
corrscore-0.1.0.dist-info/METADATA,sha256=eLeYUCpW9opAOLApSmqF1RmnhVuMtMgTmhLmv-LLjRU,5889
|
|
10
|
+
corrscore-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
11
|
+
corrscore-0.1.0.dist-info/licenses/LICENSE,sha256=jB6MybiFM3YMWm7FRRZsoRPq7FOuzYut1XkoQOVGpYU,1068
|
|
12
|
+
corrscore-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vinh Nguyen
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|