corrscore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
corrscore/__init__.py ADDED
@@ -0,0 +1,21 @@
1
+ from .backtest import BacktestResult, backtest_zero_overlap
2
+ from .bootstrap import BootstrapResult, circular_block_bootstrap
3
+ from .diebold_mariano import DieboldMarianoResult, diebold_mariano
4
+ from .mcs import MCSResult, model_confidence_set
5
+ from .scoring import matrix_energy_score, matrix_geodesic_variogram_score, matrix_variogram_score
6
+ from .utils import asymmetric_weighted_mean
7
+
8
+ __all__ = [
9
+ "matrix_energy_score",
10
+ "matrix_variogram_score",
11
+ "matrix_geodesic_variogram_score",
12
+ "backtest_zero_overlap",
13
+ "BacktestResult",
14
+ "circular_block_bootstrap",
15
+ "BootstrapResult",
16
+ "diebold_mariano",
17
+ "DieboldMarianoResult",
18
+ "model_confidence_set",
19
+ "MCSResult",
20
+ "asymmetric_weighted_mean",
21
+ ]
corrscore/backtest.py ADDED
@@ -0,0 +1,148 @@
1
+ """The zero-overlap walk-forward backtest driver.
2
+
3
+ Design note: an earlier sketch of this API had a `window` parameter
4
+ defaulting to `horizon`. Working through the actual mechanics precisely
5
+ during implementation surfaced a cleaner, more honestly-scoped contract,
6
+ documented here rather than silently substituted. This harness does not,
7
+ and cannot, police how much history a caller's `forecast_fn` consults
8
+ internally -- that model is opaque to the harness (it might be a
9
+ full-history discounted filter, a short trailing window, or anything
10
+ else), exactly the same responsibility boundary scikit-learn's
11
+ `TimeSeriesSplit` leaves to the caller. What the harness genuinely *can*
12
+ and does enforce, unconditionally, is that `ground_truth_fn` is always
13
+ called with a start point strictly after the origin
14
+ (`origin + 1 + purge_gap`), never a window that reaches back before or
15
+ across it. That guards against a real class of bug this package exists
16
+ to prevent: computing "ground truth" as a window *ending* near the
17
+ origin rather than a window *starting* strictly after it, which silently
18
+ leaks estimation-window information into the evaluation. This API makes
19
+ that specific mistake structurally impossible to reproduce, since the
20
+ caller never controls the start point passed to `ground_truth_fn`.
21
+ """
22
+ from __future__ import annotations
23
+
24
+ from typing import Any, Callable, Mapping, NamedTuple, Sequence
25
+
26
+ import numpy as np
27
+ import numpy.typing as npt
28
+
29
+ from .scoring import matrix_energy_score
30
+
31
+ ForecastFn = Callable[[int], Mapping[str, Any]]
32
+ GroundTruthFn = Callable[[int, int], npt.ArrayLike]
33
+ ScoreFn = Callable[[Mapping[str, Any], npt.ArrayLike], float]
34
+ SeverityFn = Callable[[npt.NDArray[np.float64]], float]
35
+
36
+ __all__ = ["BacktestResult", "backtest_zero_overlap"]
37
+
38
+
39
+ class BacktestResult(NamedTuple):
40
+ """Result of `backtest_zero_overlap`.
41
+
42
+ Attributes
43
+ ----------
44
+ origins : list of int
45
+ The origins actually scored, in order.
46
+ scores : dict of str -> ndarray
47
+ Per-model, per-origin scores, one array per key of the
48
+ `forecast_fns` mapping passed in, aligned with `origins`.
49
+ severity : ndarray or None
50
+ Per-origin realized-severity values (`severity_fn(y)` at each
51
+ origin), or None if `severity_fn` was not supplied. Intended for
52
+ `corrscore.asymmetric_weighted_mean`.
53
+ horizon : int
54
+ purge_gap : int
55
+ """
56
+
57
+ origins: list[int]
58
+ scores: dict[str, npt.NDArray[np.float64]]
59
+ severity: npt.NDArray[np.float64] | None
60
+ horizon: int
61
+ purge_gap: int
62
+
63
+
64
+ def backtest_zero_overlap(
65
+ forecast_fns: Mapping[str, ForecastFn] | ForecastFn,
66
+ ground_truth_fn: GroundTruthFn,
67
+ origins: Sequence[int],
68
+ horizon: int,
69
+ purge_gap: int = 0,
70
+ score_fn: ScoreFn = matrix_energy_score,
71
+ severity_fn: SeverityFn | None = None,
72
+ ) -> BacktestResult:
73
+ """Score one or more forecasting methods against a proper,
74
+ zero-overlap-by-construction ground truth.
75
+
76
+ For each `origin` in `origins`: computes
77
+ `y = ground_truth_fn(origin + 1 + purge_gap, origin + 1 + purge_gap
78
+ + horizon)`, then scores each `forecast_fns[name](origin)` against
79
+ `y` via `score_fn`. `purge_gap=0` (the default) gives zero shared
80
+ days between the forecast origin and the ground-truth window by
81
+ construction; a larger `purge_gap` is a deliberate relaxation the
82
+ caller must opt into explicitly -- never a silent default.
83
+
84
+ Parameters
85
+ ----------
86
+ forecast_fns : callable, or dict of str -> callable
87
+ Each callable maps an origin (int) to a forecast dict in the
88
+ shape `matrix_energy_score`/`matrix_variogram_score` expect
89
+ (see `corrscore.scoring`). A single callable is treated as
90
+ `{"model": forecast_fns}`. Sharing one `ground_truth_fn` call
91
+ per origin across every model is deliberate: ground truth is
92
+ model-independent and often the more expensive computation, and
93
+ this shape is exactly what `corrscore.circular_block_bootstrap`,
94
+ `corrscore.diebold_mariano`, and `corrscore.model_confidence_set`
95
+ expect as input (`result.scores[name]`).
96
+ ground_truth_fn : callable
97
+ `(start, end) -> K x K matrix`. Called only with
98
+ `start = origin + 1 + purge_gap`, `end = start + horizon` --
99
+ never anything else. It is the caller's responsibility that
100
+ this function computes a genuinely forward-looking realized
101
+ estimate from `[start, end)`, not a trailing one -- the timing
102
+ guard above prevents overlap, but not a `ground_truth_fn` that
103
+ is itself defined as a trailing window.
104
+ origins : sequence of int
105
+ horizon : int
106
+ purge_gap : int, default=0
107
+ score_fn : callable, default=matrix_energy_score
108
+ severity_fn : callable, optional
109
+ `y -> float`, a per-origin realized-severity summary (e.g. mean
110
+ absolute off-diagonal correlation) for later use with
111
+ `corrscore.asymmetric_weighted_mean`.
112
+
113
+ Returns
114
+ -------
115
+ BacktestResult
116
+ """
117
+ if callable(forecast_fns):
118
+ forecast_fns = {"model": forecast_fns}
119
+ if purge_gap < 0:
120
+ raise ValueError(f"purge_gap must be >= 0, got {purge_gap}")
121
+ if horizon < 1:
122
+ raise ValueError(f"horizon must be >= 1, got {horizon}")
123
+
124
+ names = list(forecast_fns.keys())
125
+ scores: dict[str, list[float]] = {name: [] for name in names}
126
+ severities: list[float] = []
127
+ used_origins: list[int] = []
128
+
129
+ for origin in origins:
130
+ start = origin + 1 + purge_gap
131
+ end = start + horizon
132
+ y = np.asarray(ground_truth_fn(start, end), dtype=float)
133
+ for name in names:
134
+ forecast = forecast_fns[name](origin)
135
+ scores[name].append(score_fn(forecast, y))
136
+ if severity_fn is not None:
137
+ severities.append(severity_fn(y))
138
+ used_origins.append(origin)
139
+
140
+ scores_arr = {name: np.array(values) for name, values in scores.items()}
141
+ severity_arr = np.array(severities) if severity_fn is not None else None
142
+ return BacktestResult(
143
+ origins=used_origins,
144
+ scores=scores_arr,
145
+ severity=severity_arr,
146
+ horizon=horizon,
147
+ purge_gap=purge_gap,
148
+ )
corrscore/bootstrap.py ADDED
@@ -0,0 +1,81 @@
1
+ """Circular block-bootstrap significance testing on paired per-origin
2
+ score differentials, via `arch.bootstrap.CircularBlockBootstrap` -- kept
3
+ as a real dependency rather than vendored, since `arch` clears this
4
+ package's "widely used, actively maintained, field-standard" bar for
5
+ reuse, and block-bootstrap correctness (edge handling, unbiased block
6
+ placement) carries real reimplementation risk for a "just wrap it
7
+ correctly" component.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from typing import NamedTuple
12
+
13
+ import numpy as np
14
+ import numpy.typing as npt
15
+ from arch.bootstrap import CircularBlockBootstrap
16
+
17
+ __all__ = ["BootstrapResult", "circular_block_bootstrap"]
18
+
19
+
20
+ class BootstrapResult(NamedTuple):
21
+ """Result of `circular_block_bootstrap` for one block length.
22
+
23
+ Attributes
24
+ ----------
25
+ obs : float
26
+ The observed mean of `scores_a - scores_b`.
27
+ ci_lo, ci_hi : float
28
+ 95% percentile confidence interval from the bootstrap
29
+ distribution of that mean.
30
+ p_value : float
31
+ Two-sided bootstrap p-value for H0: true mean difference = 0.
32
+ significant : bool
33
+ `ci_lo > 0 or ci_hi < 0` -- whether zero falls outside the 95%
34
+ interval.
35
+ block_len : int
36
+ n_boot : int
37
+ """
38
+
39
+ obs: float
40
+ ci_lo: float
41
+ ci_hi: float
42
+ p_value: float
43
+ significant: bool
44
+ block_len: int
45
+ n_boot: int
46
+
47
+
48
+ def circular_block_bootstrap(
49
+ scores_a: npt.ArrayLike,
50
+ scores_b: npt.ArrayLike,
51
+ block_lengths: list[int],
52
+ n_boot: int = 2000,
53
+ seed: int | np.random.Generator | None = None,
54
+ ) -> dict[int, BootstrapResult]:
55
+ """Percentile circular block bootstrap on the paired differential
56
+ `d_i = scores_a[i] - scores_b[i]`, swept across `block_lengths` as a
57
+ sensitivity check rather than relying on one automatically "optimal"
58
+ block length.
59
+
60
+ Returns a dict keyed by each entry of `block_lengths`. The most
61
+ conservative (largest-p) entry is the one worth reporting as the
62
+ headline result.
63
+ """
64
+ diffs = np.asarray(scores_a, dtype=float) - np.asarray(scores_b, dtype=float)
65
+ results: dict[int, BootstrapResult] = {}
66
+ for block_len in block_lengths:
67
+ bs = CircularBlockBootstrap(block_len, diffs, seed=seed)
68
+ boot_means = bs.apply(lambda z: np.mean(z), n_boot).ravel()
69
+ ci_lo, ci_hi = np.percentile(boot_means, [2.5, 97.5])
70
+ tail_frac = min(float(np.mean(boot_means <= 0)), float(np.mean(boot_means >= 0)))
71
+ p_value = min(1.0, 2.0 * tail_frac)
72
+ results[block_len] = BootstrapResult(
73
+ obs=float(diffs.mean()),
74
+ ci_lo=float(ci_lo),
75
+ ci_hi=float(ci_hi),
76
+ p_value=p_value,
77
+ significant=bool(ci_lo > 0 or ci_hi < 0),
78
+ block_len=block_len,
79
+ n_boot=n_boot,
80
+ )
81
+ return results
@@ -0,0 +1,129 @@
1
+ """The Diebold-Mariano test (Diebold & Mariano, 1995), vendored rather
2
+ than depending on the small, thinly-adopted `dieboldmariano` PyPI
3
+ package.
4
+
5
+ Deliberately mirrors R's `forecast::dm.test` (Hyndman et al.) formula
6
+ by formula -- not because that package is depended on, but because it
7
+ is the field's de facto reference implementation and this module's own
8
+ test suite cross-checks against its output on fixed synthetic data as a
9
+ development-time oracle (`tests/_reference/dm_test_oracle_values.py`),
10
+ per this package's "oracle, not dependency" policy. In particular: the
11
+ Harvey, Leybourne & Newbold (1997) small-sample correction is always
12
+ applied (R's `dm.test` has no toggle for it either) and the p-value
13
+ comes from a Student-t distribution with `n - 1` degrees of freedom, not
14
+ the plain asymptotic normal.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ from typing import NamedTuple
19
+
20
+ import numpy as np
21
+ import numpy.typing as npt
22
+ from scipy.stats import t as student_t
23
+
24
+ __all__ = ["DieboldMarianoResult", "diebold_mariano"]
25
+
26
+
27
+ class DieboldMarianoResult(NamedTuple):
28
+ """Result of `diebold_mariano`.
29
+
30
+ Attributes
31
+ ----------
32
+ statistic : float
33
+ The Harvey-Leybourne-Newbold-corrected DM statistic.
34
+ p_value : float
35
+ Two-sided p-value, Student-t with `n - 1` degrees of freedom.
36
+ mean_diff : float
37
+ Mean of `loss_a - loss_b` (uncorrected).
38
+ n : int
39
+ """
40
+
41
+ statistic: float
42
+ p_value: float
43
+ mean_diff: float
44
+ n: int
45
+
46
+
47
+ def _autocovariance(d: npt.NDArray[np.float64], lag: int) -> float:
48
+ """Sample autocovariance at `lag`, normalized by the FULL sample
49
+ size `n` (not `n - lag`) at every lag -- the convention R's `acf()`
50
+ uses (and that `forecast::dm.test` relies on for its long-run-
51
+ variance estimate), not the `n - lag`-normalized "unbiased" variant.
52
+ Confirmed to matter here: with `n - lag` normalization this
53
+ function's h=1 oracle cases (lag=0 only, where the two conventions
54
+ coincide) passed, while every h>1 case (which needs lag>0
55
+ autocovariances) silently disagreed with R's own output -- exactly
56
+ how the discrepancy was actually caught."""
57
+ n = d.size
58
+ dbar = d.mean()
59
+ if lag == 0:
60
+ return float(np.mean((d - dbar) ** 2))
61
+ return float(np.sum((d[lag:] - dbar) * (d[:-lag] - dbar)) / n)
62
+
63
+
64
+ def diebold_mariano(
65
+ loss_a: npt.ArrayLike,
66
+ loss_b: npt.ArrayLike,
67
+ h: int = 1,
68
+ varestimator: str = "acf",
69
+ ) -> DieboldMarianoResult:
70
+ """Diebold-Mariano test on the paired loss differential
71
+ `d_t = loss_a[t] - loss_b[t]`.
72
+
73
+ Parameters
74
+ ----------
75
+ loss_a, loss_b : array-like
76
+ Per-origin losses (e.g. `BacktestResult.scores[name]`) -- NOT
77
+ raw forecast errors; unlike R's `dm.test(e1, e2, power=...)`,
78
+ this function does not apply a power transform, since the
79
+ objects being compared here are already non-negative scores
80
+ (energy score, variogram score). Passing already-nonnegative
81
+ losses with R's `power=1` reproduces this function's `d`
82
+ exactly (see the reference-oracle test suite).
83
+ h : int, default=1
84
+ Forecast horizon. Controls the long-run-variance truncation lag
85
+ (`h - 1`) and the Harvey-Leybourne-Newbold correction factor.
86
+ varestimator : {"acf", "bartlett"}, default="acf"
87
+ "acf": unweighted sum of sample autocovariances up to lag
88
+ `h - 1` (R's own default, and the only option when `h == 1`,
89
+ where the two coincide). "bartlett": Bartlett-kernel-weighted
90
+ sum (`1 - k/h` weights), matching R's `varestimator="bartlett"`.
91
+
92
+ Returns
93
+ -------
94
+ DieboldMarianoResult
95
+ """
96
+ d = np.asarray(loss_a, dtype=float) - np.asarray(loss_b, dtype=float)
97
+ n = d.size
98
+ if h < 1:
99
+ raise ValueError(f"h must be >= 1, got {h}")
100
+ if h > n:
101
+ raise ValueError(f"h ({h}) cannot exceed the number of observations ({n})")
102
+
103
+ gamma0 = _autocovariance(d, 0)
104
+ if varestimator == "acf" or h == 1:
105
+ long_run_var = gamma0 + 2.0 * sum(_autocovariance(d, k) for k in range(1, h))
106
+ elif varestimator == "bartlett":
107
+ long_run_var = gamma0 + 2.0 * sum(
108
+ (1.0 - k / h) * _autocovariance(d, k) for k in range(1, h)
109
+ )
110
+ else:
111
+ raise ValueError(f"Unknown varestimator: {varestimator!r}")
112
+ long_run_var /= n
113
+
114
+ if long_run_var <= 0:
115
+ raise ValueError(
116
+ "Estimated long-run variance of the loss differential is <= 0 "
117
+ "(a degenerate or perfectly-periodic differential); the DM "
118
+ "statistic is undefined. R's dm.test falls back to h=1 with a "
119
+ "warning in this situation -- retry with h=1 or varestimator="
120
+ "'bartlett' explicitly rather than silently doing so here."
121
+ )
122
+
123
+ dbar = float(d.mean())
124
+ raw_statistic = dbar / np.sqrt(long_run_var)
125
+ correction = np.sqrt((n + 1 - 2 * h + (h / n) * (h - 1)) / n)
126
+ statistic = float(raw_statistic * correction)
127
+ p_value = float(2.0 * student_t.cdf(-abs(statistic), df=n - 1))
128
+
129
+ return DieboldMarianoResult(statistic=statistic, p_value=p_value, mean_diff=dbar, n=n)
corrscore/mcs.py ADDED
@@ -0,0 +1,143 @@
1
+ """The Model Confidence Set (Hansen, Lunde & Nason, 2011), vendored
2
+ rather than depending on `model-confidence-set` (JLDC's GitHub-only,
3
+ no-PyPI-release Python port): reproduced directly from the paper, with
4
+ R's actively-maintained CRAN `MCS` package (Catania & Bernardi) and the
5
+ JLDC port used only as development-time correctness references, never
6
+ imported at runtime.
7
+
8
+ Implements the range statistic (T_R) elimination algorithm: repeatedly
9
+ test whether the current candidate set is statistically distinguishable
10
+ from its own best member; if so, drop the single worst-performing model
11
+ and repeat, until the surviving set cannot be rejected. The null
12
+ distribution and each round's studentizing standard errors both come
13
+ from the SAME joint circular block bootstrap of the loss matrix
14
+ (reusing `arch.bootstrap.CircularBlockBootstrap`, matching this
15
+ package's dependency policy).
16
+
17
+ Honest scope note: this is this package's own reasonable
18
+ operationalization of Hansen et al.'s range statistic and elimination
19
+ rule, not a literal line-by-line transcription of any specific existing
20
+ implementation's internal choices (e.g. R's `MCS` package's automatic
21
+ block-length selection is not reproduced -- `block_len` is a required,
22
+ caller-chosen parameter here, swept manually if robustness to it
23
+ matters, the same discipline `circular_block_bootstrap` already uses).
24
+ The test suite cross-checks this implementation's *verdict* (which
25
+ models survive) against R's `MCS::MCSprocedure` on fixed synthetic
26
+ data where the correct verdict is unambiguous by construction, not
27
+ byte-exact statistic/p-value agreement -- see
28
+ `tests/test_mcs.py` for why that is the honest bar for this
29
+ specific dependency.
30
+ """
31
+ from __future__ import annotations
32
+
33
+ from typing import Mapping, NamedTuple
34
+
35
+ import numpy as np
36
+ import numpy.typing as npt
37
+ from arch.bootstrap import CircularBlockBootstrap
38
+
39
+ __all__ = ["MCSResult", "model_confidence_set"]
40
+
41
+
42
+ class MCSResult(NamedTuple):
43
+ """Result of `model_confidence_set`.
44
+
45
+ Attributes
46
+ ----------
47
+ survivors : list of str
48
+ Models statistically indistinguishable from the best one at
49
+ `alpha`, in no particular order.
50
+ alpha : float
51
+ eliminated : list of (str, float)
52
+ Models removed, in elimination order, paired with the round's
53
+ p-value at the moment of that model's removal.
54
+ final_p_value : float
55
+ The surviving set's own p-value (>= alpha; or 1.0 if only one
56
+ model ever remained, a case in which rejection is trivially
57
+ impossible).
58
+ """
59
+
60
+ survivors: list[str]
61
+ alpha: float
62
+ eliminated: list[tuple[str, float]]
63
+ final_p_value: float
64
+
65
+
66
+ def _round_statistics(
67
+ sub_loss: npt.NDArray[np.float64], block_len: int, n_boot: int, seed: int | np.random.Generator | None
68
+ ) -> tuple[npt.NDArray[np.float64], float]:
69
+ """One elimination round: pairwise studentized statistics for the
70
+ current model set, and the bootstrap p-value for the joint range
71
+ null. Returns `(t_obs, p_value)`, `t_obs` shape (m, m)."""
72
+ m = sub_loss.shape[1]
73
+ dbar = np.array([[np.mean(sub_loss[:, i] - sub_loss[:, j]) for j in range(m)] for i in range(m)])
74
+
75
+ bs = CircularBlockBootstrap(block_len, sub_loss, seed=seed)
76
+ boot_dbar = np.empty((n_boot, m, m))
77
+ for rep, (pos, _kw) in enumerate(bs.bootstrap(n_boot)):
78
+ resampled = pos[0]
79
+ boot_dbar[rep] = np.array(
80
+ [[np.mean(resampled[:, i] - resampled[:, j]) for j in range(m)] for i in range(m)]
81
+ )
82
+
83
+ se = boot_dbar.std(axis=0, ddof=1)
84
+ se[se == 0] = np.inf # identical-loss pairs (including the diagonal) -> t_ij = 0, never inf/nan
85
+ t_obs = dbar / se
86
+ t_range_obs = float(np.abs(t_obs).max())
87
+
88
+ t_boot = (boot_dbar - dbar[None, :, :]) / se[None, :, :]
89
+ t_range_boot = np.abs(t_boot).max(axis=(1, 2))
90
+ p_value = float(np.mean(t_range_boot >= t_range_obs))
91
+ return t_obs, p_value
92
+
93
+
94
+ def model_confidence_set(
95
+ scores: Mapping[str, npt.ArrayLike],
96
+ alpha: float = 0.10,
97
+ block_len: int = 5,
98
+ n_boot: int = 1000,
99
+ seed: int | np.random.Generator | None = None,
100
+ ) -> MCSResult:
101
+ """The Model Confidence Set via the range statistic.
102
+
103
+ Parameters
104
+ ----------
105
+ scores : dict of str -> array-like
106
+ Per-model, per-origin losses, all the same length (e.g.
107
+ `BacktestResult.scores`). At least two models required.
108
+ alpha : float, default=0.10
109
+ block_len : int, default=5
110
+ Circular block-bootstrap block length. Required and
111
+ caller-chosen (see module docstring) -- rerun with different
112
+ values to check robustness, matching `circular_block_bootstrap`.
113
+ n_boot : int, default=1000
114
+ seed : int, np.random.Generator, or None
115
+
116
+ Returns
117
+ -------
118
+ MCSResult
119
+ """
120
+ names = list(scores.keys())
121
+ if len(names) < 2:
122
+ raise ValueError("model_confidence_set needs at least two models")
123
+ loss = np.column_stack([np.asarray(scores[name], dtype=float) for name in names])
124
+
125
+ current = list(range(len(names)))
126
+ eliminated: list[tuple[str, float]] = []
127
+ p_value = 1.0
128
+
129
+ while True:
130
+ if len(current) == 1:
131
+ p_value = 1.0
132
+ break
133
+ sub_names = [names[i] for i in current]
134
+ t_obs, p_value = _round_statistics(loss[:, current], block_len, n_boot, seed)
135
+ if p_value >= alpha:
136
+ break
137
+ avg_t = t_obs.mean(axis=1)
138
+ worst_local = int(np.argmax(avg_t))
139
+ eliminated.append((sub_names[worst_local], p_value))
140
+ current.pop(worst_local)
141
+
142
+ survivors = [names[i] for i in current]
143
+ return MCSResult(survivors=survivors, alpha=alpha, eliminated=eliminated, final_p_value=p_value)
corrscore/py.typed ADDED
File without changes
corrscore/scoring.py ADDED
@@ -0,0 +1,373 @@
1
+ """Matrix-aware proper scoring rules for correlation/covariance-matrix
2
+ forecasts: the energy score (Gneiting & Raftery, 2007) and the variogram
3
+ score (Scheuerer & Hamill, 2015), both adapted to score a full K x K
4
+ matrix rather than the vector-valued observation either was originally
5
+ published for.
6
+
7
+ Forecast representation (see `matrix_energy_score` and
8
+ `matrix_variogram_score` docstrings for the exact shape of each): a
9
+ plain dict tagged by `"kind"`, dispatching across a closed-form
10
+ tractability spectrum --
11
+
12
+ - "point": a single deterministic matrix (M=1).
13
+ - "mixture": any number of discrete atoms -- exact,
14
+ O(K_atoms^2) cost, no simulation, for
15
+ any number of atoms (not hard-coded to
16
+ two regimes).
17
+ - "isotropic_gaussian_mixture": discrete atoms each with an isotropic
18
+ Gaussian scatter -- exact for the
19
+ ENERGY score via the confluent
20
+ hypergeometric mean-norm formula; only
21
+ ever reachable through this explicit
22
+ kind, never inferred from a plain
23
+ ensemble automatically, because the
24
+ isotropy assumption is a real one.
25
+ - "ensemble": a general Monte Carlo draw set -- the
26
+ only kind with no closed form; used for
27
+ anything else, including simulated
28
+ paths from a regime-switching
29
+ correlation model.
30
+
31
+ `matrix_variogram_score` has a narrower closed form than the energy
32
+ score: exact for "point" and "mixture" (both are fully determined by a
33
+ finite set of deterministic atoms), Monte-Carlo-estimated for "ensemble"
34
+ and "isotropic_gaussian_mixture" -- no closed form for a fractional
35
+ absolute moment of a Gaussian difference has been derived for this
36
+ package; sampling is the honest v1 answer for those two kinds.
37
+
38
+ `matrix_geodesic_variogram_score` is a geometric (Fisher-Rao-aware)
39
+ variant of `matrix_variogram_score`: every free entry is passed through
40
+ `phi`, the signed Fisher-Rao arc length from rho=0 (a correlation entry's
41
+ distance-to-independence, treating the entry as its own isolated 2x2
42
+ correlation matrix under the affine-invariant metric), before the
43
+ ordinary variogram-score machinery runs on the transformed values. This
44
+ is `matrix_variogram_score` itself, unmodified, called on
45
+ phi-transformed entries -- not a new formula. It needs no new propriety
46
+ argument: |a-b|^p is already conditionally negative definite (of
47
+ negative type) on all of R for any real a, b (the same classical fact
48
+ that already licenses the flat variogram score), and composing a fixed
49
+ measurable transform with an already-valid kernel changes nothing about
50
+ that -- this construction is in fact proper for *any* fixed measurable
51
+ per-entry transform, not only ones with a metric interpretation.
52
+ Validated on synthetic matrices, on a real 16-factor equity-factor
53
+ panel, and -- the one comparison that matters most -- on real,
54
+ walk-forward regime-switching correlation forecast ensembles, where it
55
+ reproduced a statistically significant discrimination advantage over
56
+ the flat variogram score's own blind spot near the boundary of valid
57
+ correlation matrices. A whole-matrix geometric energy-score analogue was
58
+ prototyped alongside this one but did not reproduce the same advantage
59
+ on that comparison, so it is not (yet) shipped in this package; this is,
60
+ for now, the one geometric scoring rule promoted from prototype to
61
+ package code.
62
+ """
63
+ from __future__ import annotations
64
+
65
+ from typing import Any, Callable, Mapping, Sequence
66
+
67
+ import numpy as np
68
+ import numpy.typing as npt
69
+ from scipy.spatial.distance import pdist
70
+ from scipy.special import gammaln, hyp1f1
71
+
72
+ Matrix = npt.NDArray[np.float64]
73
+ Forecast = Mapping[str, Any]
74
+
75
+ __all__ = ["matrix_energy_score", "matrix_variogram_score", "matrix_geodesic_variogram_score"]
76
+
77
+
78
+ def _frobenius(a: npt.ArrayLike, b: npt.ArrayLike) -> float:
79
+ """Full K x K Frobenius distance -- the standard convention for a
80
+ matrix-valued energy score, not the upper-triangle-only convention
81
+ `matrix_variogram_score` uses (see that function's own docstring for
82
+ why the two conventions deliberately differ)."""
83
+ return float(np.linalg.norm(np.asarray(a, dtype=float) - np.asarray(b, dtype=float), ord="fro"))
84
+
85
+
86
+ def _upper(m: npt.ArrayLike) -> npt.NDArray[np.float64]:
87
+ """The K(K-1)/2 free upper-triangle entries (diagonal, always 1 for
88
+ a correlation matrix, and the mirrored lower triangle both excluded,
89
+ since they carry no independent information)."""
90
+ arr = np.asarray(m, dtype=float)
91
+ k = arr.shape[0]
92
+ return arr[np.triu_indices(k, k=1)]
93
+
94
+
95
+ def _expected_norm_isotropic_gaussian(norm_mu_sq: float, sigma: float, n: int) -> float:
96
+ """E[||Z||] for Z ~ N(mu, sigma^2 I_n), via the noncentral-chi mean
97
+ formula (Johnson, Kotz & Balakrishnan, 1994), evaluated through the
98
+ confluent hypergeometric function `scipy.special.hyp1f1` rather than
99
+ simulated.
100
+
101
+ `norm_mu_sq` is ||mu||^2 -- the formula only depends on mu through
102
+ this scalar, by spherical symmetry. `n == 0` (a degenerate,
103
+ parameter-free space, e.g. K=1) returns 0 directly.
104
+ """
105
+ if n <= 0:
106
+ return 0.0
107
+ if sigma <= 0.0:
108
+ return float(np.sqrt(norm_mu_sq))
109
+ log_ratio = gammaln((n + 1) / 2.0) - gammaln(n / 2.0)
110
+ coef = sigma * np.sqrt(2.0) * np.exp(log_ratio)
111
+ return float(coef * hyp1f1(-0.5, n / 2.0, -norm_mu_sq / (2.0 * sigma**2)))
112
+
113
+
114
+ def _expected_matrix_norm_isotropic(diff_free_entries: npt.NDArray[np.float64], sigma: float) -> float:
115
+ """E[||A||_F] where A is the (implicit) symmetric, zero-diagonal K x
116
+ K matrix built by placing an isotropic Gaussian free-entry vector u
117
+ ~ N(diff_free_entries, sigma^2 I_n) (n = K(K-1)/2) into both the
118
+ upper and lower triangle. Each free entry appears twice in A (once
119
+ per triangle), so ||A||_F = sqrt(2) * ||u||_2 -- this reuses
120
+ `_expected_norm_isotropic_gaussian` above with that bookkeeping
121
+ factor folded in, not a separate derivation."""
122
+ n = diff_free_entries.size
123
+ norm_mu_sq = float(diff_free_entries @ diff_free_entries)
124
+ return np.sqrt(2.0) * _expected_norm_isotropic_gaussian(norm_mu_sq, sigma, n)
125
+
126
+
127
+ def _energy_score_point(q: npt.ArrayLike, y: npt.ArrayLike) -> float:
128
+ return _frobenius(q, y)
129
+
130
+
131
+ def _energy_score_mixture(components: Sequence[tuple[float, npt.ArrayLike]], y: npt.ArrayLike) -> float:
132
+ weights = np.array([float(p) for p, _ in components])
133
+ mats = [np.asarray(q, dtype=float) for _, q in components]
134
+ k = len(mats)
135
+ term1 = sum(weights[i] * _frobenius(mats[i], y) for i in range(k))
136
+ term2 = 0.0
137
+ for i in range(k):
138
+ for j in range(k):
139
+ term2 += weights[i] * weights[j] * _frobenius(mats[i], mats[j])
140
+ return float(term1 - 0.5 * term2)
141
+
142
+
143
+ def _energy_score_isotropic_gaussian_mixture(
144
+ components: Sequence[tuple[float, npt.ArrayLike, float]], y: npt.ArrayLike
145
+ ) -> float:
146
+ weights = np.array([float(p) for p, _, _ in components])
147
+ mats_u = [_upper(q) for _, q, _ in components]
148
+ sigmas = [float(s) for _, _, s in components]
149
+ y_u = _upper(y)
150
+ k = len(components)
151
+
152
+ term1 = 0.0
153
+ for i in range(k):
154
+ term1 += weights[i] * _expected_matrix_norm_isotropic(mats_u[i] - y_u, sigmas[i])
155
+
156
+ term2 = 0.0
157
+ for i in range(k):
158
+ for j in range(k):
159
+ combined_sigma = float(np.sqrt(sigmas[i] ** 2 + sigmas[j] ** 2))
160
+ term2 += weights[i] * weights[j] * _expected_matrix_norm_isotropic(mats_u[i] - mats_u[j], combined_sigma)
161
+
162
+ return float(term1 - 0.5 * term2)
163
+
164
+
165
+ def _energy_score_ensemble(draws: Sequence[npt.ArrayLike], y: npt.ArrayLike) -> float:
166
+ flat = np.stack([np.asarray(d, dtype=float).ravel() for d in draws])
167
+ y_flat = np.asarray(y, dtype=float).ravel()
168
+ m = flat.shape[0]
169
+ term1 = float(np.mean(np.linalg.norm(flat - y_flat, axis=1)))
170
+ if m < 2:
171
+ return term1
172
+ pairwise = pdist(flat, metric="euclidean")
173
+ term2 = float(pairwise.sum() / (m**2))
174
+ return term1 - term2
175
+
176
+
177
+ def matrix_energy_score(forecast: Forecast, y: npt.ArrayLike) -> float:
178
+ """The energy score (Gneiting & Raftery, 2007), dispatched across the
179
+ closed-form spectrum (module docstring) by `forecast["kind"]`:
180
+
181
+ - {"kind": "point", "Q": Q} -> the M=1 special case, plain
182
+ Frobenius distance.
183
+ - {"kind": "mixture", "components": [(p_1, Q_1), ..., (p_K, Q_K)]}
184
+ -> Tier 1, exact. Any number of atoms, not just two.
185
+ - {"kind": "isotropic_gaussian_mixture",
186
+ "components": [(p_1, Q_1, sigma_1), ...]} -> Tier 2, exact.
187
+ `sigma_k` is the per-component scatter in the K(K-1)/2-dimensional
188
+ free-entry space, not in the full K x K ambient space.
189
+ - {"kind": "ensemble", "draws": [Q_1, ..., Q_M]} -> Tier 3, the
190
+ general Monte Carlo form, O(M^2) pairwise distances.
191
+
192
+ `y` is the realized K x K correlation (or covariance) matrix.
193
+ """
194
+ kind = forecast["kind"]
195
+ if kind == "point":
196
+ return _energy_score_point(forecast["Q"], y)
197
+ if kind == "mixture":
198
+ return _energy_score_mixture(forecast["components"], y)
199
+ if kind == "isotropic_gaussian_mixture":
200
+ return _energy_score_isotropic_gaussian_mixture(forecast["components"], y)
201
+ if kind == "ensemble":
202
+ return _energy_score_ensemble(forecast["draws"], y)
203
+ raise ValueError(f"Unknown forecast kind: {kind!r}")
204
+
205
+
206
+ def _variogram_diff_p_point(q: npt.ArrayLike, p: float) -> npt.NDArray[np.float64]:
207
+ q_u = _upper(q)
208
+ return np.abs(q_u[:, None] - q_u[None, :]) ** p
209
+
210
+
211
+ def _variogram_diff_p_mixture(components: Sequence[tuple[float, npt.ArrayLike]], p: float) -> npt.NDArray[np.float64]:
212
+ """Exact: a discrete mixture's entries are fully determined once the
213
+ atom is known, so E|X_i - X_j|^p is just the atom-weighted average
214
+ of the deterministic per-atom values -- no sampling needed."""
215
+ weights = [float(w) for w, _ in components]
216
+ total = None
217
+ for w, q in zip(weights, (c[1] for c in components)):
218
+ term = w * _variogram_diff_p_point(q, p)
219
+ total = term if total is None else total + term
220
+ assert total is not None
221
+ return total
222
+
223
+
224
+ def _sample_upper_triangle(forecast: Forecast, n_samples: int, rng: np.random.Generator) -> npt.NDArray[np.float64]:
225
+ """Draws used only by `matrix_variogram_score`'s Monte Carlo path
226
+ (see module docstring for why "ensemble" and
227
+ "isotropic_gaussian_mixture" have no closed form here). "ensemble"
228
+ reuses its existing draws directly (deterministic, no extra
229
+ sampling); "isotropic_gaussian_mixture" is genuinely sampled."""
230
+ kind = forecast["kind"]
231
+ if kind == "ensemble":
232
+ return np.stack([_upper(d) for d in forecast["draws"]])
233
+ if kind == "isotropic_gaussian_mixture":
234
+ components = forecast["components"]
235
+ weights = np.array([float(w) for w, _, _ in components])
236
+ weights = weights / weights.sum()
237
+ atom_idx = rng.choice(len(components), size=n_samples, p=weights)
238
+ n_free = _upper(components[0][1]).size
239
+ out = np.empty((n_samples, n_free))
240
+ for k, (_, q, sigma) in enumerate(components):
241
+ mask = atom_idx == k
242
+ count = int(mask.sum())
243
+ if count == 0:
244
+ continue
245
+ noise = rng.normal(0.0, float(sigma), size=(count, n_free)) if sigma > 0 else 0.0
246
+ out[mask] = _upper(q) + noise
247
+ return out
248
+ raise ValueError(f"Forecast kind {kind!r} has no Monte Carlo sampling path")
249
+
250
+
251
+ def matrix_variogram_score(
252
+ forecast: Forecast,
253
+ y: npt.ArrayLike,
254
+ p: float = 0.5,
255
+ weights: npt.ArrayLike | None = None,
256
+ n_samples: int = 500,
257
+ random_state: int | np.random.Generator | None = None,
258
+ ) -> float:
259
+ """The variogram score (Scheuerer & Hamill, 2015), adapted to index
260
+ i, j over the K(K-1)/2 free upper-triangle entries of the
261
+ correlation matrix, rather than the K original series the published
262
+ formula was defined for.
263
+
264
+ Exact (no sampling) for "point" and "mixture" forecasts. For
265
+ "ensemble" and "isotropic_gaussian_mixture", `E|X_i - X_j|^p` is
266
+ estimated by Monte Carlo (`n_samples` draws) -- no closed form for
267
+ this fractional absolute moment has been derived for this package
268
+ (unlike `matrix_energy_score`'s Tier 2 path); see the module
269
+ docstring. Cost and memory scale as O(n_samples * n_entries^2),
270
+ where n_entries = K(K-1)/2 -- the default `n_samples=500` keeps this
271
+ modest through K=16 (n_entries=120); pass a smaller value for larger
272
+ K if memory becomes a concern.
273
+
274
+ `weights` (if given) must be an (n_entries, n_entries) array;
275
+ defaults to uniform (all-ones), matching Scheuerer & Hamill's own
276
+ default and the published formula literally.
277
+ """
278
+ y_u = _upper(y)
279
+ n = y_u.size
280
+ y_diff_p = np.abs(y_u[:, None] - y_u[None, :]) ** p
281
+ w = np.ones((n, n)) if weights is None else np.asarray(weights, dtype=float)
282
+
283
+ kind = forecast["kind"]
284
+ if kind == "point":
285
+ exp_diff_p = _variogram_diff_p_point(forecast["Q"], p)
286
+ elif kind == "mixture":
287
+ exp_diff_p = _variogram_diff_p_mixture(forecast["components"], p)
288
+ elif kind in ("ensemble", "isotropic_gaussian_mixture"):
289
+ rng = random_state if isinstance(random_state, np.random.Generator) else np.random.default_rng(random_state)
290
+ samples = _sample_upper_triangle(forecast, n_samples, rng)
291
+ exp_diff_p = (np.abs(samples[:, :, None] - samples[:, None, :]) ** p).mean(axis=0)
292
+ else:
293
+ raise ValueError(f"Unknown forecast kind: {kind!r}")
294
+
295
+ return float(np.sum(w * (y_diff_p - exp_diff_p) ** 2))
296
+
297
+
298
+ def _phi(rho: npt.ArrayLike) -> npt.NDArray[np.float64]:
299
+ """Signed Fisher-Rao arc length from rho=0: treats a single
300
+ correlation entry as its own isolated 2x2 correlation matrix
301
+ C(rho) = [[1,rho],[rho,1]] under the affine-invariant metric, and
302
+ returns sign(rho) * d_FR(I, C(rho)). Closed form (derived and
303
+ verified to machine precision against the general eigenvalue-based
304
+ Fisher-Rao formula):
305
+
306
+ phi(rho) = sign(rho) * sqrt(0.5 * (log(1+|rho|)^2 + log(1-|rho|)^2))
307
+
308
+ Smooth, odd, and strictly increasing on (-1, 1) -> R (verified
309
+ numerically); a close cousin of the century-old Fisher z-transform
310
+ arctanh(rho), diverging slightly faster as rho -> +-1.
311
+ """
312
+ rho = np.asarray(rho, dtype=float)
313
+ a = np.abs(rho)
314
+ d = np.sqrt(0.5 * (np.log1p(a) ** 2 + np.log1p(-a) ** 2))
315
+ return np.sign(rho) * d
316
+
317
+
318
+ def _phi_transform_matrix(q: npt.ArrayLike) -> npt.NDArray[np.float64]:
319
+ """`_phi` applied entrywise off-diagonal. The diagonal is zeroed
320
+ before transforming, not left at the input's own (always exactly
321
+ 1.0) diagonal: `_upper` never reads it, but `_phi(1.0)` is -inf
322
+ (log1p(-1)), which is harmless numerically (never read) but raises a
323
+ spurious divide-by-zero warning on every call otherwise."""
324
+ q = np.asarray(q, dtype=float).copy()
325
+ np.fill_diagonal(q, 0.0)
326
+ return _phi(q)
327
+
328
+
329
+ def _phi_transform_forecast(forecast: Forecast) -> Forecast:
330
+ kind = forecast["kind"]
331
+ if kind == "point":
332
+ return {"kind": "point", "Q": _phi_transform_matrix(forecast["Q"])}
333
+ if kind == "mixture":
334
+ return {"kind": "mixture", "components": [(w, _phi_transform_matrix(q)) for w, q in forecast["components"]]}
335
+ if kind == "ensemble":
336
+ return {"kind": "ensemble", "draws": [_phi_transform_matrix(d) for d in forecast["draws"]]}
337
+ raise ValueError(f"Unknown forecast kind: {kind!r}")
338
+
339
+
340
+ def matrix_geodesic_variogram_score(
341
+ forecast: Forecast,
342
+ y: npt.ArrayLike,
343
+ p: float = 0.5,
344
+ weights: npt.ArrayLike | None = None,
345
+ n_samples: int = 500,
346
+ random_state: int | np.random.Generator | None = None,
347
+ ) -> float:
348
+ """The geometric (Fisher-Rao-aware) variogram score: `matrix_
349
+ variogram_score` itself, unmodified, called on every free entry
350
+ passed through `_phi` first (see module docstring). Same forecast
351
+ dict shapes, same `p`/`weights`/`n_samples`/`random_state` semantics,
352
+ same closed-form-vs-Monte-Carlo split by `forecast["kind"]` as
353
+ `matrix_variogram_score` -- this function only changes what happens
354
+ to the raw entries before that machinery runs.
355
+
356
+ Proper for the same reason `matrix_variogram_score` is (it elicits a
357
+ pairwise-moment vector for any fixed measurable per-entry transform)
358
+ -- no new argument specific to `_phi` is needed. Not strictly proper,
359
+ for the identical reason the flat variogram score is not, though the
360
+ two scores' blind spots are provably different (a common shift in
361
+ phi-space vs. a common shift in raw rho-space) -- running both
362
+ alongside each other is a real, not just pragmatic, recommendation.
363
+
364
+ Validated on real, walk-forward regime-switching correlation forecast
365
+ ensembles: a statistically significant discrimination advantage over
366
+ the flat variogram score on that comparison, unlike a separate
367
+ whole-matrix geodesic energy score prototype, which did not reproduce
368
+ an advantage there and has accordingly not (yet) been promoted into
369
+ this package.
370
+ """
371
+ y_t = _phi_transform_matrix(y)
372
+ forecast_t = _phi_transform_forecast(forecast)
373
+ return matrix_variogram_score(forecast_t, y_t, p=p, weights=weights, n_samples=n_samples, random_state=random_state)
corrscore/utils.py ADDED
@@ -0,0 +1,37 @@
1
+ """Outcome-severity-weighted aggregation -- a small, general-purpose
2
+ utility for weighting per-origin scores by how severe the realized
3
+ outcome was, surfaced while dogfooding this package against a real
4
+ regime-detection backtest.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import numpy as np
9
+ import numpy.typing as npt
10
+
11
+ __all__ = ["asymmetric_weighted_mean"]
12
+
13
+
14
+ def asymmetric_weighted_mean(values: npt.ArrayLike, severity: npt.ArrayLike, kappa: float = 1.0) -> float:
15
+ """Weight per-origin `values` (typically per-origin scores from a
16
+ `BacktestResult`) by realized outcome severity: origins at or above
17
+ the median `severity` are weighted `kappa` times as heavily as
18
+ origins below it.
19
+
20
+ `severity` should be a non-circular, outcome-derived measure -- e.g.
21
+ the mean absolute off-diagonal correlation of the realized
22
+ ground-truth window (`BacktestResult.severity`, via
23
+ `backtest_zero_overlap`'s `severity_fn`) -- not any model's own
24
+ regime call, to avoid biasing which origins get up-weighted toward
25
+ whichever model is being evaluated.
26
+
27
+ `kappa=1.0` (the default) reduces to a plain mean.
28
+ """
29
+ values_arr = np.asarray(values, dtype=float)
30
+ severity_arr = np.asarray(severity, dtype=float)
31
+ if values_arr.shape != severity_arr.shape:
32
+ raise ValueError(
33
+ f"values and severity must have the same shape, got {values_arr.shape} and {severity_arr.shape}"
34
+ )
35
+ median = np.median(severity_arr)
36
+ weights = np.where(severity_arr >= median, kappa, 1.0)
37
+ return float(np.sum(values_arr * weights) / np.sum(weights))
@@ -0,0 +1,129 @@
1
+ Metadata-Version: 2.5
2
+ Name: corrscore
3
+ Version: 0.1.0
4
+ Summary: Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts: zero-overlap walk-forward evaluation, energy/variogram scoring, and significance testing (block bootstrap, Diebold-Mariano, Model Confidence Set).
5
+ Author-email: Vinh Nguyen <vinhnguyen3455@gmail.com>
6
+ License: MIT
7
+ License-File: LICENSE
8
+ Classifier: Intended Audience :: Financial and Insurance Industry
9
+ Classifier: Intended Audience :: Science/Research
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Topic :: Office/Business :: Financial :: Investment
16
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
17
+ Requires-Python: >=3.10
18
+ Requires-Dist: arch>=6.3
19
+ Requires-Dist: numpy>=1.24
20
+ Requires-Dist: scipy>=1.10
21
+ Provides-Extra: dev
22
+ Requires-Dist: hypothesis>=6.100; extra == 'dev'
23
+ Requires-Dist: mypy>=1.10; extra == 'dev'
24
+ Requires-Dist: pytest>=8; extra == 'dev'
25
+ Description-Content-Type: text/markdown
26
+
27
+ # corrscore
28
+
29
+ Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts, in Python.
30
+
31
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
32
+
33
+ ## Why this exists
34
+
35
+ Forecasting a correlation or covariance matrix is common in risk management and portfolio
36
+ construction — but evaluating that forecast correctly is not routine. Two mistakes are easy to
37
+ make and hard to notice:
38
+
39
+ 1. **Naive matrix-comparison metrics aren't proper scoring rules.** A metric that isn't a proper
40
+ scoring rule can reward a forecaster for hedging toward a "safe" answer instead of reporting
41
+ their honest best guess — the evaluation itself creates bad incentives.
42
+ 2. **Walk-forward evaluation windows are easy to overlap with the estimation window**, silently
43
+ leaking future information into a backtest and inflating apparent skill.
44
+
45
+ `corrscore` evaluates a forecast; it never produces one. It doesn't fit a model, doesn't implement
46
+ any particular correlation-dynamics model, and doesn't fetch or clean data — it's a focused
47
+ evaluation layer you drop on top of whatever you're already forecasting with.
48
+
49
+ ## What it offers
50
+
51
+ - **Matrix-aware energy and variogram scores.** The two standard proper scoring rules from the
52
+ forecast-verification literature, generalized from their usual vector-valued form to score a
53
+ full `K x K` correlation/covariance matrix directly.
54
+ - **A geometric variant of the variogram score**, aware of the fact that correlation matrices live
55
+ on a curved space rather than flat Euclidean space — sharper at detecting forecast danger as a
56
+ matrix approaches the boundary of validity (near-singular, highly correlated regimes).
57
+ - **Four forecast representations**, not just point forecasts: a single deterministic matrix, a
58
+ discrete mixture of any number of atoms, an isotropic-Gaussian mixture, or a general Monte Carlo
59
+ ensemble — with closed-form scoring wherever one exists, Monte Carlo estimation only where it
60
+ doesn't.
61
+ - **A backtesting harness (`backtest_zero_overlap`)** that makes the specific, easy-to-make
62
+ lookahead bug — ground truth computed from a window that overlaps the forecast origin —
63
+ structurally impossible to reproduce, rather than something you have to remember to get right.
64
+ - **Significance testing**, not just point comparisons: a circular block bootstrap for
65
+ serially-dependent score differentials, the Diebold-Mariano test, and the Model Confidence Set
66
+ — so "is model A really better than model B" has an actual answer.
67
+
68
+ ```python
69
+ from corrscore import matrix_energy_score, backtest_zero_overlap
70
+ from corrscore import circular_block_bootstrap, diebold_mariano, model_confidence_set
71
+
72
+ result = backtest_zero_overlap(
73
+ forecast_fns={"naive": naive_forecast, "filter": my_model.forecast},
74
+ ground_truth_fn=realized_correlation, # (start, end) -> K x K matrix
75
+ origins=origins,
76
+ horizon=10,
77
+ )
78
+ sig = circular_block_bootstrap(result.scores["filter"], result.scores["naive"], block_lengths=[1, 3, 6, 10])
79
+ mcs = model_confidence_set(result.scores, alpha=0.10)
80
+ ```
81
+
82
+ ## Install
83
+
84
+ ```bash
85
+ pip install corrscore
86
+ ```
87
+
88
+ Requires Python 3.10-3.12. Runtime dependencies are `numpy`, `scipy`, and `arch` (for the circular
89
+ block bootstrap) — nothing else.
90
+
91
+ ## Validation
92
+
93
+ Every scoring rule and test in this package is checked against an independent source of truth, not
94
+ just its own self-consistency: `diebold_mariano` and `model_confidence_set` are cross-checked
95
+ against live-generated R oracles (`forecast::dm.test` byte-exact, `MCS::MCSprocedure`
96
+ verdict-matched), and every closed-form scoring formula is checked against brute-force Monte Carlo
97
+ simulation of the object it claims to score. Fully type-hinted and `mypy`-clean. See `tests/` for
98
+ the full suite.
99
+
100
+ ## Related work
101
+
102
+ No existing package (Python or R) treats a correlation/covariance matrix as the forecast object
103
+ with purge-aware walk-forward splitting and a proper scoring rule built in — see
104
+ [`docs/survey/`](docs/survey/) for the full landscape survey and the specific reuse-vs.-vendor
105
+ decision behind each dependency.
106
+
107
+ ## Citation
108
+
109
+ A companion paper describing the package's design and methodology is available as a working
110
+ paper on SSRN: [Nguyen (2026), "corrscore: Matrix-Aware Proper Scoring Rules and Significance
111
+ Testing for Correlation and Covariance Forecasts in
112
+ Python"](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=7358478). See
113
+ [`CITATION.cff`](CITATION.cff) for a machine-readable citation.
114
+
115
+ ## Development
116
+
117
+ ```bash
118
+ git clone https://github.com/vinhnguyen3455/corrscore
119
+ cd corrscore
120
+ pip install -e ".[dev]"
121
+ pytest -q
122
+ mypy src/corrscore
123
+ ```
124
+
125
+ Issues and pull requests welcome.
126
+
127
+ ## License
128
+
129
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,12 @@
1
+ corrscore/__init__.py,sha256=xjKePWKScFs2h7EjlYW9HPJIPoNTZU5evw7M4TFGC-E,728
2
+ corrscore/backtest.py,sha256=QfYoZNOFNAB-8vQ_ukS07ZzdWaAk9VXJGhsE9m20Wfk,6143
3
+ corrscore/bootstrap.py,sha256=qJkLDh95JucUQ5NyCsO4zDDPy_Xm3K7X0uaj4KWFbZU,2801
4
+ corrscore/diebold_mariano.py,sha256=Oi3aZDFhZAG5HY3Q5CpjOnHk7-kbejUPwQWIKu-RrJU,5093
5
+ corrscore/mcs.py,sha256=ki6KH47dpFgCaotrCyKa_B9nR2f98cRJ46SD-MZmEsw,5636
6
+ corrscore/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ corrscore/scoring.py,sha256=0QJrtjylgaGovnoq7Mdz-oe8J1aDFeVhGPNsRsEVVuc,17546
8
+ corrscore/utils.py,sha256=_CV08KOct3AWVcLV36CmcTkXuiC6qv3ZQu5osJs1xwo,1587
9
+ corrscore-0.1.0.dist-info/METADATA,sha256=eLeYUCpW9opAOLApSmqF1RmnhVuMtMgTmhLmv-LLjRU,5889
10
+ corrscore-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
11
+ corrscore-0.1.0.dist-info/licenses/LICENSE,sha256=jB6MybiFM3YMWm7FRRZsoRPq7FOuzYut1XkoQOVGpYU,1068
12
+ corrscore-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Vinh Nguyen
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.