corrscore 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corrscore-0.1.0/.github/workflows/publish.yml +36 -0
- corrscore-0.1.0/.github/workflows/test.yml +23 -0
- corrscore-0.1.0/.gitignore +9 -0
- corrscore-0.1.0/CITATION.cff +22 -0
- corrscore-0.1.0/LICENSE +21 -0
- corrscore-0.1.0/PKG-INFO +129 -0
- corrscore-0.1.0/README.md +103 -0
- corrscore-0.1.0/docs/survey/README.md +56 -0
- corrscore-0.1.0/docs/survey/related-software-survey.md +36 -0
- corrscore-0.1.0/pyproject.toml +44 -0
- corrscore-0.1.0/src/corrscore/__init__.py +21 -0
- corrscore-0.1.0/src/corrscore/backtest.py +148 -0
- corrscore-0.1.0/src/corrscore/bootstrap.py +81 -0
- corrscore-0.1.0/src/corrscore/diebold_mariano.py +129 -0
- corrscore-0.1.0/src/corrscore/mcs.py +143 -0
- corrscore-0.1.0/src/corrscore/py.typed +0 -0
- corrscore-0.1.0/src/corrscore/scoring.py +373 -0
- corrscore-0.1.0/src/corrscore/utils.py +37 -0
- corrscore-0.1.0/tests/_reference/dm_test_oracle_values.py +44 -0
- corrscore-0.1.0/tests/_reference/gen_dm_oracle.R +31 -0
- corrscore-0.1.0/tests/_reference/gen_mcs_case.R +23 -0
- corrscore-0.1.0/tests/_reference/mcs_oracle_case.py +18 -0
- corrscore-0.1.0/tests/test_backtest.py +125 -0
- corrscore-0.1.0/tests/test_bootstrap.py +41 -0
- corrscore-0.1.0/tests/test_diebold_mariano.py +95 -0
- corrscore-0.1.0/tests/test_mcs.py +72 -0
- corrscore-0.1.0/tests/test_scoring.py +350 -0
- corrscore-0.1.0/tests/test_utils.py +50 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
build:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v4
|
|
12
|
+
- uses: actions/setup-python@v5
|
|
13
|
+
with:
|
|
14
|
+
python-version: "3.12"
|
|
15
|
+
- name: Install build backend
|
|
16
|
+
run: pip install build
|
|
17
|
+
- name: Build sdist and wheel
|
|
18
|
+
run: python -m build
|
|
19
|
+
- uses: actions/upload-artifact@v4
|
|
20
|
+
with:
|
|
21
|
+
name: dist
|
|
22
|
+
path: dist/
|
|
23
|
+
|
|
24
|
+
publish:
|
|
25
|
+
needs: build
|
|
26
|
+
runs-on: ubuntu-latest
|
|
27
|
+
environment: pypi
|
|
28
|
+
permissions:
|
|
29
|
+
id-token: write
|
|
30
|
+
steps:
|
|
31
|
+
- uses: actions/download-artifact@v4
|
|
32
|
+
with:
|
|
33
|
+
name: dist
|
|
34
|
+
path: dist/
|
|
35
|
+
- name: Publish to PyPI
|
|
36
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: test
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
strategy:
|
|
11
|
+
matrix:
|
|
12
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: ${{ matrix.python-version }}
|
|
18
|
+
- name: Install package with dev dependencies
|
|
19
|
+
run: pip install -e ".[dev]"
|
|
20
|
+
- name: Run tests
|
|
21
|
+
run: pytest -q
|
|
22
|
+
- name: Type-check
|
|
23
|
+
run: mypy src/corrscore
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use this software, please cite the accompanying paper below."
|
|
3
|
+
title: "corrscore: Matrix-Aware Proper Scoring Rules and Significance Testing for Correlation and Covariance Forecasts in Python"
|
|
4
|
+
type: software
|
|
5
|
+
authors:
|
|
6
|
+
- family-names: Nguyen
|
|
7
|
+
given-names: Vinh
|
|
8
|
+
# orcid: "https://orcid.org/XXXX-XXXX-XXXX-XXXX" # fill in
|
|
9
|
+
url: "https://github.com/vinhnguyen3455/corrscore"
|
|
10
|
+
license: MIT
|
|
11
|
+
version: 0.1.0
|
|
12
|
+
date-released: 2026-08-27
|
|
13
|
+
preferred-citation:
|
|
14
|
+
type: unpublished
|
|
15
|
+
title: "corrscore: Matrix-Aware Proper Scoring Rules and Significance Testing for Correlation and Covariance Forecasts in Python"
|
|
16
|
+
authors:
|
|
17
|
+
- family-names: Nguyen
|
|
18
|
+
given-names: Vinh
|
|
19
|
+
# orcid: "https://orcid.org/XXXX-XXXX-XXXX-XXXX" # fill in
|
|
20
|
+
year: 2026
|
|
21
|
+
url: "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=7358478"
|
|
22
|
+
notes: "Working paper, SSRN. Also pending submission to arXiv (q-fin.CP) -- update this record with the arXiv identifier once endorsement clears."
|
corrscore-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vinh Nguyen
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
corrscore-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: corrscore
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts: zero-overlap walk-forward evaluation, energy/variogram scoring, and significance testing (block bootstrap, Diebold-Mariano, Model Confidence Set).
|
|
5
|
+
Author-email: Vinh Nguyen <vinhnguyen3455@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Classifier: Intended Audience :: Financial and Insurance Industry
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Office/Business :: Financial :: Investment
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Requires-Dist: arch>=6.3
|
|
19
|
+
Requires-Dist: numpy>=1.24
|
|
20
|
+
Requires-Dist: scipy>=1.10
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
23
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
24
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# corrscore
|
|
28
|
+
|
|
29
|
+
Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts, in Python.
|
|
30
|
+
|
|
31
|
+
[](LICENSE)
|
|
32
|
+
|
|
33
|
+
## Why this exists
|
|
34
|
+
|
|
35
|
+
Forecasting a correlation or covariance matrix is common in risk management and portfolio
|
|
36
|
+
construction — but evaluating that forecast correctly is not routine. Two mistakes are easy to
|
|
37
|
+
make and hard to notice:
|
|
38
|
+
|
|
39
|
+
1. **Naive matrix-comparison metrics aren't proper scoring rules.** A metric that isn't a proper
|
|
40
|
+
scoring rule can reward a forecaster for hedging toward a "safe" answer instead of reporting
|
|
41
|
+
their honest best guess — the evaluation itself creates bad incentives.
|
|
42
|
+
2. **Walk-forward evaluation windows are easy to overlap with the estimation window**, silently
|
|
43
|
+
leaking future information into a backtest and inflating apparent skill.
|
|
44
|
+
|
|
45
|
+
`corrscore` evaluates a forecast; it never produces one. It doesn't fit a model, doesn't implement
|
|
46
|
+
any particular correlation-dynamics model, and doesn't fetch or clean data — it's a focused
|
|
47
|
+
evaluation layer you drop on top of whatever you're already forecasting with.
|
|
48
|
+
|
|
49
|
+
## What it offers
|
|
50
|
+
|
|
51
|
+
- **Matrix-aware energy and variogram scores.** The two standard proper scoring rules from the
|
|
52
|
+
forecast-verification literature, generalized from their usual vector-valued form to score a
|
|
53
|
+
full `K x K` correlation/covariance matrix directly.
|
|
54
|
+
- **A geometric variant of the variogram score**, aware of the fact that correlation matrices live
|
|
55
|
+
on a curved space rather than flat Euclidean space — sharper at detecting forecast danger as a
|
|
56
|
+
matrix approaches the boundary of validity (near-singular, highly correlated regimes).
|
|
57
|
+
- **Four forecast representations**, not just point forecasts: a single deterministic matrix, a
|
|
58
|
+
discrete mixture of any number of atoms, an isotropic-Gaussian mixture, or a general Monte Carlo
|
|
59
|
+
ensemble — with closed-form scoring wherever one exists, Monte Carlo estimation only where it
|
|
60
|
+
doesn't.
|
|
61
|
+
- **A backtesting harness (`backtest_zero_overlap`)** that makes the specific, easy-to-make
|
|
62
|
+
lookahead bug — ground truth computed from a window that overlaps the forecast origin —
|
|
63
|
+
structurally impossible to reproduce, rather than something you have to remember to get right.
|
|
64
|
+
- **Significance testing**, not just point comparisons: a circular block bootstrap for
|
|
65
|
+
serially-dependent score differentials, the Diebold-Mariano test, and the Model Confidence Set
|
|
66
|
+
— so "is model A really better than model B" has an actual answer.
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from corrscore import matrix_energy_score, backtest_zero_overlap
|
|
70
|
+
from corrscore import circular_block_bootstrap, diebold_mariano, model_confidence_set
|
|
71
|
+
|
|
72
|
+
result = backtest_zero_overlap(
|
|
73
|
+
forecast_fns={"naive": naive_forecast, "filter": my_model.forecast},
|
|
74
|
+
ground_truth_fn=realized_correlation, # (start, end) -> K x K matrix
|
|
75
|
+
origins=origins,
|
|
76
|
+
horizon=10,
|
|
77
|
+
)
|
|
78
|
+
sig = circular_block_bootstrap(result.scores["filter"], result.scores["naive"], block_lengths=[1, 3, 6, 10])
|
|
79
|
+
mcs = model_confidence_set(result.scores, alpha=0.10)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Install
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install corrscore
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Requires Python 3.10-3.12. Runtime dependencies are `numpy`, `scipy`, and `arch` (for the circular
|
|
89
|
+
block bootstrap) — nothing else.
|
|
90
|
+
|
|
91
|
+
## Validation
|
|
92
|
+
|
|
93
|
+
Every scoring rule and test in this package is checked against an independent source of truth, not
|
|
94
|
+
just its own self-consistency: `diebold_mariano` and `model_confidence_set` are cross-checked
|
|
95
|
+
against live-generated R oracles (`forecast::dm.test` byte-exact, `MCS::MCSprocedure`
|
|
96
|
+
verdict-matched), and every closed-form scoring formula is checked against brute-force Monte Carlo
|
|
97
|
+
simulation of the object it claims to score. Fully type-hinted and `mypy`-clean. See `tests/` for
|
|
98
|
+
the full suite.
|
|
99
|
+
|
|
100
|
+
## Related work
|
|
101
|
+
|
|
102
|
+
No existing package (Python or R) treats a correlation/covariance matrix as the forecast object
|
|
103
|
+
with purge-aware walk-forward splitting and a proper scoring rule built in — see
|
|
104
|
+
[`docs/survey/`](docs/survey/) for the full landscape survey and the specific reuse-vs.-vendor
|
|
105
|
+
decision behind each dependency.
|
|
106
|
+
|
|
107
|
+
## Citation
|
|
108
|
+
|
|
109
|
+
A companion paper describing the package's design and methodology is available as a working
|
|
110
|
+
paper on SSRN: [Nguyen (2026), "corrscore: Matrix-Aware Proper Scoring Rules and Significance
|
|
111
|
+
Testing for Correlation and Covariance Forecasts in
|
|
112
|
+
Python"](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=7358478). See
|
|
113
|
+
[`CITATION.cff`](CITATION.cff) for a machine-readable citation.
|
|
114
|
+
|
|
115
|
+
## Development
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
git clone https://github.com/vinhnguyen3455/corrscore
|
|
119
|
+
cd corrscore
|
|
120
|
+
pip install -e ".[dev]"
|
|
121
|
+
pytest -q
|
|
122
|
+
mypy src/corrscore
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Issues and pull requests welcome.
|
|
126
|
+
|
|
127
|
+
## License
|
|
128
|
+
|
|
129
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# corrscore
|
|
2
|
+
|
|
3
|
+
Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts, in Python.
|
|
4
|
+
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
|
|
7
|
+
## Why this exists
|
|
8
|
+
|
|
9
|
+
Forecasting a correlation or covariance matrix is common in risk management and portfolio
|
|
10
|
+
construction — but evaluating that forecast correctly is not routine. Two mistakes are easy to
|
|
11
|
+
make and hard to notice:
|
|
12
|
+
|
|
13
|
+
1. **Naive matrix-comparison metrics aren't proper scoring rules.** A metric that isn't a proper
|
|
14
|
+
scoring rule can reward a forecaster for hedging toward a "safe" answer instead of reporting
|
|
15
|
+
their honest best guess — the evaluation itself creates bad incentives.
|
|
16
|
+
2. **Walk-forward evaluation windows are easy to overlap with the estimation window**, silently
|
|
17
|
+
leaking future information into a backtest and inflating apparent skill.
|
|
18
|
+
|
|
19
|
+
`corrscore` evaluates a forecast; it never produces one. It doesn't fit a model, doesn't implement
|
|
20
|
+
any particular correlation-dynamics model, and doesn't fetch or clean data — it's a focused
|
|
21
|
+
evaluation layer you drop on top of whatever you're already forecasting with.
|
|
22
|
+
|
|
23
|
+
## What it offers
|
|
24
|
+
|
|
25
|
+
- **Matrix-aware energy and variogram scores.** The two standard proper scoring rules from the
|
|
26
|
+
forecast-verification literature, generalized from their usual vector-valued form to score a
|
|
27
|
+
full `K x K` correlation/covariance matrix directly.
|
|
28
|
+
- **A geometric variant of the variogram score**, aware of the fact that correlation matrices live
|
|
29
|
+
on a curved space rather than flat Euclidean space — sharper at detecting forecast danger as a
|
|
30
|
+
matrix approaches the boundary of validity (near-singular, highly correlated regimes).
|
|
31
|
+
- **Four forecast representations**, not just point forecasts: a single deterministic matrix, a
|
|
32
|
+
discrete mixture of any number of atoms, an isotropic-Gaussian mixture, or a general Monte Carlo
|
|
33
|
+
ensemble — with closed-form scoring wherever one exists, Monte Carlo estimation only where it
|
|
34
|
+
doesn't.
|
|
35
|
+
- **A backtesting harness (`backtest_zero_overlap`)** that makes the specific, easy-to-make
|
|
36
|
+
lookahead bug — ground truth computed from a window that overlaps the forecast origin —
|
|
37
|
+
structurally impossible to reproduce, rather than something you have to remember to get right.
|
|
38
|
+
- **Significance testing**, not just point comparisons: a circular block bootstrap for
|
|
39
|
+
serially-dependent score differentials, the Diebold-Mariano test, and the Model Confidence Set
|
|
40
|
+
— so "is model A really better than model B" has an actual answer.
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
from corrscore import matrix_energy_score, backtest_zero_overlap
|
|
44
|
+
from corrscore import circular_block_bootstrap, diebold_mariano, model_confidence_set
|
|
45
|
+
|
|
46
|
+
result = backtest_zero_overlap(
|
|
47
|
+
forecast_fns={"naive": naive_forecast, "filter": my_model.forecast},
|
|
48
|
+
ground_truth_fn=realized_correlation, # (start, end) -> K x K matrix
|
|
49
|
+
origins=origins,
|
|
50
|
+
horizon=10,
|
|
51
|
+
)
|
|
52
|
+
sig = circular_block_bootstrap(result.scores["filter"], result.scores["naive"], block_lengths=[1, 3, 6, 10])
|
|
53
|
+
mcs = model_confidence_set(result.scores, alpha=0.10)
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Install
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
pip install corrscore
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Requires Python 3.10-3.12. Runtime dependencies are `numpy`, `scipy`, and `arch` (for the circular
|
|
63
|
+
block bootstrap) — nothing else.
|
|
64
|
+
|
|
65
|
+
## Validation
|
|
66
|
+
|
|
67
|
+
Every scoring rule and test in this package is checked against an independent source of truth, not
|
|
68
|
+
just its own self-consistency: `diebold_mariano` and `model_confidence_set` are cross-checked
|
|
69
|
+
against live-generated R oracles (`forecast::dm.test` byte-exact, `MCS::MCSprocedure`
|
|
70
|
+
verdict-matched), and every closed-form scoring formula is checked against brute-force Monte Carlo
|
|
71
|
+
simulation of the object it claims to score. Fully type-hinted and `mypy`-clean. See `tests/` for
|
|
72
|
+
the full suite.
|
|
73
|
+
|
|
74
|
+
## Related work
|
|
75
|
+
|
|
76
|
+
No existing package (Python or R) treats a correlation/covariance matrix as the forecast object
|
|
77
|
+
with purge-aware walk-forward splitting and a proper scoring rule built in — see
|
|
78
|
+
[`docs/survey/`](docs/survey/) for the full landscape survey and the specific reuse-vs.-vendor
|
|
79
|
+
decision behind each dependency.
|
|
80
|
+
|
|
81
|
+
## Citation
|
|
82
|
+
|
|
83
|
+
A companion paper describing the package's design and methodology is available as a working
|
|
84
|
+
paper on SSRN: [Nguyen (2026), "corrscore: Matrix-Aware Proper Scoring Rules and Significance
|
|
85
|
+
Testing for Correlation and Covariance Forecasts in
|
|
86
|
+
Python"](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=7358478). See
|
|
87
|
+
[`CITATION.cff`](CITATION.cff) for a machine-readable citation.
|
|
88
|
+
|
|
89
|
+
## Development
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
git clone https://github.com/vinhnguyen3455/corrscore
|
|
93
|
+
cd corrscore
|
|
94
|
+
pip install -e ".[dev]"
|
|
95
|
+
pytest -q
|
|
96
|
+
mypy src/corrscore
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Issues and pull requests welcome.
|
|
100
|
+
|
|
101
|
+
## License
|
|
102
|
+
|
|
103
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Related-software survey — findings (complete 2026-08-25)
|
|
2
|
+
|
|
3
|
+
This package's scope rests on one survey, run live against PyPI/CRAN/GitHub
|
|
4
|
+
(license, last-activity, and — the actual point of the exercise — an
|
|
5
|
+
adversarial search trying to falsify the claimed gap) rather than assumed
|
|
6
|
+
from general familiarity. See
|
|
7
|
+
[`related-software-survey.md`](related-software-survey.md) in this same
|
|
8
|
+
directory for the full package-by-package table.
|
|
9
|
+
|
|
10
|
+
## What this settles, concretely
|
|
11
|
+
|
|
12
|
+
1. **The gap is real and survives an adversarial search.** No package, in
|
|
13
|
+
either Python or R, treats a correlation/covariance matrix as the
|
|
14
|
+
forecast object with purge-aware walk-forward splitting and a proper
|
|
15
|
+
scoring rule built in — every existing tool operates one layer down (a
|
|
16
|
+
scalar loss, a vector observation, or a generic index-based CV
|
|
17
|
+
splitter). Confirmed by targeted searches for the specific claim
|
|
18
|
+
("correlation matrix forecast evaluation package," GitHub topic search
|
|
19
|
+
on `covariance-matrix`), not just an absence of a name that happened to
|
|
20
|
+
come to mind.
|
|
21
|
+
|
|
22
|
+
2. **`arch` (Sheppard) is the one dependency that clears this package's own
|
|
23
|
+
bar** (widely used, actively maintained, field-standard) — kept
|
|
24
|
+
directly for `CircularBlockBootstrap` rather than reimplemented.
|
|
25
|
+
Everything else found in the survey either isn't clearly established
|
|
26
|
+
enough (`scoringrules`: 98★, and its own PyPI/GitHub release metadata
|
|
27
|
+
disagreed on dates during this check — an unresolved integrity flag)
|
|
28
|
+
or isn't installable at all as a real dependency (`model-confidence-set`:
|
|
29
|
+
GitHub-only, no PyPI release, 21★; `mlfinlab`: gone commercial).
|
|
30
|
+
|
|
31
|
+
3. **R is comparatively well-served; Python is the real gap.** R's
|
|
32
|
+
`scoringRules` (CRAN, 2024-09-18), `MCS` (Catania, CRAN, **2026-03-19** —
|
|
33
|
+
genuinely current), `multDM`, and base `forecast::dm.test` together
|
|
34
|
+
cover almost everything this package implements — just not assembled
|
|
35
|
+
into one matrix-aware, zero-overlap pipeline. That absence of assembly,
|
|
36
|
+
not absence of building blocks, is why an R companion package remains a
|
|
37
|
+
plausible but non-urgent future extension, not a v1 requirement.
|
|
38
|
+
|
|
39
|
+
4. **Dependency policy, applied concretely per package**: `README.md`'s
|
|
40
|
+
own "Dependency policy" table and `src/corrscore/`'s module docstrings
|
|
41
|
+
(`scoring.py`, `diebold_mariano.py`, `mcs.py`) each state the specific
|
|
42
|
+
reuse-vs.-vendor decision and why, rather than leaving it implicit.
|
|
43
|
+
`model_confidence_set` in particular is checked against R's
|
|
44
|
+
`MCS::MCSprocedure` verdict on a fixed synthetic case
|
|
45
|
+
(`tests/_reference/mcs_oracle_case.py`) rather than claimed correct by
|
|
46
|
+
construction — it was the clearest case in the survey for vendoring
|
|
47
|
+
over depending, and also the implementation with the most room for a
|
|
48
|
+
subtle bug, so the highest-value place to have an external check.
|
|
49
|
+
|
|
50
|
+
## What this survey did not cover
|
|
51
|
+
|
|
52
|
+
Whether `matrix_variogram_score`'s specific entry-indexing convention
|
|
53
|
+
matches any convention a reader coming from the meteorology/
|
|
54
|
+
forecast-verification literature would expect by default — there is no
|
|
55
|
+
existing matrix-valued precedent to check against, which is the whole
|
|
56
|
+
reason a convention had to be chosen rather than adopted.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Related-software survey
|
|
2
|
+
|
|
3
|
+
Live-verified 2026-08-25 against PyPI/CRAN package pages, GitHub license/activity
|
|
4
|
+
checks, and source inspection where relevant.
|
|
5
|
+
|
|
6
|
+
| Package | Ecosystem | License | Last verified activity | Relevant to this harness |
|
|
7
|
+
|---|---|---|---|---|
|
|
8
|
+
| `scoringrules` | Python/PyPI | Apache-2.0 | v0.11.0; GitHub and PyPI release-date metadata disagree — unresolved | Energy + variogram score, vector-valued only, no matrix awareness |
|
|
9
|
+
| `properscoring` | Python/PyPI | Apache-2.0 | v0.1, 2015 — dead | Superseded by `scoringrules` |
|
|
10
|
+
| `arch` (Sheppard) | Python/PyPI | NCSA | v8.0.0, actively maintained | `arch.bootstrap.CircularBlockBootstrap` — the canonical Python implementation this package reuses directly |
|
|
11
|
+
| `tscv` / `timeseriescv` | Python/PyPI | BSD-3 / MIT | 2023 / 2018, stale | Gap-based CV splitters, not matrix-aware; `backtest_zero_overlap`'s own mechanics are thin enough not to depend on either |
|
|
12
|
+
| `mlfinlab` (Hudson & Thames) | Python/PyPI | closed-source, paid | confirmed live | López de Prado's own purged/embargoed K-fold code — no longer usable as a free dependency |
|
|
13
|
+
| `dieboldmariano` | Python/PyPI | MIT | v1.1.0, Jan 2025 | Standalone DM test, scalar loss series — vendored instead (see below) |
|
|
14
|
+
| `model-confidence-set` (JLDC) | Python/GitHub only | MIT | 21★, no PyPI release | Only maintained Python MCS port found — also the flimsiest dependency in the whole survey; vendored instead |
|
|
15
|
+
| `scoringRules` | R/CRAN | GPL-2/3 | v1.1.3, 2024-09-18 | Vector-valued energy score only, same matrix gap as the Python side |
|
|
16
|
+
| `MCS` (Catania) | R/CRAN | GPL-2 | v0.2.0, **2026-03-19** | Canonical Hansen et al. (2011) MCS — the best-maintained MCS implementation in either language; used as a development-time oracle |
|
|
17
|
+
| `multDM` | R/CRAN | GPL-3 | v1.1.5, 2025-03-08 | Multivariate DM test |
|
|
18
|
+
| `forecast::dm.test` | R (base ecosystem) | — | long-established | Canonical scalar DM test in R; used as this package's `diebold_mariano` development-time oracle |
|
|
19
|
+
|
|
20
|
+
**The gap check, adversarially searched, not falsified.** Targeted searches
|
|
21
|
+
("correlation matrix forecast evaluation package," "covariance forecast
|
|
22
|
+
backtesting python," "regime-switching correlation model validation
|
|
23
|
+
software," GitHub topic search on `covariance-matrix`) turned up no
|
|
24
|
+
package, in either language, that treats a correlation or covariance
|
|
25
|
+
matrix as the forecast object with purge-aware walk-forward splitting and
|
|
26
|
+
a proper scoring rule built in.
|
|
27
|
+
|
|
28
|
+
## Dependency decisions (mirrors `README.md`'s own table)
|
|
29
|
+
|
|
30
|
+
| Package | Role in `corrscore` | Decision |
|
|
31
|
+
|---|---|---|
|
|
32
|
+
| `numpy`, `scipy` | array math, `hyp1f1`/`gammaln`, `pdist` | dependency |
|
|
33
|
+
| `arch` | `CircularBlockBootstrap` | dependency |
|
|
34
|
+
| `scoringrules` | energy/variogram score math | vendored (`scoring.py`) |
|
|
35
|
+
| `dieboldmariano` | DM test | vendored (`diebold_mariano.py`), oracle-checked against `forecast::dm.test` |
|
|
36
|
+
| `model-confidence-set` | Model Confidence Set | vendored (`mcs.py`), verdict-checked against `MCS::MCSprocedure` |
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "corrscore"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Matrix-aware proper-scoring-rule backtesting for correlation and covariance forecasts: zero-overlap walk-forward evaluation, energy/variogram scoring, and significance testing (block bootstrap, Diebold-Mariano, Model Confidence Set)."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
authors = [
|
|
12
|
+
{ name = "Vinh Nguyen", email = "vinhnguyen3455@gmail.com" },
|
|
13
|
+
]
|
|
14
|
+
requires-python = ">=3.10"
|
|
15
|
+
dependencies = [
|
|
16
|
+
"numpy>=1.24",
|
|
17
|
+
"scipy>=1.10",
|
|
18
|
+
"arch>=6.3",
|
|
19
|
+
]
|
|
20
|
+
classifiers = [
|
|
21
|
+
"License :: OSI Approved :: MIT License",
|
|
22
|
+
"Programming Language :: Python :: 3",
|
|
23
|
+
"Programming Language :: Python :: 3.10",
|
|
24
|
+
"Programming Language :: Python :: 3.11",
|
|
25
|
+
"Programming Language :: Python :: 3.12",
|
|
26
|
+
"Intended Audience :: Financial and Insurance Industry",
|
|
27
|
+
"Intended Audience :: Science/Research",
|
|
28
|
+
"Topic :: Office/Business :: Financial :: Investment",
|
|
29
|
+
"Topic :: Scientific/Engineering :: Mathematics",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[project.optional-dependencies]
|
|
33
|
+
dev = [
|
|
34
|
+
"pytest>=8",
|
|
35
|
+
"hypothesis>=6.100",
|
|
36
|
+
"mypy>=1.10",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[tool.hatch.build.targets.wheel]
|
|
40
|
+
packages = ["src/corrscore"]
|
|
41
|
+
|
|
42
|
+
[tool.mypy]
|
|
43
|
+
warn_unused_configs = true
|
|
44
|
+
ignore_missing_imports = true
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
from .backtest import BacktestResult, backtest_zero_overlap
|
|
2
|
+
from .bootstrap import BootstrapResult, circular_block_bootstrap
|
|
3
|
+
from .diebold_mariano import DieboldMarianoResult, diebold_mariano
|
|
4
|
+
from .mcs import MCSResult, model_confidence_set
|
|
5
|
+
from .scoring import matrix_energy_score, matrix_geodesic_variogram_score, matrix_variogram_score
|
|
6
|
+
from .utils import asymmetric_weighted_mean
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"matrix_energy_score",
|
|
10
|
+
"matrix_variogram_score",
|
|
11
|
+
"matrix_geodesic_variogram_score",
|
|
12
|
+
"backtest_zero_overlap",
|
|
13
|
+
"BacktestResult",
|
|
14
|
+
"circular_block_bootstrap",
|
|
15
|
+
"BootstrapResult",
|
|
16
|
+
"diebold_mariano",
|
|
17
|
+
"DieboldMarianoResult",
|
|
18
|
+
"model_confidence_set",
|
|
19
|
+
"MCSResult",
|
|
20
|
+
"asymmetric_weighted_mean",
|
|
21
|
+
]
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""The zero-overlap walk-forward backtest driver.
|
|
2
|
+
|
|
3
|
+
Design note: an earlier sketch of this API had a `window` parameter
|
|
4
|
+
defaulting to `horizon`. Working through the actual mechanics precisely
|
|
5
|
+
during implementation surfaced a cleaner, more honestly-scoped contract,
|
|
6
|
+
documented here rather than silently substituted. This harness does not,
|
|
7
|
+
and cannot, police how much history a caller's `forecast_fn` consults
|
|
8
|
+
internally -- that model is opaque to the harness (it might be a
|
|
9
|
+
full-history discounted filter, a short trailing window, or anything
|
|
10
|
+
else), exactly the same responsibility boundary scikit-learn's
|
|
11
|
+
`TimeSeriesSplit` leaves to the caller. What the harness genuinely *can*
|
|
12
|
+
and does enforce, unconditionally, is that `ground_truth_fn` is always
|
|
13
|
+
called with a start point strictly after the origin
|
|
14
|
+
(`origin + 1 + purge_gap`), never a window that reaches back before or
|
|
15
|
+
across it. That guards against a real class of bug this package exists
|
|
16
|
+
to prevent: computing "ground truth" as a window *ending* near the
|
|
17
|
+
origin rather than a window *starting* strictly after it, which silently
|
|
18
|
+
leaks estimation-window information into the evaluation. This API makes
|
|
19
|
+
that specific mistake structurally impossible to reproduce, since the
|
|
20
|
+
caller never controls the start point passed to `ground_truth_fn`.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from typing import Any, Callable, Mapping, NamedTuple, Sequence
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
import numpy.typing as npt
|
|
28
|
+
|
|
29
|
+
from .scoring import matrix_energy_score
|
|
30
|
+
|
|
31
|
+
ForecastFn = Callable[[int], Mapping[str, Any]]
|
|
32
|
+
GroundTruthFn = Callable[[int, int], npt.ArrayLike]
|
|
33
|
+
ScoreFn = Callable[[Mapping[str, Any], npt.ArrayLike], float]
|
|
34
|
+
SeverityFn = Callable[[npt.NDArray[np.float64]], float]
|
|
35
|
+
|
|
36
|
+
__all__ = ["BacktestResult", "backtest_zero_overlap"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class BacktestResult(NamedTuple):
|
|
40
|
+
"""Result of `backtest_zero_overlap`.
|
|
41
|
+
|
|
42
|
+
Attributes
|
|
43
|
+
----------
|
|
44
|
+
origins : list of int
|
|
45
|
+
The origins actually scored, in order.
|
|
46
|
+
scores : dict of str -> ndarray
|
|
47
|
+
Per-model, per-origin scores, one array per key of the
|
|
48
|
+
`forecast_fns` mapping passed in, aligned with `origins`.
|
|
49
|
+
severity : ndarray or None
|
|
50
|
+
Per-origin realized-severity values (`severity_fn(y)` at each
|
|
51
|
+
origin), or None if `severity_fn` was not supplied. Intended for
|
|
52
|
+
`corrscore.asymmetric_weighted_mean`.
|
|
53
|
+
horizon : int
|
|
54
|
+
purge_gap : int
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
origins: list[int]
|
|
58
|
+
scores: dict[str, npt.NDArray[np.float64]]
|
|
59
|
+
severity: npt.NDArray[np.float64] | None
|
|
60
|
+
horizon: int
|
|
61
|
+
purge_gap: int
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def backtest_zero_overlap(
|
|
65
|
+
forecast_fns: Mapping[str, ForecastFn] | ForecastFn,
|
|
66
|
+
ground_truth_fn: GroundTruthFn,
|
|
67
|
+
origins: Sequence[int],
|
|
68
|
+
horizon: int,
|
|
69
|
+
purge_gap: int = 0,
|
|
70
|
+
score_fn: ScoreFn = matrix_energy_score,
|
|
71
|
+
severity_fn: SeverityFn | None = None,
|
|
72
|
+
) -> BacktestResult:
|
|
73
|
+
"""Score one or more forecasting methods against a proper,
|
|
74
|
+
zero-overlap-by-construction ground truth.
|
|
75
|
+
|
|
76
|
+
For each `origin` in `origins`: computes
|
|
77
|
+
`y = ground_truth_fn(origin + 1 + purge_gap, origin + 1 + purge_gap
|
|
78
|
+
+ horizon)`, then scores each `forecast_fns[name](origin)` against
|
|
79
|
+
`y` via `score_fn`. `purge_gap=0` (the default) gives zero shared
|
|
80
|
+
days between the forecast origin and the ground-truth window by
|
|
81
|
+
construction; a larger `purge_gap` is a deliberate relaxation the
|
|
82
|
+
caller must opt into explicitly -- never a silent default.
|
|
83
|
+
|
|
84
|
+
Parameters
|
|
85
|
+
----------
|
|
86
|
+
forecast_fns : callable, or dict of str -> callable
|
|
87
|
+
Each callable maps an origin (int) to a forecast dict in the
|
|
88
|
+
shape `matrix_energy_score`/`matrix_variogram_score` expect
|
|
89
|
+
(see `corrscore.scoring`). A single callable is treated as
|
|
90
|
+
`{"model": forecast_fns}`. Sharing one `ground_truth_fn` call
|
|
91
|
+
per origin across every model is deliberate: ground truth is
|
|
92
|
+
model-independent and often the more expensive computation, and
|
|
93
|
+
this shape is exactly what `corrscore.circular_block_bootstrap`,
|
|
94
|
+
`corrscore.diebold_mariano`, and `corrscore.model_confidence_set`
|
|
95
|
+
expect as input (`result.scores[name]`).
|
|
96
|
+
ground_truth_fn : callable
|
|
97
|
+
`(start, end) -> K x K matrix`. Called only with
|
|
98
|
+
`start = origin + 1 + purge_gap`, `end = start + horizon` --
|
|
99
|
+
never anything else. It is the caller's responsibility that
|
|
100
|
+
this function computes a genuinely forward-looking realized
|
|
101
|
+
estimate from `[start, end)`, not a trailing one -- the timing
|
|
102
|
+
guard above prevents overlap, but not a `ground_truth_fn` that
|
|
103
|
+
is itself defined as a trailing window.
|
|
104
|
+
origins : sequence of int
|
|
105
|
+
horizon : int
|
|
106
|
+
purge_gap : int, default=0
|
|
107
|
+
score_fn : callable, default=matrix_energy_score
|
|
108
|
+
severity_fn : callable, optional
|
|
109
|
+
`y -> float`, a per-origin realized-severity summary (e.g. mean
|
|
110
|
+
absolute off-diagonal correlation) for later use with
|
|
111
|
+
`corrscore.asymmetric_weighted_mean`.
|
|
112
|
+
|
|
113
|
+
Returns
|
|
114
|
+
-------
|
|
115
|
+
BacktestResult
|
|
116
|
+
"""
|
|
117
|
+
if callable(forecast_fns):
|
|
118
|
+
forecast_fns = {"model": forecast_fns}
|
|
119
|
+
if purge_gap < 0:
|
|
120
|
+
raise ValueError(f"purge_gap must be >= 0, got {purge_gap}")
|
|
121
|
+
if horizon < 1:
|
|
122
|
+
raise ValueError(f"horizon must be >= 1, got {horizon}")
|
|
123
|
+
|
|
124
|
+
names = list(forecast_fns.keys())
|
|
125
|
+
scores: dict[str, list[float]] = {name: [] for name in names}
|
|
126
|
+
severities: list[float] = []
|
|
127
|
+
used_origins: list[int] = []
|
|
128
|
+
|
|
129
|
+
for origin in origins:
|
|
130
|
+
start = origin + 1 + purge_gap
|
|
131
|
+
end = start + horizon
|
|
132
|
+
y = np.asarray(ground_truth_fn(start, end), dtype=float)
|
|
133
|
+
for name in names:
|
|
134
|
+
forecast = forecast_fns[name](origin)
|
|
135
|
+
scores[name].append(score_fn(forecast, y))
|
|
136
|
+
if severity_fn is not None:
|
|
137
|
+
severities.append(severity_fn(y))
|
|
138
|
+
used_origins.append(origin)
|
|
139
|
+
|
|
140
|
+
scores_arr = {name: np.array(values) for name, values in scores.items()}
|
|
141
|
+
severity_arr = np.array(severities) if severity_fn is not None else None
|
|
142
|
+
return BacktestResult(
|
|
143
|
+
origins=used_origins,
|
|
144
|
+
scores=scores_arr,
|
|
145
|
+
severity=severity_arr,
|
|
146
|
+
horizon=horizon,
|
|
147
|
+
purge_gap=purge_gap,
|
|
148
|
+
)
|