squeeze-kernel 0.5.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/PKG-INFO +13 -2
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/README.md +11 -1
- squeeze_kernel-0.7.0/pyproject.toml +60 -0
- squeeze_kernel-0.5.0/pyproject.toml → squeeze_kernel-0.7.0/pyproject.toml.orig +10 -2
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/src/squeeze_kernel/__init__.py +1 -1
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/src/squeeze_kernel/batch.py +21 -31
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/src/squeeze_kernel/estimator.py +416 -108
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.7.0}/src/squeeze_kernel/kernels.py +10 -9
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: squeeze-kernel
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage
|
|
5
5
|
Keywords: covariance,correlation,ewma,kernel,risk,streaming
|
|
6
6
|
Author: Robert Kende
|
|
@@ -19,6 +19,7 @@ Requires-Dist: numpy>=1.24
|
|
|
19
19
|
Requires-Dist: pytest>=7.0 ; extra == 'dev'
|
|
20
20
|
Requires-Dist: ruff>=0.7 ; extra == 'dev'
|
|
21
21
|
Requires-Dist: scipy>=1.11 ; extra == 'dev'
|
|
22
|
+
Requires-Dist: mypy>=1.10 ; extra == 'dev'
|
|
22
23
|
Requires-Dist: scipy>=1.11 ; extra == 'full'
|
|
23
24
|
Requires-Python: >=3.10
|
|
24
25
|
Project-URL: Homepage, https://github.com/r0k3/squeeze-kernel
|
|
@@ -116,7 +117,7 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
116
117
|
|
|
117
118
|
## Advanced options
|
|
118
119
|
|
|
119
|
-
**
|
|
120
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
120
121
|
|
|
121
122
|
```python
|
|
122
123
|
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
@@ -144,6 +145,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
144
145
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
145
146
|
```
|
|
146
147
|
|
|
148
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
152
|
+
# ... update loop ...
|
|
153
|
+
m = est.usable_mask
|
|
154
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
155
|
+
```
|
|
156
|
+
|
|
147
157
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
148
158
|
|
|
149
159
|
## How it works
|
|
@@ -162,6 +172,7 @@ The complete update is a natural-gradient step on the Gaussian log-likelihood, w
|
|
|
162
172
|
uv sync --extra full --extra dev
|
|
163
173
|
uv run python -m pytest # test suite
|
|
164
174
|
uv run python -m ruff check . # lint
|
|
175
|
+
uv run mypy # strict type check (src/squeeze_kernel)
|
|
165
176
|
uv build # build sdist + wheel
|
|
166
177
|
```
|
|
167
178
|
|
|
@@ -86,7 +86,7 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
86
86
|
|
|
87
87
|
## Advanced options
|
|
88
88
|
|
|
89
|
-
**
|
|
89
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
90
90
|
|
|
91
91
|
```python
|
|
92
92
|
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
@@ -114,6 +114,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
114
114
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
115
115
|
```
|
|
116
116
|
|
|
117
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
121
|
+
# ... update loop ...
|
|
122
|
+
m = est.usable_mask
|
|
123
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
124
|
+
```
|
|
125
|
+
|
|
117
126
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
118
127
|
|
|
119
128
|
## How it works
|
|
@@ -132,6 +141,7 @@ The complete update is a natural-gradient step on the Gaussian log-likelihood, w
|
|
|
132
141
|
uv sync --extra full --extra dev
|
|
133
142
|
uv run python -m pytest # test suite
|
|
134
143
|
uv run python -m ruff check . # lint
|
|
144
|
+
uv run mypy # strict type check (src/squeeze_kernel)
|
|
135
145
|
uv build # build sdist + wheel
|
|
136
146
|
```
|
|
137
147
|
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.10.6,<0.11.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "squeeze-kernel"
|
|
7
|
+
version = "0.7.0"
|
|
8
|
+
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
keywords = [
|
|
13
|
+
"covariance",
|
|
14
|
+
"correlation",
|
|
15
|
+
"ewma",
|
|
16
|
+
"kernel",
|
|
17
|
+
"risk",
|
|
18
|
+
"streaming",
|
|
19
|
+
]
|
|
20
|
+
classifiers = [
|
|
21
|
+
"Development Status :: 4 - Beta",
|
|
22
|
+
"Intended Audience :: Science/Research",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Mathematics",
|
|
24
|
+
"License :: OSI Approved :: MIT License",
|
|
25
|
+
"Programming Language :: Python :: 3",
|
|
26
|
+
"Programming Language :: Python :: 3.10",
|
|
27
|
+
"Programming Language :: Python :: 3.11",
|
|
28
|
+
"Programming Language :: Python :: 3.12",
|
|
29
|
+
"Programming Language :: Python :: 3.13",
|
|
30
|
+
"Operating System :: OS Independent",
|
|
31
|
+
]
|
|
32
|
+
dependencies = ["numpy>=1.24"]
|
|
33
|
+
|
|
34
|
+
[[project.authors]]
|
|
35
|
+
name = "Robert Kende"
|
|
36
|
+
|
|
37
|
+
[project.urls]
|
|
38
|
+
Homepage = "https://github.com/r0k3/squeeze-kernel"
|
|
39
|
+
Repository = "https://github.com/r0k3/squeeze-kernel"
|
|
40
|
+
Issues = "https://github.com/r0k3/squeeze-kernel/issues"
|
|
41
|
+
|
|
42
|
+
[project.optional-dependencies]
|
|
43
|
+
full = ["scipy>=1.11"]
|
|
44
|
+
dev = [
|
|
45
|
+
"pytest>=7.0",
|
|
46
|
+
"ruff>=0.7",
|
|
47
|
+
"scipy>=1.11",
|
|
48
|
+
"mypy>=1.10",
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
[tool.pytest.ini_options]
|
|
52
|
+
testpaths = ["tests"]
|
|
53
|
+
|
|
54
|
+
[tool.mypy]
|
|
55
|
+
files = ["src/squeeze_kernel"]
|
|
56
|
+
strict = true
|
|
57
|
+
|
|
58
|
+
[[tool.mypy.overrides]]
|
|
59
|
+
module = "scipy.*"
|
|
60
|
+
ignore_missing_imports = true
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "squeeze-kernel"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.7.0"
|
|
8
8
|
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -32,7 +32,15 @@ Issues = "https://github.com/r0k3/squeeze-kernel/issues"
|
|
|
32
32
|
|
|
33
33
|
[project.optional-dependencies]
|
|
34
34
|
full = ["scipy>=1.11"]
|
|
35
|
-
dev = ["pytest>=7.0", "ruff>=0.7", "scipy>=1.11"]
|
|
35
|
+
dev = ["pytest>=7.0", "ruff>=0.7", "scipy>=1.11", "mypy>=1.10"]
|
|
36
36
|
|
|
37
37
|
[tool.pytest.ini_options]
|
|
38
38
|
testpaths = ["tests"]
|
|
39
|
+
|
|
40
|
+
[tool.mypy]
|
|
41
|
+
files = ["src/squeeze_kernel"]
|
|
42
|
+
strict = true
|
|
43
|
+
|
|
44
|
+
[[tool.mypy.overrides]]
|
|
45
|
+
module = "scipy.*"
|
|
46
|
+
ignore_missing_imports = true
|
|
@@ -2,69 +2,59 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
5
7
|
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
6
9
|
|
|
7
10
|
from squeeze_kernel.estimator import SqueezeKernelEstimator
|
|
8
|
-
from squeeze_kernel.kernels import KernelFn
|
|
9
11
|
|
|
10
12
|
|
|
11
13
|
def estimate_squeeze_cov(
|
|
12
|
-
returns,
|
|
14
|
+
returns: ArrayLike,
|
|
13
15
|
*,
|
|
14
|
-
lambda_vol: float = 0.98,
|
|
15
|
-
lambda_corr: float = 0.996,
|
|
16
|
-
kappa: float | None = None,
|
|
17
|
-
kernel_fn: KernelFn | None = None,
|
|
18
|
-
kernel_kwargs: dict[str, object] | None = None,
|
|
19
|
-
epsilon: float = 1e-8,
|
|
20
|
-
shrinkage: str | float = "auto",
|
|
21
|
-
shrinkage_delta: float = 0.10,
|
|
22
|
-
impute_missing: bool = False,
|
|
23
|
-
impute_threshold: float = 0.6,
|
|
24
16
|
with_corr: bool = True,
|
|
25
17
|
with_weights: bool = False,
|
|
18
|
+
**estimator_kwargs: Any,
|
|
26
19
|
) -> tuple[np.ndarray, np.ndarray | None, np.ndarray | None]:
|
|
27
20
|
"""Estimate streaming covariance over an entire returns panel.
|
|
28
21
|
|
|
22
|
+
Runs one :class:`SqueezeKernelEstimator` over the panel day by day and
|
|
23
|
+
collects the estimate path. ``n_assets`` is taken from the panel shape;
|
|
24
|
+
every other keyword argument is forwarded unchanged to the estimator, so
|
|
25
|
+
batch mode reaches the full estimator surface — ``kappa``, ``shrinkage``,
|
|
26
|
+
``shrinkage_target``, ``corr_half_lives``, ``weight_statistic``,
|
|
27
|
+
``vol_anchor_phi``, ``min_obs``, custom kernels, and so on (see the
|
|
28
|
+
estimator's docstring for the complete list and defaults).
|
|
29
|
+
|
|
29
30
|
Parameters
|
|
30
31
|
----------
|
|
31
32
|
returns : array-like, shape (T, n)
|
|
32
33
|
2D return matrix. May contain NaN for missing observations.
|
|
33
|
-
kappa : float, optional
|
|
34
|
-
Saturation parameter for the default Fisher kernel.
|
|
35
|
-
kernel_fn : callable, optional
|
|
36
|
-
Custom kernel ``(d2, *, n_observed, **kw) -> float``.
|
|
37
|
-
kernel_kwargs : dict, optional
|
|
38
|
-
Extra keyword arguments forwarded to ``kernel_fn``.
|
|
39
34
|
with_corr : bool
|
|
40
35
|
If True, also return the correlation tensor.
|
|
41
36
|
with_weights : bool
|
|
42
37
|
If True, also return per-timestamp kernel weights.
|
|
38
|
+
**estimator_kwargs
|
|
39
|
+
Passed through to ``SqueezeKernelEstimator``.
|
|
43
40
|
|
|
44
41
|
Returns
|
|
45
42
|
-------
|
|
46
43
|
cov : ndarray, shape (T, n, n)
|
|
44
|
+
The covariance estimate after each day.
|
|
47
45
|
corr : ndarray or None, shape (T, n, n)
|
|
48
46
|
weights : ndarray or None, shape (T,)
|
|
49
47
|
"""
|
|
50
48
|
values = np.asarray(returns, dtype=np.float64)
|
|
51
49
|
if values.ndim != 2:
|
|
52
50
|
raise ValueError(f"Expected 2D returns, got shape {values.shape}.")
|
|
51
|
+
if "n_assets" in estimator_kwargs:
|
|
52
|
+
raise ValueError(
|
|
53
|
+
"n_assets is derived from the panel shape and cannot be passed."
|
|
54
|
+
)
|
|
53
55
|
t_total, n_assets = values.shape
|
|
54
56
|
|
|
55
|
-
est = SqueezeKernelEstimator(
|
|
56
|
-
n_assets,
|
|
57
|
-
lambda_vol=lambda_vol,
|
|
58
|
-
lambda_corr=lambda_corr,
|
|
59
|
-
kappa=kappa,
|
|
60
|
-
kernel_fn=kernel_fn,
|
|
61
|
-
kernel_kwargs=kernel_kwargs,
|
|
62
|
-
epsilon=epsilon,
|
|
63
|
-
shrinkage=shrinkage,
|
|
64
|
-
shrinkage_delta=shrinkage_delta,
|
|
65
|
-
impute_missing=impute_missing,
|
|
66
|
-
impute_threshold=impute_threshold,
|
|
67
|
-
)
|
|
57
|
+
est = SqueezeKernelEstimator(n_assets, **estimator_kwargs)
|
|
68
58
|
|
|
69
59
|
cov = np.empty((t_total, n_assets, n_assets), dtype=np.float64)
|
|
70
60
|
corr = np.empty_like(cov) if with_corr else None
|
|
@@ -6,14 +6,130 @@ from typing import TYPE_CHECKING
|
|
|
6
6
|
|
|
7
7
|
import numpy as np
|
|
8
8
|
|
|
9
|
+
try:
|
|
10
|
+
# SciPy's direct LAPACK bindings factorise an SPD matrix ~4x faster than
|
|
11
|
+
# the NumPy slogdet+solve route (one dpotrf vs two LU factorisations) and
|
|
12
|
+
# compute the identical quantities; the detector falls back to the
|
|
13
|
+
# NumPy-only path when SciPy is absent.
|
|
14
|
+
from scipy.linalg import cho_factor as _cho_factor, cho_solve as _cho_solve
|
|
15
|
+
except ImportError: # pragma: no cover
|
|
16
|
+
_cho_factor = _cho_solve = None
|
|
17
|
+
|
|
9
18
|
from squeeze_kernel.kernels import (
|
|
10
19
|
KernelFn, kernel_fisher, calibrate_kappa, extract_d2_series,
|
|
11
20
|
)
|
|
21
|
+
from numpy.typing import ArrayLike
|
|
12
22
|
|
|
13
23
|
if TYPE_CHECKING:
|
|
14
24
|
from collections.abc import Sequence
|
|
15
25
|
|
|
16
26
|
|
|
27
|
+
class _SurpriseDetector:
|
|
28
|
+
"""Two-sided Page CUSUM on the studentised fast-vs-slow per-rung
|
|
29
|
+
predictive-score drift.
|
|
30
|
+
|
|
31
|
+
Gates the blend weights of the scale-free correlation ladder: on each
|
|
32
|
+
update the day is scored under every rung's previous (t-1) covariance
|
|
33
|
+
— causal, since ``record`` is called with the rung states *after* the
|
|
34
|
+
update — and the drift of the centred rung scores advances the CUSUM.
|
|
35
|
+
An alarm sets a half-magnitude tilt of the theta-prior toward the
|
|
36
|
+
inverse-horizon vector (fast alarm) or the square-root-horizon vector
|
|
37
|
+
(slow alarm); on all other days the tilt decays at the fastest rung's
|
|
38
|
+
half-life. State: five scalars (``gp``, ``gm``, ``tilt``, ``scale``
|
|
39
|
+
and the cached ``prev_sig`` rung covariances).
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
_DRIFT = 0.5
|
|
43
|
+
_THRESHOLD = 4.9721088583 # Siegmund ARL approximation at ~2 years
|
|
44
|
+
_SNAP = 0.5
|
|
45
|
+
_CLIP = 3.0
|
|
46
|
+
_JITTER = 1e-12
|
|
47
|
+
|
|
48
|
+
def __init__(self, half_lives: np.ndarray) -> None:
|
|
49
|
+
hl = half_lives
|
|
50
|
+
self.pi_fast = (1.0 / hl) / (1.0 / hl).sum()
|
|
51
|
+
self.pi_slow = hl ** 0.5 / (hl ** 0.5).sum()
|
|
52
|
+
self._lam_tilt = 2.0 ** (-1.0 / float(hl.min()))
|
|
53
|
+
self._gamma_scale = 2.0 ** (-1.0 / float(np.median(hl)))
|
|
54
|
+
self._k_fast = int(np.argmin(hl))
|
|
55
|
+
self._k_slow = int(np.argmax(hl))
|
|
56
|
+
self.gp = 0.0
|
|
57
|
+
self.gm = 0.0
|
|
58
|
+
self.tilt = 0.0
|
|
59
|
+
self.scale = 1.0
|
|
60
|
+
self.prev_sig: list[np.ndarray] | None = None
|
|
61
|
+
|
|
62
|
+
def score(self, r_t: np.ndarray, finite: np.ndarray) -> np.ndarray | None:
|
|
63
|
+
"""Per-rung Gaussian log-likelihood of ``r_t`` under ``prev_sig``,
|
|
64
|
+
restricted to the observed assets. Returns None when there is no
|
|
65
|
+
previous state, nothing is observed, or the day is degenerate
|
|
66
|
+
(e.g. a fresh listing); a degenerate day also decays the tilt."""
|
|
67
|
+
if self.prev_sig is None or not finite.any():
|
|
68
|
+
return None
|
|
69
|
+
oidx = np.flatnonzero(finite)
|
|
70
|
+
r_o = r_t[oidx]
|
|
71
|
+
ell = np.empty(len(self.prev_sig))
|
|
72
|
+
for k, sig in enumerate(self.prev_sig):
|
|
73
|
+
sub = sig[np.ix_(oidx, oidx)]
|
|
74
|
+
sub = (sub + sub.T) * 0.5
|
|
75
|
+
if _cho_factor is not None:
|
|
76
|
+
# One Cholesky per rung: logdet from the factor's
|
|
77
|
+
# diagonal, quadratic form via triangular solves.
|
|
78
|
+
try:
|
|
79
|
+
cf = _cho_factor(sub, lower=True, check_finite=False)
|
|
80
|
+
except np.linalg.LinAlgError:
|
|
81
|
+
self.tilt *= self._lam_tilt
|
|
82
|
+
return None
|
|
83
|
+
logdet = 2.0 * np.log(np.diagonal(cf[0])).sum()
|
|
84
|
+
quad = float(r_o @ _cho_solve(cf, r_o, check_finite=False))
|
|
85
|
+
else:
|
|
86
|
+
sign, logdet = np.linalg.slogdet(sub)
|
|
87
|
+
if sign <= 0:
|
|
88
|
+
self.tilt *= self._lam_tilt
|
|
89
|
+
return None
|
|
90
|
+
try:
|
|
91
|
+
quad = float(r_o @ np.linalg.solve(sub, r_o))
|
|
92
|
+
except np.linalg.LinAlgError:
|
|
93
|
+
self.tilt *= self._lam_tilt
|
|
94
|
+
return None
|
|
95
|
+
ell[k] = -0.5 * (oidx.size * np.log(2 * np.pi) + logdet + quad)
|
|
96
|
+
return ell
|
|
97
|
+
|
|
98
|
+
def advance(self, ell: np.ndarray | None) -> None:
|
|
99
|
+
"""Advance the CUSUM with the rung scores for this day. ``None``
|
|
100
|
+
(no clean score) is a no-op: the degenerate-day tilt decay already
|
|
101
|
+
happened inside ``score``."""
|
|
102
|
+
if ell is None:
|
|
103
|
+
return
|
|
104
|
+
dd = ell - ell.mean()
|
|
105
|
+
rms = float(np.sqrt((dd @ dd) / ell.size))
|
|
106
|
+
self.scale = (self._gamma_scale * self.scale
|
|
107
|
+
+ (1.0 - self._gamma_scale) * rms)
|
|
108
|
+
zc = np.clip(dd / (self.scale + self._JITTER), -self._CLIP, self._CLIP)
|
|
109
|
+
zfs = float(zc[self._k_fast] - zc[self._k_slow])
|
|
110
|
+
self.gp = max(0.0, self.gp + zfs - self._DRIFT)
|
|
111
|
+
self.gm = max(0.0, self.gm - zfs - self._DRIFT)
|
|
112
|
+
if self.gp > self._THRESHOLD:
|
|
113
|
+
self.tilt, self.gp = self._SNAP, 0.0
|
|
114
|
+
elif self.gm > self._THRESHOLD:
|
|
115
|
+
self.tilt, self.gm = -self._SNAP, 0.0
|
|
116
|
+
else:
|
|
117
|
+
self.tilt *= self._lam_tilt
|
|
118
|
+
|
|
119
|
+
def record(self, rung_covs: list[np.ndarray]) -> None:
|
|
120
|
+
"""Cache this step's rung covariances as next step's forecasts."""
|
|
121
|
+
self.prev_sig = rung_covs
|
|
122
|
+
|
|
123
|
+
def blend(self, prior: np.ndarray) -> np.ndarray:
|
|
124
|
+
"""Blend weights for the rung covariances: the theta-prior on
|
|
125
|
+
non-alarmed days, tilted half-magnitude toward the horizon vectors
|
|
126
|
+
while an alarm is live. Convex in both regimes."""
|
|
127
|
+
t = self.tilt
|
|
128
|
+
if t >= 0:
|
|
129
|
+
return np.asarray((1.0 - t) * prior + t * self.pi_fast)
|
|
130
|
+
return np.asarray((1.0 + t) * prior + (-t) * self.pi_slow)
|
|
131
|
+
|
|
132
|
+
|
|
17
133
|
class SqueezeKernelEstimator:
|
|
18
134
|
"""Streaming robust covariance estimator with pluggable kernel weighting.
|
|
19
135
|
|
|
@@ -113,27 +229,63 @@ class SqueezeKernelEstimator:
|
|
|
113
229
|
benchmark: +0.14 (negligible) at n=100, −4.2 at n=200, −25.0 at
|
|
114
230
|
n=300. Recommended when n approaches the effective sample size.
|
|
115
231
|
corr_half_lives : sequence of float, optional
|
|
116
|
-
Scale-free correlation memory
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
``(43, 173, 693)``): each
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
232
|
+
Scale-free correlation memory with surprise-gated adaptive blend
|
|
233
|
+
weights. When set, the single correlation timescale is replaced
|
|
234
|
+
by a positive combination of EWMAs on the given geometric
|
|
235
|
+
half-life ladder (in trading days, e.g. ``(43, 173, 693)``): each
|
|
236
|
+
scale is normalised and adaptively shrunk against its own
|
|
237
|
+
effective sample size, and the resulting covariances are blended
|
|
238
|
+
with weights resting at the prior ∝ half-life\\ :sup:`corr_theta`.
|
|
239
|
+
By Bernstein's theorem the ladder approximates the power-law
|
|
240
|
+
memory of financial correlations (the streaming analogue of HAR).
|
|
241
|
+
For ladders of two or more rungs the blend weights are gated by a
|
|
242
|
+
sequential surprise detector: a two-sided Page CUSUM on the
|
|
243
|
+
studentised fast-vs-slow per-rung predictive-score drift
|
|
244
|
+
(reference drift 0.5, threshold 4.9721 = Siegmund average-run-
|
|
245
|
+
length ~2 years); an alarm applies a half-magnitude tilt of the
|
|
246
|
+
theta-prior toward the inverse-horizon vector (fast alarms,
|
|
247
|
+
w ~ 1/h) or the square-root-horizon vector (slow alarms,
|
|
248
|
+
w ~ h^0.5), decaying at the fastest rung's half-life. Weights
|
|
249
|
+
equal the prior on all non-alarmed days and the blend stays
|
|
250
|
+
convex, so PSD is structural throughout. The detector adds five
|
|
251
|
+
scalars of state and one Cholesky per rung per update for the
|
|
252
|
+
scores; it is validated across S&P panels (held-out +0.5-0.6
|
|
253
|
+
NLL/day over the detector-off blend), an external industry panel
|
|
254
|
+
incl. an out-of-time seal (+0.53/day, p=1e-4), a multi-asset
|
|
255
|
+
futures panel, and synthetic regime/null suites, with a
|
|
256
|
+
sign-stable one-at-a-time sensitivity sweep over all structural
|
|
257
|
+
constants. ``None`` (default) is the published single-scale
|
|
258
|
+
estimator, bit-for-bit; a one-element ladder reduces to a
|
|
259
|
+
single-scale estimator at that half-life (no detector).
|
|
260
|
+
Mutually exclusive with ``lambda_corr_fast``; composes with
|
|
261
|
+
``shrinkage_target='cluster'``. Cost is O(K·n²) per update plus
|
|
262
|
+
the detector's per-rung score factorisations. Held-out one-step
|
|
263
|
+
NLL on the S&P-500 benchmark improves on the single-scale
|
|
264
|
+
estimator at every dimension (−4.6 at n=100, −11.1 at n=300
|
|
265
|
+
before the cluster target); the $90\\%$ model confidence set
|
|
266
|
+
collapses to this configuration alone. Recommended default: the
|
|
267
|
+
base-centred ladder ``(43, 173, 693)`` with ``corr_theta=0.25``.
|
|
134
268
|
corr_theta : float
|
|
135
|
-
Long-memory exponent controlling the
|
|
136
|
-
0.25). Only used when ``corr_half_lives`` is set.
|
|
269
|
+
Long-memory exponent controlling the resting blend weights
|
|
270
|
+
(default 0.25). Only used when ``corr_half_lives`` is set.
|
|
271
|
+
min_obs : int or None
|
|
272
|
+
Usability gate for newly listed assets. When set, the property
|
|
273
|
+
``usable_mask`` marks an asset usable only once it has delivered
|
|
274
|
+
at least ``min_obs`` finite observations. The gate is purely
|
|
275
|
+
diagnostic: state evolution and ``get_cov``/``get_corr`` are
|
|
276
|
+
unchanged (the states keep warming during the gated window, so an
|
|
277
|
+
asset is fully warm when the gate lifts). Motivation: on an
|
|
278
|
+
expanding multi-asset universe, forecast rows for assets in their
|
|
279
|
+
first ~100 observations are dominated by the single-observation
|
|
280
|
+
variance initialisation and are unusable for scoring or portfolio
|
|
281
|
+
construction (measured ≈ +2,800 NLL/day on days whose scored set
|
|
282
|
+
included such assets). Deployment recipe::
|
|
283
|
+
|
|
284
|
+
m = est.usable_mask
|
|
285
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
286
|
+
|
|
287
|
+
Recommended ``min_obs`` ≈ 60–100 for daily data. ``None``
|
|
288
|
+
(default) disables the gate (``usable_mask`` is all-True).
|
|
137
289
|
|
|
138
290
|
Examples
|
|
139
291
|
--------
|
|
@@ -167,6 +319,7 @@ class SqueezeKernelEstimator:
|
|
|
167
319
|
shrinkage_target: str = "equicorrelation",
|
|
168
320
|
corr_half_lives: "Sequence[float] | None" = None,
|
|
169
321
|
corr_theta: float = 0.25,
|
|
322
|
+
min_obs: int | None = None,
|
|
170
323
|
):
|
|
171
324
|
self.n_assets = n_assets
|
|
172
325
|
self.lambda_vol = lambda_vol
|
|
@@ -192,6 +345,10 @@ class SqueezeKernelEstimator:
|
|
|
192
345
|
if shrinkage_target not in ("equicorrelation", "cluster"):
|
|
193
346
|
raise ValueError("shrinkage_target must be 'equicorrelation' or 'cluster'.")
|
|
194
347
|
self.shrinkage_target = shrinkage_target
|
|
348
|
+
if min_obs is not None and (not isinstance(min_obs, int) or min_obs < 1):
|
|
349
|
+
raise ValueError("min_obs must be a positive integer or None.")
|
|
350
|
+
self.min_obs = min_obs
|
|
351
|
+
self._obs_count = np.zeros(n_assets, dtype=np.int64)
|
|
195
352
|
|
|
196
353
|
# Scale-free correlation memory (opt-in): replace the single correlation
|
|
197
354
|
# timescale by a positive combination of EWMAs on a geometric half-life
|
|
@@ -201,8 +358,10 @@ class SqueezeKernelEstimator:
|
|
|
201
358
|
self.corr_theta = corr_theta
|
|
202
359
|
self._corr_lam: np.ndarray | None = None
|
|
203
360
|
self._corr_w: np.ndarray | None = None
|
|
204
|
-
self.
|
|
361
|
+
self._detector: _SurpriseDetector | None = None
|
|
362
|
+
self._Q_list: list[np.ndarray] | None = None
|
|
205
363
|
self._S_list: list[float] | None = None
|
|
364
|
+
self._adaptive = False
|
|
206
365
|
if corr_half_lives is not None:
|
|
207
366
|
hl = np.asarray(corr_half_lives, dtype=np.float64)
|
|
208
367
|
if hl.ndim != 1 or hl.size < 1 or np.any(hl <= 0.0):
|
|
@@ -218,8 +377,13 @@ class SqueezeKernelEstimator:
|
|
|
218
377
|
self._corr_lam = 2.0 ** (-1.0 / hl)
|
|
219
378
|
w = hl ** corr_theta
|
|
220
379
|
self._corr_w = w / w.sum()
|
|
221
|
-
self.
|
|
380
|
+
self._Q_list = [np.eye(n_assets, dtype=np.float64) for _ in hl]
|
|
222
381
|
self._S_list = [float(epsilon) for _ in hl]
|
|
382
|
+
# Surprise-gated blend weights (integral for K >= 2): see
|
|
383
|
+
# ``_SurpriseDetector``.
|
|
384
|
+
self._adaptive = hl.size >= 2
|
|
385
|
+
if self._adaptive:
|
|
386
|
+
self._detector = _SurpriseDetector(hl)
|
|
223
387
|
|
|
224
388
|
# Resolve shrinkage
|
|
225
389
|
if isinstance(shrinkage, str):
|
|
@@ -231,26 +395,38 @@ class SqueezeKernelEstimator:
|
|
|
231
395
|
self._kernel_fn, self._kernel_kwargs = _resolve_kernel(kappa, kernel_fn, kernel_kwargs)
|
|
232
396
|
self.kappa = self._kernel_kwargs.get("kappa") if self._kernel_fn is kernel_fisher else None
|
|
233
397
|
|
|
234
|
-
# State
|
|
398
|
+
# State. The correlation memory is stored NORMALISED: Q_t = M_t / S_t
|
|
399
|
+
# with the recursion Q_t = (1 - eta_t) Q_{t-1} + eta_t z_t z_t',
|
|
400
|
+
# eta_t = w_t / S_t after S_t <- lam S_{t-1} + w_t. This is
|
|
401
|
+
# algebraically identical to the raw-mass form (M init eps*I, S init
|
|
402
|
+
# eps => Q init I), keeps the matrix state well scaled, and makes the
|
|
403
|
+
# PSD convex-combination recursion explicit.
|
|
235
404
|
self._var_t: np.ndarray | None = None
|
|
236
405
|
self._var_init: np.ndarray | None = None
|
|
237
406
|
self._var_anchor: np.ndarray | None = None
|
|
238
|
-
self.
|
|
407
|
+
self._vol_t: np.ndarray | None = None
|
|
408
|
+
# No single-scale state is allocated in ladder mode.
|
|
409
|
+
self._Q_t = (np.eye(n_assets, dtype=np.float64)
|
|
410
|
+
if self._corr_lam is None else None)
|
|
239
411
|
self._S_t = float(epsilon)
|
|
240
412
|
self._cov: np.ndarray | None = None
|
|
241
413
|
self._corr: np.ndarray | None = None
|
|
242
414
|
self._last_weight: float = 0.0
|
|
415
|
+
# Extraction (normalise + shrink + vol application) is deferred until
|
|
416
|
+
# get_cov()/get_corr(); _dirty marks state newer than _cov/_corr.
|
|
417
|
+
self._dirty = False
|
|
243
418
|
|
|
244
419
|
# Cached scratch buffers reused per ``update()`` to avoid per-step
|
|
245
420
|
# allocator churn. These are intentionally module-private and
|
|
246
421
|
# never escape the estimator.
|
|
247
422
|
self._scratch_outer = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
248
423
|
self._scratch_corr = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
424
|
+
self._scratch_had = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
249
425
|
self._n_off = float(n_assets * (n_assets - 1)) if n_assets > 1 else 1.0
|
|
250
426
|
|
|
251
427
|
# ── Public API ────────────────────────────────────────────────────────
|
|
252
428
|
|
|
253
|
-
def update(self, r_t) -> float:
|
|
429
|
+
def update(self, r_t: ArrayLike) -> float:
|
|
254
430
|
"""Process one return vector and update the covariance estimate.
|
|
255
431
|
|
|
256
432
|
Parameters
|
|
@@ -270,25 +446,42 @@ class SqueezeKernelEstimator:
|
|
|
270
446
|
raise ValueError(f"Expected shape ({n},), got {r_t.shape}.")
|
|
271
447
|
|
|
272
448
|
finite = np.isfinite(r_t)
|
|
449
|
+
self._obs_count[finite] += 1
|
|
450
|
+
|
|
451
|
+
# ── Adaptive-weight detector: score r_t under yesterday's per-rung
|
|
452
|
+
# forecasts, then advance the CUSUM (weights used below therefore
|
|
453
|
+
# reflect information through r_t only — causal). ──
|
|
454
|
+
ell = None
|
|
455
|
+
detector = self._detector
|
|
456
|
+
if detector is not None:
|
|
457
|
+
ell = detector.score(r_t, finite)
|
|
458
|
+
detector.advance(ell)
|
|
273
459
|
|
|
274
460
|
# ── Volatility update ──
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
461
|
+
# Locals alias the variance states: the guard initialises the pair
|
|
462
|
+
# together, so both are non-None afterwards, and local aliases keep
|
|
463
|
+
# the narrowing visible to mypy.
|
|
464
|
+
var_t = self._var_t
|
|
465
|
+
var_init = self._var_init
|
|
466
|
+
if var_t is None or var_init is None:
|
|
467
|
+
var_t = np.zeros(n, dtype=np.float64)
|
|
468
|
+
var_init = np.zeros(n, dtype=bool)
|
|
469
|
+
self._var_t = var_t
|
|
470
|
+
self._var_init = var_init
|
|
278
471
|
if self.vol_anchor_phi is not None:
|
|
279
472
|
self._var_anchor = np.zeros(n, dtype=np.float64)
|
|
280
473
|
|
|
281
|
-
first = finite & ~
|
|
282
|
-
repeat = finite &
|
|
474
|
+
first = finite & ~var_init
|
|
475
|
+
repeat = finite & var_init
|
|
283
476
|
if np.any(first):
|
|
284
|
-
|
|
285
|
-
|
|
477
|
+
var_t[first] = r_t[first] ** 2 + eps
|
|
478
|
+
var_init[first] = True
|
|
286
479
|
if self._var_anchor is not None:
|
|
287
|
-
self._var_anchor[first] =
|
|
480
|
+
self._var_anchor[first] = var_t[first]
|
|
288
481
|
if np.any(repeat):
|
|
289
482
|
if self.vol_anchor_phi is None:
|
|
290
|
-
|
|
291
|
-
self.lambda_vol *
|
|
483
|
+
var_t[repeat] = (
|
|
484
|
+
self.lambda_vol * var_t[repeat]
|
|
292
485
|
+ (1.0 - self.lambda_vol) * r_t[repeat] ** 2
|
|
293
486
|
)
|
|
294
487
|
else:
|
|
@@ -296,20 +489,23 @@ class SqueezeKernelEstimator:
|
|
|
296
489
|
# per-asset anchor before the measurement update, then update
|
|
297
490
|
# the anchor itself (order matters and matches the validated
|
|
298
491
|
# experiment: prediction uses the *old* anchor).
|
|
492
|
+
var_anchor = self._var_anchor
|
|
493
|
+
assert var_anchor is not None
|
|
494
|
+
# allocated together with the anchor on the first update
|
|
299
495
|
phi = self.vol_anchor_phi
|
|
300
496
|
lam_bar = self.vol_anchor_decay
|
|
301
|
-
anchor =
|
|
302
|
-
v_pred = anchor + phi * (
|
|
303
|
-
|
|
497
|
+
anchor = var_anchor[repeat]
|
|
498
|
+
v_pred = anchor + phi * (var_t[repeat] - anchor)
|
|
499
|
+
var_t[repeat] = (
|
|
304
500
|
self.lambda_vol * v_pred
|
|
305
501
|
+ (1.0 - self.lambda_vol) * r_t[repeat] ** 2
|
|
306
502
|
)
|
|
307
|
-
|
|
503
|
+
var_anchor[repeat] = (
|
|
308
504
|
lam_bar * anchor + (1.0 - lam_bar) * r_t[repeat] ** 2
|
|
309
505
|
)
|
|
310
506
|
|
|
311
507
|
vol_t = np.zeros(n, dtype=np.float64)
|
|
312
|
-
vol_t[
|
|
508
|
+
vol_t[var_init] = np.sqrt(var_t[var_init])
|
|
313
509
|
|
|
314
510
|
# ── Standardized returns ──
|
|
315
511
|
z_t = np.zeros(n, dtype=np.float64)
|
|
@@ -317,12 +513,17 @@ class SqueezeKernelEstimator:
|
|
|
317
513
|
if n_obs > 0:
|
|
318
514
|
z_t[finite] = r_t[finite] / (vol_t[finite] + eps)
|
|
319
515
|
d2 = float(z_t[finite] @ z_t[finite]) / n_obs
|
|
320
|
-
if self.weight_statistic == "mahalanobis" and self.
|
|
516
|
+
if self.weight_statistic == "mahalanobis" and self._vol_t is not None:
|
|
321
517
|
# Score-exact surprise against the estimator's own previous
|
|
322
518
|
# correlation; falls back to the marginal d² on the first
|
|
323
|
-
# step or a (rare) singular observed submatrix.
|
|
519
|
+
# step or a (rare) singular observed submatrix. Extraction is
|
|
520
|
+
# lazy, so bring _corr up to the t-1 state first.
|
|
521
|
+
if self._dirty:
|
|
522
|
+
self._materialize()
|
|
523
|
+
corr_prev = self._corr
|
|
524
|
+
assert corr_prev is not None # _materialize always sets it
|
|
324
525
|
try:
|
|
325
|
-
c_sub =
|
|
526
|
+
c_sub = corr_prev[np.ix_(finite, finite)]
|
|
326
527
|
d2 = float(z_t[finite] @ np.linalg.solve(c_sub, z_t[finite])) / n_obs
|
|
327
528
|
except np.linalg.LinAlgError:
|
|
328
529
|
pass
|
|
@@ -334,65 +535,93 @@ class SqueezeKernelEstimator:
|
|
|
334
535
|
if self.impute_missing and 0 < n_obs < n:
|
|
335
536
|
self._impute(z_t, finite)
|
|
336
537
|
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
538
|
+
add = w_t > 0.0 and n_obs > 0
|
|
539
|
+
if add:
|
|
540
|
+
# np.multiply.outer with out= avoids the temporary that
|
|
541
|
+
# np.outer otherwise allocates each step. zz' is computed once
|
|
542
|
+
# and shared by every rung.
|
|
543
|
+
np.multiply.outer(z_t, z_t, out=self._scratch_outer)
|
|
544
|
+
corr_lam = self._corr_lam
|
|
545
|
+
if corr_lam is None:
|
|
546
|
+
# ── Single-scale correlation EWMA (published path) ──
|
|
547
|
+
Q_t = self._Q_t
|
|
548
|
+
assert Q_t is not None # the single-scale state exists exactly
|
|
549
|
+
lam_c = self.lambda_corr # when the ladder is not configured
|
|
340
550
|
if self.lambda_corr_fast is not None:
|
|
341
551
|
# Score-driven memory: stress days (w_t → 1) shorten the memory
|
|
342
552
|
# toward lambda_corr_fast; calm days keep the slow decay.
|
|
343
553
|
lam_c = self.lambda_corr + (self.lambda_corr_fast - self.lambda_corr) * w_t
|
|
344
554
|
self._S_t = lam_c * self._S_t + w_t
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
#
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
self.
|
|
351
|
-
|
|
555
|
+
if add:
|
|
556
|
+
# Q <- (1 - eta) Q + eta zz'. With w_t = 0 both S and M decay
|
|
557
|
+
# by lam_c, so Q is unchanged — no matrix work at all.
|
|
558
|
+
eta = w_t / self._S_t
|
|
559
|
+
Q_t *= 1.0 - eta
|
|
560
|
+
self._scratch_outer *= eta
|
|
561
|
+
Q_t += self._scratch_outer
|
|
352
562
|
else:
|
|
353
563
|
# ── Scale-free ladder (Mode A) ──
|
|
354
|
-
# Update K correlation
|
|
355
|
-
# ladder; normalise
|
|
356
|
-
#
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
564
|
+
# Update K normalised correlation states on the geometric
|
|
565
|
+
# half-life ladder; extraction (normalise + shrink + blend)
|
|
566
|
+
# happens lazily in _materialize().
|
|
567
|
+
Q_list = self._Q_list
|
|
568
|
+
S_list = self._S_list
|
|
569
|
+
corr_w = self._corr_w
|
|
570
|
+
assert Q_list is not None and S_list is not None
|
|
571
|
+
assert corr_w is not None # allocated together with corr_lam
|
|
361
572
|
s_eff = 0.0
|
|
362
|
-
for k in range(
|
|
363
|
-
|
|
364
|
-
self._M_list[k] *= self._corr_lam[k]
|
|
573
|
+
for k in range(corr_lam.size):
|
|
574
|
+
S_list[k] = corr_lam[k] * S_list[k] + w_t
|
|
365
575
|
if add:
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
self._cov = cov
|
|
576
|
+
eta = w_t / S_list[k]
|
|
577
|
+
Q_list[k] *= 1.0 - eta
|
|
578
|
+
np.multiply(self._scratch_outer, eta, out=self._scratch_had)
|
|
579
|
+
Q_list[k] += self._scratch_had
|
|
580
|
+
s_eff += corr_w[k] * S_list[k]
|
|
372
581
|
self._S_t = s_eff # blended effective size (for the property)
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
582
|
+
self._vol_t = vol_t
|
|
583
|
+
self._dirty = True
|
|
584
|
+
if self._adaptive:
|
|
585
|
+
self._materialize_adaptive()
|
|
376
586
|
self._last_weight = w_t
|
|
377
587
|
return w_t
|
|
378
588
|
|
|
379
589
|
def get_cov(self) -> np.ndarray:
|
|
380
590
|
"""Return the current covariance matrix estimate (n x n)."""
|
|
381
|
-
if self.
|
|
591
|
+
if self._vol_t is None:
|
|
382
592
|
raise RuntimeError("Call update() at least once before get_cov().")
|
|
383
|
-
|
|
593
|
+
if self._dirty:
|
|
594
|
+
self._materialize()
|
|
595
|
+
cov = self._cov
|
|
596
|
+
assert cov is not None
|
|
597
|
+
return cov.copy()
|
|
384
598
|
|
|
385
599
|
def get_corr(self) -> np.ndarray:
|
|
386
600
|
"""Return the current correlation matrix estimate (n x n)."""
|
|
387
|
-
if self.
|
|
601
|
+
if self._vol_t is None:
|
|
388
602
|
raise RuntimeError("Call update() at least once before get_corr().")
|
|
389
|
-
|
|
603
|
+
if self._dirty:
|
|
604
|
+
self._materialize()
|
|
605
|
+
corr = self._corr
|
|
606
|
+
assert corr is not None
|
|
607
|
+
return corr.copy()
|
|
390
608
|
|
|
391
609
|
@property
|
|
392
610
|
def weight(self) -> float:
|
|
393
611
|
"""Kernel weight assigned to the most recent observation."""
|
|
394
612
|
return self._last_weight
|
|
395
613
|
|
|
614
|
+
@property
|
|
615
|
+
def usable_mask(self) -> np.ndarray:
|
|
616
|
+
"""Boolean mask of assets with at least ``min_obs`` observations.
|
|
617
|
+
|
|
618
|
+
All-True when ``min_obs`` is None. Purely diagnostic — estimates
|
|
619
|
+
are not affected; subset the outputs with it (see class docstring).
|
|
620
|
+
"""
|
|
621
|
+
if self.min_obs is None:
|
|
622
|
+
return np.ones(self.n_assets, dtype=bool)
|
|
623
|
+
return self._obs_count >= self.min_obs
|
|
624
|
+
|
|
396
625
|
@property
|
|
397
626
|
def effective_sample_size(self) -> float:
|
|
398
627
|
"""Kernel-weighted effective sample size S_t."""
|
|
@@ -408,7 +637,7 @@ class SqueezeKernelEstimator:
|
|
|
408
637
|
|
|
409
638
|
@staticmethod
|
|
410
639
|
def calibrate_kappa(
|
|
411
|
-
returns, target_weight: float = 0.5, lambda_vol: float = 0.98,
|
|
640
|
+
returns: ArrayLike, target_weight: float = 0.5, lambda_vol: float = 0.98,
|
|
412
641
|
) -> float:
|
|
413
642
|
"""Calibrate κ from burn-in data so E[w_t] ≈ target_weight.
|
|
414
643
|
|
|
@@ -432,9 +661,13 @@ class SqueezeKernelEstimator:
|
|
|
432
661
|
# ── Private helpers ───────────────────────────────────────────────────
|
|
433
662
|
|
|
434
663
|
def _impute(self, z_t: np.ndarray, finite: np.ndarray) -> None:
|
|
664
|
+
if self._Q_t is None:
|
|
665
|
+
# Ladder mode: imputation reads the single-scale state, which
|
|
666
|
+
# was never updated on this path — historically a silent no-op
|
|
667
|
+
# (all correlations below threshold); keep it an explicit one.
|
|
668
|
+
return
|
|
435
669
|
eps = self.epsilon
|
|
436
|
-
|
|
437
|
-
sigma_z = self._M_t / denom
|
|
670
|
+
sigma_z = self._Q_t
|
|
438
671
|
diag_z = np.diag(sigma_z)
|
|
439
672
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
440
673
|
missing = ~finite
|
|
@@ -453,24 +686,22 @@ class SqueezeKernelEstimator:
|
|
|
453
686
|
if den > 0.0:
|
|
454
687
|
z_t[i] = num / den
|
|
455
688
|
|
|
456
|
-
def
|
|
689
|
+
def _shrunk_corr_from_Q(self, Q: np.ndarray, S_t: float) -> np.ndarray:
|
|
690
|
+
"""Normalise one Q state to a correlation and shrink it in place.
|
|
691
|
+
|
|
692
|
+
Returns ``self._scratch_corr`` — valid only until the next call.
|
|
693
|
+
"""
|
|
457
694
|
eps = self.epsilon
|
|
458
|
-
M_t = self._M_t if M_t is None else M_t
|
|
459
|
-
S_t = max(self._S_t if S_in is None else S_in, eps)
|
|
460
695
|
n = self.n_assets
|
|
696
|
+
S_t = max(S_t, eps)
|
|
461
697
|
|
|
462
|
-
#
|
|
463
|
-
#
|
|
464
|
-
|
|
465
|
-
diag_z = np.diagonal(M_t).copy()
|
|
466
|
-
diag_z /= S_t # in-place
|
|
698
|
+
# corr_ij = Q_ij * inv_diag_i * inv_diag_j; the scalar S_t cancels
|
|
699
|
+
# in the normalisation, so Q needs no rescaling pass.
|
|
700
|
+
diag_z = np.diagonal(Q).copy()
|
|
467
701
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
468
|
-
# corr_ij = (M_ij / S_t) * inv_diag_i * inv_diag_j; this writes
|
|
469
|
-
# the rescaled outer-product into _scratch_corr in one pass.
|
|
470
702
|
np.multiply.outer(inv_diag, inv_diag, out=self._scratch_corr)
|
|
471
703
|
corr = self._scratch_corr
|
|
472
|
-
corr *=
|
|
473
|
-
corr *= (1.0 / S_t) # absorb the M_t / S_t scale
|
|
704
|
+
corr *= Q # in-place
|
|
474
705
|
np.fill_diagonal(corr, np.where(diag_z > eps, 1.0, 0.0))
|
|
475
706
|
|
|
476
707
|
# Adaptive shrinkage: blend toward the equicorrelation target
|
|
@@ -484,6 +715,9 @@ class SqueezeKernelEstimator:
|
|
|
484
715
|
# Off-diagonal mean: O(n^2) sum, no mask allocation.
|
|
485
716
|
rho_bar = (corr.sum() - corr.trace()) / self._n_off
|
|
486
717
|
if self.shrinkage_target == "equicorrelation" or rho_bar <= 0.0:
|
|
718
|
+
# rho_bar <= 0 also covers the q = meanoff(C o C) = 0 corner:
|
|
719
|
+
# C o C has nonnegative entries, so q = 0 forces C = I and
|
|
720
|
+
# hence rho_bar = 0 — the equicorrelation fallback applies.
|
|
487
721
|
corr *= (1.0 - alpha)
|
|
488
722
|
corr += alpha * rho_bar
|
|
489
723
|
np.fill_diagonal(corr, 1.0)
|
|
@@ -495,27 +729,95 @@ class SqueezeKernelEstimator:
|
|
|
495
729
|
# level-matched so the target carries the same average
|
|
496
730
|
# correlation mass as the equicorrelation target. As
|
|
497
731
|
# alpha -> 0 this reduces exactly to the published estimator.
|
|
498
|
-
had =
|
|
732
|
+
had = self._scratch_had
|
|
733
|
+
np.multiply(corr, corr, out=had) # Hadamard square, O(n^2)
|
|
499
734
|
mean_off = (had.sum() - np.trace(had)) / self._n_off
|
|
500
735
|
gamma = min(1.0, rho_bar / max(mean_off, eps))
|
|
501
736
|
corr *= (1.0 - alpha)
|
|
502
737
|
corr += (alpha * (1.0 - alpha)) * rho_bar
|
|
503
|
-
|
|
738
|
+
had *= alpha * alpha * gamma
|
|
739
|
+
corr += had
|
|
504
740
|
np.fill_diagonal(corr, 1.0)
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
741
|
+
return corr
|
|
742
|
+
|
|
743
|
+
def _materialize_adaptive(self) -> None:
|
|
744
|
+
"""Adaptive-weight extraction: build per-rung shrunk covariances
|
|
745
|
+
(kept for the next update's detector scores), blend with the
|
|
746
|
+
CUSUM-tilted weights, derive _cov/_corr. Runs eagerly."""
|
|
747
|
+
detector = self._detector
|
|
748
|
+
corr_lam = self._corr_lam
|
|
749
|
+
Q_list = self._Q_list
|
|
750
|
+
S_list = self._S_list
|
|
751
|
+
corr_w = self._corr_w
|
|
752
|
+
vol_t = self._vol_t
|
|
753
|
+
assert detector is not None # only called when the detector exists
|
|
754
|
+
assert corr_lam is not None and Q_list is not None
|
|
755
|
+
assert S_list is not None and corr_w is not None
|
|
756
|
+
assert vol_t is not None # set on every update before this runs
|
|
757
|
+
eps = self.epsilon
|
|
758
|
+
vv = np.multiply.outer(vol_t, vol_t)
|
|
759
|
+
sig_k = []
|
|
760
|
+
for k in range(corr_lam.size):
|
|
761
|
+
corr_k = self._shrunk_corr_from_Q(Q_list[k], S_list[k])
|
|
762
|
+
sig_k.append(corr_k * vv)
|
|
763
|
+
detector.record(sig_k)
|
|
764
|
+
w = detector.blend(corr_w)
|
|
765
|
+
cov = w[0] * sig_k[0]
|
|
766
|
+
for k in range(1, len(sig_k)):
|
|
767
|
+
cov = cov + w[k] * sig_k[k]
|
|
768
|
+
cov = (cov + cov.T) * 0.5
|
|
769
|
+
self._cov = cov
|
|
770
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
771
|
+
self._corr = cov / np.outer(d, d)
|
|
772
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
773
|
+
self._dirty = False
|
|
774
|
+
|
|
775
|
+
def _materialize(self) -> None:
|
|
776
|
+
"""Extract _cov/_corr from the current state (lazy, on demand)."""
|
|
777
|
+
eps = self.epsilon
|
|
778
|
+
vol_t = self._vol_t
|
|
779
|
+
assert vol_t is not None # get_cov/get_corr guard this before calling
|
|
780
|
+
if self._corr_lam is None:
|
|
781
|
+
# ── Single scale: shrunk correlation IS the correlation output ──
|
|
782
|
+
Q_t = self._Q_t
|
|
783
|
+
assert Q_t is not None # the ladder is not configured on this path
|
|
784
|
+
corr = self._shrunk_corr_from_Q(Q_t, self._S_t)
|
|
785
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
786
|
+
cov = corr * self._scratch_outer
|
|
787
|
+
cov += cov.T
|
|
788
|
+
cov *= 0.5
|
|
789
|
+
corr_out = corr.copy()
|
|
790
|
+
corr_out += corr_out.T
|
|
791
|
+
corr_out *= 0.5
|
|
792
|
+
self._cov, self._corr = cov, corr_out
|
|
793
|
+
else:
|
|
794
|
+
# ── Ladder: blend per-rung shrunk correlations, then apply the
|
|
795
|
+
# (shared) volatilities once — algebraically identical to
|
|
796
|
+
# blending per-rung covariances, K-1 fewer O(n^2) passes.
|
|
797
|
+
corr_lam = self._corr_lam
|
|
798
|
+
Q_list = self._Q_list
|
|
799
|
+
S_list = self._S_list
|
|
800
|
+
corr_w = self._corr_w
|
|
801
|
+
assert Q_list is not None and S_list is not None
|
|
802
|
+
assert corr_w is not None # allocated together with corr_lam
|
|
803
|
+
mix = np.zeros((self.n_assets, self.n_assets), dtype=np.float64)
|
|
804
|
+
for k in range(corr_lam.size):
|
|
805
|
+
corr_k = self._shrunk_corr_from_Q(Q_list[k], S_list[k])
|
|
806
|
+
corr_k *= corr_w[k]
|
|
807
|
+
mix += corr_k
|
|
808
|
+
cov = mix
|
|
809
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
810
|
+
cov *= self._scratch_outer
|
|
811
|
+
cov += cov.T
|
|
812
|
+
cov *= 0.5
|
|
813
|
+
self._cov = cov
|
|
814
|
+
# Correlation is re-derived from the blended covariance (not the
|
|
815
|
+
# blended correlation mix) to keep the published dead-asset
|
|
816
|
+
# semantics: rows of never-observed assets renormalise to zero.
|
|
817
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
818
|
+
self._corr = cov / np.outer(d, d)
|
|
819
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
820
|
+
self._dirty = False
|
|
519
821
|
|
|
520
822
|
|
|
521
823
|
# ── Kernel resolution ─────────────────────────────────────────────────────────
|
|
@@ -537,7 +839,13 @@ def _resolve_kernel(
|
|
|
537
839
|
if "kappa" in kw and kappa is not None:
|
|
538
840
|
raise ValueError("Pass kappa either as a top-level argument or in kernel_kwargs, not both.")
|
|
539
841
|
|
|
540
|
-
|
|
842
|
+
if "kappa" in kw:
|
|
843
|
+
kappa_src = kw["kappa"]
|
|
844
|
+
if isinstance(kappa_src, bool) or not isinstance(kappa_src, (int, float)):
|
|
845
|
+
raise ValueError("kappa in kernel_kwargs must be a number.")
|
|
846
|
+
resolved_kappa = float(kappa_src)
|
|
847
|
+
else:
|
|
848
|
+
resolved_kappa = 0.25 if kappa is None else float(kappa)
|
|
541
849
|
if resolved_kappa <= 0.0:
|
|
542
850
|
raise ValueError("kappa must be > 0.")
|
|
543
851
|
kw["kappa"] = resolved_kappa
|
|
@@ -5,10 +5,13 @@ from __future__ import annotations
|
|
|
5
5
|
import math
|
|
6
6
|
from typing import Callable
|
|
7
7
|
|
|
8
|
+
import numpy as np
|
|
9
|
+
from numpy.typing import ArrayLike
|
|
10
|
+
|
|
8
11
|
KernelFn = Callable[..., float]
|
|
9
12
|
|
|
10
13
|
|
|
11
|
-
def kernel_fisher(d2: float, /, *, kappa: float, **kw) -> float:
|
|
14
|
+
def kernel_fisher(d2: float, /, *, kappa: float, **kw: object) -> float:
|
|
12
15
|
"""Fisher information saturation kernel: w = d² / (d² + κ).
|
|
13
16
|
|
|
14
17
|
Motivated by the signal-to-noise structure of the Gaussian score.
|
|
@@ -19,14 +22,14 @@ def kernel_fisher(d2: float, /, *, kappa: float, **kw) -> float:
|
|
|
19
22
|
return d2 / (d2 + kappa)
|
|
20
23
|
|
|
21
24
|
|
|
22
|
-
def kernel_exponential(d2: float, /, *, gamma: float, **kw) -> float:
|
|
25
|
+
def kernel_exponential(d2: float, /, *, gamma: float, **kw: object) -> float:
|
|
23
26
|
"""Rayleigh survival (exponential) kernel: w = 1 − exp(−d²/(2γ²))."""
|
|
24
27
|
if gamma <= 0.0:
|
|
25
28
|
raise ValueError("gamma must be > 0.")
|
|
26
29
|
return 1.0 - math.exp(-d2 / (2.0 * gamma * gamma))
|
|
27
30
|
|
|
28
31
|
|
|
29
|
-
def kernel_chi2_cdf(d2: float, /, *, n_observed: int, **kw) -> float:
|
|
32
|
+
def kernel_chi2_cdf(d2: float, /, *, n_observed: int, **kw: object) -> float:
|
|
30
33
|
"""Chi-squared CDF kernel: w = F_χ²_N(N·d²). Requires scipy."""
|
|
31
34
|
try:
|
|
32
35
|
from scipy.stats import chi2
|
|
@@ -38,7 +41,7 @@ def kernel_chi2_cdf(d2: float, /, *, n_observed: int, **kw) -> float:
|
|
|
38
41
|
return float(chi2.cdf(n_observed * d2, df=n_observed))
|
|
39
42
|
|
|
40
43
|
|
|
41
|
-
def calibrate_kappa(d2_samples, target_mean_weight: float = 0.5) -> float:
|
|
44
|
+
def calibrate_kappa(d2_samples: ArrayLike, target_mean_weight: float = 0.5) -> float:
|
|
42
45
|
"""Find κ such that E[d²/(d²+κ)] ≈ target_mean_weight on burn-in data.
|
|
43
46
|
|
|
44
47
|
This implements the closed-form initializer from Proposition 1 of the paper,
|
|
@@ -56,7 +59,6 @@ def calibrate_kappa(d2_samples, target_mean_weight: float = 0.5) -> float:
|
|
|
56
59
|
float
|
|
57
60
|
Calibrated κ value.
|
|
58
61
|
"""
|
|
59
|
-
import numpy as np
|
|
60
62
|
try:
|
|
61
63
|
from scipy.optimize import brentq
|
|
62
64
|
except ImportError as exc:
|
|
@@ -75,7 +77,7 @@ def calibrate_kappa(d2_samples, target_mean_weight: float = 0.5) -> float:
|
|
|
75
77
|
kappa_init = mu * (1.0 / target_mean_weight - 1.0)
|
|
76
78
|
|
|
77
79
|
# Refine via root finding
|
|
78
|
-
def residual(kappa):
|
|
80
|
+
def residual(kappa: float) -> float:
|
|
79
81
|
return float(np.mean(d2 / (d2 + kappa))) - target_mean_weight
|
|
80
82
|
|
|
81
83
|
lo = max(kappa_init * 0.01, 1e-6)
|
|
@@ -90,8 +92,8 @@ def calibrate_kappa(d2_samples, target_mean_weight: float = 0.5) -> float:
|
|
|
90
92
|
|
|
91
93
|
|
|
92
94
|
def extract_d2_series(
|
|
93
|
-
returns, lambda_vol: float = 0.98, epsilon: float = 1e-8,
|
|
94
|
-
):
|
|
95
|
+
returns: ArrayLike, lambda_vol: float = 0.98, epsilon: float = 1e-8,
|
|
96
|
+
) -> np.ndarray:
|
|
95
97
|
"""Extract the d̄²_t series from a returns panel for κ calibration.
|
|
96
98
|
|
|
97
99
|
Parameters
|
|
@@ -106,7 +108,6 @@ def extract_d2_series(
|
|
|
106
108
|
ndarray, shape (T,)
|
|
107
109
|
Average squared standardized return at each timestep.
|
|
108
110
|
"""
|
|
109
|
-
import numpy as np
|
|
110
111
|
|
|
111
112
|
x = np.asarray(returns, dtype=np.float64)
|
|
112
113
|
t_total, n = x.shape
|