squeeze-kernel 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/PKG-INFO +21 -3
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/README.md +20 -2
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/pyproject.toml +1 -1
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/__init__.py +1 -1
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/estimator.py +343 -48
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/batch.py +0 -0
- {squeeze_kernel-0.4.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/kernels.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: squeeze-kernel
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage
|
|
5
5
|
Keywords: covariance,correlation,ewma,kernel,risk,streaming
|
|
6
6
|
Author: Robert Kende
|
|
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
|
|
|
37
37
|
|
|
38
38
|
A **streaming covariance estimator for panels of daily financial returns**. One `O(n²)` update per day, positive semi-definite **by construction** at every step, missing values handled **natively**, and defaults that require no tuning. Only dependency: NumPy.
|
|
39
39
|
|
|
40
|
-
Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage"* (Kende, 2026).
|
|
40
|
+
Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage"* (Kende, 2026) — [SSRN abstract 6455918](https://ssrn.com/abstract=6455918).
|
|
41
41
|
|
|
42
42
|
## Why
|
|
43
43
|
|
|
@@ -116,6 +116,14 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
116
116
|
|
|
117
117
|
## Advanced options
|
|
118
118
|
|
|
119
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
123
|
+
# maximal variant at high dimension:
|
|
124
|
+
est = SqueezeKernelEstimator(n_assets=300, corr_half_lives=(43, 173, 693), shrinkage_target="cluster")
|
|
125
|
+
```
|
|
126
|
+
|
|
119
127
|
**Score-exact weighting** (`weight_statistic="mahalanobis"`, use with `kappa=1.0`): drives the kernel with the Mahalanobis surprise `z'C⁻¹z/N` against the estimator's own correlation instead of the marginal dispersion. Improves accuracy in the moderate-concentration regime — use only when `n / T_eff ≲ 0.5` (e.g. n ≤ 100 at the default `lambda_corr`); at higher concentration the estimated inverse degrades it and the default is strictly better.
|
|
120
128
|
|
|
121
129
|
```python
|
|
@@ -136,6 +144,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
136
144
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
137
145
|
```
|
|
138
146
|
|
|
147
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
151
|
+
# ... update loop ...
|
|
152
|
+
m = est.usable_mask
|
|
153
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
154
|
+
```
|
|
155
|
+
|
|
139
156
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
140
157
|
|
|
141
158
|
## How it works
|
|
@@ -165,7 +182,8 @@ Releases: publishing a GitHub release from a `v*` tag triggers the [publish work
|
|
|
165
182
|
@article{kende2026squeeze,
|
|
166
183
|
title = {The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage},
|
|
167
184
|
author = {Kende, Robert},
|
|
168
|
-
year = {2026}
|
|
185
|
+
year = {2026},
|
|
186
|
+
note = {Available at SSRN: \url{https://ssrn.com/abstract=6455918}}
|
|
169
187
|
}
|
|
170
188
|
```
|
|
171
189
|
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
A **streaming covariance estimator for panels of daily financial returns**. One `O(n²)` update per day, positive semi-definite **by construction** at every step, missing values handled **natively**, and defaults that require no tuning. Only dependency: NumPy.
|
|
9
9
|
|
|
10
|
-
Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage"* (Kende, 2026).
|
|
10
|
+
Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage"* (Kende, 2026) — [SSRN abstract 6455918](https://ssrn.com/abstract=6455918).
|
|
11
11
|
|
|
12
12
|
## Why
|
|
13
13
|
|
|
@@ -86,6 +86,14 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
86
86
|
|
|
87
87
|
## Advanced options
|
|
88
88
|
|
|
89
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
93
|
+
# maximal variant at high dimension:
|
|
94
|
+
est = SqueezeKernelEstimator(n_assets=300, corr_half_lives=(43, 173, 693), shrinkage_target="cluster")
|
|
95
|
+
```
|
|
96
|
+
|
|
89
97
|
**Score-exact weighting** (`weight_statistic="mahalanobis"`, use with `kappa=1.0`): drives the kernel with the Mahalanobis surprise `z'C⁻¹z/N` against the estimator's own correlation instead of the marginal dispersion. Improves accuracy in the moderate-concentration regime — use only when `n / T_eff ≲ 0.5` (e.g. n ≤ 100 at the default `lambda_corr`); at higher concentration the estimated inverse degrades it and the default is strictly better.
|
|
90
98
|
|
|
91
99
|
```python
|
|
@@ -106,6 +114,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
106
114
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
107
115
|
```
|
|
108
116
|
|
|
117
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
121
|
+
# ... update loop ...
|
|
122
|
+
m = est.usable_mask
|
|
123
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
124
|
+
```
|
|
125
|
+
|
|
109
126
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
110
127
|
|
|
111
128
|
## How it works
|
|
@@ -135,7 +152,8 @@ Releases: publishing a GitHub release from a `v*` tag triggers the [publish work
|
|
|
135
152
|
@article{kende2026squeeze,
|
|
136
153
|
title = {The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking with Adaptive Shrinkage},
|
|
137
154
|
author = {Kende, Robert},
|
|
138
|
-
year = {2026}
|
|
155
|
+
year = {2026},
|
|
156
|
+
note = {Available at SSRN: \url{https://ssrn.com/abstract=6455918}}
|
|
139
157
|
}
|
|
140
158
|
```
|
|
141
159
|
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "squeeze-kernel"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.6.0"
|
|
8
8
|
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -2,12 +2,26 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
5
7
|
import numpy as np
|
|
6
8
|
|
|
9
|
+
try:
|
|
10
|
+
# SciPy's direct LAPACK bindings factorise an SPD matrix ~4x faster than
|
|
11
|
+
# the NumPy slogdet+solve route (one dpotrf vs two LU factorisations) and
|
|
12
|
+
# compute the identical quantities; the detector falls back to the
|
|
13
|
+
# NumPy-only path when SciPy is absent.
|
|
14
|
+
from scipy.linalg import cho_factor as _cho_factor, cho_solve as _cho_solve
|
|
15
|
+
except ImportError: # pragma: no cover
|
|
16
|
+
_cho_factor = _cho_solve = None
|
|
17
|
+
|
|
7
18
|
from squeeze_kernel.kernels import (
|
|
8
19
|
KernelFn, kernel_fisher, calibrate_kappa, extract_d2_series,
|
|
9
20
|
)
|
|
10
21
|
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
from collections.abc import Sequence
|
|
24
|
+
|
|
11
25
|
|
|
12
26
|
class SqueezeKernelEstimator:
|
|
13
27
|
"""Streaming robust covariance estimator with pluggable kernel weighting.
|
|
@@ -107,6 +121,64 @@ class SqueezeKernelEstimator:
|
|
|
107
121
|
concentration is unchanged. Held-out one-step NLL on the S&P-500
|
|
108
122
|
benchmark: +0.14 (negligible) at n=100, −4.2 at n=200, −25.0 at
|
|
109
123
|
n=300. Recommended when n approaches the effective sample size.
|
|
124
|
+
corr_half_lives : sequence of float, optional
|
|
125
|
+
Scale-free correlation memory with surprise-gated adaptive blend
|
|
126
|
+
weights. When set, the single correlation timescale is replaced
|
|
127
|
+
by a positive combination of EWMAs on the given geometric
|
|
128
|
+
half-life ladder (in trading days, e.g. ``(43, 173, 693)``): each
|
|
129
|
+
scale is normalised and adaptively shrunk against its own
|
|
130
|
+
effective sample size, and the resulting covariances are blended
|
|
131
|
+
with weights resting at the prior ∝ half-life\\ :sup:`corr_theta`.
|
|
132
|
+
By Bernstein's theorem the ladder approximates the power-law
|
|
133
|
+
memory of financial correlations (the streaming analogue of HAR).
|
|
134
|
+
For ladders of two or more rungs the blend weights are gated by a
|
|
135
|
+
sequential surprise detector: a two-sided Page CUSUM on the
|
|
136
|
+
studentised fast-vs-slow per-rung predictive-score drift
|
|
137
|
+
(reference drift 0.5, threshold 4.9721 = Siegmund average-run-
|
|
138
|
+
length ~2 years); an alarm applies a half-magnitude tilt of the
|
|
139
|
+
theta-prior toward the inverse-horizon vector (fast alarms,
|
|
140
|
+
w ~ 1/h) or the square-root-horizon vector (slow alarms,
|
|
141
|
+
w ~ h^0.5), decaying at the fastest rung's half-life. Weights
|
|
142
|
+
equal the prior on all non-alarmed days and the blend stays
|
|
143
|
+
convex, so PSD is structural throughout. The detector adds five
|
|
144
|
+
scalars of state and one Cholesky per rung per update for the
|
|
145
|
+
scores; it is validated across S&P panels (held-out +0.5-0.6
|
|
146
|
+
NLL/day over the detector-off blend), an external industry panel
|
|
147
|
+
incl. an out-of-time seal (+0.53/day, p=1e-4), a multi-asset
|
|
148
|
+
futures panel, and synthetic regime/null suites, with a
|
|
149
|
+
sign-stable one-at-a-time sensitivity sweep over all structural
|
|
150
|
+
constants. ``None`` (default) is the published single-scale
|
|
151
|
+
estimator, bit-for-bit; a one-element ladder reduces to a
|
|
152
|
+
single-scale estimator at that half-life (no detector).
|
|
153
|
+
Mutually exclusive with ``lambda_corr_fast``; composes with
|
|
154
|
+
``shrinkage_target='cluster'``. Cost is O(K·n²) per update plus
|
|
155
|
+
the detector's per-rung score factorisations. Held-out one-step
|
|
156
|
+
NLL on the S&P-500 benchmark improves on the single-scale
|
|
157
|
+
estimator at every dimension (−4.6 at n=100, −11.1 at n=300
|
|
158
|
+
before the cluster target); the $90\\%$ model confidence set
|
|
159
|
+
collapses to this configuration alone. Recommended default: the
|
|
160
|
+
base-centred ladder ``(43, 173, 693)`` with ``corr_theta=0.25``.
|
|
161
|
+
corr_theta : float
|
|
162
|
+
Long-memory exponent controlling the resting blend weights
|
|
163
|
+
(default 0.25). Only used when ``corr_half_lives`` is set.
|
|
164
|
+
min_obs : int or None
|
|
165
|
+
Usability gate for newly listed assets. When set, the property
|
|
166
|
+
``usable_mask`` marks an asset usable only once it has delivered
|
|
167
|
+
at least ``min_obs`` finite observations. The gate is purely
|
|
168
|
+
diagnostic: state evolution and ``get_cov``/``get_corr`` are
|
|
169
|
+
unchanged (the states keep warming during the gated window, so an
|
|
170
|
+
asset is fully warm when the gate lifts). Motivation: on an
|
|
171
|
+
expanding multi-asset universe, forecast rows for assets in their
|
|
172
|
+
first ~100 observations are dominated by the single-observation
|
|
173
|
+
variance initialisation and are unusable for scoring or portfolio
|
|
174
|
+
construction (measured ≈ +2,800 NLL/day on days whose scored set
|
|
175
|
+
included such assets). Deployment recipe::
|
|
176
|
+
|
|
177
|
+
m = est.usable_mask
|
|
178
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
179
|
+
|
|
180
|
+
Recommended ``min_obs`` ≈ 60–100 for daily data. ``None``
|
|
181
|
+
(default) disables the gate (``usable_mask`` is all-True).
|
|
110
182
|
|
|
111
183
|
Examples
|
|
112
184
|
--------
|
|
@@ -138,6 +210,9 @@ class SqueezeKernelEstimator:
|
|
|
138
210
|
vol_anchor_phi: float | None = None,
|
|
139
211
|
vol_anchor_decay: float = 0.999,
|
|
140
212
|
shrinkage_target: str = "equicorrelation",
|
|
213
|
+
corr_half_lives: "Sequence[float] | None" = None,
|
|
214
|
+
corr_theta: float = 0.25,
|
|
215
|
+
min_obs: int | None = None,
|
|
141
216
|
):
|
|
142
217
|
self.n_assets = n_assets
|
|
143
218
|
self.lambda_vol = lambda_vol
|
|
@@ -163,6 +238,54 @@ class SqueezeKernelEstimator:
|
|
|
163
238
|
if shrinkage_target not in ("equicorrelation", "cluster"):
|
|
164
239
|
raise ValueError("shrinkage_target must be 'equicorrelation' or 'cluster'.")
|
|
165
240
|
self.shrinkage_target = shrinkage_target
|
|
241
|
+
if min_obs is not None and (not isinstance(min_obs, int) or min_obs < 1):
|
|
242
|
+
raise ValueError("min_obs must be a positive integer or None.")
|
|
243
|
+
self.min_obs = min_obs
|
|
244
|
+
self._obs_count = np.zeros(n_assets, dtype=np.int64)
|
|
245
|
+
|
|
246
|
+
# Scale-free correlation memory (opt-in): replace the single correlation
|
|
247
|
+
# timescale by a positive combination of EWMAs on a geometric half-life
|
|
248
|
+
# ladder, blended per-scale (Mode A). None => single-scale, published
|
|
249
|
+
# behaviour bit-for-bit. See ``corr_half_lives`` in the class docstring.
|
|
250
|
+
self.corr_half_lives = None
|
|
251
|
+
self.corr_theta = corr_theta
|
|
252
|
+
self._corr_lam: np.ndarray | None = None
|
|
253
|
+
self._corr_w: np.ndarray | None = None
|
|
254
|
+
self._Q_list: list[np.ndarray] | None = None
|
|
255
|
+
self._S_list: list[float] | None = None
|
|
256
|
+
self._adaptive = False
|
|
257
|
+
if corr_half_lives is not None:
|
|
258
|
+
hl = np.asarray(corr_half_lives, dtype=np.float64)
|
|
259
|
+
if hl.ndim != 1 or hl.size < 1 or np.any(hl <= 0.0):
|
|
260
|
+
raise ValueError("corr_half_lives must be a non-empty sequence of positive half-lives.")
|
|
261
|
+
if corr_theta < 0.0:
|
|
262
|
+
raise ValueError("corr_theta must be >= 0.")
|
|
263
|
+
if lambda_corr_fast is not None:
|
|
264
|
+
raise ValueError(
|
|
265
|
+
"corr_half_lives and lambda_corr_fast are mutually exclusive "
|
|
266
|
+
"correlation-memory mechanisms; set at most one."
|
|
267
|
+
)
|
|
268
|
+
self.corr_half_lives = hl
|
|
269
|
+
self._corr_lam = 2.0 ** (-1.0 / hl)
|
|
270
|
+
w = hl ** corr_theta
|
|
271
|
+
self._corr_w = w / w.sum()
|
|
272
|
+
self._Q_list = [np.eye(n_assets, dtype=np.float64) for _ in hl]
|
|
273
|
+
self._S_list = [float(epsilon) for _ in hl]
|
|
274
|
+
# Surprise-gated blend weights (integral for K >= 2): two-sided
|
|
275
|
+
# Page CUSUM on the studentised fast-vs-slow per-rung
|
|
276
|
+
# predictive-score drift; drift 0.5, threshold from Siegmund's
|
|
277
|
+
# ARL approximation at ARL0 = 504 trading days (~2 years).
|
|
278
|
+
self._adaptive = hl.size >= 2
|
|
279
|
+
if self._adaptive:
|
|
280
|
+
self._aw_pi_fast = (1.0 / hl) / (1.0 / hl).sum()
|
|
281
|
+
self._aw_pi_slow = hl ** 0.5 / (hl ** 0.5).sum()
|
|
282
|
+
self._aw_lam_tilt = 2.0 ** (-1.0 / float(hl.min()))
|
|
283
|
+
self._aw_gamma_scale = 2.0 ** (-1.0 / float(np.median(hl)))
|
|
284
|
+
self._aw_drift, self._aw_b, self._aw_snap = 0.5, 4.9721088583, 0.5
|
|
285
|
+
self._aw_gp = self._aw_gm = 0.0
|
|
286
|
+
self._aw_tilt = 0.0
|
|
287
|
+
self._aw_scale = 1.0
|
|
288
|
+
self._aw_prev_sig: list[np.ndarray] | None = None
|
|
166
289
|
|
|
167
290
|
# Resolve shrinkage
|
|
168
291
|
if isinstance(shrinkage, str):
|
|
@@ -174,21 +297,33 @@ class SqueezeKernelEstimator:
|
|
|
174
297
|
self._kernel_fn, self._kernel_kwargs = _resolve_kernel(kappa, kernel_fn, kernel_kwargs)
|
|
175
298
|
self.kappa = self._kernel_kwargs.get("kappa") if self._kernel_fn is kernel_fisher else None
|
|
176
299
|
|
|
177
|
-
# State
|
|
300
|
+
# State. The correlation memory is stored NORMALISED: Q_t = M_t / S_t
|
|
301
|
+
# with the recursion Q_t = (1 - eta_t) Q_{t-1} + eta_t z_t z_t',
|
|
302
|
+
# eta_t = w_t / S_t after S_t <- lam S_{t-1} + w_t. This is
|
|
303
|
+
# algebraically identical to the raw-mass form (M init eps*I, S init
|
|
304
|
+
# eps => Q init I), keeps the matrix state well scaled, and makes the
|
|
305
|
+
# PSD convex-combination recursion explicit.
|
|
178
306
|
self._var_t: np.ndarray | None = None
|
|
179
307
|
self._var_init: np.ndarray | None = None
|
|
180
308
|
self._var_anchor: np.ndarray | None = None
|
|
181
|
-
self.
|
|
309
|
+
self._vol_t: np.ndarray | None = None
|
|
310
|
+
# No single-scale state is allocated in ladder mode.
|
|
311
|
+
self._Q_t = (np.eye(n_assets, dtype=np.float64)
|
|
312
|
+
if self._corr_lam is None else None)
|
|
182
313
|
self._S_t = float(epsilon)
|
|
183
314
|
self._cov: np.ndarray | None = None
|
|
184
315
|
self._corr: np.ndarray | None = None
|
|
185
316
|
self._last_weight: float = 0.0
|
|
317
|
+
# Extraction (normalise + shrink + vol application) is deferred until
|
|
318
|
+
# get_cov()/get_corr(); _dirty marks state newer than _cov/_corr.
|
|
319
|
+
self._dirty = False
|
|
186
320
|
|
|
187
321
|
# Cached scratch buffers reused per ``update()`` to avoid per-step
|
|
188
322
|
# allocator churn. These are intentionally module-private and
|
|
189
323
|
# never escape the estimator.
|
|
190
324
|
self._scratch_outer = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
191
325
|
self._scratch_corr = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
326
|
+
self._scratch_had = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
192
327
|
self._n_off = float(n_assets * (n_assets - 1)) if n_assets > 1 else 1.0
|
|
193
328
|
|
|
194
329
|
# ── Public API ────────────────────────────────────────────────────────
|
|
@@ -213,6 +348,63 @@ class SqueezeKernelEstimator:
|
|
|
213
348
|
raise ValueError(f"Expected shape ({n},), got {r_t.shape}.")
|
|
214
349
|
|
|
215
350
|
finite = np.isfinite(r_t)
|
|
351
|
+
self._obs_count[finite] += 1
|
|
352
|
+
|
|
353
|
+
# ── Adaptive-weight detector: score r_t under yesterday's per-rung
|
|
354
|
+
# forecasts, then advance the CUSUM (weights used below therefore
|
|
355
|
+
# reflect information through r_t only — causal). ──
|
|
356
|
+
if (self._adaptive and self._aw_prev_sig is not None
|
|
357
|
+
and finite.any()):
|
|
358
|
+
oidx = np.flatnonzero(finite)
|
|
359
|
+
r_o = r_t[oidx]
|
|
360
|
+
ell = np.empty(len(self._aw_prev_sig))
|
|
361
|
+
ok = True
|
|
362
|
+
for k, sig in enumerate(self._aw_prev_sig):
|
|
363
|
+
sub = sig[np.ix_(oidx, oidx)]
|
|
364
|
+
sub = (sub + sub.T) * 0.5
|
|
365
|
+
if _cho_factor is not None:
|
|
366
|
+
# One Cholesky per rung: logdet from the factor's
|
|
367
|
+
# diagonal, quadratic form via triangular solves.
|
|
368
|
+
try:
|
|
369
|
+
cf = _cho_factor(sub, lower=True, check_finite=False)
|
|
370
|
+
except np.linalg.LinAlgError:
|
|
371
|
+
ok = False # degenerate day (e.g. a fresh
|
|
372
|
+
break # listing): no clean score
|
|
373
|
+
logdet = 2.0 * np.log(np.diagonal(cf[0])).sum()
|
|
374
|
+
quad = float(r_o @ _cho_solve(cf, r_o, check_finite=False))
|
|
375
|
+
else:
|
|
376
|
+
sign, logdet = np.linalg.slogdet(sub)
|
|
377
|
+
if sign <= 0:
|
|
378
|
+
ok = False # degenerate day (e.g. a fresh
|
|
379
|
+
break # listing): no clean score
|
|
380
|
+
try:
|
|
381
|
+
quad = float(r_o @ np.linalg.solve(sub, r_o))
|
|
382
|
+
except np.linalg.LinAlgError:
|
|
383
|
+
ok = False
|
|
384
|
+
break
|
|
385
|
+
ell[k] = -0.5 * (oidx.size * np.log(2 * np.pi) + logdet + quad)
|
|
386
|
+
if not ok:
|
|
387
|
+
self._aw_tilt *= self._aw_lam_tilt
|
|
388
|
+
ell = None
|
|
389
|
+
else:
|
|
390
|
+
ell = None
|
|
391
|
+
if ell is not None:
|
|
392
|
+
dd = ell - ell.mean()
|
|
393
|
+
rms = float(np.sqrt((dd @ dd) / ell.size))
|
|
394
|
+
self._aw_scale = (self._aw_gamma_scale * self._aw_scale
|
|
395
|
+
+ (1.0 - self._aw_gamma_scale) * rms)
|
|
396
|
+
zc = np.clip(dd / (self._aw_scale + 1e-12), -3.0, 3.0)
|
|
397
|
+
k_fast = int(np.argmin(self.corr_half_lives))
|
|
398
|
+
k_slow = int(np.argmax(self.corr_half_lives))
|
|
399
|
+
zfs = float(zc[k_fast] - zc[k_slow])
|
|
400
|
+
self._aw_gp = max(0.0, self._aw_gp + zfs - self._aw_drift)
|
|
401
|
+
self._aw_gm = max(0.0, self._aw_gm - zfs - self._aw_drift)
|
|
402
|
+
if self._aw_gp > self._aw_b:
|
|
403
|
+
self._aw_tilt, self._aw_gp = self._aw_snap, 0.0
|
|
404
|
+
elif self._aw_gm > self._aw_b:
|
|
405
|
+
self._aw_tilt, self._aw_gm = -self._aw_snap, 0.0
|
|
406
|
+
else:
|
|
407
|
+
self._aw_tilt *= self._aw_lam_tilt
|
|
216
408
|
|
|
217
409
|
# ── Volatility update ──
|
|
218
410
|
if self._var_t is None:
|
|
@@ -260,10 +452,13 @@ class SqueezeKernelEstimator:
|
|
|
260
452
|
if n_obs > 0:
|
|
261
453
|
z_t[finite] = r_t[finite] / (vol_t[finite] + eps)
|
|
262
454
|
d2 = float(z_t[finite] @ z_t[finite]) / n_obs
|
|
263
|
-
if self.weight_statistic == "mahalanobis" and self.
|
|
455
|
+
if self.weight_statistic == "mahalanobis" and self._vol_t is not None:
|
|
264
456
|
# Score-exact surprise against the estimator's own previous
|
|
265
457
|
# correlation; falls back to the marginal d² on the first
|
|
266
|
-
# step or a (rare) singular observed submatrix.
|
|
458
|
+
# step or a (rare) singular observed submatrix. Extraction is
|
|
459
|
+
# lazy, so bring _corr up to the t-1 state first.
|
|
460
|
+
if self._dirty:
|
|
461
|
+
self._materialize()
|
|
267
462
|
try:
|
|
268
463
|
c_sub = self._corr[np.ix_(finite, finite)]
|
|
269
464
|
d2 = float(z_t[finite] @ np.linalg.solve(c_sub, z_t[finite])) / n_obs
|
|
@@ -277,35 +472,63 @@ class SqueezeKernelEstimator:
|
|
|
277
472
|
if self.impute_missing and 0 < n_obs < n:
|
|
278
473
|
self._impute(z_t, finite)
|
|
279
474
|
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
if self.lambda_corr_fast is not None:
|
|
283
|
-
# Score-driven memory: stress days (w_t → 1) shorten the memory
|
|
284
|
-
# toward lambda_corr_fast; calm days keep the slow decay.
|
|
285
|
-
lam_c = self.lambda_corr + (self.lambda_corr_fast - self.lambda_corr) * w_t
|
|
286
|
-
self._S_t = lam_c * self._S_t + w_t
|
|
287
|
-
self._M_t *= lam_c
|
|
288
|
-
if w_t > 0.0 and n_obs > 0:
|
|
475
|
+
add = w_t > 0.0 and n_obs > 0
|
|
476
|
+
if add:
|
|
289
477
|
# np.multiply.outer with out= avoids the temporary that
|
|
290
|
-
# np.outer otherwise allocates each step.
|
|
478
|
+
# np.outer otherwise allocates each step. zz' is computed once
|
|
479
|
+
# and shared by every rung.
|
|
291
480
|
np.multiply.outer(z_t, z_t, out=self._scratch_outer)
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
481
|
+
if self._corr_lam is None:
|
|
482
|
+
# ── Single-scale correlation EWMA (published path) ──
|
|
483
|
+
lam_c = self.lambda_corr
|
|
484
|
+
if self.lambda_corr_fast is not None:
|
|
485
|
+
# Score-driven memory: stress days (w_t → 1) shorten the memory
|
|
486
|
+
# toward lambda_corr_fast; calm days keep the slow decay.
|
|
487
|
+
lam_c = self.lambda_corr + (self.lambda_corr_fast - self.lambda_corr) * w_t
|
|
488
|
+
self._S_t = lam_c * self._S_t + w_t
|
|
489
|
+
if add:
|
|
490
|
+
# Q <- (1 - eta) Q + eta zz'. With w_t = 0 both S and M decay
|
|
491
|
+
# by lam_c, so Q is unchanged — no matrix work at all.
|
|
492
|
+
eta = w_t / self._S_t
|
|
493
|
+
self._Q_t *= 1.0 - eta
|
|
494
|
+
self._scratch_outer *= eta
|
|
495
|
+
self._Q_t += self._scratch_outer
|
|
496
|
+
else:
|
|
497
|
+
# ── Scale-free ladder (Mode A) ──
|
|
498
|
+
# Update K normalised correlation states on the geometric
|
|
499
|
+
# half-life ladder; extraction (normalise + shrink + blend)
|
|
500
|
+
# happens lazily in _materialize().
|
|
501
|
+
s_eff = 0.0
|
|
502
|
+
for k in range(self._corr_lam.size):
|
|
503
|
+
self._S_list[k] = self._corr_lam[k] * self._S_list[k] + w_t
|
|
504
|
+
if add:
|
|
505
|
+
eta = w_t / self._S_list[k]
|
|
506
|
+
self._Q_list[k] *= 1.0 - eta
|
|
507
|
+
np.multiply(self._scratch_outer, eta, out=self._scratch_had)
|
|
508
|
+
self._Q_list[k] += self._scratch_had
|
|
509
|
+
s_eff += self._corr_w[k] * self._S_list[k]
|
|
510
|
+
self._S_t = s_eff # blended effective size (for the property)
|
|
511
|
+
self._vol_t = vol_t
|
|
512
|
+
self._dirty = True
|
|
513
|
+
if self._adaptive:
|
|
514
|
+
self._materialize_adaptive()
|
|
296
515
|
self._last_weight = w_t
|
|
297
516
|
return w_t
|
|
298
517
|
|
|
299
518
|
def get_cov(self) -> np.ndarray:
|
|
300
519
|
"""Return the current covariance matrix estimate (n x n)."""
|
|
301
|
-
if self.
|
|
520
|
+
if self._vol_t is None:
|
|
302
521
|
raise RuntimeError("Call update() at least once before get_cov().")
|
|
522
|
+
if self._dirty:
|
|
523
|
+
self._materialize()
|
|
303
524
|
return self._cov.copy()
|
|
304
525
|
|
|
305
526
|
def get_corr(self) -> np.ndarray:
|
|
306
527
|
"""Return the current correlation matrix estimate (n x n)."""
|
|
307
|
-
if self.
|
|
528
|
+
if self._vol_t is None:
|
|
308
529
|
raise RuntimeError("Call update() at least once before get_corr().")
|
|
530
|
+
if self._dirty:
|
|
531
|
+
self._materialize()
|
|
309
532
|
return self._corr.copy()
|
|
310
533
|
|
|
311
534
|
@property
|
|
@@ -313,6 +536,17 @@ class SqueezeKernelEstimator:
|
|
|
313
536
|
"""Kernel weight assigned to the most recent observation."""
|
|
314
537
|
return self._last_weight
|
|
315
538
|
|
|
539
|
+
@property
|
|
540
|
+
def usable_mask(self) -> np.ndarray:
|
|
541
|
+
"""Boolean mask of assets with at least ``min_obs`` observations.
|
|
542
|
+
|
|
543
|
+
All-True when ``min_obs`` is None. Purely diagnostic — estimates
|
|
544
|
+
are not affected; subset the outputs with it (see class docstring).
|
|
545
|
+
"""
|
|
546
|
+
if self.min_obs is None:
|
|
547
|
+
return np.ones(self.n_assets, dtype=bool)
|
|
548
|
+
return self._obs_count >= self.min_obs
|
|
549
|
+
|
|
316
550
|
@property
|
|
317
551
|
def effective_sample_size(self) -> float:
|
|
318
552
|
"""Kernel-weighted effective sample size S_t."""
|
|
@@ -352,9 +586,13 @@ class SqueezeKernelEstimator:
|
|
|
352
586
|
# ── Private helpers ───────────────────────────────────────────────────
|
|
353
587
|
|
|
354
588
|
def _impute(self, z_t: np.ndarray, finite: np.ndarray) -> None:
|
|
589
|
+
if self._Q_t is None:
|
|
590
|
+
# Ladder mode: imputation reads the single-scale state, which
|
|
591
|
+
# was never updated on this path — historically a silent no-op
|
|
592
|
+
# (all correlations below threshold); keep it an explicit one.
|
|
593
|
+
return
|
|
355
594
|
eps = self.epsilon
|
|
356
|
-
|
|
357
|
-
sigma_z = self._M_t / denom
|
|
595
|
+
sigma_z = self._Q_t
|
|
358
596
|
diag_z = np.diag(sigma_z)
|
|
359
597
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
360
598
|
missing = ~finite
|
|
@@ -373,23 +611,22 @@ class SqueezeKernelEstimator:
|
|
|
373
611
|
if den > 0.0:
|
|
374
612
|
z_t[i] = num / den
|
|
375
613
|
|
|
376
|
-
def
|
|
614
|
+
def _shrunk_corr_from_Q(self, Q: np.ndarray, S_t: float) -> np.ndarray:
|
|
615
|
+
"""Normalise one Q state to a correlation and shrink it in place.
|
|
616
|
+
|
|
617
|
+
Returns ``self._scratch_corr`` — valid only until the next call.
|
|
618
|
+
"""
|
|
377
619
|
eps = self.epsilon
|
|
378
|
-
S_t = max(self._S_t, eps)
|
|
379
620
|
n = self.n_assets
|
|
621
|
+
S_t = max(S_t, eps)
|
|
380
622
|
|
|
381
|
-
#
|
|
382
|
-
#
|
|
383
|
-
|
|
384
|
-
diag_z = np.diagonal(self._M_t).copy()
|
|
385
|
-
diag_z /= S_t # in-place
|
|
623
|
+
# corr_ij = Q_ij * inv_diag_i * inv_diag_j; the scalar S_t cancels
|
|
624
|
+
# in the normalisation, so Q needs no rescaling pass.
|
|
625
|
+
diag_z = np.diagonal(Q).copy()
|
|
386
626
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
387
|
-
# corr_ij = (M_ij / S_t) * inv_diag_i * inv_diag_j; this writes
|
|
388
|
-
# the rescaled outer-product into _scratch_corr in one pass.
|
|
389
627
|
np.multiply.outer(inv_diag, inv_diag, out=self._scratch_corr)
|
|
390
628
|
corr = self._scratch_corr
|
|
391
|
-
corr *=
|
|
392
|
-
corr *= (1.0 / S_t) # absorb the M_t / S_t scale
|
|
629
|
+
corr *= Q # in-place
|
|
393
630
|
np.fill_diagonal(corr, np.where(diag_z > eps, 1.0, 0.0))
|
|
394
631
|
|
|
395
632
|
# Adaptive shrinkage: blend toward the equicorrelation target
|
|
@@ -403,6 +640,9 @@ class SqueezeKernelEstimator:
|
|
|
403
640
|
# Off-diagonal mean: O(n^2) sum, no mask allocation.
|
|
404
641
|
rho_bar = (corr.sum() - corr.trace()) / self._n_off
|
|
405
642
|
if self.shrinkage_target == "equicorrelation" or rho_bar <= 0.0:
|
|
643
|
+
# rho_bar <= 0 also covers the q = meanoff(C o C) = 0 corner:
|
|
644
|
+
# C o C has nonnegative entries, so q = 0 forces C = I and
|
|
645
|
+
# hence rho_bar = 0 — the equicorrelation fallback applies.
|
|
406
646
|
corr *= (1.0 - alpha)
|
|
407
647
|
corr += alpha * rho_bar
|
|
408
648
|
np.fill_diagonal(corr, 1.0)
|
|
@@ -414,27 +654,82 @@ class SqueezeKernelEstimator:
|
|
|
414
654
|
# level-matched so the target carries the same average
|
|
415
655
|
# correlation mass as the equicorrelation target. As
|
|
416
656
|
# alpha -> 0 this reduces exactly to the published estimator.
|
|
417
|
-
had =
|
|
657
|
+
had = self._scratch_had
|
|
658
|
+
np.multiply(corr, corr, out=had) # Hadamard square, O(n^2)
|
|
418
659
|
mean_off = (had.sum() - np.trace(had)) / self._n_off
|
|
419
660
|
gamma = min(1.0, rho_bar / max(mean_off, eps))
|
|
420
661
|
corr *= (1.0 - alpha)
|
|
421
662
|
corr += (alpha * (1.0 - alpha)) * rho_bar
|
|
422
|
-
|
|
663
|
+
had *= alpha * alpha * gamma
|
|
664
|
+
corr += had
|
|
423
665
|
np.fill_diagonal(corr, 1.0)
|
|
666
|
+
return corr
|
|
424
667
|
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
668
|
+
def _materialize_adaptive(self) -> None:
|
|
669
|
+
"""Adaptive-weight extraction: build per-rung shrunk covariances
|
|
670
|
+
(kept for the next update's detector scores), blend with the
|
|
671
|
+
CUSUM-tilted weights, derive _cov/_corr. Runs eagerly."""
|
|
672
|
+
eps = self.epsilon
|
|
673
|
+
vol_t = self._vol_t
|
|
674
|
+
vv = np.multiply.outer(vol_t, vol_t)
|
|
675
|
+
sig_k = []
|
|
676
|
+
for k in range(self._corr_lam.size):
|
|
677
|
+
corr_k = self._shrunk_corr_from_Q(self._Q_list[k], self._S_list[k])
|
|
678
|
+
sig_k.append(corr_k * vv)
|
|
679
|
+
self._aw_prev_sig = sig_k
|
|
680
|
+
t = self._aw_tilt
|
|
681
|
+
pi = self._corr_w
|
|
682
|
+
if t >= 0:
|
|
683
|
+
w = (1.0 - t) * pi + t * self._aw_pi_fast
|
|
684
|
+
else:
|
|
685
|
+
w = (1.0 + t) * pi + (-t) * self._aw_pi_slow
|
|
686
|
+
cov = w[0] * sig_k[0]
|
|
687
|
+
for k in range(1, len(sig_k)):
|
|
688
|
+
cov = cov + w[k] * sig_k[k]
|
|
689
|
+
cov = (cov + cov.T) * 0.5
|
|
690
|
+
self._cov = cov
|
|
691
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
692
|
+
self._corr = cov / np.outer(d, d)
|
|
693
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
694
|
+
self._dirty = False
|
|
695
|
+
|
|
696
|
+
def _materialize(self) -> None:
|
|
697
|
+
"""Extract _cov/_corr from the current state (lazy, on demand)."""
|
|
698
|
+
eps = self.epsilon
|
|
699
|
+
vol_t = self._vol_t
|
|
700
|
+
if self._corr_lam is None:
|
|
701
|
+
# ── Single scale: shrunk correlation IS the correlation output ──
|
|
702
|
+
corr = self._shrunk_corr_from_Q(self._Q_t, self._S_t)
|
|
703
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
704
|
+
cov = corr * self._scratch_outer
|
|
705
|
+
cov += cov.T
|
|
706
|
+
cov *= 0.5
|
|
707
|
+
corr_out = corr.copy()
|
|
708
|
+
corr_out += corr_out.T
|
|
709
|
+
corr_out *= 0.5
|
|
710
|
+
self._cov, self._corr = cov, corr_out
|
|
711
|
+
else:
|
|
712
|
+
# ── Ladder: blend per-rung shrunk correlations, then apply the
|
|
713
|
+
# (shared) volatilities once — algebraically identical to
|
|
714
|
+
# blending per-rung covariances, K-1 fewer O(n^2) passes.
|
|
715
|
+
mix = np.zeros((self.n_assets, self.n_assets), dtype=np.float64)
|
|
716
|
+
for k in range(self._corr_lam.size):
|
|
717
|
+
corr_k = self._shrunk_corr_from_Q(self._Q_list[k], self._S_list[k])
|
|
718
|
+
corr_k *= self._corr_w[k]
|
|
719
|
+
mix += corr_k
|
|
720
|
+
cov = mix
|
|
721
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
722
|
+
cov *= self._scratch_outer
|
|
723
|
+
cov += cov.T
|
|
724
|
+
cov *= 0.5
|
|
725
|
+
self._cov = cov
|
|
726
|
+
# Correlation is re-derived from the blended covariance (not the
|
|
727
|
+
# blended correlation mix) to keep the published dead-asset
|
|
728
|
+
# semantics: rows of never-observed assets renormalise to zero.
|
|
729
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
730
|
+
self._corr = cov / np.outer(d, d)
|
|
731
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
732
|
+
self._dirty = False
|
|
438
733
|
|
|
439
734
|
|
|
440
735
|
# ── Kernel resolution ─────────────────────────────────────────────────────────
|
|
File without changes
|
|
File without changes
|