squeeze-kernel 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/PKG-INFO +11 -2
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/README.md +10 -1
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/pyproject.toml +1 -1
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/__init__.py +1 -1
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/estimator.py +295 -81
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/batch.py +0 -0
- {squeeze_kernel-0.5.0 → squeeze_kernel-0.6.0}/src/squeeze_kernel/kernels.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: squeeze-kernel
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage
|
|
5
5
|
Keywords: covariance,correlation,ewma,kernel,risk,streaming
|
|
6
6
|
Author: Robert Kende
|
|
@@ -116,7 +116,7 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
116
116
|
|
|
117
117
|
## Advanced options
|
|
118
118
|
|
|
119
|
-
**
|
|
119
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
120
120
|
|
|
121
121
|
```python
|
|
122
122
|
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
@@ -144,6 +144,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
144
144
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
145
145
|
```
|
|
146
146
|
|
|
147
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
151
|
+
# ... update loop ...
|
|
152
|
+
m = est.usable_mask
|
|
153
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
154
|
+
```
|
|
155
|
+
|
|
147
156
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
148
157
|
|
|
149
158
|
## How it works
|
|
@@ -86,7 +86,7 @@ kappa = SqueezeKernelEstimator.calibrate_kappa(burn_in_returns, target_weight=0.
|
|
|
86
86
|
|
|
87
87
|
## Advanced options
|
|
88
88
|
|
|
89
|
-
**
|
|
89
|
+
**Adaptive scale-free correlation memory** (`corr_half_lives=(43, 173, 693)`, `corr_theta=0.25`): replaces the single correlation timescale with a positive combination of EWMAs on a geometric half-life ladder — each scale normalized and adaptively shrunk against its own effective sample size, then the covariances blended with weights resting at the prior ∝ half-life^`corr_theta`. By Bernstein's theorem the ladder approximates the power-law memory of financial correlations (the streaming analogue of HAR). For ladders of two or more rungs the blend weights are gated by a sequential surprise detector: a two-sided Page CUSUM on the studentized fast-vs-slow per-rung predictive-score drift (threshold set by Siegmund's average-run-length approximation at ~2 years, no tuned parameters) tilts the weights toward the fast or slow end of the ladder when one side accumulates statistically forced evidence, decaying back at the fastest rung's half-life. Weights equal the prior on all non-alarmed days, the blend stays convex, so PSD holds by construction; `None` (default) reproduces the published single-scale estimator exactly. This is the paper's **headline configuration**: on the S&P 500 benchmark it leads every tested method at every universe size (held-out one-step NLL −4.6 vs single-scale at n=100, −11.1 at n=300 before the cluster target), the 90% model confidence set collapses to it alone, and the detector's margin is confirmed out-of-time on an external industry panel (+0.53 NLL/day, p=1×10⁻⁴). Cost is O(K·n²) per update plus one Cholesky per rung per day for the detector scores. Composes with `shrinkage_target="cluster"`; mutually exclusive with `lambda_corr_fast`.
|
|
90
90
|
|
|
91
91
|
```python
|
|
92
92
|
est = SqueezeKernelEstimator(n_assets=100, corr_half_lives=(43, 173, 693), corr_theta=0.25)
|
|
@@ -114,6 +114,15 @@ est = SqueezeKernelEstimator(n_assets=100, vol_anchor_phi=0.995)
|
|
|
114
114
|
est = SqueezeKernelEstimator(n_assets=300, shrinkage_target="cluster")
|
|
115
115
|
```
|
|
116
116
|
|
|
117
|
+
**New-listing usability gate** (`min_obs=60`): on an expanding universe, an asset's forecast rows are dominated by its single-observation variance initialization for its first weeks of life and are unusable for scoring or portfolio construction (measured ≈ +2,800 NLL/day on days whose scored set included such assets, on a 42-instrument multi-asset panel). `min_obs` gates nothing inside the estimator — states warm normally, all outputs are unchanged — it exposes a `usable_mask` property marking assets with at least `min_obs` finite observations, so deployments subset with it:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
est = SqueezeKernelEstimator(n_assets=42, min_obs=60)
|
|
121
|
+
# ... update loop ...
|
|
122
|
+
m = est.usable_mask
|
|
123
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
124
|
+
```
|
|
125
|
+
|
|
117
126
|
**Alternative kernels**: pass `kernel_fn=kernel_exponential` (with `kernel_kwargs={"gamma": ...}`) or `kernel_chi2_cdf`, or any callable `(d2, *, n_observed, **kw) -> float` mapping to `[0, 1)`. The PSD guarantee holds for any such kernel.
|
|
118
127
|
|
|
119
128
|
## How it works
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "squeeze-kernel"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.6.0"
|
|
8
8
|
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -6,6 +6,15 @@ from typing import TYPE_CHECKING
|
|
|
6
6
|
|
|
7
7
|
import numpy as np
|
|
8
8
|
|
|
9
|
+
try:
|
|
10
|
+
# SciPy's direct LAPACK bindings factorise an SPD matrix ~4x faster than
|
|
11
|
+
# the NumPy slogdet+solve route (one dpotrf vs two LU factorisations) and
|
|
12
|
+
# compute the identical quantities; the detector falls back to the
|
|
13
|
+
# NumPy-only path when SciPy is absent.
|
|
14
|
+
from scipy.linalg import cho_factor as _cho_factor, cho_solve as _cho_solve
|
|
15
|
+
except ImportError: # pragma: no cover
|
|
16
|
+
_cho_factor = _cho_solve = None
|
|
17
|
+
|
|
9
18
|
from squeeze_kernel.kernels import (
|
|
10
19
|
KernelFn, kernel_fisher, calibrate_kappa, extract_d2_series,
|
|
11
20
|
)
|
|
@@ -113,27 +122,63 @@ class SqueezeKernelEstimator:
|
|
|
113
122
|
benchmark: +0.14 (negligible) at n=100, −4.2 at n=200, −25.0 at
|
|
114
123
|
n=300. Recommended when n approaches the effective sample size.
|
|
115
124
|
corr_half_lives : sequence of float, optional
|
|
116
|
-
Scale-free correlation memory
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
``(43, 173, 693)``): each
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
125
|
+
Scale-free correlation memory with surprise-gated adaptive blend
|
|
126
|
+
weights. When set, the single correlation timescale is replaced
|
|
127
|
+
by a positive combination of EWMAs on the given geometric
|
|
128
|
+
half-life ladder (in trading days, e.g. ``(43, 173, 693)``): each
|
|
129
|
+
scale is normalised and adaptively shrunk against its own
|
|
130
|
+
effective sample size, and the resulting covariances are blended
|
|
131
|
+
with weights resting at the prior ∝ half-life\\ :sup:`corr_theta`.
|
|
132
|
+
By Bernstein's theorem the ladder approximates the power-law
|
|
133
|
+
memory of financial correlations (the streaming analogue of HAR).
|
|
134
|
+
For ladders of two or more rungs the blend weights are gated by a
|
|
135
|
+
sequential surprise detector: a two-sided Page CUSUM on the
|
|
136
|
+
studentised fast-vs-slow per-rung predictive-score drift
|
|
137
|
+
(reference drift 0.5, threshold 4.9721 = Siegmund average-run-
|
|
138
|
+
length ~2 years); an alarm applies a half-magnitude tilt of the
|
|
139
|
+
theta-prior toward the inverse-horizon vector (fast alarms,
|
|
140
|
+
w ~ 1/h) or the square-root-horizon vector (slow alarms,
|
|
141
|
+
w ~ h^0.5), decaying at the fastest rung's half-life. Weights
|
|
142
|
+
equal the prior on all non-alarmed days and the blend stays
|
|
143
|
+
convex, so PSD is structural throughout. The detector adds five
|
|
144
|
+
scalars of state and one Cholesky per rung per update for the
|
|
145
|
+
scores; it is validated across S&P panels (held-out +0.5-0.6
|
|
146
|
+
NLL/day over the detector-off blend), an external industry panel
|
|
147
|
+
incl. an out-of-time seal (+0.53/day, p=1e-4), a multi-asset
|
|
148
|
+
futures panel, and synthetic regime/null suites, with a
|
|
149
|
+
sign-stable one-at-a-time sensitivity sweep over all structural
|
|
150
|
+
constants. ``None`` (default) is the published single-scale
|
|
151
|
+
estimator, bit-for-bit; a one-element ladder reduces to a
|
|
152
|
+
single-scale estimator at that half-life (no detector).
|
|
153
|
+
Mutually exclusive with ``lambda_corr_fast``; composes with
|
|
154
|
+
``shrinkage_target='cluster'``. Cost is O(K·n²) per update plus
|
|
155
|
+
the detector's per-rung score factorisations. Held-out one-step
|
|
156
|
+
NLL on the S&P-500 benchmark improves on the single-scale
|
|
157
|
+
estimator at every dimension (−4.6 at n=100, −11.1 at n=300
|
|
158
|
+
before the cluster target); the $90\\%$ model confidence set
|
|
159
|
+
collapses to this configuration alone. Recommended default: the
|
|
160
|
+
base-centred ladder ``(43, 173, 693)`` with ``corr_theta=0.25``.
|
|
134
161
|
corr_theta : float
|
|
135
|
-
Long-memory exponent controlling the
|
|
136
|
-
0.25). Only used when ``corr_half_lives`` is set.
|
|
162
|
+
Long-memory exponent controlling the resting blend weights
|
|
163
|
+
(default 0.25). Only used when ``corr_half_lives`` is set.
|
|
164
|
+
min_obs : int or None
|
|
165
|
+
Usability gate for newly listed assets. When set, the property
|
|
166
|
+
``usable_mask`` marks an asset usable only once it has delivered
|
|
167
|
+
at least ``min_obs`` finite observations. The gate is purely
|
|
168
|
+
diagnostic: state evolution and ``get_cov``/``get_corr`` are
|
|
169
|
+
unchanged (the states keep warming during the gated window, so an
|
|
170
|
+
asset is fully warm when the gate lifts). Motivation: on an
|
|
171
|
+
expanding multi-asset universe, forecast rows for assets in their
|
|
172
|
+
first ~100 observations are dominated by the single-observation
|
|
173
|
+
variance initialisation and are unusable for scoring or portfolio
|
|
174
|
+
construction (measured ≈ +2,800 NLL/day on days whose scored set
|
|
175
|
+
included such assets). Deployment recipe::
|
|
176
|
+
|
|
177
|
+
m = est.usable_mask
|
|
178
|
+
cov_usable = est.get_cov()[np.ix_(m, m)]
|
|
179
|
+
|
|
180
|
+
Recommended ``min_obs`` ≈ 60–100 for daily data. ``None``
|
|
181
|
+
(default) disables the gate (``usable_mask`` is all-True).
|
|
137
182
|
|
|
138
183
|
Examples
|
|
139
184
|
--------
|
|
@@ -167,6 +212,7 @@ class SqueezeKernelEstimator:
|
|
|
167
212
|
shrinkage_target: str = "equicorrelation",
|
|
168
213
|
corr_half_lives: "Sequence[float] | None" = None,
|
|
169
214
|
corr_theta: float = 0.25,
|
|
215
|
+
min_obs: int | None = None,
|
|
170
216
|
):
|
|
171
217
|
self.n_assets = n_assets
|
|
172
218
|
self.lambda_vol = lambda_vol
|
|
@@ -192,6 +238,10 @@ class SqueezeKernelEstimator:
|
|
|
192
238
|
if shrinkage_target not in ("equicorrelation", "cluster"):
|
|
193
239
|
raise ValueError("shrinkage_target must be 'equicorrelation' or 'cluster'.")
|
|
194
240
|
self.shrinkage_target = shrinkage_target
|
|
241
|
+
if min_obs is not None and (not isinstance(min_obs, int) or min_obs < 1):
|
|
242
|
+
raise ValueError("min_obs must be a positive integer or None.")
|
|
243
|
+
self.min_obs = min_obs
|
|
244
|
+
self._obs_count = np.zeros(n_assets, dtype=np.int64)
|
|
195
245
|
|
|
196
246
|
# Scale-free correlation memory (opt-in): replace the single correlation
|
|
197
247
|
# timescale by a positive combination of EWMAs on a geometric half-life
|
|
@@ -201,8 +251,9 @@ class SqueezeKernelEstimator:
|
|
|
201
251
|
self.corr_theta = corr_theta
|
|
202
252
|
self._corr_lam: np.ndarray | None = None
|
|
203
253
|
self._corr_w: np.ndarray | None = None
|
|
204
|
-
self.
|
|
254
|
+
self._Q_list: list[np.ndarray] | None = None
|
|
205
255
|
self._S_list: list[float] | None = None
|
|
256
|
+
self._adaptive = False
|
|
206
257
|
if corr_half_lives is not None:
|
|
207
258
|
hl = np.asarray(corr_half_lives, dtype=np.float64)
|
|
208
259
|
if hl.ndim != 1 or hl.size < 1 or np.any(hl <= 0.0):
|
|
@@ -218,8 +269,23 @@ class SqueezeKernelEstimator:
|
|
|
218
269
|
self._corr_lam = 2.0 ** (-1.0 / hl)
|
|
219
270
|
w = hl ** corr_theta
|
|
220
271
|
self._corr_w = w / w.sum()
|
|
221
|
-
self.
|
|
272
|
+
self._Q_list = [np.eye(n_assets, dtype=np.float64) for _ in hl]
|
|
222
273
|
self._S_list = [float(epsilon) for _ in hl]
|
|
274
|
+
# Surprise-gated blend weights (integral for K >= 2): two-sided
|
|
275
|
+
# Page CUSUM on the studentised fast-vs-slow per-rung
|
|
276
|
+
# predictive-score drift; drift 0.5, threshold from Siegmund's
|
|
277
|
+
# ARL approximation at ARL0 = 504 trading days (~2 years).
|
|
278
|
+
self._adaptive = hl.size >= 2
|
|
279
|
+
if self._adaptive:
|
|
280
|
+
self._aw_pi_fast = (1.0 / hl) / (1.0 / hl).sum()
|
|
281
|
+
self._aw_pi_slow = hl ** 0.5 / (hl ** 0.5).sum()
|
|
282
|
+
self._aw_lam_tilt = 2.0 ** (-1.0 / float(hl.min()))
|
|
283
|
+
self._aw_gamma_scale = 2.0 ** (-1.0 / float(np.median(hl)))
|
|
284
|
+
self._aw_drift, self._aw_b, self._aw_snap = 0.5, 4.9721088583, 0.5
|
|
285
|
+
self._aw_gp = self._aw_gm = 0.0
|
|
286
|
+
self._aw_tilt = 0.0
|
|
287
|
+
self._aw_scale = 1.0
|
|
288
|
+
self._aw_prev_sig: list[np.ndarray] | None = None
|
|
223
289
|
|
|
224
290
|
# Resolve shrinkage
|
|
225
291
|
if isinstance(shrinkage, str):
|
|
@@ -231,21 +297,33 @@ class SqueezeKernelEstimator:
|
|
|
231
297
|
self._kernel_fn, self._kernel_kwargs = _resolve_kernel(kappa, kernel_fn, kernel_kwargs)
|
|
232
298
|
self.kappa = self._kernel_kwargs.get("kappa") if self._kernel_fn is kernel_fisher else None
|
|
233
299
|
|
|
234
|
-
# State
|
|
300
|
+
# State. The correlation memory is stored NORMALISED: Q_t = M_t / S_t
|
|
301
|
+
# with the recursion Q_t = (1 - eta_t) Q_{t-1} + eta_t z_t z_t',
|
|
302
|
+
# eta_t = w_t / S_t after S_t <- lam S_{t-1} + w_t. This is
|
|
303
|
+
# algebraically identical to the raw-mass form (M init eps*I, S init
|
|
304
|
+
# eps => Q init I), keeps the matrix state well scaled, and makes the
|
|
305
|
+
# PSD convex-combination recursion explicit.
|
|
235
306
|
self._var_t: np.ndarray | None = None
|
|
236
307
|
self._var_init: np.ndarray | None = None
|
|
237
308
|
self._var_anchor: np.ndarray | None = None
|
|
238
|
-
self.
|
|
309
|
+
self._vol_t: np.ndarray | None = None
|
|
310
|
+
# No single-scale state is allocated in ladder mode.
|
|
311
|
+
self._Q_t = (np.eye(n_assets, dtype=np.float64)
|
|
312
|
+
if self._corr_lam is None else None)
|
|
239
313
|
self._S_t = float(epsilon)
|
|
240
314
|
self._cov: np.ndarray | None = None
|
|
241
315
|
self._corr: np.ndarray | None = None
|
|
242
316
|
self._last_weight: float = 0.0
|
|
317
|
+
# Extraction (normalise + shrink + vol application) is deferred until
|
|
318
|
+
# get_cov()/get_corr(); _dirty marks state newer than _cov/_corr.
|
|
319
|
+
self._dirty = False
|
|
243
320
|
|
|
244
321
|
# Cached scratch buffers reused per ``update()`` to avoid per-step
|
|
245
322
|
# allocator churn. These are intentionally module-private and
|
|
246
323
|
# never escape the estimator.
|
|
247
324
|
self._scratch_outer = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
248
325
|
self._scratch_corr = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
326
|
+
self._scratch_had = np.empty((n_assets, n_assets), dtype=np.float64)
|
|
249
327
|
self._n_off = float(n_assets * (n_assets - 1)) if n_assets > 1 else 1.0
|
|
250
328
|
|
|
251
329
|
# ── Public API ────────────────────────────────────────────────────────
|
|
@@ -270,6 +348,63 @@ class SqueezeKernelEstimator:
|
|
|
270
348
|
raise ValueError(f"Expected shape ({n},), got {r_t.shape}.")
|
|
271
349
|
|
|
272
350
|
finite = np.isfinite(r_t)
|
|
351
|
+
self._obs_count[finite] += 1
|
|
352
|
+
|
|
353
|
+
# ── Adaptive-weight detector: score r_t under yesterday's per-rung
|
|
354
|
+
# forecasts, then advance the CUSUM (weights used below therefore
|
|
355
|
+
# reflect information through r_t only — causal). ──
|
|
356
|
+
if (self._adaptive and self._aw_prev_sig is not None
|
|
357
|
+
and finite.any()):
|
|
358
|
+
oidx = np.flatnonzero(finite)
|
|
359
|
+
r_o = r_t[oidx]
|
|
360
|
+
ell = np.empty(len(self._aw_prev_sig))
|
|
361
|
+
ok = True
|
|
362
|
+
for k, sig in enumerate(self._aw_prev_sig):
|
|
363
|
+
sub = sig[np.ix_(oidx, oidx)]
|
|
364
|
+
sub = (sub + sub.T) * 0.5
|
|
365
|
+
if _cho_factor is not None:
|
|
366
|
+
# One Cholesky per rung: logdet from the factor's
|
|
367
|
+
# diagonal, quadratic form via triangular solves.
|
|
368
|
+
try:
|
|
369
|
+
cf = _cho_factor(sub, lower=True, check_finite=False)
|
|
370
|
+
except np.linalg.LinAlgError:
|
|
371
|
+
ok = False # degenerate day (e.g. a fresh
|
|
372
|
+
break # listing): no clean score
|
|
373
|
+
logdet = 2.0 * np.log(np.diagonal(cf[0])).sum()
|
|
374
|
+
quad = float(r_o @ _cho_solve(cf, r_o, check_finite=False))
|
|
375
|
+
else:
|
|
376
|
+
sign, logdet = np.linalg.slogdet(sub)
|
|
377
|
+
if sign <= 0:
|
|
378
|
+
ok = False # degenerate day (e.g. a fresh
|
|
379
|
+
break # listing): no clean score
|
|
380
|
+
try:
|
|
381
|
+
quad = float(r_o @ np.linalg.solve(sub, r_o))
|
|
382
|
+
except np.linalg.LinAlgError:
|
|
383
|
+
ok = False
|
|
384
|
+
break
|
|
385
|
+
ell[k] = -0.5 * (oidx.size * np.log(2 * np.pi) + logdet + quad)
|
|
386
|
+
if not ok:
|
|
387
|
+
self._aw_tilt *= self._aw_lam_tilt
|
|
388
|
+
ell = None
|
|
389
|
+
else:
|
|
390
|
+
ell = None
|
|
391
|
+
if ell is not None:
|
|
392
|
+
dd = ell - ell.mean()
|
|
393
|
+
rms = float(np.sqrt((dd @ dd) / ell.size))
|
|
394
|
+
self._aw_scale = (self._aw_gamma_scale * self._aw_scale
|
|
395
|
+
+ (1.0 - self._aw_gamma_scale) * rms)
|
|
396
|
+
zc = np.clip(dd / (self._aw_scale + 1e-12), -3.0, 3.0)
|
|
397
|
+
k_fast = int(np.argmin(self.corr_half_lives))
|
|
398
|
+
k_slow = int(np.argmax(self.corr_half_lives))
|
|
399
|
+
zfs = float(zc[k_fast] - zc[k_slow])
|
|
400
|
+
self._aw_gp = max(0.0, self._aw_gp + zfs - self._aw_drift)
|
|
401
|
+
self._aw_gm = max(0.0, self._aw_gm - zfs - self._aw_drift)
|
|
402
|
+
if self._aw_gp > self._aw_b:
|
|
403
|
+
self._aw_tilt, self._aw_gp = self._aw_snap, 0.0
|
|
404
|
+
elif self._aw_gm > self._aw_b:
|
|
405
|
+
self._aw_tilt, self._aw_gm = -self._aw_snap, 0.0
|
|
406
|
+
else:
|
|
407
|
+
self._aw_tilt *= self._aw_lam_tilt
|
|
273
408
|
|
|
274
409
|
# ── Volatility update ──
|
|
275
410
|
if self._var_t is None:
|
|
@@ -317,10 +452,13 @@ class SqueezeKernelEstimator:
|
|
|
317
452
|
if n_obs > 0:
|
|
318
453
|
z_t[finite] = r_t[finite] / (vol_t[finite] + eps)
|
|
319
454
|
d2 = float(z_t[finite] @ z_t[finite]) / n_obs
|
|
320
|
-
if self.weight_statistic == "mahalanobis" and self.
|
|
455
|
+
if self.weight_statistic == "mahalanobis" and self._vol_t is not None:
|
|
321
456
|
# Score-exact surprise against the estimator's own previous
|
|
322
457
|
# correlation; falls back to the marginal d² on the first
|
|
323
|
-
# step or a (rare) singular observed submatrix.
|
|
458
|
+
# step or a (rare) singular observed submatrix. Extraction is
|
|
459
|
+
# lazy, so bring _corr up to the t-1 state first.
|
|
460
|
+
if self._dirty:
|
|
461
|
+
self._materialize()
|
|
324
462
|
try:
|
|
325
463
|
c_sub = self._corr[np.ix_(finite, finite)]
|
|
326
464
|
d2 = float(z_t[finite] @ np.linalg.solve(c_sub, z_t[finite])) / n_obs
|
|
@@ -334,58 +472,63 @@ class SqueezeKernelEstimator:
|
|
|
334
472
|
if self.impute_missing and 0 < n_obs < n:
|
|
335
473
|
self._impute(z_t, finite)
|
|
336
474
|
|
|
475
|
+
add = w_t > 0.0 and n_obs > 0
|
|
476
|
+
if add:
|
|
477
|
+
# np.multiply.outer with out= avoids the temporary that
|
|
478
|
+
# np.outer otherwise allocates each step. zz' is computed once
|
|
479
|
+
# and shared by every rung.
|
|
480
|
+
np.multiply.outer(z_t, z_t, out=self._scratch_outer)
|
|
337
481
|
if self._corr_lam is None:
|
|
338
|
-
# ── Single-scale correlation EWMA (published path
|
|
482
|
+
# ── Single-scale correlation EWMA (published path) ──
|
|
339
483
|
lam_c = self.lambda_corr
|
|
340
484
|
if self.lambda_corr_fast is not None:
|
|
341
485
|
# Score-driven memory: stress days (w_t → 1) shorten the memory
|
|
342
486
|
# toward lambda_corr_fast; calm days keep the slow decay.
|
|
343
487
|
lam_c = self.lambda_corr + (self.lambda_corr_fast - self.lambda_corr) * w_t
|
|
344
488
|
self._S_t = lam_c * self._S_t + w_t
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
#
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
self.
|
|
351
|
-
|
|
489
|
+
if add:
|
|
490
|
+
# Q <- (1 - eta) Q + eta zz'. With w_t = 0 both S and M decay
|
|
491
|
+
# by lam_c, so Q is unchanged — no matrix work at all.
|
|
492
|
+
eta = w_t / self._S_t
|
|
493
|
+
self._Q_t *= 1.0 - eta
|
|
494
|
+
self._scratch_outer *= eta
|
|
495
|
+
self._Q_t += self._scratch_outer
|
|
352
496
|
else:
|
|
353
497
|
# ── Scale-free ladder (Mode A) ──
|
|
354
|
-
# Update K correlation
|
|
355
|
-
# ladder; normalise
|
|
356
|
-
#
|
|
357
|
-
add = w_t > 0.0 and n_obs > 0
|
|
358
|
-
if add:
|
|
359
|
-
np.multiply.outer(z_t, z_t, out=self._scratch_outer)
|
|
360
|
-
cov = np.zeros((n, n), dtype=np.float64)
|
|
498
|
+
# Update K normalised correlation states on the geometric
|
|
499
|
+
# half-life ladder; extraction (normalise + shrink + blend)
|
|
500
|
+
# happens lazily in _materialize().
|
|
361
501
|
s_eff = 0.0
|
|
362
502
|
for k in range(self._corr_lam.size):
|
|
363
503
|
self._S_list[k] = self._corr_lam[k] * self._S_list[k] + w_t
|
|
364
|
-
self._M_list[k] *= self._corr_lam[k]
|
|
365
504
|
if add:
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
505
|
+
eta = w_t / self._S_list[k]
|
|
506
|
+
self._Q_list[k] *= 1.0 - eta
|
|
507
|
+
np.multiply(self._scratch_outer, eta, out=self._scratch_had)
|
|
508
|
+
self._Q_list[k] += self._scratch_had
|
|
369
509
|
s_eff += self._corr_w[k] * self._S_list[k]
|
|
370
|
-
cov = 0.5 * (cov + cov.T)
|
|
371
|
-
self._cov = cov
|
|
372
510
|
self._S_t = s_eff # blended effective size (for the property)
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
511
|
+
self._vol_t = vol_t
|
|
512
|
+
self._dirty = True
|
|
513
|
+
if self._adaptive:
|
|
514
|
+
self._materialize_adaptive()
|
|
376
515
|
self._last_weight = w_t
|
|
377
516
|
return w_t
|
|
378
517
|
|
|
379
518
|
def get_cov(self) -> np.ndarray:
|
|
380
519
|
"""Return the current covariance matrix estimate (n x n)."""
|
|
381
|
-
if self.
|
|
520
|
+
if self._vol_t is None:
|
|
382
521
|
raise RuntimeError("Call update() at least once before get_cov().")
|
|
522
|
+
if self._dirty:
|
|
523
|
+
self._materialize()
|
|
383
524
|
return self._cov.copy()
|
|
384
525
|
|
|
385
526
|
def get_corr(self) -> np.ndarray:
|
|
386
527
|
"""Return the current correlation matrix estimate (n x n)."""
|
|
387
|
-
if self.
|
|
528
|
+
if self._vol_t is None:
|
|
388
529
|
raise RuntimeError("Call update() at least once before get_corr().")
|
|
530
|
+
if self._dirty:
|
|
531
|
+
self._materialize()
|
|
389
532
|
return self._corr.copy()
|
|
390
533
|
|
|
391
534
|
@property
|
|
@@ -393,6 +536,17 @@ class SqueezeKernelEstimator:
|
|
|
393
536
|
"""Kernel weight assigned to the most recent observation."""
|
|
394
537
|
return self._last_weight
|
|
395
538
|
|
|
539
|
+
@property
|
|
540
|
+
def usable_mask(self) -> np.ndarray:
|
|
541
|
+
"""Boolean mask of assets with at least ``min_obs`` observations.
|
|
542
|
+
|
|
543
|
+
All-True when ``min_obs`` is None. Purely diagnostic — estimates
|
|
544
|
+
are not affected; subset the outputs with it (see class docstring).
|
|
545
|
+
"""
|
|
546
|
+
if self.min_obs is None:
|
|
547
|
+
return np.ones(self.n_assets, dtype=bool)
|
|
548
|
+
return self._obs_count >= self.min_obs
|
|
549
|
+
|
|
396
550
|
@property
|
|
397
551
|
def effective_sample_size(self) -> float:
|
|
398
552
|
"""Kernel-weighted effective sample size S_t."""
|
|
@@ -432,9 +586,13 @@ class SqueezeKernelEstimator:
|
|
|
432
586
|
# ── Private helpers ───────────────────────────────────────────────────
|
|
433
587
|
|
|
434
588
|
def _impute(self, z_t: np.ndarray, finite: np.ndarray) -> None:
|
|
589
|
+
if self._Q_t is None:
|
|
590
|
+
# Ladder mode: imputation reads the single-scale state, which
|
|
591
|
+
# was never updated on this path — historically a silent no-op
|
|
592
|
+
# (all correlations below threshold); keep it an explicit one.
|
|
593
|
+
return
|
|
435
594
|
eps = self.epsilon
|
|
436
|
-
|
|
437
|
-
sigma_z = self._M_t / denom
|
|
595
|
+
sigma_z = self._Q_t
|
|
438
596
|
diag_z = np.diag(sigma_z)
|
|
439
597
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
440
598
|
missing = ~finite
|
|
@@ -453,24 +611,22 @@ class SqueezeKernelEstimator:
|
|
|
453
611
|
if den > 0.0:
|
|
454
612
|
z_t[i] = num / den
|
|
455
613
|
|
|
456
|
-
def
|
|
614
|
+
def _shrunk_corr_from_Q(self, Q: np.ndarray, S_t: float) -> np.ndarray:
|
|
615
|
+
"""Normalise one Q state to a correlation and shrink it in place.
|
|
616
|
+
|
|
617
|
+
Returns ``self._scratch_corr`` — valid only until the next call.
|
|
618
|
+
"""
|
|
457
619
|
eps = self.epsilon
|
|
458
|
-
M_t = self._M_t if M_t is None else M_t
|
|
459
|
-
S_t = max(self._S_t if S_in is None else S_in, eps)
|
|
460
620
|
n = self.n_assets
|
|
621
|
+
S_t = max(S_t, eps)
|
|
461
622
|
|
|
462
|
-
#
|
|
463
|
-
#
|
|
464
|
-
|
|
465
|
-
diag_z = np.diagonal(M_t).copy()
|
|
466
|
-
diag_z /= S_t # in-place
|
|
623
|
+
# corr_ij = Q_ij * inv_diag_i * inv_diag_j; the scalar S_t cancels
|
|
624
|
+
# in the normalisation, so Q needs no rescaling pass.
|
|
625
|
+
diag_z = np.diagonal(Q).copy()
|
|
467
626
|
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
468
|
-
# corr_ij = (M_ij / S_t) * inv_diag_i * inv_diag_j; this writes
|
|
469
|
-
# the rescaled outer-product into _scratch_corr in one pass.
|
|
470
627
|
np.multiply.outer(inv_diag, inv_diag, out=self._scratch_corr)
|
|
471
628
|
corr = self._scratch_corr
|
|
472
|
-
corr *=
|
|
473
|
-
corr *= (1.0 / S_t) # absorb the M_t / S_t scale
|
|
629
|
+
corr *= Q # in-place
|
|
474
630
|
np.fill_diagonal(corr, np.where(diag_z > eps, 1.0, 0.0))
|
|
475
631
|
|
|
476
632
|
# Adaptive shrinkage: blend toward the equicorrelation target
|
|
@@ -484,6 +640,9 @@ class SqueezeKernelEstimator:
|
|
|
484
640
|
# Off-diagonal mean: O(n^2) sum, no mask allocation.
|
|
485
641
|
rho_bar = (corr.sum() - corr.trace()) / self._n_off
|
|
486
642
|
if self.shrinkage_target == "equicorrelation" or rho_bar <= 0.0:
|
|
643
|
+
# rho_bar <= 0 also covers the q = meanoff(C o C) = 0 corner:
|
|
644
|
+
# C o C has nonnegative entries, so q = 0 forces C = I and
|
|
645
|
+
# hence rho_bar = 0 — the equicorrelation fallback applies.
|
|
487
646
|
corr *= (1.0 - alpha)
|
|
488
647
|
corr += alpha * rho_bar
|
|
489
648
|
np.fill_diagonal(corr, 1.0)
|
|
@@ -495,27 +654,82 @@ class SqueezeKernelEstimator:
|
|
|
495
654
|
# level-matched so the target carries the same average
|
|
496
655
|
# correlation mass as the equicorrelation target. As
|
|
497
656
|
# alpha -> 0 this reduces exactly to the published estimator.
|
|
498
|
-
had =
|
|
657
|
+
had = self._scratch_had
|
|
658
|
+
np.multiply(corr, corr, out=had) # Hadamard square, O(n^2)
|
|
499
659
|
mean_off = (had.sum() - np.trace(had)) / self._n_off
|
|
500
660
|
gamma = min(1.0, rho_bar / max(mean_off, eps))
|
|
501
661
|
corr *= (1.0 - alpha)
|
|
502
662
|
corr += (alpha * (1.0 - alpha)) * rho_bar
|
|
503
|
-
|
|
663
|
+
had *= alpha * alpha * gamma
|
|
664
|
+
corr += had
|
|
504
665
|
np.fill_diagonal(corr, 1.0)
|
|
666
|
+
return corr
|
|
505
667
|
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
668
|
+
def _materialize_adaptive(self) -> None:
|
|
669
|
+
"""Adaptive-weight extraction: build per-rung shrunk covariances
|
|
670
|
+
(kept for the next update's detector scores), blend with the
|
|
671
|
+
CUSUM-tilted weights, derive _cov/_corr. Runs eagerly."""
|
|
672
|
+
eps = self.epsilon
|
|
673
|
+
vol_t = self._vol_t
|
|
674
|
+
vv = np.multiply.outer(vol_t, vol_t)
|
|
675
|
+
sig_k = []
|
|
676
|
+
for k in range(self._corr_lam.size):
|
|
677
|
+
corr_k = self._shrunk_corr_from_Q(self._Q_list[k], self._S_list[k])
|
|
678
|
+
sig_k.append(corr_k * vv)
|
|
679
|
+
self._aw_prev_sig = sig_k
|
|
680
|
+
t = self._aw_tilt
|
|
681
|
+
pi = self._corr_w
|
|
682
|
+
if t >= 0:
|
|
683
|
+
w = (1.0 - t) * pi + t * self._aw_pi_fast
|
|
684
|
+
else:
|
|
685
|
+
w = (1.0 + t) * pi + (-t) * self._aw_pi_slow
|
|
686
|
+
cov = w[0] * sig_k[0]
|
|
687
|
+
for k in range(1, len(sig_k)):
|
|
688
|
+
cov = cov + w[k] * sig_k[k]
|
|
689
|
+
cov = (cov + cov.T) * 0.5
|
|
690
|
+
self._cov = cov
|
|
691
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
692
|
+
self._corr = cov / np.outer(d, d)
|
|
693
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
694
|
+
self._dirty = False
|
|
695
|
+
|
|
696
|
+
def _materialize(self) -> None:
|
|
697
|
+
"""Extract _cov/_corr from the current state (lazy, on demand)."""
|
|
698
|
+
eps = self.epsilon
|
|
699
|
+
vol_t = self._vol_t
|
|
700
|
+
if self._corr_lam is None:
|
|
701
|
+
# ── Single scale: shrunk correlation IS the correlation output ──
|
|
702
|
+
corr = self._shrunk_corr_from_Q(self._Q_t, self._S_t)
|
|
703
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
704
|
+
cov = corr * self._scratch_outer
|
|
705
|
+
cov += cov.T
|
|
706
|
+
cov *= 0.5
|
|
707
|
+
corr_out = corr.copy()
|
|
708
|
+
corr_out += corr_out.T
|
|
709
|
+
corr_out *= 0.5
|
|
710
|
+
self._cov, self._corr = cov, corr_out
|
|
711
|
+
else:
|
|
712
|
+
# ── Ladder: blend per-rung shrunk correlations, then apply the
|
|
713
|
+
# (shared) volatilities once — algebraically identical to
|
|
714
|
+
# blending per-rung covariances, K-1 fewer O(n^2) passes.
|
|
715
|
+
mix = np.zeros((self.n_assets, self.n_assets), dtype=np.float64)
|
|
716
|
+
for k in range(self._corr_lam.size):
|
|
717
|
+
corr_k = self._shrunk_corr_from_Q(self._Q_list[k], self._S_list[k])
|
|
718
|
+
corr_k *= self._corr_w[k]
|
|
719
|
+
mix += corr_k
|
|
720
|
+
cov = mix
|
|
721
|
+
np.multiply.outer(vol_t, vol_t, out=self._scratch_outer)
|
|
722
|
+
cov *= self._scratch_outer
|
|
723
|
+
cov += cov.T
|
|
724
|
+
cov *= 0.5
|
|
725
|
+
self._cov = cov
|
|
726
|
+
# Correlation is re-derived from the blended covariance (not the
|
|
727
|
+
# blended correlation mix) to keep the published dead-asset
|
|
728
|
+
# semantics: rows of never-observed assets renormalise to zero.
|
|
729
|
+
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
730
|
+
self._corr = cov / np.outer(d, d)
|
|
731
|
+
np.fill_diagonal(self._corr, 1.0)
|
|
732
|
+
self._dirty = False
|
|
519
733
|
|
|
520
734
|
|
|
521
735
|
# ── Kernel resolution ─────────────────────────────────────────────────────────
|
|
File without changes
|
|
File without changes
|