squeeze-kernel 2.0.0__tar.gz → 3.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/PKG-INFO +13 -12
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/README.md +12 -11
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/pyproject.toml +1 -1
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/pyproject.toml.orig +1 -1
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/src/squeeze_kernel/__init__.py +1 -1
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/src/squeeze_kernel/core.py +21 -5
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/src/squeeze_kernel/estimator.py +372 -48
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/src/squeeze_kernel/batch.py +0 -0
- {squeeze_kernel-2.0.0 → squeeze_kernel-3.1.0}/src/squeeze_kernel/kernels.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: squeeze-kernel
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.1.0
|
|
4
4
|
Summary: Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage
|
|
5
5
|
Keywords: covariance,correlation,ewma,kernel,risk,streaming
|
|
6
6
|
Author: Robert Kende
|
|
@@ -36,7 +36,7 @@ Description-Content-Type: text/markdown
|
|
|
36
36
|
[](https://pypi.org/project/squeeze-kernel/)
|
|
37
37
|
[](LICENSE)
|
|
38
38
|
|
|
39
|
-
A **streaming covariance estimator for panels of financial returns** whose entire public surface is **one number** — the decay `lam` of the anchor correlation timescale. Every other quantity is derived from it, fixed by a structural argument, or computed online from the estimator's own state.
|
|
39
|
+
A **streaming covariance estimator for panels of financial returns** whose entire public surface is **one number** — the decay `lam` of the anchor correlation timescale. Every other quantity is derived from it, fixed by a structural argument, or computed online from the estimator's own state. An `O(Kn²)` state update per day, positive semi-definite **by construction**, missing values handled **natively**, no tuning, no refits. Only dependency: NumPy.
|
|
40
40
|
|
|
41
41
|
```python
|
|
42
42
|
from squeeze_kernel import SqueezeKernel
|
|
@@ -53,12 +53,12 @@ Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking wi
|
|
|
53
53
|
|
|
54
54
|
Markets do not keep calendar time. Following Mandelbrot, the estimator treats a panel as a collection of partially coupled markets, **each advancing on its own activity-driven clock** — and reads those clocks from the panel's own correlation structure, so a hot cluster (say precious metals and FX) advances its correlation state while an idle one (agriculture) does not, without anyone identifying a cluster. On those clocks it runs a single recursion that:
|
|
55
55
|
|
|
56
|
-
- **is PSD at every step, structurally** — the correlation state evolves by a diagonal-congruence flow (a congruence plus a rank-one term); no eigenvalue clipping, no nearest-PSD repair, no
|
|
57
|
-
- **learns in market time and forgets in calendar time** — observations enter with a saturating, self-studentising weight (no day counts more than one unit of trading time); memory decays at fixed per-day rates on a geometric ladder of three timescales `(lam⁴, lam, lam^¼)
|
|
58
|
-
- **regularises itself** — each timescale's shrinkage intensity is computed from two online statistics, the concentration `n/ν` (dimension per unit trading time) and the de-noised fraction of correlation dispersion the target explains; the target is the Hadamard square of the running correlation (cluster-respecting, PSD by the Schur product theorem);
|
|
59
|
-
- **adapts its memory to regime breaks** —
|
|
56
|
+
- **is PSD at every step, structurally** — the correlation state evolves by a diagonal-congruence flow (a congruence plus a rank-one term); no eigenvalue clipping, no nearest-PSD repair, and no factorisation anywhere in the state update. (The adaptive timescale weights are the one exception: they read each timescale's predictive likelihood, which costs one Cholesky per timescale per day. Turn them off and the estimator is pure `O(Kn²)`.)
|
|
57
|
+
- **learns in market time and forgets in calendar time** — observations enter with a saturating, self-studentising weight (no day counts more than one unit of trading time); memory decays at fixed per-day rates on a geometric ladder of three timescales `(lam⁴, lam, lam^¼)`. Pairs accrue covariance at the geometric mean of their two clock increments, which is the most positive semi-definiteness allows and exactly the Cauchy–Schwarz bound on how far two assets' clocks can overlap;
|
|
58
|
+
- **regularises itself** — each timescale's shrinkage intensity is computed from two online statistics, the concentration `n/ν` (dimension per unit trading time) and the de-noised fraction of correlation dispersion the target explains; the target is the Hadamard square of the running correlation, which is exactly the correlation matrix of the *squared* returns (cluster-respecting, PSD by the Schur product theorem, and sign-blind by construction — the signs are carried by the unshrunk term);
|
|
59
|
+
- **adapts its memory to regime breaks, in both stages** — the timescale mix moves by an exponentiated-gradient step on the *blend's* own log score (not on any single timescale's, which would select rather than blend), at a temperature calibrated so that uninformative evidence leaves the mix within a factor e of its prior; the marginal variance is tracked on its own three-rung ladder, two octaves below the correlation ladder, and pooled panel-wide by the same rule on saturated evidence, so one spike day cannot hand a stale rung weeks of weight. Nothing in either mixture is fitted: the ladders are derived from `lam`, the evidence memory is the fastest rung's, and the temperature is the null's;
|
|
60
60
|
- **ingests missing values natively** — listings, delistings, halts enter as `NaN`;
|
|
61
|
-
- **is fast** — a thirty-year daily pass at n=300
|
|
61
|
+
- **is fast** — one Cholesky and one triangular solve per day; a thirty-year daily pass at n=300 runs in about two minutes single-threaded, well under daily rolling-window refits.
|
|
62
62
|
|
|
63
63
|
**Evidence.** On thirty years of S&P 500 constituents against an eleven-method field (EWMA, DCC, Ledoit–Wolf, OAS, nonlinear shrinkage, RMT filtering, Gerber, IEWMA, CM-IEWMA, and the published v1 estimator) it leads at every universe size from 50 to 300 and is the **sole member of the 90% model confidence set at every size**. Carried **zero-shot** to a diversified panel of 121 futures across eight asset classes it beats the same field *calibrated on that panel's own history* — matched-backbone IEWMA by 6.9 NLL/day (p = 4·10⁻⁴), calibrated DCC by 17.9 — out-of-time.
|
|
64
64
|
|
|
@@ -102,7 +102,7 @@ sk = SqueezeKernel() # lam=0.996 (anchor half-life ~173 days)
|
|
|
102
102
|
for r_t in returns:
|
|
103
103
|
w = sk.update(r_t) # returns the day's kernel weight
|
|
104
104
|
cov, corr = sk.covariance(), sk.correlation()
|
|
105
|
-
sk.state() # kernel scale, per-timescale effective sizes,
|
|
105
|
+
sk.state() # kernel scale, per-timescale effective sizes, mixture tilt
|
|
106
106
|
|
|
107
107
|
# batch mode: full panel in, covariance path out
|
|
108
108
|
cov_path, corr_path, weights = estimate_squeeze_cov(returns, with_weights=True)
|
|
@@ -117,15 +117,16 @@ Missing values: pass `NaN` (or `mask=` on `update`). Newly listed, delisted or h
|
|
|
117
117
|
| timescale ladder | decays `(lam⁴, lam, lam^¼)` — half-lives `(h/4, h, 4h)`, `h = -1/log2(lam)` |
|
|
118
118
|
| kernel scale | state: `κ_t = ⅓ · EWMA(activity)` at the anchor rate |
|
|
119
119
|
| shrinkage intensity | per timescale, `α = min(1,c) · g̃²/(g̃² + (1−g̃)²·max(0, 1/c − 1))` from the online concentration `c = n/ν` and target-fit `g̃` |
|
|
120
|
-
| timescale weights | prior ∝ √h,
|
|
121
|
-
|
|
|
122
|
-
|
|
|
120
|
+
| timescale weights | prior ∝ √h, moved by exponentiated gradient on the blend log score at the null temperature, fast-rung memory |
|
|
121
|
+
| volatility memory | ladder `(h/16, h/4, h)`, pooled panel-wide by the same rule on tanh-saturated evidence, uniform prior |
|
|
122
|
+
| structural constants | K=3, b=4, θ=½, κ-scale ⅓, Schur power 2 — each bracketed by ablation in the paper |
|
|
123
|
+
| fitted constants | none (the 2.x volatility clock `λ_v = 0.98` is replaced by the ladder above) |
|
|
123
124
|
|
|
124
125
|
`from squeeze_kernel import CONSTANTS` exposes the structural constants for research. The published v1 estimator (all its knobs) remains available as `SqueezeKernelEstimator` / `SqueezeKernel.v1(...)`; every 2.0 mechanism is also an estimator-level switch for ablation. See [MIGRATION.md](MIGRATION.md).
|
|
125
126
|
|
|
126
127
|
## How it works
|
|
127
128
|
|
|
128
|
-
One daily update: variance
|
|
129
|
+
One daily update: variance on a three-rung ladder per asset, pooled by panel-wide self-adapting weights → standardised surprise → per-asset clock increments from the Schur-square-weighted neighbourhood mean of squared surprises → diagonal-congruence update of each timescale's correlation state on those clocks → per-timescale shrinkage as a learned pool of the raw correlation, its equicorrelation level and its Hadamard square, with the self-tuning intensity rule as the prior and the blend gradient as the correction → blend across timescales, moved by the gradient of the blend's own log score → covariance. The paper gives the derivations, guarantees (PSD, conditioning floor, exact reductions to the published special cases), and the full evaluation.
|
|
129
130
|
|
|
130
131
|
## Development
|
|
131
132
|
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
[](https://pypi.org/project/squeeze-kernel/)
|
|
6
6
|
[](LICENSE)
|
|
7
7
|
|
|
8
|
-
A **streaming covariance estimator for panels of financial returns** whose entire public surface is **one number** — the decay `lam` of the anchor correlation timescale. Every other quantity is derived from it, fixed by a structural argument, or computed online from the estimator's own state.
|
|
8
|
+
A **streaming covariance estimator for panels of financial returns** whose entire public surface is **one number** — the decay `lam` of the anchor correlation timescale. Every other quantity is derived from it, fixed by a structural argument, or computed online from the estimator's own state. An `O(Kn²)` state update per day, positive semi-definite **by construction**, missing values handled **natively**, no tuning, no refits. Only dependency: NumPy.
|
|
9
9
|
|
|
10
10
|
```python
|
|
11
11
|
from squeeze_kernel import SqueezeKernel
|
|
@@ -22,12 +22,12 @@ Reference: *"The Squeeze Kernel Covariance Estimator: Dual-Timescale Tracking wi
|
|
|
22
22
|
|
|
23
23
|
Markets do not keep calendar time. Following Mandelbrot, the estimator treats a panel as a collection of partially coupled markets, **each advancing on its own activity-driven clock** — and reads those clocks from the panel's own correlation structure, so a hot cluster (say precious metals and FX) advances its correlation state while an idle one (agriculture) does not, without anyone identifying a cluster. On those clocks it runs a single recursion that:
|
|
24
24
|
|
|
25
|
-
- **is PSD at every step, structurally** — the correlation state evolves by a diagonal-congruence flow (a congruence plus a rank-one term); no eigenvalue clipping, no nearest-PSD repair, no
|
|
26
|
-
- **learns in market time and forgets in calendar time** — observations enter with a saturating, self-studentising weight (no day counts more than one unit of trading time); memory decays at fixed per-day rates on a geometric ladder of three timescales `(lam⁴, lam, lam^¼)
|
|
27
|
-
- **regularises itself** — each timescale's shrinkage intensity is computed from two online statistics, the concentration `n/ν` (dimension per unit trading time) and the de-noised fraction of correlation dispersion the target explains; the target is the Hadamard square of the running correlation (cluster-respecting, PSD by the Schur product theorem);
|
|
28
|
-
- **adapts its memory to regime breaks** —
|
|
25
|
+
- **is PSD at every step, structurally** — the correlation state evolves by a diagonal-congruence flow (a congruence plus a rank-one term); no eigenvalue clipping, no nearest-PSD repair, and no factorisation anywhere in the state update. (The adaptive timescale weights are the one exception: they read each timescale's predictive likelihood, which costs one Cholesky per timescale per day. Turn them off and the estimator is pure `O(Kn²)`.)
|
|
26
|
+
- **learns in market time and forgets in calendar time** — observations enter with a saturating, self-studentising weight (no day counts more than one unit of trading time); memory decays at fixed per-day rates on a geometric ladder of three timescales `(lam⁴, lam, lam^¼)`. Pairs accrue covariance at the geometric mean of their two clock increments, which is the most positive semi-definiteness allows and exactly the Cauchy–Schwarz bound on how far two assets' clocks can overlap;
|
|
27
|
+
- **regularises itself** — each timescale's shrinkage intensity is computed from two online statistics, the concentration `n/ν` (dimension per unit trading time) and the de-noised fraction of correlation dispersion the target explains; the target is the Hadamard square of the running correlation, which is exactly the correlation matrix of the *squared* returns (cluster-respecting, PSD by the Schur product theorem, and sign-blind by construction — the signs are carried by the unshrunk term);
|
|
28
|
+
- **adapts its memory to regime breaks, in both stages** — the timescale mix moves by an exponentiated-gradient step on the *blend's* own log score (not on any single timescale's, which would select rather than blend), at a temperature calibrated so that uninformative evidence leaves the mix within a factor e of its prior; the marginal variance is tracked on its own three-rung ladder, two octaves below the correlation ladder, and pooled panel-wide by the same rule on saturated evidence, so one spike day cannot hand a stale rung weeks of weight. Nothing in either mixture is fitted: the ladders are derived from `lam`, the evidence memory is the fastest rung's, and the temperature is the null's;
|
|
29
29
|
- **ingests missing values natively** — listings, delistings, halts enter as `NaN`;
|
|
30
|
-
- **is fast** — a thirty-year daily pass at n=300
|
|
30
|
+
- **is fast** — one Cholesky and one triangular solve per day; a thirty-year daily pass at n=300 runs in about two minutes single-threaded, well under daily rolling-window refits.
|
|
31
31
|
|
|
32
32
|
**Evidence.** On thirty years of S&P 500 constituents against an eleven-method field (EWMA, DCC, Ledoit–Wolf, OAS, nonlinear shrinkage, RMT filtering, Gerber, IEWMA, CM-IEWMA, and the published v1 estimator) it leads at every universe size from 50 to 300 and is the **sole member of the 90% model confidence set at every size**. Carried **zero-shot** to a diversified panel of 121 futures across eight asset classes it beats the same field *calibrated on that panel's own history* — matched-backbone IEWMA by 6.9 NLL/day (p = 4·10⁻⁴), calibrated DCC by 17.9 — out-of-time.
|
|
33
33
|
|
|
@@ -71,7 +71,7 @@ sk = SqueezeKernel() # lam=0.996 (anchor half-life ~173 days)
|
|
|
71
71
|
for r_t in returns:
|
|
72
72
|
w = sk.update(r_t) # returns the day's kernel weight
|
|
73
73
|
cov, corr = sk.covariance(), sk.correlation()
|
|
74
|
-
sk.state() # kernel scale, per-timescale effective sizes,
|
|
74
|
+
sk.state() # kernel scale, per-timescale effective sizes, mixture tilt
|
|
75
75
|
|
|
76
76
|
# batch mode: full panel in, covariance path out
|
|
77
77
|
cov_path, corr_path, weights = estimate_squeeze_cov(returns, with_weights=True)
|
|
@@ -86,15 +86,16 @@ Missing values: pass `NaN` (or `mask=` on `update`). Newly listed, delisted or h
|
|
|
86
86
|
| timescale ladder | decays `(lam⁴, lam, lam^¼)` — half-lives `(h/4, h, 4h)`, `h = -1/log2(lam)` |
|
|
87
87
|
| kernel scale | state: `κ_t = ⅓ · EWMA(activity)` at the anchor rate |
|
|
88
88
|
| shrinkage intensity | per timescale, `α = min(1,c) · g̃²/(g̃² + (1−g̃)²·max(0, 1/c − 1))` from the online concentration `c = n/ν` and target-fit `g̃` |
|
|
89
|
-
| timescale weights | prior ∝ √h,
|
|
90
|
-
|
|
|
91
|
-
|
|
|
89
|
+
| timescale weights | prior ∝ √h, moved by exponentiated gradient on the blend log score at the null temperature, fast-rung memory |
|
|
90
|
+
| volatility memory | ladder `(h/16, h/4, h)`, pooled panel-wide by the same rule on tanh-saturated evidence, uniform prior |
|
|
91
|
+
| structural constants | K=3, b=4, θ=½, κ-scale ⅓, Schur power 2 — each bracketed by ablation in the paper |
|
|
92
|
+
| fitted constants | none (the 2.x volatility clock `λ_v = 0.98` is replaced by the ladder above) |
|
|
92
93
|
|
|
93
94
|
`from squeeze_kernel import CONSTANTS` exposes the structural constants for research. The published v1 estimator (all its knobs) remains available as `SqueezeKernelEstimator` / `SqueezeKernel.v1(...)`; every 2.0 mechanism is also an estimator-level switch for ablation. See [MIGRATION.md](MIGRATION.md).
|
|
94
95
|
|
|
95
96
|
## How it works
|
|
96
97
|
|
|
97
|
-
One daily update: variance
|
|
98
|
+
One daily update: variance on a three-rung ladder per asset, pooled by panel-wide self-adapting weights → standardised surprise → per-asset clock increments from the Schur-square-weighted neighbourhood mean of squared surprises → diagonal-congruence update of each timescale's correlation state on those clocks → per-timescale shrinkage as a learned pool of the raw correlation, its equicorrelation level and its Hadamard square, with the self-tuning intensity rule as the prior and the blend gradient as the correction → blend across timescales, moved by the gradient of the blend's own log score → covariance. The paper gives the derivations, guarantees (PSD, conditioning floor, exact reductions to the published special cases), and the full evaluation.
|
|
98
99
|
|
|
99
100
|
## Development
|
|
100
101
|
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "squeeze-kernel"
|
|
7
|
-
version = "
|
|
7
|
+
version = "3.1.0"
|
|
8
8
|
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "squeeze-kernel"
|
|
7
|
-
version = "
|
|
7
|
+
version = "3.1.0"
|
|
8
8
|
description = "Streaming, PSD-by-construction covariance estimator with Fisher-kernel weighting and adaptive shrinkage"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -11,9 +11,21 @@ Configuration (research record: squeeze_cov V2_STATUS.md, branch v2):
|
|
|
11
11
|
- correlation ladder at ``half_life * (43/173, 1, 693/173)`` — the
|
|
12
12
|
canonical rungs at the default half-life; rung weights ``pi ~ sqrt(h)``
|
|
13
13
|
(theta = 1/2, structural),
|
|
14
|
-
- volatility
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
- volatility on a ladder at ``half_life * (1/16, 1/4, 1)``, pooled
|
|
15
|
+
panel-wide by Bayes at the null-calibrated temperature on saturated
|
|
16
|
+
evidence with the fastest rung's memory and a uniform prior; this
|
|
17
|
+
replaced the 2.x fitted constant ``lambda_vol = 0.98`` (still accepted
|
|
18
|
+
by ``SqueezeKernelEstimator`` and ignored when ``vol_ladder=True``),
|
|
19
|
+
- timescale weights moved by exponentiated gradient on the blend's log
|
|
20
|
+
score at the null temperature (fast-rung memory) from the prior
|
|
21
|
+
``pi ~ sqrt(h)``; replaced the 2.x CUSUM detector,
|
|
22
|
+
- (3.1) each timescale's shrunk correlation is a learned convex
|
|
23
|
+
combination of three PSD unit-diagonal experts -- its raw
|
|
24
|
+
correlation, the equicorrelation matrix at its mean level, and its
|
|
25
|
+
Schur square -- whose prior is the intensity rule's morph
|
|
26
|
+
``(1-a, a(1-a), a^2)`` and whose weights the same blend gradient
|
|
27
|
+
corrects online (studentised within the timescale, same temperature
|
|
28
|
+
and memory). At the prior the estimator is the 3.0 one,
|
|
17
29
|
- kernel scale as state, not parameter: ``kappa_t = (1/3) EWMA_h(d^2)``
|
|
18
30
|
(chi-squared-null constant),
|
|
19
31
|
- self-tuning shrinkage intensity per rung from the online concentration
|
|
@@ -31,7 +43,7 @@ from dataclasses import dataclass
|
|
|
31
43
|
import numpy as np
|
|
32
44
|
from numpy.typing import ArrayLike
|
|
33
45
|
|
|
34
|
-
from .estimator import SqueezeKernelEstimator
|
|
46
|
+
from .estimator import SqueezeKernelEstimator, _BlendGradient
|
|
35
47
|
|
|
36
48
|
__all__ = ["SqueezeKernel", "StructuralConstants", "CONSTANTS"]
|
|
37
49
|
|
|
@@ -45,7 +57,7 @@ class StructuralConstants:
|
|
|
45
57
|
|
|
46
58
|
b: float = 4.0 # ladder spacing: rungs (h/b, h, h*b)
|
|
47
59
|
theta: float = 0.5 # rung weights ~ h^theta
|
|
48
|
-
lambda_vol: float = 0.98 #
|
|
60
|
+
lambda_vol: float = 0.98 # 2.x constant; unused by the 3.0 front-end (vol ladder)
|
|
49
61
|
kappa_c: float = 1.0 / 3.0 # kernel scale: kappa = kappa_c * EWMA(activity)
|
|
50
62
|
schur_p: int = 2 # Hadamard power of the cluster target
|
|
51
63
|
epsilon: float = 1e-8
|
|
@@ -106,6 +118,9 @@ class SqueezeKernel:
|
|
|
106
118
|
level_match=False,
|
|
107
119
|
detector=True,
|
|
108
120
|
clock="asset",
|
|
121
|
+
vol_ladder=True,
|
|
122
|
+
weights="eg_blend",
|
|
123
|
+
split_learn=True,
|
|
109
124
|
epsilon=c.epsilon,
|
|
110
125
|
)
|
|
111
126
|
|
|
@@ -161,6 +176,7 @@ class SqueezeKernel:
|
|
|
161
176
|
"rung_S": S,
|
|
162
177
|
"rung_nu": nu,
|
|
163
178
|
"detector_tilt": 0.0 if det is None else det.tilt,
|
|
179
|
+
"split": det.v.copy() if isinstance(det, _BlendGradient) and det.split else None,
|
|
164
180
|
}
|
|
165
181
|
|
|
166
182
|
# ── v1 escape hatch ──────────────────────────────────────────────────
|
|
@@ -128,6 +128,206 @@ class _SurpriseDetector:
|
|
|
128
128
|
return np.asarray((1.0 + t) * prior + (-t) * self.pi_slow)
|
|
129
129
|
|
|
130
130
|
|
|
131
|
+
class _BlendGradient:
|
|
132
|
+
"""Exponentiated gradient on the BLEND's log score, at the
|
|
133
|
+
null-calibrated temperature.
|
|
134
|
+
|
|
135
|
+
The emitted covariance is the linear pool Sigma(w) = sum_k w_k Sigma_k
|
|
136
|
+
of the rung covariances. Its Gaussian log score has gradient
|
|
137
|
+
|
|
138
|
+
d ell / d w_k = -1/2 tr(Sigma^-1 Sigma_k) + 1/2 u' Sigma_k u,
|
|
139
|
+
u = Sigma^-1 r,
|
|
140
|
+
|
|
141
|
+
so the update moves the blend rather than selecting a rung (Bayesian
|
|
142
|
+
model averaging over rung likelihoods selects, and collapses). The
|
|
143
|
+
gradient is studentised across rungs by its running rms and scaled by
|
|
144
|
+
sqrt(2 eps), eps the fixed-share rate: the accumulated log-odds then
|
|
145
|
+
have unit variance under uninformative evidence, so the mixture stays
|
|
146
|
+
within a factor e of its prior when there is nothing to learn and
|
|
147
|
+
moves linearly in a persistent advantage. Evidence memory: the
|
|
148
|
+
fastest rung (a regime can change as fast as the fastest rung can
|
|
149
|
+
follow). State: the weights, one scalar scale, and the cached rung
|
|
150
|
+
covariances of the previous step. Causal: ``score`` reads the
|
|
151
|
+
previous step's covariances, ``advance`` updates the weights used for
|
|
152
|
+
the next blend.
|
|
153
|
+
"""
|
|
154
|
+
|
|
155
|
+
_JITTER = 1e-12
|
|
156
|
+
|
|
157
|
+
def __init__(self, half_lives: np.ndarray, prior: np.ndarray,
|
|
158
|
+
split: bool = False, n_experts: int = 3) -> None:
|
|
159
|
+
hl = np.asarray(half_lives, dtype=np.float64)
|
|
160
|
+
self.prior = np.asarray(prior, dtype=np.float64)
|
|
161
|
+
self.w = self.prior.copy()
|
|
162
|
+
self.eps = float(1.0 - 2.0 ** (-1.0 / float(hl.min())))
|
|
163
|
+
self._gamma_scale = 2.0 ** (-1.0 / float(np.median(hl)))
|
|
164
|
+
self.scale = 1.0
|
|
165
|
+
self.prev_sig: list[np.ndarray] | None = None
|
|
166
|
+
self.prev_w: np.ndarray | None = None
|
|
167
|
+
# Learned shrinkage split (3.1): each timescale's correlation is a
|
|
168
|
+
# convex combination of three PSD unit-diagonal experts -- raw,
|
|
169
|
+
# equicorrelation at its mean level, Schur square -- whose prior
|
|
170
|
+
# is the intensity rule's morph (1-a, a(1-a), a^2) and whose
|
|
171
|
+
# weights are moved by the same gradient step as the timescale
|
|
172
|
+
# weights, studentised within the timescale. Starts uniform and
|
|
173
|
+
# forgets toward the prior at the fixed-share rate.
|
|
174
|
+
self.split = bool(split)
|
|
175
|
+
self.n_experts = int(n_experts)
|
|
176
|
+
self.v = np.full((hl.size, self.n_experts), 1.0 / self.n_experts)
|
|
177
|
+
self.scale_v = np.ones(hl.size)
|
|
178
|
+
self.prev_comps: list[tuple[np.ndarray, ...]] | None = None
|
|
179
|
+
self.prev_prior: np.ndarray | None = None
|
|
180
|
+
self.prev_cov: np.ndarray | None = None
|
|
181
|
+
self.prev_vol: np.ndarray | None = None
|
|
182
|
+
self.prev_v: np.ndarray | None = None
|
|
183
|
+
self._gc: np.ndarray | None = None
|
|
184
|
+
|
|
185
|
+
def score(self, r_t: np.ndarray, finite: np.ndarray) -> np.ndarray | None:
|
|
186
|
+
"""Gradient of the previous blend's log score at ``r_t`` (observed
|
|
187
|
+
subvector), one entry per rung; ``None`` on a degenerate day."""
|
|
188
|
+
if self.prev_sig is None or self.prev_w is None:
|
|
189
|
+
return None
|
|
190
|
+
idx = np.flatnonzero(finite)
|
|
191
|
+
if idx.size < 2:
|
|
192
|
+
return None
|
|
193
|
+
K = len(self.prev_sig)
|
|
194
|
+
full = idx.size == self.prev_sig[0].shape[0]
|
|
195
|
+
# Gate exactly as the research engine does: the day counts only if
|
|
196
|
+
# every rung's observed sub-block is positive definite (a newly
|
|
197
|
+
# listed asset enters with a zero row and fails this); otherwise
|
|
198
|
+
# the mixture keeps its weights for the day. With the split the
|
|
199
|
+
# rung matrices held here are correlations (vol > 0 on observed
|
|
200
|
+
# assets, so PD of the correlation block <=> PD of the covariance).
|
|
201
|
+
for k in range(K):
|
|
202
|
+
sub = self.prev_sig[k] if full else self.prev_sig[k][np.ix_(idx, idx)]
|
|
203
|
+
if not self.split:
|
|
204
|
+
sub = 0.5 * (sub + sub.T)
|
|
205
|
+
try:
|
|
206
|
+
if _cho_factor is not None:
|
|
207
|
+
_cho_factor(sub, lower=True, check_finite=False)
|
|
208
|
+
else:
|
|
209
|
+
np.linalg.cholesky(sub)
|
|
210
|
+
except (np.linalg.LinAlgError, ValueError):
|
|
211
|
+
return None
|
|
212
|
+
if self.split:
|
|
213
|
+
prev_vol, prev_cov = self.prev_vol, self.prev_cov
|
|
214
|
+
assert prev_vol is not None and prev_cov is not None
|
|
215
|
+
# an asset observed today but without a variance state at the
|
|
216
|
+
# forecast has a zero covariance row: abstain, as the covariance
|
|
217
|
+
# factorisation did
|
|
218
|
+
if not full and np.any(prev_vol[idx] <= 0.0):
|
|
219
|
+
return None
|
|
220
|
+
sig = prev_cov if full else prev_cov[np.ix_(idx, idx)]
|
|
221
|
+
sig = 0.5 * (sig + sig.T)
|
|
222
|
+
else:
|
|
223
|
+
sig = np.zeros((idx.size, idx.size))
|
|
224
|
+
for k in range(K):
|
|
225
|
+
sig += self.prev_w[k] * self.prev_sig[k][np.ix_(idx, idx)]
|
|
226
|
+
sig = 0.5 * (sig + sig.T)
|
|
227
|
+
# Floor the spectrum before factorising, exactly as the research
|
|
228
|
+
# engine's scoring path does: a near-singular early blend would
|
|
229
|
+
# otherwise hand the running scale one enormous gradient and
|
|
230
|
+
# silence the mixer for years.
|
|
231
|
+
min_eig = float(np.linalg.eigvalsh(sig).min())
|
|
232
|
+
if min_eig < 1e-8:
|
|
233
|
+
sig = sig + np.eye(idx.size) * (1e-8 - min_eig)
|
|
234
|
+
try:
|
|
235
|
+
if _cho_factor is not None:
|
|
236
|
+
cf = _cho_factor(sig, lower=True, check_finite=False)
|
|
237
|
+
u = _cho_solve(cf, r_t[idx], check_finite=False)
|
|
238
|
+
sinv = _cho_solve(cf, np.eye(idx.size), check_finite=False)
|
|
239
|
+
else:
|
|
240
|
+
sinv = np.linalg.inv(sig)
|
|
241
|
+
u = sinv @ r_t[idx]
|
|
242
|
+
except (np.linalg.LinAlgError, ValueError):
|
|
243
|
+
return None
|
|
244
|
+
g = np.empty(K)
|
|
245
|
+
self._gc = None
|
|
246
|
+
if self.split and self.prev_comps is not None:
|
|
247
|
+
prev_comps, prev_vol, prev_v = self.prev_comps, self.prev_vol, self.prev_v
|
|
248
|
+
assert prev_vol is not None and prev_v is not None
|
|
249
|
+
# Gradients on the expert CORRELATIONS:
|
|
250
|
+
# tr(Sinv (T o vv)) = sum((Sinv o vv) * T),
|
|
251
|
+
# u'(T o vv) u = (u o s)' T (u o s);
|
|
252
|
+
# the rung gradient follows by linearity of the pool.
|
|
253
|
+
vo = prev_vol if full else prev_vol[idx]
|
|
254
|
+
mw = sinv * np.multiply.outer(vo, vo)
|
|
255
|
+
ut = u * vo
|
|
256
|
+
J = self.n_experts
|
|
257
|
+
gc = np.empty((K, J))
|
|
258
|
+
for k in range(K):
|
|
259
|
+
for j in range(J):
|
|
260
|
+
t_ = prev_comps[k][j]
|
|
261
|
+
ts = t_ if full else t_[np.ix_(idx, idx)]
|
|
262
|
+
gc[k, j] = -0.5 * float((mw * ts).sum()) + 0.5 * float(ut @ ts @ ut)
|
|
263
|
+
g[k] = float(prev_v[k] @ gc[k])
|
|
264
|
+
if not np.all(np.isfinite(g)):
|
|
265
|
+
return None
|
|
266
|
+
self._gc = gc
|
|
267
|
+
return g
|
|
268
|
+
for k in range(K):
|
|
269
|
+
sk = self.prev_sig[k][np.ix_(idx, idx)]
|
|
270
|
+
g[k] = -0.5 * float((sinv * sk).sum()) + 0.5 * float(u @ sk @ u)
|
|
271
|
+
if not np.all(np.isfinite(g)):
|
|
272
|
+
return None
|
|
273
|
+
return g
|
|
274
|
+
|
|
275
|
+
def advance(self, g: np.ndarray | None) -> None:
|
|
276
|
+
if g is not None:
|
|
277
|
+
dd = g - g.mean()
|
|
278
|
+
rms = float(np.sqrt((dd @ dd) / dd.size))
|
|
279
|
+
self.scale = (self._gamma_scale * self.scale
|
|
280
|
+
+ (1.0 - self._gamma_scale) * rms)
|
|
281
|
+
ell = (dd / (self.scale + self._JITTER)) * np.sqrt(2.0 * self.eps)
|
|
282
|
+
lw = np.log(np.maximum(self.w, 1e-300)) + ell
|
|
283
|
+
lw -= lw.max()
|
|
284
|
+
w = np.exp(lw)
|
|
285
|
+
self.w = w / w.sum()
|
|
286
|
+
if self.split and self._gc is not None and self.prev_prior is not None:
|
|
287
|
+
for k in range(self.v.shape[0]):
|
|
288
|
+
gc = self._gc[k]
|
|
289
|
+
if not np.all(np.isfinite(gc)):
|
|
290
|
+
continue
|
|
291
|
+
ddv = gc - gc.mean()
|
|
292
|
+
rmsv = float(np.sqrt((ddv @ ddv) / float(self.n_experts)))
|
|
293
|
+
self.scale_v[k] = (self._gamma_scale * self.scale_v[k]
|
|
294
|
+
+ (1.0 - self._gamma_scale) * rmsv)
|
|
295
|
+
lv = (np.log(np.maximum(self.v[k], 1e-300))
|
|
296
|
+
+ (ddv / (self.scale_v[k] + self._JITTER)) * np.sqrt(2.0 * self.eps))
|
|
297
|
+
lv -= lv.max()
|
|
298
|
+
vk = np.exp(lv)
|
|
299
|
+
vk /= vk.sum()
|
|
300
|
+
self.v[k] = (1.0 - self.eps) * vk + self.eps * self.prev_prior[k]
|
|
301
|
+
self.w = (1.0 - self.eps) * self.w + self.eps * self.prior
|
|
302
|
+
|
|
303
|
+
def record(self, rung_covs: list[np.ndarray],
|
|
304
|
+
comps: list[tuple[np.ndarray, ...]] | None = None,
|
|
305
|
+
prior: np.ndarray | None = None,
|
|
306
|
+
cov: np.ndarray | None = None,
|
|
307
|
+
vol: np.ndarray | None = None) -> None:
|
|
308
|
+
"""Cache this step's forecast for next step's ``score``. Without
|
|
309
|
+
the split: the rung covariances and the weights that blend them.
|
|
310
|
+
With the split: the rung CORRELATIONS (mixed), the expert
|
|
311
|
+
correlations, the rule's prior split, the emitted covariance and
|
|
312
|
+
the volatility vector; nothing per expert is materialised."""
|
|
313
|
+
self.prev_sig = rung_covs
|
|
314
|
+
self.prev_w = self.w.copy()
|
|
315
|
+
self.prev_comps = comps
|
|
316
|
+
self.prev_prior = None if prior is None else prior.copy()
|
|
317
|
+
self.prev_cov = cov
|
|
318
|
+
self.prev_vol = vol
|
|
319
|
+
self.prev_v = self.v.copy()
|
|
320
|
+
|
|
321
|
+
def blend(self, prior: np.ndarray) -> np.ndarray:
|
|
322
|
+
return np.asarray(self.w)
|
|
323
|
+
|
|
324
|
+
@property
|
|
325
|
+
def tilt(self) -> float:
|
|
326
|
+
"""Diagnostic analogue of the CUSUM tilt: signed departure of the
|
|
327
|
+
blend from its prior toward the fast (+) or slow (-) rung."""
|
|
328
|
+
return float(self.w[0] - self.prior[0] - (self.w[-1] - self.prior[-1]))
|
|
329
|
+
|
|
330
|
+
|
|
131
331
|
class SqueezeKernelEstimator:
|
|
132
332
|
"""The Squeeze Kernel estimator with its full switch surface.
|
|
133
333
|
|
|
@@ -219,8 +419,18 @@ class SqueezeKernelEstimator:
|
|
|
219
419
|
level_match: bool = True,
|
|
220
420
|
detector: bool = True,
|
|
221
421
|
clock: str = "global",
|
|
422
|
+
vol_ladder: bool = False,
|
|
423
|
+
weights: str = "cusum",
|
|
424
|
+
split_learn: bool = False,
|
|
222
425
|
):
|
|
223
426
|
self.n_assets = n_assets
|
|
427
|
+
if weights not in ("cusum", "eg_blend"):
|
|
428
|
+
raise ValueError("weights must be 'cusum' or 'eg_blend'.")
|
|
429
|
+
self.weights = weights
|
|
430
|
+
if split_learn and weights != "eg_blend":
|
|
431
|
+
raise ValueError("split_learn requires weights='eg_blend'.")
|
|
432
|
+
self.split_learn = bool(split_learn)
|
|
433
|
+
self.vol_ladder = bool(vol_ladder)
|
|
224
434
|
self.lambda_vol = lambda_vol
|
|
225
435
|
self.lambda_corr = lambda_corr
|
|
226
436
|
self.epsilon = epsilon
|
|
@@ -288,7 +498,7 @@ class SqueezeKernelEstimator:
|
|
|
288
498
|
self.corr_theta = corr_theta
|
|
289
499
|
self._corr_lam: np.ndarray | None = None
|
|
290
500
|
self._corr_w: np.ndarray | None = None
|
|
291
|
-
self._detector: _SurpriseDetector | None = None
|
|
501
|
+
self._detector: _SurpriseDetector | _BlendGradient | None = None
|
|
292
502
|
self._Q_list: list[np.ndarray] | None = None
|
|
293
503
|
self._S_list: list[float] | None = None
|
|
294
504
|
self._adaptive = False
|
|
@@ -311,7 +521,10 @@ class SqueezeKernelEstimator:
|
|
|
311
521
|
# weights; the v2 ablation switch).
|
|
312
522
|
self._adaptive = detector and hl.size >= 2
|
|
313
523
|
if self._adaptive:
|
|
314
|
-
self._detector =
|
|
524
|
+
self._detector = (_BlendGradient(hl, self._corr_w, split=self.split_learn,
|
|
525
|
+
n_experts=2 if shrinkage_target == "equicorrelation" else 3)
|
|
526
|
+
if weights == "eg_blend"
|
|
527
|
+
else _SurpriseDetector(hl))
|
|
315
528
|
|
|
316
529
|
# Resolve shrinkage
|
|
317
530
|
if isinstance(shrinkage, str):
|
|
@@ -328,6 +541,24 @@ class SqueezeKernelEstimator:
|
|
|
328
541
|
self._var_t: np.ndarray | None = None
|
|
329
542
|
self._var_init: np.ndarray | None = None
|
|
330
543
|
self._vol_t: np.ndarray | None = None
|
|
544
|
+
# Self-adapting volatility memory: a ladder of decays two octaves
|
|
545
|
+
# below the correlation ladder, pooled with panel-wide weights that
|
|
546
|
+
# are Bayes at the null-calibrated temperature on SATURATED
|
|
547
|
+
# (tanh) evidence -- one day cannot hand a stale rung weeks of
|
|
548
|
+
# weight -- with the fastest rung's memory and a uniform prior.
|
|
549
|
+
# Replaces the fitted constant ``lambda_vol``.
|
|
550
|
+
self._var_l: np.ndarray | None = None
|
|
551
|
+
if self.vol_ladder:
|
|
552
|
+
if self.corr_half_lives is None:
|
|
553
|
+
raise ValueError("vol_ladder requires corr_half_lives.")
|
|
554
|
+
h = float(np.median(self.corr_half_lives))
|
|
555
|
+
self._hl_v = np.array([h / 16.0, h / 4.0, h])
|
|
556
|
+
self._lv_l = 2.0 ** (-1.0 / self._hl_v)
|
|
557
|
+
self._eps_v = float(1.0 - self._lv_l[0])
|
|
558
|
+
self._pi_v = np.full(3, 1.0 / 3.0)
|
|
559
|
+
self._gamma_v = 2.0 ** (-1.0 / h)
|
|
560
|
+
self._w_p = self._pi_v.copy()
|
|
561
|
+
self._scale_p = 1.0
|
|
331
562
|
# No single-scale state is allocated in ladder mode.
|
|
332
563
|
self._Q_t = (np.eye(n_assets, dtype=np.float64)
|
|
333
564
|
if self._corr_lam is None else None)
|
|
@@ -397,11 +628,36 @@ class SqueezeKernelEstimator:
|
|
|
397
628
|
if np.any(first):
|
|
398
629
|
var_t[first] = r_t[first] ** 2 + eps
|
|
399
630
|
var_init[first] = True
|
|
631
|
+
if self.vol_ladder:
|
|
632
|
+
if self._var_l is None:
|
|
633
|
+
self._var_l = np.full((3, n), eps)
|
|
634
|
+
self._var_l[:, first] = r_t[first] ** 2 + eps
|
|
400
635
|
if np.any(repeat):
|
|
401
|
-
|
|
402
|
-
self.
|
|
403
|
-
|
|
404
|
-
|
|
636
|
+
if self.vol_ladder:
|
|
637
|
+
assert self._var_l is not None
|
|
638
|
+
r2 = r_t[repeat] ** 2
|
|
639
|
+
vr = self._var_l[:, repeat]
|
|
640
|
+
vr_safe = np.maximum(vr, eps)
|
|
641
|
+
ell_v = -0.5 * (np.log(vr_safe) + r2 / vr_safe)
|
|
642
|
+
ep = ell_v.sum(axis=1)
|
|
643
|
+
dp = ep - ep.mean()
|
|
644
|
+
rp = float(np.sqrt((dp @ dp) / 3.0))
|
|
645
|
+
self._scale_p = (self._gamma_v * self._scale_p
|
|
646
|
+
+ (1.0 - self._gamma_v) * rp)
|
|
647
|
+
zp = np.tanh(dp / (self._scale_p + 1e-12))
|
|
648
|
+
lw = np.log(np.maximum(self._w_p, 1e-300)) + zp * np.sqrt(2.0 * self._eps_v)
|
|
649
|
+
lw -= lw.max()
|
|
650
|
+
w_p = np.exp(lw)
|
|
651
|
+
w_p /= w_p.sum()
|
|
652
|
+
self._w_p = (1.0 - self._eps_v) * w_p + self._eps_v * self._pi_v
|
|
653
|
+
lv = self._lv_l[:, None]
|
|
654
|
+
self._var_l[:, repeat] = lv * vr + (1.0 - lv) * r2
|
|
655
|
+
var_t[repeat] = (self._w_p[:, None] * self._var_l[:, repeat]).sum(axis=0)
|
|
656
|
+
else:
|
|
657
|
+
var_t[repeat] = (
|
|
658
|
+
self.lambda_vol * var_t[repeat]
|
|
659
|
+
+ (1.0 - self.lambda_vol) * r_t[repeat] ** 2
|
|
660
|
+
)
|
|
405
661
|
|
|
406
662
|
vol_t = np.zeros(n, dtype=np.float64)
|
|
407
663
|
vol_t[var_init] = np.sqrt(var_t[var_init])
|
|
@@ -577,6 +833,47 @@ class SqueezeKernelEstimator:
|
|
|
577
833
|
# T = (1 - rho_bar) I + rho_bar 11'. We avoid materialising T by
|
|
578
834
|
# blending the off-diagonal toward rho_bar in place and resetting
|
|
579
835
|
# the diagonal to 1.
|
|
836
|
+
alpha = self._intensity(corr, S_t, J_t)
|
|
837
|
+
if alpha > 0.0 and n > 1:
|
|
838
|
+
# Off-diagonal mean: O(n^2) sum, no mask allocation.
|
|
839
|
+
rho_bar = (corr.sum() - corr.trace()) / self._n_off
|
|
840
|
+
if self.shrinkage_target == "equicorrelation" or rho_bar <= 0.0:
|
|
841
|
+
# rho_bar <= 0 also covers the q = meanoff(C o C) = 0 corner:
|
|
842
|
+
# C o C has nonnegative entries, so q = 0 forces C = I and
|
|
843
|
+
# hence rho_bar = 0 — the equicorrelation fallback applies.
|
|
844
|
+
corr *= (1.0 - alpha)
|
|
845
|
+
corr += alpha * rho_bar
|
|
846
|
+
np.fill_diagonal(corr, 1.0)
|
|
847
|
+
else:
|
|
848
|
+
# Cluster (concentration-morphing) target:
|
|
849
|
+
# T = (1-alpha) T_equi + alpha [(1-gamma) I + gamma (C o C)]
|
|
850
|
+
# C o C is the Hadamard square of the raw correlation (PSD by
|
|
851
|
+
# the Schur product theorem, unit diagonal for free); gamma is
|
|
852
|
+
# level-matched so the target carries the same average
|
|
853
|
+
# correlation mass as the equicorrelation target. As
|
|
854
|
+
# alpha -> 0 this reduces exactly to the published estimator.
|
|
855
|
+
had = self._scratch_had
|
|
856
|
+
np.multiply(corr, corr, out=had) # Hadamard square, O(n^2)
|
|
857
|
+
if self.level_match:
|
|
858
|
+
mean_off = (had.sum() - np.trace(had)) / self._n_off
|
|
859
|
+
gamma = min(1.0, rho_bar / max(mean_off, eps))
|
|
860
|
+
else:
|
|
861
|
+
gamma = 1.0 # v2: pure Schur-square target
|
|
862
|
+
corr *= (1.0 - alpha)
|
|
863
|
+
corr += (alpha * (1.0 - alpha)) * rho_bar
|
|
864
|
+
had *= alpha * alpha * gamma
|
|
865
|
+
corr += had
|
|
866
|
+
np.fill_diagonal(corr, 1.0)
|
|
867
|
+
return corr
|
|
868
|
+
|
|
869
|
+
def _intensity(self, corr: np.ndarray, S_t: float,
|
|
870
|
+
J_t: float | None = None) -> float:
|
|
871
|
+
"""Shrinkage intensity of one timescale from its normalised raw
|
|
872
|
+
correlation ``corr`` and its information masses (published rule or
|
|
873
|
+
the self-tuning rule); ``corr`` is read, not modified."""
|
|
874
|
+
eps = self.epsilon
|
|
875
|
+
n = self.n_assets
|
|
876
|
+
S_t = max(S_t, eps)
|
|
580
877
|
alpha = self._shrinkage_alpha
|
|
581
878
|
if alpha < 0:
|
|
582
879
|
if self.alpha_rule != "published" and J_t is not None:
|
|
@@ -615,37 +912,29 @@ class SqueezeKernelEstimator:
|
|
|
615
912
|
alpha = min(1.0, c_k) / (1.0 + r_ * r_ * s_)
|
|
616
913
|
else:
|
|
617
914
|
alpha = max(0.0, min(1.0, n / (2.0 * S_t) - self.shrinkage_delta))
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
else:
|
|
642
|
-
gamma = 1.0 # v2: pure Schur-square target
|
|
643
|
-
corr *= (1.0 - alpha)
|
|
644
|
-
corr += (alpha * (1.0 - alpha)) * rho_bar
|
|
645
|
-
had *= alpha * alpha * gamma
|
|
646
|
-
corr += had
|
|
647
|
-
np.fill_diagonal(corr, 1.0)
|
|
648
|
-
return corr
|
|
915
|
+
return alpha
|
|
916
|
+
|
|
917
|
+
def _rung_experts(self, Q: np.ndarray, S_t: float, J_t: float
|
|
918
|
+
) -> tuple[tuple[np.ndarray, ...], np.ndarray]:
|
|
919
|
+
"""The three PSD unit-diagonal experts of one timescale -- raw
|
|
920
|
+
correlation, equicorrelation at its mean level, Schur square --
|
|
921
|
+
and the rule's prior split (1-a, a(1-a), a^2)."""
|
|
922
|
+
eps = self.epsilon
|
|
923
|
+
n = self.n_assets
|
|
924
|
+
diag_z = np.diagonal(Q).copy()
|
|
925
|
+
inv_diag = 1.0 / np.sqrt(np.maximum(diag_z, eps))
|
|
926
|
+
corr = np.multiply.outer(inv_diag, inv_diag)
|
|
927
|
+
corr *= Q
|
|
928
|
+
np.fill_diagonal(corr, np.where(diag_z > eps, 1.0, 0.0))
|
|
929
|
+
a = self._intensity(corr, S_t, J_t)
|
|
930
|
+
rho = (corr.sum() - n) / self._n_off
|
|
931
|
+
rho = float(min(max(rho, -0.99 / max(n - 1, 1)), 0.999))
|
|
932
|
+
E = np.full((n, n), rho)
|
|
933
|
+
np.fill_diagonal(E, 1.0)
|
|
934
|
+
if self.shrinkage_target == "equicorrelation":
|
|
935
|
+
# target off: the pool is the equicorrelation shrinkage alone
|
|
936
|
+
return (corr, E), np.array([1.0 - a, a])
|
|
937
|
+
return (corr, E, corr * corr), np.array([1.0 - a, a * (1.0 - a), a * a])
|
|
649
938
|
|
|
650
939
|
def _materialize_adaptive(self) -> None:
|
|
651
940
|
"""Adaptive-weight extraction: build per-rung shrunk covariances
|
|
@@ -671,21 +960,56 @@ class SqueezeKernelEstimator:
|
|
|
671
960
|
if self.clock == "asset" and buf is None:
|
|
672
961
|
buf = np.empty((K, self.n_assets, self.n_assets))
|
|
673
962
|
self._corr_blend_buf = buf
|
|
963
|
+
comps: list[tuple[np.ndarray, ...]] | None = None
|
|
964
|
+
prior: np.ndarray | None = None
|
|
965
|
+
v: np.ndarray | None = None
|
|
966
|
+
if self.split_learn:
|
|
967
|
+
assert isinstance(detector, _BlendGradient) # the split rides on the blend gradient
|
|
968
|
+
comps = []
|
|
969
|
+
v = detector.v
|
|
970
|
+
n_exp = 2 if self.shrinkage_target == "equicorrelation" else 3
|
|
971
|
+
if v.shape[1] != n_exp:
|
|
972
|
+
# the target was switched after construction; the split has
|
|
973
|
+
# not learned anything yet, so re-shape its state
|
|
974
|
+
assert detector.prev_comps is None, "cannot switch the target mid-stream"
|
|
975
|
+
detector.n_experts = n_exp
|
|
976
|
+
detector.v = np.full((K, n_exp), 1.0 / n_exp)
|
|
977
|
+
v = detector.v
|
|
978
|
+
prior = np.empty(v.shape)
|
|
674
979
|
for k in range(K):
|
|
675
|
-
|
|
980
|
+
if self.split_learn:
|
|
981
|
+
assert comps is not None and prior is not None and v is not None
|
|
982
|
+
experts, prior[k] = self._rung_experts(Q_list[k], S_list[k], J_list[k])
|
|
983
|
+
corr_k = v[k, 0] * experts[0]
|
|
984
|
+
for j in range(1, len(experts)):
|
|
985
|
+
corr_k = corr_k + v[k, j] * experts[j]
|
|
986
|
+
comps.append(experts)
|
|
987
|
+
sig_k.append(corr_k) # mixed correlation; vv applied once below
|
|
988
|
+
else:
|
|
989
|
+
corr_k = self._shrunk_corr_from_Q(Q_list[k], S_list[k], J_list[k])
|
|
990
|
+
sig_k.append(corr_k * vv)
|
|
676
991
|
if buf is not None:
|
|
677
992
|
buf[k][:] = corr_k
|
|
678
|
-
sig_k.append(corr_k * vv)
|
|
679
|
-
detector.record(sig_k)
|
|
680
993
|
w = detector.blend(corr_w)
|
|
681
|
-
if
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
cov =
|
|
688
|
-
|
|
994
|
+
if self.split_learn:
|
|
995
|
+
cb = w[0] * sig_k[0]
|
|
996
|
+
for k in range(1, K):
|
|
997
|
+
cb = cb + w[k] * sig_k[k]
|
|
998
|
+
if buf is not None:
|
|
999
|
+
self._C_prev = cb
|
|
1000
|
+
cov = cb * vv
|
|
1001
|
+
cov = (cov + cov.T) * 0.5
|
|
1002
|
+
detector.record(sig_k, comps, prior, cov, vol_t) # type: ignore[call-arg]
|
|
1003
|
+
else:
|
|
1004
|
+
detector.record(sig_k)
|
|
1005
|
+
if buf is not None:
|
|
1006
|
+
# next day's neighborhood weighting: the emitted blend of the
|
|
1007
|
+
# shrunk rung correlations
|
|
1008
|
+
self._C_prev = np.tensordot(w, buf, axes=1)
|
|
1009
|
+
cov = w[0] * sig_k[0]
|
|
1010
|
+
for k in range(1, len(sig_k)):
|
|
1011
|
+
cov = cov + w[k] * sig_k[k]
|
|
1012
|
+
cov = (cov + cov.T) * 0.5
|
|
689
1013
|
self._cov = cov
|
|
690
1014
|
d = np.sqrt(np.maximum(np.diagonal(cov), eps))
|
|
691
1015
|
self._corr = cov / np.outer(d, d)
|
|
File without changes
|
|
File without changes
|