factor-qc 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- factor_qc/__init__.py +34 -0
- factor_qc/__main__.py +4 -0
- factor_qc/cli.py +114 -0
- factor_qc/gate.py +230 -0
- factor_qc/stats.py +493 -0
- factor_qc-0.1.0.dist-info/METADATA +168 -0
- factor_qc-0.1.0.dist-info/RECORD +11 -0
- factor_qc-0.1.0.dist-info/WHEEL +5 -0
- factor_qc-0.1.0.dist-info/entry_points.txt +3 -0
- factor_qc-0.1.0.dist-info/licenses/LICENSE +21 -0
- factor_qc-0.1.0.dist-info/top_level.txt +1 -0
factor_qc/__init__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""factor-qc: fail-closed quality gate for backtests.
|
|
2
|
+
|
|
3
|
+
One numpy-only engine for the standard backtest-overfit statistics —
|
|
4
|
+
Deflated Sharpe Ratio, Probability of Backtest Overfitting (CSCV),
|
|
5
|
+
Harvey-Liu multiple-testing haircut, Minimum Track Record Length — wrapped
|
|
6
|
+
in a gate that refuses to judge a backtest that does not declare how many
|
|
7
|
+
configurations were tried.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .gate import run_gate
|
|
11
|
+
from .stats import (
|
|
12
|
+
build_check_artifact,
|
|
13
|
+
build_overfit_report,
|
|
14
|
+
deflated_sharpe_ratio,
|
|
15
|
+
haircut_sharpe,
|
|
16
|
+
minimum_track_record_length,
|
|
17
|
+
probability_of_backtest_overfitting,
|
|
18
|
+
sharpe_ratio,
|
|
19
|
+
skew_kurt,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
__version__ = "0.1.0"
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"build_check_artifact",
|
|
26
|
+
"build_overfit_report",
|
|
27
|
+
"deflated_sharpe_ratio",
|
|
28
|
+
"haircut_sharpe",
|
|
29
|
+
"minimum_track_record_length",
|
|
30
|
+
"probability_of_backtest_overfitting",
|
|
31
|
+
"run_gate",
|
|
32
|
+
"sharpe_ratio",
|
|
33
|
+
"skew_kurt",
|
|
34
|
+
]
|
factor_qc/__main__.py
ADDED
factor_qc/cli.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Command-line interface for factor-qc.
|
|
2
|
+
|
|
3
|
+
Subcommands:
|
|
4
|
+
|
|
5
|
+
- ``check`` run the fail-closed gate on a returns series
|
|
6
|
+
- ``version`` print version
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
import sys
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
|
|
19
|
+
from . import __version__
|
|
20
|
+
from .gate import run_gate
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _load_returns(path: str) -> np.ndarray:
|
|
24
|
+
value = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
25
|
+
if not isinstance(value, list) or not all(isinstance(x, (int, float)) for x in value):
|
|
26
|
+
raise ValueError(f"returns must be a JSON list of numbers: {path}")
|
|
27
|
+
return np.asarray(value, dtype=float)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _load_trials(path: str) -> np.ndarray:
|
|
31
|
+
value = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
32
|
+
if not isinstance(value, list) or not value:
|
|
33
|
+
raise ValueError(f"trials must be a non-empty JSON list of lists: {path}")
|
|
34
|
+
rows = []
|
|
35
|
+
for row in value:
|
|
36
|
+
if not isinstance(row, list):
|
|
37
|
+
raise ValueError(f"trials rows must be lists: {path}")
|
|
38
|
+
rows.append([float(x) for x in row])
|
|
39
|
+
matrix = np.asarray(rows, dtype=float)
|
|
40
|
+
if matrix.ndim != 2:
|
|
41
|
+
raise ValueError(f"trials must be 2D (T x N): {path}")
|
|
42
|
+
return matrix
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
46
|
+
parser = argparse.ArgumentParser(
|
|
47
|
+
prog="qc",
|
|
48
|
+
description="Fail-closed quality gate for backtests.",
|
|
49
|
+
)
|
|
50
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
51
|
+
|
|
52
|
+
check = sub.add_parser("check", help="run the fail-closed gate")
|
|
53
|
+
check.add_argument("--returns", required=True, help="JSON list of per-period returns")
|
|
54
|
+
check.add_argument("--trials", default=None, help="JSON 2D matrix (T x N) of trial returns")
|
|
55
|
+
check.add_argument(
|
|
56
|
+
"--n-trials",
|
|
57
|
+
type=int,
|
|
58
|
+
default=None,
|
|
59
|
+
help="honest number of configurations tried (required, fail-closed)",
|
|
60
|
+
)
|
|
61
|
+
check.add_argument("--periods-per-year", type=int, default=252)
|
|
62
|
+
check.add_argument(
|
|
63
|
+
"--n-blocks",
|
|
64
|
+
type=int,
|
|
65
|
+
default=16,
|
|
66
|
+
help="CSCV blocks for PBO (default 16 = 12,870 splits; use 8 or 10 for speed)",
|
|
67
|
+
)
|
|
68
|
+
check.add_argument("--json", action="store_true", help="machine-readable output")
|
|
69
|
+
|
|
70
|
+
sub.add_parser("version", help="print version")
|
|
71
|
+
return parser
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def main(argv: list[str] | None = None) -> int:
|
|
75
|
+
parser = build_parser()
|
|
76
|
+
args = parser.parse_args(argv)
|
|
77
|
+
|
|
78
|
+
if args.command == "version":
|
|
79
|
+
print(__version__)
|
|
80
|
+
return 0
|
|
81
|
+
|
|
82
|
+
if args.command == "check":
|
|
83
|
+
returns = _load_returns(args.returns)
|
|
84
|
+
trials = _load_trials(args.trials) if args.trials else None
|
|
85
|
+
if args.n_trials is not None and args.n_trials < 1:
|
|
86
|
+
parser.error("--n-trials must be >= 1")
|
|
87
|
+
body: dict[str, Any] = run_gate(
|
|
88
|
+
returns,
|
|
89
|
+
args.n_trials,
|
|
90
|
+
trials_matrix=trials,
|
|
91
|
+
periods_per_year=args.periods_per_year,
|
|
92
|
+
n_blocks=args.n_blocks,
|
|
93
|
+
)
|
|
94
|
+
if args.json:
|
|
95
|
+
print(json.dumps(body, ensure_ascii=False, indent=2))
|
|
96
|
+
else:
|
|
97
|
+
print(body["verdict"])
|
|
98
|
+
if body["report"] is not None:
|
|
99
|
+
print(body["report_text"])
|
|
100
|
+
for check in body["checks"]:
|
|
101
|
+
marker = "PASS" if check["passed"] else "FAIL"
|
|
102
|
+
print(
|
|
103
|
+
f" [{marker}] {check['severity']} {check['check_id']}: "
|
|
104
|
+
f"{check['title']} (value={check['value']}, "
|
|
105
|
+
f"threshold={check['threshold']})"
|
|
106
|
+
)
|
|
107
|
+
return 0 if body["passed"] else 1
|
|
108
|
+
|
|
109
|
+
parser.error(f"unknown command: {args.command}")
|
|
110
|
+
return 2
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
if __name__ == "__main__":
|
|
114
|
+
sys.exit(main())
|
factor_qc/gate.py
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""The fail-closed quality gate.
|
|
2
|
+
|
|
3
|
+
``factor_qc.stats`` computes; this module decides — and its default answer
|
|
4
|
+
is *no*. A backtest that refuses to declare how many configurations were
|
|
5
|
+
tried is refused outright: without an honest ``n_trials`` there is no
|
|
6
|
+
deflation benchmark, no haircut and no track-record floor, and any verdict
|
|
7
|
+
would be theatre.
|
|
8
|
+
|
|
9
|
+
Checks are graded by severity:
|
|
10
|
+
|
|
11
|
+
- **P0 (fatal)** — the candidate must not pass:
|
|
12
|
+
- Deflated Sharpe Ratio below threshold (selection-bias corrected),
|
|
13
|
+
- Probability of Backtest Overfitting above threshold (when a trials
|
|
14
|
+
matrix is provided),
|
|
15
|
+
- multiple-testing haircut annual Sharpe below the floor,
|
|
16
|
+
- Minimum Track Record Length longer than the available sample.
|
|
17
|
+
- **P1 (warning)** — proceed with eyes open:
|
|
18
|
+
- PSR vs zero below 0.95 (weak evidence even before deflation),
|
|
19
|
+
- sample shorter than one year (252 obs),
|
|
20
|
+
- trials count aggressive relative to sample (n_trials > n_obs / 5).
|
|
21
|
+
- **P2 (info)** — recorded, no action:
|
|
22
|
+
- non-normal return moments (skew / kurtosis),
|
|
23
|
+
- very few trials (n_trials < 5).
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
import numpy as np
|
|
31
|
+
|
|
32
|
+
from .stats import (
|
|
33
|
+
DSR_THRESHOLD,
|
|
34
|
+
PBO_THRESHOLD,
|
|
35
|
+
build_overfit_report,
|
|
36
|
+
minimum_track_record_length,
|
|
37
|
+
probabilistic_sharpe_ratio,
|
|
38
|
+
render_report_text,
|
|
39
|
+
sharpe_ratio,
|
|
40
|
+
skew_kurt,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
SAFETY = {
|
|
44
|
+
"production_effect": False,
|
|
45
|
+
"changes_probability": False,
|
|
46
|
+
"allow_real_trade": False,
|
|
47
|
+
}
|
|
48
|
+
MIN_YEAR_OBS = 252
|
|
49
|
+
ADJUSTED_SHARPE_FLOOR = 0.5
|
|
50
|
+
PSR_ZERO_THRESHOLD = 0.95
|
|
51
|
+
TRIAL_AGGRESSION_DENOM = 5.0
|
|
52
|
+
MIN_TRIALS_INFO = 5
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def run_gate(
|
|
56
|
+
returns: np.ndarray,
|
|
57
|
+
n_trials: int | None,
|
|
58
|
+
*,
|
|
59
|
+
trials_matrix: np.ndarray | None = None,
|
|
60
|
+
periods_per_year: int = 252,
|
|
61
|
+
dsr_threshold: float = DSR_THRESHOLD,
|
|
62
|
+
pbo_threshold: float = PBO_THRESHOLD,
|
|
63
|
+
adjusted_sharpe_floor: float = ADJUSTED_SHARPE_FLOOR,
|
|
64
|
+
require_declared_trials: bool = True,
|
|
65
|
+
n_blocks: int = 16,
|
|
66
|
+
) -> dict[str, Any]:
|
|
67
|
+
"""Run the fail-closed gate. Returns a dict with ``passed``, ``checks``
|
|
68
|
+
and the underlying statistics report.
|
|
69
|
+
|
|
70
|
+
When ``require_declared_trials`` is true (default) and ``n_trials`` is
|
|
71
|
+
None, the gate refuses: ``passed=False`` with a single P0 blocker
|
|
72
|
+
``n_trials_declaration_required``.
|
|
73
|
+
|
|
74
|
+
``n_blocks`` controls the CSCV granularity of the PBO check; note that
|
|
75
|
+
CSCV enumerates C(n_blocks, n_blocks/2) splits (12,870 for the default
|
|
76
|
+
16), so smaller values (e.g. 8 or 10) run much faster on large trial
|
|
77
|
+
matrices.
|
|
78
|
+
"""
|
|
79
|
+
r = np.asarray(returns, dtype=float)
|
|
80
|
+
r = r[~np.isnan(r)]
|
|
81
|
+
checks: list[dict[str, Any]] = []
|
|
82
|
+
|
|
83
|
+
def add(check_id: str, severity: str, title: str, value: Any, threshold: Any, passed: bool) -> None:
|
|
84
|
+
checks.append(
|
|
85
|
+
{
|
|
86
|
+
"check_id": check_id,
|
|
87
|
+
"severity": severity,
|
|
88
|
+
"title": title,
|
|
89
|
+
"value": value,
|
|
90
|
+
"threshold": threshold,
|
|
91
|
+
"passed": passed,
|
|
92
|
+
}
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
if n_trials is None:
|
|
96
|
+
if require_declared_trials:
|
|
97
|
+
add(
|
|
98
|
+
"n_trials_declaration_required",
|
|
99
|
+
"P0",
|
|
100
|
+
"honest n_trials declaration",
|
|
101
|
+
None,
|
|
102
|
+
"declared integer >= 1",
|
|
103
|
+
False,
|
|
104
|
+
)
|
|
105
|
+
return {
|
|
106
|
+
"passed": False,
|
|
107
|
+
"verdict": "FAIL - n_trials must be declared before a backtest "
|
|
108
|
+
"can be judged (fail-closed)",
|
|
109
|
+
"checks": checks,
|
|
110
|
+
"report": None,
|
|
111
|
+
"safety": SAFETY,
|
|
112
|
+
}
|
|
113
|
+
n_trials = 1
|
|
114
|
+
|
|
115
|
+
n_obs = int(r.size)
|
|
116
|
+
report = build_overfit_report(
|
|
117
|
+
r,
|
|
118
|
+
n_trials,
|
|
119
|
+
trials_matrix=trials_matrix,
|
|
120
|
+
periods_per_year=periods_per_year,
|
|
121
|
+
dsr_threshold=dsr_threshold,
|
|
122
|
+
pbo_threshold=pbo_threshold,
|
|
123
|
+
n_blocks=n_blocks,
|
|
124
|
+
)
|
|
125
|
+
dsr = report["deflated_sharpe_ratio"]
|
|
126
|
+
pbo = report["pbo"]
|
|
127
|
+
ann = np.sqrt(periods_per_year)
|
|
128
|
+
sr_pp = sharpe_ratio(r)
|
|
129
|
+
skew, kurt = skew_kurt(r)
|
|
130
|
+
mintrl = minimum_track_record_length(sr_pp, 0.0, skew, kurt)
|
|
131
|
+
psr_zero = probabilistic_sharpe_ratio(sr_pp, 0.0, n_obs, skew, kurt)
|
|
132
|
+
adjusted_annual = report["haircut"]["adjusted_sharpe_annual"]
|
|
133
|
+
|
|
134
|
+
# P0: fatal
|
|
135
|
+
add(
|
|
136
|
+
"dsr",
|
|
137
|
+
"P0",
|
|
138
|
+
"deflated sharpe ratio >= threshold",
|
|
139
|
+
round(dsr, 4),
|
|
140
|
+
dsr_threshold,
|
|
141
|
+
bool(dsr >= dsr_threshold),
|
|
142
|
+
)
|
|
143
|
+
if pbo is not None:
|
|
144
|
+
add(
|
|
145
|
+
"pbo",
|
|
146
|
+
"P0",
|
|
147
|
+
"probability of backtest overfitting <= threshold",
|
|
148
|
+
round(pbo["pbo"], 4),
|
|
149
|
+
pbo_threshold,
|
|
150
|
+
bool(pbo["pbo"] <= pbo_threshold),
|
|
151
|
+
)
|
|
152
|
+
add(
|
|
153
|
+
"haircut_sharpe",
|
|
154
|
+
"P0",
|
|
155
|
+
"multiple-testing haircut annual sharpe >= floor",
|
|
156
|
+
round(adjusted_annual, 4),
|
|
157
|
+
adjusted_sharpe_floor,
|
|
158
|
+
bool(adjusted_annual >= adjusted_sharpe_floor),
|
|
159
|
+
)
|
|
160
|
+
add(
|
|
161
|
+
"mintrl",
|
|
162
|
+
"P0",
|
|
163
|
+
"minimum track record length <= sample",
|
|
164
|
+
round(mintrl, 1) if np.isfinite(mintrl) else None,
|
|
165
|
+
f"<= {n_obs}",
|
|
166
|
+
bool(np.isfinite(mintrl) and mintrl <= n_obs),
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
# P1: warnings
|
|
170
|
+
add(
|
|
171
|
+
"psr_vs_zero",
|
|
172
|
+
"P1",
|
|
173
|
+
"probabilistic sharpe vs zero >= threshold",
|
|
174
|
+
round(psr_zero, 4) if np.isfinite(psr_zero) else None,
|
|
175
|
+
PSR_ZERO_THRESHOLD,
|
|
176
|
+
bool(np.isfinite(psr_zero) and psr_zero >= PSR_ZERO_THRESHOLD),
|
|
177
|
+
)
|
|
178
|
+
add(
|
|
179
|
+
"sample_length",
|
|
180
|
+
"P1",
|
|
181
|
+
"sample >= one year of observations",
|
|
182
|
+
n_obs,
|
|
183
|
+
MIN_YEAR_OBS,
|
|
184
|
+
bool(n_obs >= MIN_YEAR_OBS),
|
|
185
|
+
)
|
|
186
|
+
add(
|
|
187
|
+
"trial_aggression",
|
|
188
|
+
"P1",
|
|
189
|
+
"trials not aggressive relative to sample",
|
|
190
|
+
n_trials,
|
|
191
|
+
f"<= {n_obs / TRIAL_AGGRESSION_DENOM:.0f}",
|
|
192
|
+
bool(n_trials <= n_obs / TRIAL_AGGRESSION_DENOM),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
# P2: info
|
|
196
|
+
add(
|
|
197
|
+
"return_moments",
|
|
198
|
+
"P2",
|
|
199
|
+
"return moments near-normal",
|
|
200
|
+
{"skew": round(skew, 4), "kurtosis": round(kurt, 4)},
|
|
201
|
+
"skew ~ 0, kurtosis ~ 3",
|
|
202
|
+
bool(abs(skew) < 0.5 and abs(kurt - 3.0) < 1.0),
|
|
203
|
+
)
|
|
204
|
+
add(
|
|
205
|
+
"trial_count",
|
|
206
|
+
"P2",
|
|
207
|
+
"trials count >= minimal",
|
|
208
|
+
n_trials,
|
|
209
|
+
MIN_TRIALS_INFO,
|
|
210
|
+
bool(n_trials >= MIN_TRIALS_INFO),
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
p0_failed = [check for check in checks if check["severity"] == "P0" and not check["passed"]]
|
|
214
|
+
passed = not p0_failed
|
|
215
|
+
if passed:
|
|
216
|
+
verdict = "PASS - survives multiple-testing correction (no P0 failures)"
|
|
217
|
+
else:
|
|
218
|
+
detail = "; ".join(
|
|
219
|
+
f"{check['check_id']}: {check['value']} vs {check['threshold']}"
|
|
220
|
+
for check in p0_failed
|
|
221
|
+
)
|
|
222
|
+
verdict = f"FAIL - P0 blocker(s): {detail}"
|
|
223
|
+
return {
|
|
224
|
+
"passed": passed,
|
|
225
|
+
"verdict": verdict,
|
|
226
|
+
"checks": checks,
|
|
227
|
+
"report": report,
|
|
228
|
+
"report_text": render_report_text(report),
|
|
229
|
+
"safety": SAFETY,
|
|
230
|
+
}
|
factor_qc/stats.py
ADDED
|
@@ -0,0 +1,493 @@
|
|
|
1
|
+
"""Independent backtest-overfit statistics (DSR / PBO / haircut / MinTRL).
|
|
2
|
+
|
|
3
|
+
Implementation of the published methods:
|
|
4
|
+
|
|
5
|
+
- Deflated / Probabilistic Sharpe Ratio and Minimum Track Record Length:
|
|
6
|
+
Bailey & Lopez de Prado (2012, 2014).
|
|
7
|
+
- Probability of Backtest Overfitting (PBO) via Combinatorially-Symmetric
|
|
8
|
+
Cross-Validation: Bailey, Borwein, Lopez de Prado & Zhu (2017).
|
|
9
|
+
- Multiple-testing haircut of the Sharpe Ratio: Harvey & Liu (2015).
|
|
10
|
+
|
|
11
|
+
Implementation notes
|
|
12
|
+
--------------------
|
|
13
|
+
Only numpy + the standard library are required (no scipy). The
|
|
14
|
+
standard-normal inverse CDF uses Acklam's rational approximation with one
|
|
15
|
+
Newton refinement. All Sharpe ratios are per-observation (not annualised);
|
|
16
|
+
the report builder annualises only for display.
|
|
17
|
+
|
|
18
|
+
This module computes; it never decides. The fail-closed gate lives in
|
|
19
|
+
``factor_qc.gate``.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import math
|
|
25
|
+
from datetime import datetime
|
|
26
|
+
from itertools import combinations
|
|
27
|
+
from typing import Any, Sequence
|
|
28
|
+
|
|
29
|
+
import numpy as np
|
|
30
|
+
|
|
31
|
+
EULER_MASCHERONI = 0.5772156649015328606
|
|
32
|
+
DSR_THRESHOLD = 0.95
|
|
33
|
+
PBO_THRESHOLD = 0.50
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# --------------------------------------------------------------------------- #
|
|
37
|
+
# Standard normal helpers (no scipy)
|
|
38
|
+
# --------------------------------------------------------------------------- #
|
|
39
|
+
def _norm_cdf(z: float) -> float:
|
|
40
|
+
return 0.5 * (1.0 + math.erf(z / math.sqrt(2.0)))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _norm_sf(z: float) -> float:
|
|
44
|
+
return 0.5 * math.erfc(z / math.sqrt(2.0))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _norm_ppf(p: float) -> float:
|
|
48
|
+
"""Inverse standard-normal CDF (Acklam's rational approximation + Newton step)."""
|
|
49
|
+
p = min(max(p, 1e-16), 1.0 - 1e-16)
|
|
50
|
+
a = [
|
|
51
|
+
-3.969683028665376e01,
|
|
52
|
+
2.209460984245205e02,
|
|
53
|
+
-2.759285104469687e02,
|
|
54
|
+
1.383577518672690e02,
|
|
55
|
+
-3.066479806614716e01,
|
|
56
|
+
2.506628277459239e00,
|
|
57
|
+
]
|
|
58
|
+
b = [
|
|
59
|
+
-5.447609879822406e01,
|
|
60
|
+
1.615858368580409e02,
|
|
61
|
+
-1.556989798598866e02,
|
|
62
|
+
6.680131188771972e01,
|
|
63
|
+
-1.328068155288572e01,
|
|
64
|
+
]
|
|
65
|
+
c = [
|
|
66
|
+
-7.784894002430293e-03,
|
|
67
|
+
-3.223964580411365e-01,
|
|
68
|
+
-2.400758277161838e00,
|
|
69
|
+
-2.549732539343734e00,
|
|
70
|
+
4.374664141464968e00,
|
|
71
|
+
2.938163982698783e00,
|
|
72
|
+
]
|
|
73
|
+
d = [
|
|
74
|
+
7.784695709041462e-03,
|
|
75
|
+
3.224671290700398e-01,
|
|
76
|
+
2.445134137142996e00,
|
|
77
|
+
3.754408661907416e00,
|
|
78
|
+
]
|
|
79
|
+
plow = 0.02425
|
|
80
|
+
if p < plow:
|
|
81
|
+
q = math.sqrt(-2.0 * math.log(p))
|
|
82
|
+
x = (((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / (
|
|
83
|
+
(((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1.0
|
|
84
|
+
)
|
|
85
|
+
elif p <= 1.0 - plow:
|
|
86
|
+
q = p - 0.5
|
|
87
|
+
r = q * q
|
|
88
|
+
x = (
|
|
89
|
+
(((((a[0] * r + a[1]) * r + a[2]) * r + a[3]) * r + a[4]) * r + a[5]) * q
|
|
90
|
+
) / (((((b[0] * r + b[1]) * r + b[2]) * r + b[3]) * r + b[4]) * r + 1.0)
|
|
91
|
+
else:
|
|
92
|
+
q = math.sqrt(-2.0 * math.log(1.0 - p))
|
|
93
|
+
x = -(((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) / (
|
|
94
|
+
(((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1.0
|
|
95
|
+
)
|
|
96
|
+
error = _norm_cdf(x) - p
|
|
97
|
+
u = error * math.sqrt(2.0 * math.pi) * math.exp(x * x / 2.0)
|
|
98
|
+
return x - u / (1.0 + x * u / 2.0)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _norm_isf(p: float) -> float:
|
|
102
|
+
return _norm_ppf(1.0 - p)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# --------------------------------------------------------------------------- #
|
|
106
|
+
# Sharpe ratio and moments
|
|
107
|
+
# --------------------------------------------------------------------------- #
|
|
108
|
+
def sharpe_ratio(returns: np.ndarray, benchmark: float = 0.0) -> float:
|
|
109
|
+
"""Per-period Sharpe ratio (ddof=1); NaN when fewer than 2 observations."""
|
|
110
|
+
r = np.asarray(returns, dtype=float)
|
|
111
|
+
r = r[~np.isnan(r)]
|
|
112
|
+
if r.size < 2:
|
|
113
|
+
return float("nan")
|
|
114
|
+
sd = r.std(ddof=1)
|
|
115
|
+
if sd == 0:
|
|
116
|
+
return float("nan")
|
|
117
|
+
return float((r.mean() - benchmark) / sd)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def skew_kurt(returns: np.ndarray) -> tuple[float, float]:
|
|
121
|
+
"""Sample skewness (g1) and non-excess kurtosis (g2, normal == 3)."""
|
|
122
|
+
r = np.asarray(returns, dtype=float)
|
|
123
|
+
r = r[~np.isnan(r)]
|
|
124
|
+
if r.size < 4:
|
|
125
|
+
return 0.0, 3.0
|
|
126
|
+
mean = r.mean()
|
|
127
|
+
sd = r.std(ddof=0)
|
|
128
|
+
if sd == 0:
|
|
129
|
+
return 0.0, 3.0
|
|
130
|
+
centered = (r - mean) / sd
|
|
131
|
+
return float(np.mean(centered ** 3)), float(np.mean(centered ** 4))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# --------------------------------------------------------------------------- #
|
|
135
|
+
# Deflated Sharpe Ratio / PSR / MinTRL
|
|
136
|
+
# --------------------------------------------------------------------------- #
|
|
137
|
+
def probabilistic_sharpe_ratio(
|
|
138
|
+
observed_sr: float,
|
|
139
|
+
benchmark_sr: float,
|
|
140
|
+
n_obs: int,
|
|
141
|
+
skew: float,
|
|
142
|
+
kurtosis: float,
|
|
143
|
+
) -> float:
|
|
144
|
+
"""P(SR > SR*) under the non-normal Sharpe estimator standard error."""
|
|
145
|
+
if n_obs < 2 or math.isnan(observed_sr):
|
|
146
|
+
return float("nan")
|
|
147
|
+
denom = 1.0 - skew * observed_sr + ((kurtosis - 1.0) / 4.0) * observed_sr ** 2
|
|
148
|
+
denom = max(denom, 1e-12)
|
|
149
|
+
se = math.sqrt(denom / (n_obs - 1))
|
|
150
|
+
return _norm_cdf((observed_sr - benchmark_sr) / se)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def expected_max_sharpe(sr_variance_across_trials: float, n_trials: int) -> float:
|
|
154
|
+
"""E[max SR] across N independent zero-true-SR trials (deflation benchmark)."""
|
|
155
|
+
if n_trials < 2:
|
|
156
|
+
return 0.0
|
|
157
|
+
variance = max(sr_variance_across_trials, 0.0)
|
|
158
|
+
z1 = _norm_ppf(1.0 - 1.0 / n_trials)
|
|
159
|
+
z2 = _norm_ppf(1.0 - 1.0 / (n_trials * math.e))
|
|
160
|
+
return float(
|
|
161
|
+
math.sqrt(variance)
|
|
162
|
+
* ((1.0 - EULER_MASCHERONI) * z1 + EULER_MASCHERONI * z2)
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def deflated_sharpe_ratio(
|
|
167
|
+
strategy_returns: np.ndarray,
|
|
168
|
+
n_trials: int,
|
|
169
|
+
*,
|
|
170
|
+
sr_variance_across_trials: float | None = None,
|
|
171
|
+
all_trial_sharpes: Sequence[float] | None = None,
|
|
172
|
+
threshold: float = DSR_THRESHOLD,
|
|
173
|
+
) -> dict[str, Any]:
|
|
174
|
+
r = np.asarray(strategy_returns, dtype=float)
|
|
175
|
+
r = r[~np.isnan(r)]
|
|
176
|
+
n = r.size
|
|
177
|
+
sr = sharpe_ratio(r)
|
|
178
|
+
skew, kurt = skew_kurt(r)
|
|
179
|
+
if sr_variance_across_trials is None:
|
|
180
|
+
if all_trial_sharpes is not None and len(all_trial_sharpes) > 1:
|
|
181
|
+
sr_variance_across_trials = float(
|
|
182
|
+
np.var(np.asarray(all_trial_sharpes, dtype=float), ddof=1)
|
|
183
|
+
)
|
|
184
|
+
else:
|
|
185
|
+
denom = 1.0 - skew * sr + ((kurt - 1.0) / 4.0) * sr ** 2
|
|
186
|
+
sr_variance_across_trials = max(denom, 1e-12) / max(n - 1, 1)
|
|
187
|
+
sr0 = expected_max_sharpe(sr_variance_across_trials, n_trials)
|
|
188
|
+
return {
|
|
189
|
+
"observed_sharpe": sr,
|
|
190
|
+
"deflated_benchmark_sr0": sr0,
|
|
191
|
+
"psr_vs_zero": probabilistic_sharpe_ratio(sr, 0.0, n, skew, kurt),
|
|
192
|
+
"deflated_sharpe_ratio": probabilistic_sharpe_ratio(sr, sr0, n, skew, kurt),
|
|
193
|
+
"n_obs": n,
|
|
194
|
+
"n_trials": n_trials,
|
|
195
|
+
"skew": skew,
|
|
196
|
+
"kurtosis": kurt,
|
|
197
|
+
"passed": bool(
|
|
198
|
+
probabilistic_sharpe_ratio(sr, sr0, n, skew, kurt) >= threshold
|
|
199
|
+
),
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def minimum_track_record_length(
|
|
204
|
+
observed_sr: float,
|
|
205
|
+
benchmark_sr: float,
|
|
206
|
+
skew: float,
|
|
207
|
+
kurtosis: float,
|
|
208
|
+
confidence: float = 0.95,
|
|
209
|
+
) -> float:
|
|
210
|
+
if observed_sr <= benchmark_sr:
|
|
211
|
+
return float("inf")
|
|
212
|
+
z = _norm_ppf(confidence)
|
|
213
|
+
num = max(
|
|
214
|
+
1.0 - skew * observed_sr + ((kurtosis - 1.0) / 4.0) * observed_sr ** 2,
|
|
215
|
+
1e-12,
|
|
216
|
+
)
|
|
217
|
+
return float(1.0 + num * (z / (observed_sr - benchmark_sr)) ** 2)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
# --------------------------------------------------------------------------- #
|
|
221
|
+
# Multiple-testing haircut (Harvey & Liu 2015)
|
|
222
|
+
# --------------------------------------------------------------------------- #
|
|
223
|
+
def _p_from_t(tstat: float) -> float:
|
|
224
|
+
return 2.0 * _norm_sf(abs(tstat))
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _adjusted_p(p: float, n_tests: int, method: str, rank: int = 1) -> float:
|
|
228
|
+
method = method.lower()
|
|
229
|
+
if method == "bonferroni":
|
|
230
|
+
return min(1.0, p * n_tests)
|
|
231
|
+
if method == "holm":
|
|
232
|
+
return min(1.0, p * (n_tests - rank + 1))
|
|
233
|
+
if method == "bhy":
|
|
234
|
+
harmonic = sum(1.0 / i for i in range(1, n_tests + 1))
|
|
235
|
+
return min(1.0, p * n_tests * harmonic / rank)
|
|
236
|
+
raise ValueError(f"unknown method: {method}")
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def haircut_sharpe(
|
|
240
|
+
observed_sharpe_per_period: float,
|
|
241
|
+
n_obs: int,
|
|
242
|
+
n_tests: int,
|
|
243
|
+
method: str = "bonferroni",
|
|
244
|
+
rank: int = 1,
|
|
245
|
+
) -> dict[str, Any]:
|
|
246
|
+
t_obs = observed_sharpe_per_period * math.sqrt(n_obs)
|
|
247
|
+
p_obs = _p_from_t(t_obs)
|
|
248
|
+
p_adj = _adjusted_p(p_obs, n_tests, method, rank)
|
|
249
|
+
# The inverse-CDF approximation is only meaningful down to ~1e-15; clip
|
|
250
|
+
# extreme significance so the adjusted Sharpe stays finite.
|
|
251
|
+
p_adj = min(1.0, max(p_adj, 1e-15))
|
|
252
|
+
t_adj = math.copysign(_norm_isf(p_adj / 2.0), observed_sharpe_per_period)
|
|
253
|
+
sr_adj = t_adj / math.sqrt(n_obs)
|
|
254
|
+
haircut = (
|
|
255
|
+
1.0 - sr_adj / observed_sharpe_per_period
|
|
256
|
+
if observed_sharpe_per_period
|
|
257
|
+
else float("nan")
|
|
258
|
+
)
|
|
259
|
+
return {
|
|
260
|
+
"method": method,
|
|
261
|
+
"observed_sharpe": observed_sharpe_per_period,
|
|
262
|
+
"adjusted_sharpe": sr_adj,
|
|
263
|
+
"haircut": haircut,
|
|
264
|
+
"observed_pvalue": p_obs,
|
|
265
|
+
"adjusted_pvalue": p_adj,
|
|
266
|
+
"n_tests": n_tests,
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# --------------------------------------------------------------------------- #
|
|
271
|
+
# Probability of Backtest Overfitting (CSCV, Bailey et al. 2017)
|
|
272
|
+
# --------------------------------------------------------------------------- #
|
|
273
|
+
def _sharpe_cols(block: np.ndarray) -> np.ndarray:
|
|
274
|
+
mean = np.nanmean(block, axis=0)
|
|
275
|
+
sd = np.nanstd(block, axis=0, ddof=1)
|
|
276
|
+
sd = np.where(sd == 0, np.nan, sd)
|
|
277
|
+
return mean / sd
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def probability_of_backtest_overfitting(
|
|
281
|
+
perf_matrix: np.ndarray,
|
|
282
|
+
n_blocks: int = 16,
|
|
283
|
+
) -> dict[str, Any]:
|
|
284
|
+
matrix = np.asarray(perf_matrix, dtype=float)
|
|
285
|
+
if matrix.ndim != 2:
|
|
286
|
+
raise ValueError("perf_matrix must be 2D (T x N)")
|
|
287
|
+
t_rows, n_strategies = matrix.shape
|
|
288
|
+
if n_strategies < 2:
|
|
289
|
+
raise ValueError("need at least 2 strategy configurations for PBO")
|
|
290
|
+
if n_blocks % 2 != 0:
|
|
291
|
+
raise ValueError("n_blocks must be even")
|
|
292
|
+
if n_blocks > t_rows:
|
|
293
|
+
raise ValueError("n_blocks cannot exceed observations")
|
|
294
|
+
|
|
295
|
+
block_idx = np.array_split(np.arange(t_rows), n_blocks)
|
|
296
|
+
blocks = list(range(n_blocks))
|
|
297
|
+
logits: list[float] = []
|
|
298
|
+
oos_ranks: list[float] = []
|
|
299
|
+
|
|
300
|
+
for is_blocks in combinations(blocks, n_blocks // 2):
|
|
301
|
+
is_set = set(is_blocks)
|
|
302
|
+
is_rows = np.concatenate([block_idx[b] for b in blocks if b in is_set])
|
|
303
|
+
oos_rows = np.concatenate([block_idx[b] for b in blocks if b not in is_set])
|
|
304
|
+
is_perf = _sharpe_cols(matrix[is_rows])
|
|
305
|
+
oos_perf = _sharpe_cols(matrix[oos_rows])
|
|
306
|
+
if np.all(np.isnan(is_perf)):
|
|
307
|
+
continue
|
|
308
|
+
n_star = int(np.nanargmax(is_perf))
|
|
309
|
+
valid = ~np.isnan(oos_perf)
|
|
310
|
+
rank = float(np.sum(oos_perf[valid] <= oos_perf[n_star]))
|
|
311
|
+
w = rank / (float(np.sum(valid)) + 1.0)
|
|
312
|
+
w = min(max(w, 1e-6), 1.0 - 1e-6)
|
|
313
|
+
logits.append(float(np.log(w / (1.0 - w))))
|
|
314
|
+
oos_ranks.append(w)
|
|
315
|
+
|
|
316
|
+
pbo = (
|
|
317
|
+
float(np.mean([1.0 if lam <= 0 else 0.0 for lam in logits]))
|
|
318
|
+
if logits
|
|
319
|
+
else float("nan")
|
|
320
|
+
)
|
|
321
|
+
return {
|
|
322
|
+
"pbo": pbo,
|
|
323
|
+
"n_splits": len(logits),
|
|
324
|
+
"n_strategies": n_strategies,
|
|
325
|
+
"n_blocks": n_blocks,
|
|
326
|
+
"median_logit": float(np.median(logits)) if logits else float("nan"),
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
# --------------------------------------------------------------------------- #
|
|
331
|
+
# Report builder
|
|
332
|
+
# --------------------------------------------------------------------------- #
|
|
333
|
+
def build_overfit_report(
|
|
334
|
+
selected_returns: np.ndarray,
|
|
335
|
+
n_trials: int,
|
|
336
|
+
*,
|
|
337
|
+
trials_matrix: np.ndarray | None = None,
|
|
338
|
+
periods_per_year: int = 252,
|
|
339
|
+
dsr_threshold: float = DSR_THRESHOLD,
|
|
340
|
+
pbo_threshold: float = PBO_THRESHOLD,
|
|
341
|
+
n_blocks: int = 16,
|
|
342
|
+
haircut_method: str = "bonferroni",
|
|
343
|
+
) -> dict[str, Any]:
|
|
344
|
+
r = np.asarray(selected_returns, dtype=float)
|
|
345
|
+
r = r[~np.isnan(r)]
|
|
346
|
+
ann = math.sqrt(periods_per_year)
|
|
347
|
+
sr_pp = sharpe_ratio(r)
|
|
348
|
+
skew, kurt = skew_kurt(r)
|
|
349
|
+
|
|
350
|
+
all_trial_sharpes = None
|
|
351
|
+
if trials_matrix is not None:
|
|
352
|
+
tm = np.asarray(trials_matrix, dtype=float)
|
|
353
|
+
all_trial_sharpes = [
|
|
354
|
+
sharpe_ratio(tm[:, column]) for column in range(tm.shape[1])
|
|
355
|
+
]
|
|
356
|
+
all_trial_sharpes = [value for value in all_trial_sharpes if not math.isnan(value)]
|
|
357
|
+
n_trials = max(n_trials, len(all_trial_sharpes))
|
|
358
|
+
|
|
359
|
+
dsr = deflated_sharpe_ratio(
|
|
360
|
+
r,
|
|
361
|
+
n_trials,
|
|
362
|
+
all_trial_sharpes=all_trial_sharpes,
|
|
363
|
+
threshold=dsr_threshold,
|
|
364
|
+
)
|
|
365
|
+
hc = haircut_sharpe(sr_pp, r.size, n_trials, method=haircut_method)
|
|
366
|
+
mintrl = minimum_track_record_length(sr_pp, 0.0, skew, kurt)
|
|
367
|
+
|
|
368
|
+
pbo_block = None
|
|
369
|
+
if trials_matrix is not None and np.asarray(trials_matrix).shape[1] >= 2:
|
|
370
|
+
pbo_block = probability_of_backtest_overfitting(
|
|
371
|
+
np.asarray(trials_matrix, dtype=float),
|
|
372
|
+
n_blocks=n_blocks,
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
flags: list[str] = []
|
|
376
|
+
if dsr["deflated_sharpe_ratio"] < dsr_threshold:
|
|
377
|
+
flags.append(f"DSR {dsr['deflated_sharpe_ratio']:.2f} < {dsr_threshold}")
|
|
378
|
+
if pbo_block is not None and pbo_block["pbo"] > pbo_threshold:
|
|
379
|
+
flags.append(f"PBO {pbo_block['pbo']:.2f} > {pbo_threshold}")
|
|
380
|
+
if hc["adjusted_sharpe"] * ann < 0.5:
|
|
381
|
+
flags.append(f"haircut Sharpe {hc['adjusted_sharpe'] * ann:.2f} < 0.5")
|
|
382
|
+
if mintrl > r.size:
|
|
383
|
+
flags.append(f"MinTRL {mintrl:.0f} > sample {r.size}")
|
|
384
|
+
|
|
385
|
+
passed = len(flags) == 0
|
|
386
|
+
verdict = (
|
|
387
|
+
"PASS - survives multiple-testing correction"
|
|
388
|
+
if passed
|
|
389
|
+
else "FAIL - likely overfit / selection-biased: " + "; ".join(flags)
|
|
390
|
+
)
|
|
391
|
+
return {
|
|
392
|
+
"verdict": verdict,
|
|
393
|
+
"passed": passed,
|
|
394
|
+
"observed_sharpe_annual": round(sr_pp * ann, 4),
|
|
395
|
+
"skew": round(skew, 4),
|
|
396
|
+
"kurtosis": round(kurt, 4),
|
|
397
|
+
"n_obs": int(r.size),
|
|
398
|
+
"n_trials": int(n_trials),
|
|
399
|
+
"deflated_sharpe_ratio": round(dsr["deflated_sharpe_ratio"], 4),
|
|
400
|
+
"deflation_benchmark_sr0_annual": round(dsr["deflated_benchmark_sr0"] * ann, 4),
|
|
401
|
+
"psr_vs_zero": round(dsr["psr_vs_zero"], 4),
|
|
402
|
+
"haircut": {
|
|
403
|
+
"method": hc["method"],
|
|
404
|
+
"adjusted_sharpe_annual": round(hc["adjusted_sharpe"] * ann, 4),
|
|
405
|
+
"haircut_pct": round(hc["haircut"], 4),
|
|
406
|
+
"observed_pvalue": hc["observed_pvalue"],
|
|
407
|
+
"adjusted_pvalue": hc["adjusted_pvalue"],
|
|
408
|
+
},
|
|
409
|
+
"minimum_track_record_length": round(mintrl, 1),
|
|
410
|
+
"pbo": pbo_block,
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def render_report_text(report: dict[str, Any]) -> str:
|
|
415
|
+
lines = [
|
|
416
|
+
"=" * 64,
|
|
417
|
+
" BACKTEST OVERFITTING REPORT",
|
|
418
|
+
"=" * 64,
|
|
419
|
+
f" Verdict : {report['verdict']}",
|
|
420
|
+
"-" * 64,
|
|
421
|
+
f" Observed Sharpe (annual) : {report['observed_sharpe_annual']}",
|
|
422
|
+
f" Trials (multiple tests) : {report['n_trials']}",
|
|
423
|
+
f" Observations : {report['n_obs']}",
|
|
424
|
+
f" Skew / Kurtosis : {report['skew']} / {report['kurtosis']}",
|
|
425
|
+
"-" * 64,
|
|
426
|
+
f" Deflated Sharpe Ratio : {report['deflated_sharpe_ratio']} "
|
|
427
|
+
f"(benchmark SR0 {report['deflation_benchmark_sr0_annual']} ann.)",
|
|
428
|
+
f" PSR vs 0 : {report['psr_vs_zero']}",
|
|
429
|
+
f" Haircut Sharpe ({report['haircut']['method']}): "
|
|
430
|
+
f"{report['haircut']['adjusted_sharpe_annual']} "
|
|
431
|
+
f"(-{report['haircut']['haircut_pct']:.0%})",
|
|
432
|
+
f" Min Track Record Length : {report['minimum_track_record_length']} obs",
|
|
433
|
+
]
|
|
434
|
+
if report["pbo"] is not None:
|
|
435
|
+
lines.append(
|
|
436
|
+
f" PBO : {report['pbo']['pbo']} "
|
|
437
|
+
f"({report['pbo']['n_splits']} CSCV splits)"
|
|
438
|
+
)
|
|
439
|
+
lines.append("=" * 64)
|
|
440
|
+
return "\n".join(lines)
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def build_check_artifact(
|
|
444
|
+
*,
|
|
445
|
+
name: str,
|
|
446
|
+
source: str,
|
|
447
|
+
selected_returns: np.ndarray,
|
|
448
|
+
n_trials: int,
|
|
449
|
+
trials_matrix: np.ndarray | None = None,
|
|
450
|
+
periods_per_year: int = 252,
|
|
451
|
+
n_blocks: int = 16,
|
|
452
|
+
returns_meta: dict[str, Any] | None = None,
|
|
453
|
+
trials_meta: dict[str, Any] | None = None,
|
|
454
|
+
generated_at: str | None = None,
|
|
455
|
+
) -> dict[str, Any]:
|
|
456
|
+
"""Provenance-wrapped check artifact (schema ``factor_qc.check.v1``)."""
|
|
457
|
+
report = build_overfit_report(
|
|
458
|
+
selected_returns,
|
|
459
|
+
n_trials,
|
|
460
|
+
trials_matrix=trials_matrix,
|
|
461
|
+
periods_per_year=periods_per_year,
|
|
462
|
+
n_blocks=n_blocks,
|
|
463
|
+
)
|
|
464
|
+
artifact: dict[str, Any] = {
|
|
465
|
+
"schema_version": "factor_qc.check.v1",
|
|
466
|
+
"generated_at": generated_at
|
|
467
|
+
or datetime.now().astimezone().isoformat(timespec="seconds"),
|
|
468
|
+
"tool": "factor_qc.stats (independent engine, numpy only)",
|
|
469
|
+
"name": name,
|
|
470
|
+
"declared": {
|
|
471
|
+
"n_trials": n_trials,
|
|
472
|
+
"periods_per_year": periods_per_year,
|
|
473
|
+
"haircut_method": report["haircut"]["method"],
|
|
474
|
+
},
|
|
475
|
+
"inputs": {
|
|
476
|
+
"returns": returns_meta or {},
|
|
477
|
+
"trials": trials_meta,
|
|
478
|
+
},
|
|
479
|
+
"source": source,
|
|
480
|
+
"boundaries": {
|
|
481
|
+
"production_effect": False,
|
|
482
|
+
"changes_probability": False,
|
|
483
|
+
"allow_real_trade": False,
|
|
484
|
+
},
|
|
485
|
+
"report": report,
|
|
486
|
+
}
|
|
487
|
+
if trials_matrix is None and report.get("pbo") is None:
|
|
488
|
+
artifact["limitations"] = [
|
|
489
|
+
"trials matrix not provided; PBO not computed (report pbo=null) and "
|
|
490
|
+
"the DSR cross-trial variance degrades to a conservative "
|
|
491
|
+
"single-trial estimate."
|
|
492
|
+
]
|
|
493
|
+
return artifact
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: factor-qc
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Fail-closed quality gate for backtests: DSR, PBO, Harvey-Liu haircut and MinTRL in one numpy-only engine, with P0/P1/P2 severity grading.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Keywords: backtest,overfitting,deflated-sharpe,pbo,mintrl,factor,qc,quant
|
|
7
|
+
Classifier: Development Status :: 3 - Alpha
|
|
8
|
+
Classifier: Environment :: Console
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Financial and Insurance Industry
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Office/Business :: Financial :: Investment
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: numpy>=1.24
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# factor-qc
|
|
25
|
+
|
|
26
|
+
A **fail-closed quality gate** for backtests: one numpy-only engine covering
|
|
27
|
+
Deflated Sharpe Ratio, Probability of Backtest Overfitting (CSCV), the
|
|
28
|
+
Harvey-Liu multiple-testing haircut and Minimum Track Record Length —
|
|
29
|
+
graded P0/P1/P2, and **it refuses to judge a backtest that will not declare
|
|
30
|
+
how many configurations were tried**. Python 3.11+, one dependency
|
|
31
|
+
(`numpy`), Windows / Linux / macOS.
|
|
32
|
+
|
|
33
|
+
**Status:** v0.1 — alpha. The statistics are battle-tested inside a
|
|
34
|
+
production research pipeline and validated against published reference
|
|
35
|
+
values, but this standalone package is new: expect the CLI to shift before
|
|
36
|
+
v1.0.
|
|
37
|
+
|
|
38
|
+
## Why this exists
|
|
39
|
+
|
|
40
|
+
The standard story: you try 200 factor configurations, the best one shows a
|
|
41
|
+
Sharpe of 1.65, you feel great. The honest story: with 200 trials of pure
|
|
42
|
+
noise, *someone* is going to show a Sharpe of 1.65 — the expected maximum of
|
|
43
|
+
200 zero-true-SR trials — and it will not be your skill, it will be your
|
|
44
|
+
selection bias.
|
|
45
|
+
|
|
46
|
+
Most backtest tooling computes statistics and prints reports. `factor-qc`
|
|
47
|
+
is a **gate**: it decides, with graded severity, whether a candidate may
|
|
48
|
+
pass — and its default answer is *no*:
|
|
49
|
+
|
|
50
|
+
- **P0 (fatal)** — DSR below threshold, PBO above threshold, haircut Sharpe
|
|
51
|
+
below floor, MinTRL longer than the sample → the candidate must not pass.
|
|
52
|
+
- **P1 (warning)** — weak PSR vs zero, short sample, aggressive trial count
|
|
53
|
+
→ proceed with eyes open.
|
|
54
|
+
- **P2 (info)** — non-normal moments, tiny trial count → recorded, no action.
|
|
55
|
+
|
|
56
|
+
## Philosophy
|
|
57
|
+
|
|
58
|
+
**Honesty is the default; the gate is fail-closed.**
|
|
59
|
+
|
|
60
|
+
The one non-negotiable input is `n_trials`: the honest count of
|
|
61
|
+
configurations you tried. Without it there is no deflation benchmark
|
|
62
|
+
([Bailey & López de Prado 2014](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2460551)),
|
|
63
|
+
no haircut ([Harvey, Liu & Zhu 2016, RFS](https://doi.org/10.1093/rfs/hhv059)),
|
|
64
|
+
no track-record floor
|
|
65
|
+
([Bailey & López de Prado 2018, JPM](https://ideas.repec.org/a/rsk/journl/0journalpm-v44n5.html))
|
|
66
|
+
and no overfitting probability
|
|
67
|
+
([Bailey, Borwein, López de Prado & Zhu 2017, JCF](https://escholarship.org/uc/item/4w1110bb)).
|
|
68
|
+
Refuse to declare, and the gate refuses to judge — that asymmetry is the
|
|
69
|
+
point. `qc check` exits non-zero on any P0 failure, so it drops into CI,
|
|
70
|
+
pre-commit hooks and research gates as a hard blocker, not a suggestion.
|
|
71
|
+
|
|
72
|
+
Two design commitments that keep it honest:
|
|
73
|
+
|
|
74
|
+
1. **numpy only, no scipy** — the standard-normal inverse CDF is Acklam's
|
|
75
|
+
rational approximation with one Newton refinement; every number in the
|
|
76
|
+
report is reproducible from the code in this repo, no hidden black box.
|
|
77
|
+
2. **PBO is optional but explicit** — without a trials matrix the gate says
|
|
78
|
+
PBO *was not computed*, and the DSR cross-trial variance degrades to a
|
|
79
|
+
conservative single-trial estimate. Absence of evidence is reported as
|
|
80
|
+
absence, never as evidence.
|
|
81
|
+
|
|
82
|
+
## Quick start
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
# install from PyPI (once published)
|
|
86
|
+
pip install factor-qc
|
|
87
|
+
|
|
88
|
+
# or run without installing anything:
|
|
89
|
+
# PYTHONPATH=src python -m factor_qc --help
|
|
90
|
+
|
|
91
|
+
python examples/demo.py # try it on reproducible synthetic cases
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Your own backtest:
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
# returns.json = JSON list of per-period returns of the selected candidate
|
|
98
|
+
# trials.json = JSON 2D matrix (T x N) of every configuration you tried
|
|
99
|
+
|
|
100
|
+
qc check --returns returns.json --trials trials.json --n-trials 200
|
|
101
|
+
# -> FAIL - P0 blocker(s): dsr: 0.63 vs 0.95; mintrl: 250.8 vs <= 1000; ...
|
|
102
|
+
|
|
103
|
+
qc check --returns returns.json --n-trials 5 --json # machine-readable
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Exit codes: `0` = no P0 failures (P1/P2 may still be failing), `1` = at
|
|
107
|
+
least one P0 failure (or missing `n_trials`), `2` = usage error. Wire it
|
|
108
|
+
into CI as a hard gate.
|
|
109
|
+
|
|
110
|
+
## Commands
|
|
111
|
+
|
|
112
|
+
| Command | What it does |
|
|
113
|
+
| --- | --- |
|
|
114
|
+
| `check` | Run the gate: DSR, PBO (when `--trials` given), haircut Sharpe, MinTRL as P0; PSR-vs-zero, sample length, trial aggression as P1; moments and trial count as P2. Human-readable or `--json` output |
|
|
115
|
+
| `version` | Print version |
|
|
116
|
+
|
|
117
|
+
Flags: `--returns` (required), `--trials` (optional), `--n-trials`
|
|
118
|
+
(required unless `require_declared_trials` is disabled in code),
|
|
119
|
+
`--periods-per-year` (default 252), `--n-blocks` (CSCV granularity, default
|
|
120
|
+
16).
|
|
121
|
+
|
|
122
|
+
## The checks
|
|
123
|
+
|
|
124
|
+
| Check | Severity | Method | Reference |
|
|
125
|
+
| --- | --- | --- | --- |
|
|
126
|
+
| `dsr` | P0 | Deflated Sharpe Ratio: P(SR > E[max SR of N trials]) under non-normal moments | [Bailey & López de Prado (2014), JPM 40(5)](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2460551) |
|
|
127
|
+
| `pbo` | P0 | Probability of Backtest Overfitting via Combinatorially-Symmetric Cross-Validation (12,870 splits at n_blocks=16) | [Bailey, Borwein, López de Prado & Zhu (2017), JCF](https://escholarship.org/uc/item/4w1110bb) |
|
|
128
|
+
| `haircut_sharpe` | P0 | Multiple-testing haircut of the Sharpe ratio (Bonferroni/Holm/BHY) | [Harvey & Liu (2015)](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2528780) |
|
|
129
|
+
| `mintrl` | P0 | Minimum Track Record Length: observations needed before SR is significant | [Bailey & López de Prado (2018), JPM 44(5)](https://ideas.repec.org/a/rsk/journl/0journalpm-v44n5.html) |
|
|
130
|
+
| `psr_vs_zero` | P1 | Probabilistic Sharpe vs zero | [Bailey & López de Prado (2012)](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2168747) |
|
|
131
|
+
| `sample_length` | P1 | ≥ 252 observations | — |
|
|
132
|
+
| `trial_aggression` | P1 | n_trials ≤ n_obs / 5 | — |
|
|
133
|
+
| `return_moments` | P2 | skew ≈ 0, kurtosis ≈ 3 | — |
|
|
134
|
+
| `trial_count` | P2 | n_trials ≥ 5 | — |
|
|
135
|
+
|
|
136
|
+
The P0 set mirrors the spirit of [Harvey, Liu & Zhu (2016),
|
|
137
|
+
"…and the Cross-Section of Expected Returns"](https://doi.org/10.1093/rfs/hhv059):
|
|
138
|
+
a factor must survive multiple-testing correction to earn the right to be
|
|
139
|
+
called a factor. The gate is the machine version of that editorial stance.
|
|
140
|
+
|
|
141
|
+
## Performance note
|
|
142
|
+
|
|
143
|
+
CSCV enumerates C(n_blocks, n_blocks/2) splits — 12,870 at the default 16.
|
|
144
|
+
On large trial matrices (T=1000, N=200) that takes minutes; use
|
|
145
|
+
`--n-blocks 8` (70 splits) or `10` (252 splits) for interactive speed at
|
|
146
|
+
slightly coarser granularity.
|
|
147
|
+
|
|
148
|
+
## Development
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
python -m pip install -e . pytest
|
|
152
|
+
python -m pytest
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
CI runs the full test suite on Ubuntu, Windows and macOS with Python 3.11
|
|
156
|
+
and 3.12. Issues are handled on weekends; pull requests are welcome.
|
|
157
|
+
|
|
158
|
+
## Related work
|
|
159
|
+
|
|
160
|
+
- [Bailey & López de Prado (2014), The Deflated Sharpe Ratio](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2460551)
|
|
161
|
+
- [Bailey, Borwein, López de Prado & Zhu (2017), The Probability of Backtest Overfitting](https://escholarship.org/uc/item/4w1110bb)
|
|
162
|
+
- [Harvey, Liu & Zhu (2016), …and the Cross-Section of Expected Returns (RFS)](https://doi.org/10.1093/rfs/hhv059)
|
|
163
|
+
- [Harvey & Liu (2021), Lucky Factors (JFE)](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=2528780)
|
|
164
|
+
- [Mobarekeh & López de Prado (2024), Backtest Overfitting in the Machine Learning Era (SSRN 4778909)](https://papers.ssrn.com/sol3/papers.cfm?abstract_id=4778909) — why OOS methods still need honest trial accounting
|
|
165
|
+
|
|
166
|
+
## License
|
|
167
|
+
|
|
168
|
+
MIT
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
factor_qc/__init__.py,sha256=DYAYqtljdvNsFaW7eLSKYmNqv0ZM5_fduNdIxuIQJaM,908
|
|
2
|
+
factor_qc/__main__.py,sha256=MHKZ_ae3fSLGTLUUMOx15fWdeOnJSHhq-zslRP5F5Lc,79
|
|
3
|
+
factor_qc/cli.py,sha256=FR59UdOvGSLySdqsZwmz07C_D5CgGLB8T6E35QwrHPA,3704
|
|
4
|
+
factor_qc/gate.py,sha256=IMWflYb8ieQ5Els3hf0kyvRV-vtNmGso5Kp9B4S_8nI,6966
|
|
5
|
+
factor_qc/stats.py,sha256=qxNhUpGJvEiAOyUOCrGFEBzyTg-ZhtYEtW2_-LLqDes,17250
|
|
6
|
+
factor_qc-0.1.0.dist-info/licenses/LICENSE,sha256=rk_db6ozKbaQzFNI1YAoOBuzKn5b7YsYH53CRpYHXAU,1079
|
|
7
|
+
factor_qc-0.1.0.dist-info/METADATA,sha256=p6LtsfDqqxWdEpBIlDN8PUkPY7C7GUILOl3Vf4JWxnA,8144
|
|
8
|
+
factor_qc-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
9
|
+
factor_qc-0.1.0.dist-info/entry_points.txt,sha256=P3KEI5brT2meq1qKSWLEPeL7lZOHqhwzWab4hAkWSiM,73
|
|
10
|
+
factor_qc-0.1.0.dist-info/top_level.txt,sha256=YdSiclAtCi5Oth-kuaO0AMi4YZVjzISbQD55j0OpvL4,10
|
|
11
|
+
factor_qc-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Factor QC contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
factor_qc
|