canli-validation-mcp 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -12
- package/package.json +6 -3
- package/src/local/api/_lib/limits.js +30 -0
- package/src/local/js/breadth-core.js +96 -0
- package/src/local/js/dsr-core.js +302 -0
- package/src/local/js/haircut-core.js +96 -0
- package/src/local/js/luck-core.js +97 -0
- package/src/local/js/moments-core.js +54 -0
- package/src/local/js/paper-evidence-core.js +167 -0
- package/src/local/js/pbo-core.js +112 -0
- package/src/local/js/receipt-statement.js +54 -0
- package/src/local/js/selection-risk-core.js +197 -0
- package/src/local/js/student-t.js +139 -0
- package/src/local/js/validate/backtest-length.js +61 -0
- package/src/local/js/validate/breadth.js +24 -0
- package/src/local/js/validate/deflated-sharpe.js +28 -0
- package/src/local/js/validate/haircut-sharpe.js +54 -0
- package/src/local/js/validate/luck-trials.js +63 -0
- package/src/local/js/validate/overfitting.js +29 -0
- package/src/local/js/validate/paper-evidence.js +11 -0
- package/src/local/js/validate/track-record.js +50 -0
- package/src/local/scripts/canonical-json.mjs +176 -0
- package/src/local/standards/paper-evidence/schema.json +429 -0
- package/src/local.mjs +45 -0
- package/src/receipt-keys.json +13 -0
- package/src/schemas.mjs +195 -51
- package/src/series-file.mjs +124 -0
- package/src/server.mjs +349 -10
package/README.md
CHANGED
|
@@ -1,12 +1,18 @@
|
|
|
1
1
|
# canli-validation-mcp
|
|
2
2
|
|
|
3
|
+
[](https://www.npmjs.com/package/canli-validation-mcp)
|
|
4
|
+
[](https://scorecard.dev/viewer/?uri=github.com/arhancanli/canli-validation-mcp)
|
|
5
|
+
[](https://glama.ai/mcp/servers/arhancanli/canli-validation-mcp)
|
|
6
|
+
|
|
3
7
|
An MCP (Model Context Protocol) server over canlicapital.com's free, keyed validation API. It
|
|
4
|
-
gives a coding agent
|
|
5
|
-
CSCV overfitting, paper-evidence conformance, breadth ceiling, minimum track record length
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
8
|
+
gives a coding agent fourteen tools: issue a free key, run the eight validators (deflated Sharpe,
|
|
9
|
+
CSCV overfitting, paper-evidence conformance, breadth ceiling, minimum track record length,
|
|
10
|
+
minimum backtest length, haircut Sharpe ratio, luck-equivalent trials),
|
|
11
|
+
audit one backtest with three of them in a single call, fetch a stored receipt, verify a
|
|
12
|
+
receipt's signature offline, read service status, and read a company's reported financial history
|
|
13
|
+
from SEC filings. Every validation result carries, beside the number, the sentences that say what
|
|
14
|
+
it does not establish and the receipt that records it, success or error, so the agent cannot see a
|
|
15
|
+
number without its limits.
|
|
10
16
|
|
|
11
17
|
This package is published to npm as [`canli-validation-mcp`](https://www.npmjs.com/package/canli-validation-mcp).
|
|
12
18
|
Also listed on the official MCP Registry (`io.github.arhancanli/canli-validation-mcp`) and
|
|
@@ -16,11 +22,18 @@ test this package itself; see "Local checkout" near the bottom.
|
|
|
16
22
|
This README describes the version in `package.json`. Unversioned `npx` runs npm's latest
|
|
17
23
|
release; `npx -y canli-validation-mcp@<version>` pins one.
|
|
18
24
|
|
|
25
|
+
## How well agents use it
|
|
26
|
+
|
|
27
|
+
A fixed benchmark gives models these tools and scores whether they pick the right one and return
|
|
28
|
+
the right answer, against ground truth computed from the same checked code. Results for three
|
|
29
|
+
models, with every run recorded, are in
|
|
30
|
+
[`bench/agent/README.md`](https://github.com/arhancanli/canlicapital/blob/main/mcp/bench/agent/README.md).
|
|
31
|
+
|
|
19
32
|
## What the API is (and is not)
|
|
20
33
|
|
|
21
34
|
The engine is the product. The service runs your submitted numbers through the same honesty
|
|
22
35
|
arithmetic canlicapital.com's own paper record runs on itself and hands back a verdict anyone can
|
|
23
|
-
recompute from the receipt. It does not accept market data, does not
|
|
36
|
+
recompute from the receipt. It signs every receipt, does not accept market data, does not
|
|
24
37
|
grade a strategy, and never saw your data source, its costs, or any lookahead in how a series was
|
|
25
38
|
built. See `docs/superpowers/specs/2026-09-05-developer-key-validation-api-design.md` in the main
|
|
26
39
|
repository for the full design.
|
|
@@ -35,9 +48,14 @@ repository for the full design.
|
|
|
35
48
|
| `validate_paper_evidence` | `POST /api/v1/validate/paper-evidence` | yes |
|
|
36
49
|
| `validate_breadth` | `POST /api/v1/validate/breadth` | yes |
|
|
37
50
|
| `validate_track_record` | `POST /api/v1/validate/track-record` | yes |
|
|
51
|
+
| `validate_backtest_length` | `POST /api/v1/validate/backtest-length` | yes |
|
|
52
|
+
| `validate_haircut_sharpe` | `POST /api/v1/validate/haircut-sharpe` | yes |
|
|
53
|
+
| `validate_luck_trials` | `POST /api/v1/validate/luck-trials` | yes |
|
|
54
|
+
| `audit_backtest` | the deflated Sharpe, track record and, with `variants`, overfitting routes, one validation each | yes |
|
|
55
|
+
| `verify_receipt` | `GET /api/v1/receipts/{id}` when given an id; the checks run locally | no |
|
|
38
56
|
| `get_receipt` | `GET /api/v1/receipts/{id}` | no |
|
|
39
57
|
| `service_status` | `GET /api/v1/validate/status` | no |
|
|
40
|
-
| `company_financial_history` | `GET /company-data/{cik}.json` | no |
|
|
58
|
+
| `company_financial_history` | `GET /company-data/{cik}.json` (a ticker resolves through `GET /api/v1/company-tickers.json`) | no |
|
|
41
59
|
|
|
42
60
|
`validate_deflated_sharpe` accepts exactly one of two input shapes, never a mix of both:
|
|
43
61
|
|
|
@@ -59,6 +77,32 @@ original SEC response and the record's own boundary sentence: these are accounti
|
|
|
59
77
|
reported to the SEC, not market prices, returns or a recommendation. Companies and concepts
|
|
60
78
|
outside the current release return an error with the available concepts listed.
|
|
61
79
|
|
|
80
|
+
## Auditing a backtest in one call
|
|
81
|
+
|
|
82
|
+
`audit_backtest` takes one strategy's return series, the number of variants tried and their Sharpe
|
|
83
|
+
dispersion, and optionally every variant's returns. It runs `validate_deflated_sharpe` on the
|
|
84
|
+
series, then `validate_track_record` on the Sharpe, skew and kurtosis that check derived, then,
|
|
85
|
+
with `variants`, `validate_overfitting`. Each check is exactly what its own tool returns, with its
|
|
86
|
+
own receipt; the boundary sentences they share are stated once. A check that refuses (a Sharpe
|
|
87
|
+
that cannot beat the benchmark has no minimum track record) is reported as that check's error.
|
|
88
|
+
The audit adds no grade of its own. Through the API it uses one validation per check.
|
|
89
|
+
|
|
90
|
+
When the server runs on your machine, `returns_file` and `variants_file` take the path of the
|
|
91
|
+
backtest's output instead of the numbers: a CSV (comma, semicolon or tab separated, with or without
|
|
92
|
+
a header; date and label columns are ignored, an unnamed or counting index column is skipped and
|
|
93
|
+
reported) or a JSON array. `returns_column` picks the column when there are several. Only the
|
|
94
|
+
numbers are read; the hosted endpoint refuses file paths. An agent copying a long series into a
|
|
95
|
+
call can drop values, and the copy costs tokens; in our agent benchmark, reading the file instead
|
|
96
|
+
took three audit questions from 4 of 9 to 8 of 9 answered correctly on gpt-5.4-mini.
|
|
97
|
+
|
|
98
|
+
## Prompts, resources and structured results
|
|
99
|
+
|
|
100
|
+
Clients that show MCP prompts offer two guided workflows: `validate_backtest` (deflated Sharpe, then
|
|
101
|
+
overfitting, then the track record needed, reported with what each number does not establish) and
|
|
102
|
+
`track_record_needed`. Two resources can be read: `canli://limits`, the boundary sentences every
|
|
103
|
+
result carries, and `canli://sources`, the papers behind each validator and how each is checked
|
|
104
|
+
against them. Every tool result carries its envelope both as text and as `structuredContent`.
|
|
105
|
+
|
|
62
106
|
## Compact context (0.3.0)
|
|
63
107
|
|
|
64
108
|
An agent pays for every token a tool returns, including whitespace it never reads. Since 0.3.0
|
|
@@ -88,8 +132,10 @@ an array of objects.
|
|
|
88
132
|
|---|---|---|
|
|
89
133
|
| `CANLI_API_BASE` | `https://canlicapital.com` | Where the API lives. Point it at a preview deployment for testing. |
|
|
90
134
|
| `CANLI_KEY` | unset | A key already issued from `POST /api/v1/keys`. When set, `get_key` sends no request and reports the key is already configured; every other tool sends it as `Authorization: Bearer <key>`. |
|
|
135
|
+
| `CANLI_FULL_ENVELOPE` | unset | `1` or `true` returns each validation's full API envelope instead of the compact result (below). |
|
|
136
|
+
| `CANLI_LOCAL` | unset | `1` or `true` runs the eight validators on this machine (private local mode, below): no key, no network, no receipt. |
|
|
91
137
|
|
|
92
|
-
If `CANLI_KEY` is not set, call `get_key` once per session before the
|
|
138
|
+
If `CANLI_KEY` is not set and local mode is off, call `get_key` once per session before the validators. The key it
|
|
93
139
|
returns lives only in this process's memory for the life of the session; it is not written to
|
|
94
140
|
disk.
|
|
95
141
|
|
|
@@ -154,6 +200,20 @@ claude mcp add canli -- npx -y canli-validation-mcp
|
|
|
154
200
|
|
|
155
201
|
Run `claude mcp list` to confirm it is registered, and `claude mcp remove canli` to remove it.
|
|
156
202
|
|
|
203
|
+
## Private local mode
|
|
204
|
+
|
|
205
|
+
Set `CANLI_LOCAL=1` and the eight validators run on your machine: nothing about the series you
|
|
206
|
+
submit is sent to canlicapital.com, no key is needed, and no receipt is stored. The computation is
|
|
207
|
+
the API's own, shipped byte for byte in `src/local` (a test fails if it drifts), so a local result
|
|
208
|
+
equals the hosted one; it names no receipt id because none was made.
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
claude mcp add canli-local --env CANLI_LOCAL=1 -- npx -y canli-validation-mcp
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
`get_receipt`, `service_status` and `company_financial_history` still read from canlicapital.com;
|
|
215
|
+
they send no series. In the Claude Desktop extension this is the "Private local mode" setting.
|
|
216
|
+
|
|
157
217
|
## Generic stdio client
|
|
158
218
|
|
|
159
219
|
Any MCP client that can spawn a process and speak stdio will work. Using the official SDK
|
|
@@ -188,7 +248,7 @@ const result = await client.callTool({
|
|
|
188
248
|
cross_trial_sharpe_sd_annualized: 0.5,
|
|
189
249
|
},
|
|
190
250
|
});
|
|
191
|
-
console.log(result.content[0].text); // the
|
|
251
|
+
console.log(result.content[0].text); // the answer, its limits and its receipt
|
|
192
252
|
|
|
193
253
|
await client.close();
|
|
194
254
|
```
|
|
@@ -208,8 +268,16 @@ Every envelope this server returns carries these sentences, verbatim, from the A
|
|
|
208
268
|
request, 20000 observations per series, 200 variants per matrix.
|
|
209
269
|
|
|
210
270
|
Each tool's description also states one of these sentences, so an agent sees the boundary before
|
|
211
|
-
it calls the tool, not only after.
|
|
212
|
-
|
|
271
|
+
it calls the tool, not only after.
|
|
272
|
+
|
|
273
|
+
## Compact results (tokens)
|
|
274
|
+
|
|
275
|
+
A validation result is the answer (`data`), the sentences above except the quota line, and the
|
|
276
|
+
receipt's id and URL; an error keeps its error. The rest of the API envelope (schema, endpoint,
|
|
277
|
+
timestamps, claim and capital class, the human page, the source-file hashes and the quota line)
|
|
278
|
+
describes the service rather than the answer, and an agent pays for every token of it on every
|
|
279
|
+
call. It stays in the stored receipt, which `get_receipt` returns in full, and in `service_status`.
|
|
280
|
+
On a breadth result this is about half the text. Set `CANLI_FULL_ENVELOPE=1` to receive every field.
|
|
213
281
|
|
|
214
282
|
## Local checkout
|
|
215
283
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "canli-validation-mcp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "MCP server for canlicapital.com's free validation API: deflated Sharpe, CSCV overfitting, paper-evidence conformance, and breadth ceiling, each returned as the full API envelope so the boundary language cannot be dropped.",
|
|
5
5
|
"private": false,
|
|
6
6
|
"type": "module",
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
"canlicapital-validation-mcp": "src/server.mjs"
|
|
10
10
|
},
|
|
11
11
|
"engines": {
|
|
12
|
-
"node": ">=20"
|
|
12
|
+
"node": ">=20.10"
|
|
13
13
|
},
|
|
14
14
|
"files": [
|
|
15
15
|
"src",
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
},
|
|
22
22
|
"dependencies": {
|
|
23
23
|
"@modelcontextprotocol/sdk": "1.30.0",
|
|
24
|
-
"zod": "4.5
|
|
24
|
+
"zod": "4.6.5"
|
|
25
25
|
},
|
|
26
26
|
"mcpName": "io.github.arhancanli/canli-validation-mcp",
|
|
27
27
|
"repository": {
|
|
@@ -32,5 +32,8 @@
|
|
|
32
32
|
"homepage": "https://canlicapital.com/developers",
|
|
33
33
|
"bugs": {
|
|
34
34
|
"url": "https://github.com/arhancanli/canlicapital/issues"
|
|
35
|
+
},
|
|
36
|
+
"devDependencies": {
|
|
37
|
+
"fast-check": "4.10.2"
|
|
35
38
|
}
|
|
36
39
|
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
// api/_lib/limits.js
|
|
2
|
+
// THE quota constants. Enforcement (handler.js), the /developers page, the OpenAPI document and
|
|
3
|
+
// public/glassbox/validation_api_limits.json all import or derive from this object, so the number a
|
|
4
|
+
// reader sees is the number the service enforces. Change a value here and nowhere else.
|
|
5
|
+
export const LIMITS = Object.freeze({
|
|
6
|
+
validations_per_key_per_day: 1000,
|
|
7
|
+
keys_per_client_per_day: 5,
|
|
8
|
+
max_key_revoke_body_bytes: 1024,
|
|
9
|
+
max_body_bytes: 1024 * 1024,
|
|
10
|
+
max_observations: 20000,
|
|
11
|
+
max_variants: 200,
|
|
12
|
+
max_cscv_combinations: 2000,
|
|
13
|
+
wall_time_seconds: 10,
|
|
14
|
+
});
|
|
15
|
+
|
|
16
|
+
export const LIMITS_TEXT = Object.freeze([
|
|
17
|
+
"This verdict is about the series exactly as submitted. The service never saw the data source, its costs, survivorship, or any lookahead in how the series was built.",
|
|
18
|
+
"A deflated Sharpe or overfitting probability above or below any threshold is not admission to anything and is not a forecast.",
|
|
19
|
+
"The receipt is content-hashed, reproducible from the open-source core it names, and signed with Ed25519 by a key published at https://canlicapital.com/.well-known/canli-receipt-keys.json.",
|
|
20
|
+
`Quotas: ${LIMITS.validations_per_key_per_day} validations per key per UTC day, ${LIMITS.keys_per_client_per_day} keys per client per UTC day, ${LIMITS.max_body_bytes} bytes per validation request, ${LIMITS.max_key_revoke_body_bytes} bytes per key revocation request, ${LIMITS.max_observations} observations per series, ${LIMITS.max_variants} variants per matrix.`,
|
|
21
|
+
]);
|
|
22
|
+
|
|
23
|
+
// What the page has never said. None of this is enforcement text: it is what happens to a key
|
|
24
|
+
// after it is issued, which a reader otherwise has to find out by trying it. Any numeral here is
|
|
25
|
+
// a LIMITS value, never a hand-typed one, so it cannot drift from what the service actually does.
|
|
26
|
+
export const KEY_LIFECYCLE_TEXT = Object.freeze([
|
|
27
|
+
"Keys do not expire once issued.",
|
|
28
|
+
"Revoke the bearer key with POST /api/v1/keys/revoke. Revocation does not use validation quota and is irreversible; repeating it leaves the original revocation time intact. Requests already admitted may finish. Confirm a successful response before assuming the key is disabled. Issue a replacement separately; there is no atomic rotate operation.",
|
|
29
|
+
`"Client", for the daily key-issuance quota, means the request's IP address hashed together with a salt that rotates every UTC day, not a stored account. A shared office or NAT IP address draws from the same pool of ${LIMITS.keys_per_client_per_day} keys a day as every other request behind it.`,
|
|
30
|
+
]);
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
// =============================================================================
|
|
2
|
+
// breadth-core.js
|
|
3
|
+
// -----------------------------------------------------------------------------
|
|
4
|
+
// The arithmetic behind /tools/breadth: what a book of N sleeves is worth, and
|
|
5
|
+
// why adding sleeves stops helping.
|
|
6
|
+
//
|
|
7
|
+
// For N equally weighted sleeves, each with per-period Sharpe s and identical
|
|
8
|
+
// pairwise correlation rho:
|
|
9
|
+
//
|
|
10
|
+
// portfolio mean = s * sigma
|
|
11
|
+
// portfolio variance = sigma^2 * (1 + (N-1) * rho) / N
|
|
12
|
+
// BOOK SHARPE = s * sqrt( N / (1 + (N-1) * rho) )
|
|
13
|
+
//
|
|
14
|
+
// The limit as N grows without bound is the part worth staring at:
|
|
15
|
+
//
|
|
16
|
+
// ceiling = s / sqrt(rho) for rho > 0
|
|
17
|
+
//
|
|
18
|
+
// It does not depend on N at all. Past a certain point, breadth is not the lever;
|
|
19
|
+
// correlation is. A project that answers a disappointing Sharpe by adding sleeves
|
|
20
|
+
// is working on the wrong number, and this file exists so that is visible rather
|
|
21
|
+
// than argued about.
|
|
22
|
+
//
|
|
23
|
+
// The assumptions are strong and stated everywhere they are used: equal weights,
|
|
24
|
+
// equal Sharpe, one shared pairwise correlation. Real books have none of those.
|
|
25
|
+
// The lab is for the SHAPE of the constraint, not for forecasting a book.
|
|
26
|
+
// =============================================================================
|
|
27
|
+
|
|
28
|
+
/** Book Sharpe for N equally weighted sleeves at shared correlation rho. */
|
|
29
|
+
export function bookSharpe({ sleeveSharpe, sleeves, correlation }) {
|
|
30
|
+
if (!Number.isInteger(sleeves) || sleeves < 1) throw new Error("sleeves must be a positive integer");
|
|
31
|
+
if (!Number.isFinite(sleeveSharpe)) throw new Error("sleeveSharpe must be finite");
|
|
32
|
+
if (!Number.isFinite(correlation)) throw new Error("correlation must be finite");
|
|
33
|
+
// The single condition, stated once. `1 + (N-1)*rho` IS the portfolio variance in
|
|
34
|
+
// units of a sleeve's variance, so it must be strictly positive:
|
|
35
|
+
//
|
|
36
|
+
// rho < -1/(N-1) the covariance matrix is not positive semidefinite and no
|
|
37
|
+
// set of real return series can produce it;
|
|
38
|
+
// rho == -1/(N-1) the matrix is PSD but singular, and the equally weighted
|
|
39
|
+
// portfolio has exactly zero variance, so its Sharpe is
|
|
40
|
+
// infinite. Mathematically real, financially a fantasy, and
|
|
41
|
+
// returning a spectacular number for it is precisely the
|
|
42
|
+
// failure mode this tool argues against.
|
|
43
|
+
//
|
|
44
|
+
// Both are refused, and the boundary case is named separately because a caller
|
|
45
|
+
// who lands on it exactly has done something interesting rather than careless.
|
|
46
|
+
const denominator = 1 + (sleeves - 1) * correlation;
|
|
47
|
+
if (denominator <= 0) {
|
|
48
|
+
const floor = sleeves > 1 ? -1 / (sleeves - 1) : -1;
|
|
49
|
+
throw new Error(
|
|
50
|
+
denominator === 0
|
|
51
|
+
? `a shared correlation of exactly ${correlation} across ${sleeves} sleeves gives the ` +
|
|
52
|
+
"equally weighted book zero variance and an infinite Sharpe. That is a degenerate " +
|
|
53
|
+
"case, not an opportunity."
|
|
54
|
+
: `a shared correlation of ${correlation} is impossible for ${sleeves} sleeves: below ` +
|
|
55
|
+
`${floor.toFixed(4)} the covariance matrix is not positive semidefinite`,
|
|
56
|
+
);
|
|
57
|
+
}
|
|
58
|
+
return sleeveSharpe * Math.sqrt(sleeves / denominator);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** The value no amount of breadth can exceed. Infinite only when rho <= 0. */
|
|
62
|
+
export function breadthCeiling({ sleeveSharpe, correlation }) {
|
|
63
|
+
if (correlation <= 0) return Number.POSITIVE_INFINITY;
|
|
64
|
+
return sleeveSharpe / Math.sqrt(correlation);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** The smallest N reaching `target`, or null when the ceiling forbids it. */
|
|
68
|
+
export function sleevesRequired({ sleeveSharpe, correlation, target, maxSleeves = 500 }) {
|
|
69
|
+
const ceiling = breadthCeiling({ sleeveSharpe, correlation });
|
|
70
|
+
if (target > ceiling) return { sleeves: null, ceiling, reachable: false };
|
|
71
|
+
for (let n = 1; n <= maxSleeves; n += 1) {
|
|
72
|
+
if (bookSharpe({ sleeveSharpe, sleeves: n, correlation }) >= target) {
|
|
73
|
+
return { sleeves: n, ceiling, reachable: true };
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return { sleeves: null, ceiling, reachable: false };
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** The curve of book Sharpe against N, for plotting. */
|
|
80
|
+
export function breadthCurve({ sleeveSharpe, correlation, maxSleeves }) {
|
|
81
|
+
const points = [];
|
|
82
|
+
for (let n = 1; n <= maxSleeves; n += 1) {
|
|
83
|
+
// Stop at the last N the correlation can actually support, rather than at the
|
|
84
|
+
// last one that does not throw: those differ by exactly the degenerate case.
|
|
85
|
+
if (1 + (n - 1) * correlation <= 0) break;
|
|
86
|
+
points.push({ sleeves: n, sharpe: bookSharpe({ sleeveSharpe, sleeves: n, correlation }) });
|
|
87
|
+
}
|
|
88
|
+
return points;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** How much of the distance to the ceiling N sleeves have actually captured. */
|
|
92
|
+
export function ceilingCaptured({ sleeveSharpe, sleeves, correlation }) {
|
|
93
|
+
const ceiling = breadthCeiling({ sleeveSharpe, correlation });
|
|
94
|
+
if (!Number.isFinite(ceiling)) return null;
|
|
95
|
+
return bookSharpe({ sleeveSharpe, sleeves, correlation }) / ceiling;
|
|
96
|
+
}
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
// Pure deflated-Sharpe arithmetic. No DOM: shared by /tools/deflated-sharpe and the
|
|
2
|
+
// validation API so the two can never disagree. Bound to
|
|
3
|
+
// public/glassbox/deflated_sharpe_calculator_contract.json by checkGoldenVectors.
|
|
4
|
+
const EULER_MASCHERONI = 0.5772156649;
|
|
5
|
+
|
|
6
|
+
// Complementary error function to full double precision (relative error at most 1.5e-14 for
|
|
7
|
+
// x >= -20 and 6e-14 to -38, measured against the C library erfc). For |x| < 1.5 it uses the series
|
|
8
|
+
// erf(x) = 2/sqrt(pi) exp(-x^2) sum 2^n x^(2n+1) / (1*3*...*(2n+1)), which has no cancellation;
|
|
9
|
+
// beyond that, the continued fraction erfc(x) = exp(-x^2)/sqrt(pi) / (x + 1/2 / (x + 1 / (x + ...)))
|
|
10
|
+
// evaluated by the modified Lentz method. It replaced the Abramowitz and Stegun 7.1.26
|
|
11
|
+
// approximation (absolute error up to 1.5e-7, no relative accuracy in the tails) on 2026-09-25.
|
|
12
|
+
function erfc(value) {
|
|
13
|
+
if (Number.isNaN(value)) return Number.NaN;
|
|
14
|
+
if (value < 0) return 2 - erfc(-value);
|
|
15
|
+
const x = value;
|
|
16
|
+
if (x < 1.5) {
|
|
17
|
+
let term = x;
|
|
18
|
+
let sum = x;
|
|
19
|
+
for (let n = 1; n < 200; n++) {
|
|
20
|
+
term *= (2 * x * x) / (2 * n + 1);
|
|
21
|
+
sum += term;
|
|
22
|
+
if (term < sum * 1e-17) break;
|
|
23
|
+
}
|
|
24
|
+
return 1 - (2 / Math.sqrt(Math.PI)) * Math.exp(-x * x) * sum;
|
|
25
|
+
}
|
|
26
|
+
if (x > 27.3) return 0;
|
|
27
|
+
// Lentz: f = b0 + a1/(b1 + a2/(b2 + ...)) with b_k = x, a_k = k/2.
|
|
28
|
+
const tiny = 1e-300;
|
|
29
|
+
let f = x;
|
|
30
|
+
let c = x;
|
|
31
|
+
let d = 0;
|
|
32
|
+
for (let k = 1; k < 500; k++) {
|
|
33
|
+
const a = k / 2;
|
|
34
|
+
d = x + a * d;
|
|
35
|
+
d = d === 0 ? tiny : d;
|
|
36
|
+
c = x + a / c;
|
|
37
|
+
c = c === 0 ? tiny : c;
|
|
38
|
+
d = 1 / d;
|
|
39
|
+
const delta = c * d;
|
|
40
|
+
f *= delta;
|
|
41
|
+
if (Math.abs(delta - 1) < 1e-16) break;
|
|
42
|
+
}
|
|
43
|
+
return Math.exp(-x * x) / (Math.sqrt(Math.PI) * f);
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export function normalCdf(value) {
|
|
47
|
+
if (value === Infinity) return 1;
|
|
48
|
+
if (value === -Infinity) return 0;
|
|
49
|
+
return 0.5 * erfc(-value / Math.SQRT2);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function acklamPpf(probability) {
|
|
53
|
+
if (!(probability > 0 && probability < 1)) {
|
|
54
|
+
throw new RangeError("Normal quantile probability must be between zero and one");
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const a = [
|
|
58
|
+
-3.969683028665376e1,
|
|
59
|
+
2.209460984245205e2,
|
|
60
|
+
-2.759285104469687e2,
|
|
61
|
+
1.38357751867269e2,
|
|
62
|
+
-3.066479806614716e1,
|
|
63
|
+
2.506628277459239,
|
|
64
|
+
];
|
|
65
|
+
const b = [
|
|
66
|
+
-5.447609879822406e1,
|
|
67
|
+
1.615858368580409e2,
|
|
68
|
+
-1.556989798598866e2,
|
|
69
|
+
6.680131188771972e1,
|
|
70
|
+
-1.328068155288572e1,
|
|
71
|
+
];
|
|
72
|
+
const c = [
|
|
73
|
+
-7.784894002430293e-3,
|
|
74
|
+
-3.223964580411365e-1,
|
|
75
|
+
-2.400758277161838,
|
|
76
|
+
-2.549732539343734,
|
|
77
|
+
4.374664141464968,
|
|
78
|
+
2.938163982698783,
|
|
79
|
+
];
|
|
80
|
+
const d = [
|
|
81
|
+
7.784695709041462e-3,
|
|
82
|
+
3.224671290700398e-1,
|
|
83
|
+
2.445134137142996,
|
|
84
|
+
3.754408661907416,
|
|
85
|
+
];
|
|
86
|
+
const low = 0.02425;
|
|
87
|
+
const high = 1 - low;
|
|
88
|
+
|
|
89
|
+
if (probability < low) {
|
|
90
|
+
const q = Math.sqrt(-2 * Math.log(probability));
|
|
91
|
+
return (
|
|
92
|
+
(((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) /
|
|
93
|
+
((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1)
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
if (probability <= high) {
|
|
97
|
+
const q = probability - 0.5;
|
|
98
|
+
const r = q * q;
|
|
99
|
+
return (
|
|
100
|
+
(((((a[0] * r + a[1]) * r + a[2]) * r + a[3]) * r + a[4]) * r + a[5]) * q /
|
|
101
|
+
(((((b[0] * r + b[1]) * r + b[2]) * r + b[3]) * r + b[4]) * r + 1)
|
|
102
|
+
);
|
|
103
|
+
}
|
|
104
|
+
const q = Math.sqrt(-2 * Math.log(1 - probability));
|
|
105
|
+
return -(
|
|
106
|
+
(((((c[0] * q + c[1]) * q + c[2]) * q + c[3]) * q + c[4]) * q + c[5]) /
|
|
107
|
+
((((d[0] * q + d[1]) * q + d[2]) * q + d[3]) * q + 1)
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Inverse normal CDF: Acklam's rational approximation (relative error near 1.15e-9) refined by two
|
|
112
|
+
// Halley steps against the full-precision normalCdf, which brings it to the precision of the CDF.
|
|
113
|
+
export function normalPpf(probability) {
|
|
114
|
+
let x = acklamPpf(probability);
|
|
115
|
+
for (let step = 0; step < 2; step++) {
|
|
116
|
+
const error = normalCdf(x) - probability;
|
|
117
|
+
const u = error * Math.sqrt(2 * Math.PI) * Math.exp((x * x) / 2);
|
|
118
|
+
if (!Number.isFinite(u)) break;
|
|
119
|
+
x -= u / (1 + (x * u) / 2);
|
|
120
|
+
}
|
|
121
|
+
return x;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function requireFinite(name, value) {
|
|
125
|
+
if (!Number.isFinite(value)) throw new RangeError(`${name} must be a finite number`);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// Expected maximum of N independent standard Normal draws (Bailey, Borwein, López de Prado and Zhu
|
|
129
|
+
// 2014, Proposition 2.1; Bailey and López de Prado 2014, the deflated Sharpe ratio): the Sharpe
|
|
130
|
+
// ratio, in units of its standard deviation, that the best of N skill-less trials is expected to show.
|
|
131
|
+
export function expectedMaxStandardNormal(trials) {
|
|
132
|
+
const n = Number(trials);
|
|
133
|
+
if (!Number.isInteger(n) || n < 2) throw new RangeError("Effective independent trials must be an integer of at least 2");
|
|
134
|
+
return (1 - EULER_MASCHERONI) * normalPpf(1 - 1 / n) + EULER_MASCHERONI * normalPpf(1 - 1 / (n * Math.E));
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Minimum Backtest Length (Bailey, Borwein, López de Prado and Zhu 2014, Theorem 3.1): the years of
|
|
138
|
+
// backtest needed so that the best of N skill-less trials is not expected to show an annualized
|
|
139
|
+
// Sharpe of targetSharpe in sample: ((1-g) Z^-1[1-1/N] + g Z^-1[1-1/(Ne)])^2 / targetSharpe^2,
|
|
140
|
+
// bounded above by 2 ln N / targetSharpe^2. Necessary, not sufficient, to avoid overfitting.
|
|
141
|
+
export function minimumBacktestLength({ trials, targetSharpe }) {
|
|
142
|
+
const target = Number(targetSharpe);
|
|
143
|
+
if (!(target > 0 && Number.isFinite(target))) throw new RangeError("The target Sharpe must be a positive number");
|
|
144
|
+
const expectedMax = expectedMaxStandardNormal(trials);
|
|
145
|
+
return {
|
|
146
|
+
years: (expectedMax / target) ** 2,
|
|
147
|
+
upper_bound_years: (2 * Math.log(Number(trials))) / target ** 2,
|
|
148
|
+
expected_max_sharpe_one_year: expectedMax,
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// The largest number of independent trials whose best is still expected to stay below targetSharpe
|
|
153
|
+
// in sample over `years` of backtest (Eq. 3.1 solved for N). The expected maximum grows with N, so
|
|
154
|
+
// a doubling search then a bisection finds it exactly. Returns 1 when even two trials are too many.
|
|
155
|
+
export function maximumIndependentTrials({ years, targetSharpe }) {
|
|
156
|
+
const y = Number(years);
|
|
157
|
+
const target = Number(targetSharpe);
|
|
158
|
+
if (!(y > 0 && Number.isFinite(y))) throw new RangeError("Backtest years must be a positive number");
|
|
159
|
+
if (!(target > 0 && Number.isFinite(target))) throw new RangeError("The target Sharpe must be a positive number");
|
|
160
|
+
const ceiling = target * Math.sqrt(y);
|
|
161
|
+
const fits = (n) => expectedMaxStandardNormal(n) <= ceiling;
|
|
162
|
+
if (!fits(2)) return 1;
|
|
163
|
+
let lo = 2;
|
|
164
|
+
let hi = 4;
|
|
165
|
+
const LIMIT = 1e15;
|
|
166
|
+
while (fits(hi)) {
|
|
167
|
+
lo = hi;
|
|
168
|
+
if (hi >= LIMIT) return LIMIT;
|
|
169
|
+
hi = Math.min(hi * 2, LIMIT);
|
|
170
|
+
}
|
|
171
|
+
while (hi - lo > 1) {
|
|
172
|
+
const mid = Math.floor((lo + hi) / 2);
|
|
173
|
+
if (fits(mid)) lo = mid;
|
|
174
|
+
else hi = mid;
|
|
175
|
+
}
|
|
176
|
+
return lo;
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
export function calculateDsr(input) {
|
|
180
|
+
const values = Object.fromEntries(
|
|
181
|
+
Object.entries(input).map(([key, value]) => [key, Number(value)]),
|
|
182
|
+
);
|
|
183
|
+
for (const [key, value] of Object.entries(values)) requireFinite(key, value);
|
|
184
|
+
if (!Number.isInteger(values.observations) || values.observations < 2) {
|
|
185
|
+
throw new RangeError("Return observations must be an integer of at least 2");
|
|
186
|
+
}
|
|
187
|
+
if (!(values.periods_per_year > 0)) {
|
|
188
|
+
throw new RangeError("Periods per year must be greater than zero");
|
|
189
|
+
}
|
|
190
|
+
if (
|
|
191
|
+
!Number.isInteger(values.effective_independent_trials) ||
|
|
192
|
+
values.effective_independent_trials < 2
|
|
193
|
+
) {
|
|
194
|
+
throw new RangeError("Effective independent trials must be an integer of at least 2");
|
|
195
|
+
}
|
|
196
|
+
if (values.cross_trial_sharpe_sd_annualized < 0) {
|
|
197
|
+
throw new RangeError("Cross-trial Sharpe dispersion cannot be negative");
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const annualizationScale = Math.sqrt(values.periods_per_year);
|
|
201
|
+
const observedSharpePerPeriod = values.observed_sharpe_annualized / annualizationScale;
|
|
202
|
+
const trialSdPerPeriod = values.cross_trial_sharpe_sd_annualized / annualizationScale;
|
|
203
|
+
const trialVariancePerPeriod = trialSdPerPeriod ** 2;
|
|
204
|
+
const quantile = expectedMaxStandardNormal(values.effective_independent_trials);
|
|
205
|
+
const expectedMaxSharpePerPeriod = trialSdPerPeriod * quantile;
|
|
206
|
+
const nonNormalityVarianceTerm =
|
|
207
|
+
1 -
|
|
208
|
+
values.skew * observedSharpePerPeriod +
|
|
209
|
+
((values.non_excess_kurtosis - 1) / 4) * observedSharpePerPeriod ** 2;
|
|
210
|
+
if (!(nonNormalityVarianceTerm > 0)) {
|
|
211
|
+
throw new RangeError(
|
|
212
|
+
"These skew, kurtosis and Sharpe inputs produce a non-positive estimator variance term",
|
|
213
|
+
);
|
|
214
|
+
}
|
|
215
|
+
const denominator = Math.sqrt(nonNormalityVarianceTerm);
|
|
216
|
+
const sampleScale = Math.sqrt(values.observations - 1);
|
|
217
|
+
const psrZ = observedSharpePerPeriod * sampleScale / denominator;
|
|
218
|
+
const dsrZ =
|
|
219
|
+
(observedSharpePerPeriod - expectedMaxSharpePerPeriod) * sampleScale / denominator;
|
|
220
|
+
|
|
221
|
+
return {
|
|
222
|
+
observed_sharpe_per_period: observedSharpePerPeriod,
|
|
223
|
+
cross_trial_sharpe_variance_per_period: trialVariancePerPeriod,
|
|
224
|
+
expected_max_sharpe_per_period: expectedMaxSharpePerPeriod,
|
|
225
|
+
expected_max_sharpe_annualized: expectedMaxSharpePerPeriod * annualizationScale,
|
|
226
|
+
psr_against_zero: normalCdf(psrZ),
|
|
227
|
+
deflated_sharpe_ratio: normalCdf(dsrZ),
|
|
228
|
+
non_normality_variance_term: nonNormalityVarianceTerm,
|
|
229
|
+
search_haircut_annualized:
|
|
230
|
+
values.observed_sharpe_annualized - expectedMaxSharpePerPeriod * annualizationScale,
|
|
231
|
+
psr_z_score: psrZ,
|
|
232
|
+
dsr_z_score: dsrZ,
|
|
233
|
+
};
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
export function checkGoldenVectors(vectors, tolerance = 8e-7) {
|
|
237
|
+
const failures = [];
|
|
238
|
+
for (const vector of vectors) {
|
|
239
|
+
const observed = calculateDsr(vector.inputs);
|
|
240
|
+
for (const [key, expected] of Object.entries(vector.outputs)) {
|
|
241
|
+
const error = Math.abs(observed[key] - expected);
|
|
242
|
+
if (error > tolerance) failures.push({ vector: vector.id, key, expected, observed: observed[key], error });
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
return failures;
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
// Probabilistic Sharpe ratio against a benchmark Sharpe, and the minimum track record length for
|
|
250
|
+
// it to clear that benchmark at a confidence level: Bailey and López de Prado, "The Sharpe Ratio
|
|
251
|
+
// Efficient Frontier", Journal of Risk 15(2), 2012, Eqs. (11) and (13). Inputs are annualized; the
|
|
252
|
+
// estimator works per period, like calculateDsr above.
|
|
253
|
+
function perPeriodInputs(values) {
|
|
254
|
+
for (const [key, value] of Object.entries(values)) requireFinite(key, value);
|
|
255
|
+
if (!(values.periods_per_year > 0)) throw new RangeError("Periods per year must be greater than zero");
|
|
256
|
+
const scale = Math.sqrt(values.periods_per_year);
|
|
257
|
+
const sr = values.observed_sharpe_annualized / scale;
|
|
258
|
+
const benchmark = values.benchmark_sharpe_annualized / scale;
|
|
259
|
+
const term = 1 - values.skew * sr + ((values.non_excess_kurtosis - 1) / 4) * sr ** 2;
|
|
260
|
+
if (!(term > 0)) {
|
|
261
|
+
throw new RangeError("These skew, kurtosis and Sharpe inputs produce a non-positive estimator variance term");
|
|
262
|
+
}
|
|
263
|
+
return { sr, benchmark, term };
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
export function probabilisticSharpe(input) {
|
|
267
|
+
const values = {
|
|
268
|
+
observed_sharpe_annualized: Number(input.observed_sharpe_annualized),
|
|
269
|
+
benchmark_sharpe_annualized: Number(input.benchmark_sharpe_annualized ?? 0),
|
|
270
|
+
observations: Number(input.observations),
|
|
271
|
+
periods_per_year: Number(input.periods_per_year),
|
|
272
|
+
skew: Number(input.skew),
|
|
273
|
+
non_excess_kurtosis: Number(input.non_excess_kurtosis),
|
|
274
|
+
};
|
|
275
|
+
if (!Number.isInteger(values.observations) || values.observations < 2) {
|
|
276
|
+
throw new RangeError("Return observations must be an integer of at least 2");
|
|
277
|
+
}
|
|
278
|
+
const { sr, benchmark, term } = perPeriodInputs(values);
|
|
279
|
+
const z = ((sr - benchmark) * Math.sqrt(values.observations - 1)) / Math.sqrt(term);
|
|
280
|
+
return { probabilistic_sharpe_ratio: normalCdf(z), z_score: z };
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
export function minimumTrackRecordLength(input) {
|
|
284
|
+
const confidence = Number(input.confidence ?? 0.95);
|
|
285
|
+
if (!(confidence > 0 && confidence < 1)) throw new RangeError("Confidence must be strictly between 0 and 1");
|
|
286
|
+
const values = {
|
|
287
|
+
observed_sharpe_annualized: Number(input.observed_sharpe_annualized),
|
|
288
|
+
benchmark_sharpe_annualized: Number(input.benchmark_sharpe_annualized ?? 0),
|
|
289
|
+
periods_per_year: Number(input.periods_per_year),
|
|
290
|
+
skew: Number(input.skew),
|
|
291
|
+
non_excess_kurtosis: Number(input.non_excess_kurtosis),
|
|
292
|
+
};
|
|
293
|
+
const { sr, benchmark, term } = perPeriodInputs(values);
|
|
294
|
+
if (!(sr > benchmark)) {
|
|
295
|
+
throw new RangeError("The observed Sharpe must exceed the benchmark; no track record length is enough otherwise");
|
|
296
|
+
}
|
|
297
|
+
const observations = 1 + term * (normalPpf(confidence) / (sr - benchmark)) ** 2;
|
|
298
|
+
if (!Number.isFinite(observations)) {
|
|
299
|
+
throw new RangeError("The observed Sharpe is too close to the benchmark; no finite track record length reaches this confidence");
|
|
300
|
+
}
|
|
301
|
+
return { observations, years: observations / values.periods_per_year, confidence };
|
|
302
|
+
}
|