@general-liquidity/sharpebench 0.19.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -4
- package/dist/index.js +25 -1
- package/dist/types.d.ts +61 -23
- package/package.json +1 -1
- package/pkg/package.json +1 -1
- package/pkg/sharpebench_bg.wasm +0 -0
package/README.md
CHANGED
|
@@ -37,17 +37,17 @@ console.log(greeks({ spot: 100, strike: 100, t_years: 1, rate: 0.05, vol: 0.2, i
|
|
|
37
37
|
|---|---|
|
|
38
38
|
| `score(submissions, config?)` | ranked `CompositeScore[]` |
|
|
39
39
|
| `scoreAgent(submission, config?)` | one `CompositeScore` (deflated Sharpe, pass^k, process, rolling worst-case Sharpe) |
|
|
40
|
-
| `selfAudit()` | `SelfAuditReport`, the
|
|
40
|
+
| `selfAudit()` | `SelfAuditReport`, results of the named anti-gaming regressions |
|
|
41
41
|
| `auditBriefing(briefing, policy?)` | `BriefingAudit`, an input-side salience-bias audit |
|
|
42
42
|
| `scoreAllocation(trajectory, policy?)` | `AllocationReport`, weight-vector validity plus L1 turnover |
|
|
43
43
|
| `greeks(params)` | `GreeksResult`, Black-Scholes price, Greeks, and local exposure flags |
|
|
44
44
|
| `canary(seed)` | `Canary`, a do-not-train contamination tripwire |
|
|
45
|
-
| `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict |
|
|
46
|
-
| `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, and
|
|
45
|
+
| `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict. `trialsSrStd` is annualized; `periodsPerYear` (default 252) says what a period is |
|
|
46
|
+
| `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, PBO and HLZ diagnostics alongside the one-series verdict |
|
|
47
47
|
| `percentileSelection(candidates, opts?)` | Point winner versus bootstrap-percentile winner and optimism gaps |
|
|
48
48
|
| `decomposeUncertainty(input)` | Aleatoric, epistemic, and distributional diagnostic legs |
|
|
49
49
|
| `crowdingHalfLife(adoption, params)` | Caller-calibrated crowding-decay prior, reported but never gating |
|
|
50
|
-
| `classifyDisqualification(submissions, config?)` | Named hard-gate and advisory reasons |
|
|
50
|
+
| `classifyDisqualification(submissions, config?)` | Named hard-gate, unavailability, and advisory reasons |
|
|
51
51
|
| `regimeCompare(a, b, regimes, opts?)` | Regime-conditional distribution comparison and pooled-sign reversal |
|
|
52
52
|
|
|
53
53
|
All inputs and outputs are fully typed (TypeScript declarations ship with the
|
|
@@ -56,6 +56,29 @@ committed golden on the Ubuntu CI host. The Rust CI separately pins the two
|
|
|
56
56
|
committed golden fields on Linux, macOS, and Windows; this is not a claim about
|
|
57
57
|
every possible input or platform.
|
|
58
58
|
|
|
59
|
+
## Unavailable results
|
|
60
|
+
|
|
61
|
+
Check error fields before interpreting numeric diagnostics. `statisticsError`
|
|
62
|
+
marks withheld deflation; `snoopingError` marks withheld fieldwise tests, whose
|
|
63
|
+
compatibility sentinels are p-values of 1 and all-false `stepDown`. These are
|
|
64
|
+
not measured results. Unavailable PBO is `null` with `pboError`.
|
|
65
|
+
|
|
66
|
+
Nonfinite numeric diagnostics serialize as `null`, including a minimum track
|
|
67
|
+
record length with no finite solution. TypeScript callers must handle nulls.
|
|
68
|
+
`percentileSelection` reports `input_error` and null winner indices when any
|
|
69
|
+
candidate is unsupported. Its `alpha_warning` reports where `alpha` sits and
|
|
70
|
+
nothing else: a refusal that had nothing to do with alpha, such as too few
|
|
71
|
+
observations or a rejected block probability, leaves the flag false.
|
|
72
|
+
|
|
73
|
+
`classifyDisqualification` returns eleven reasons in three groups: five mirror
|
|
74
|
+
the scorer's hard eligibility gates, two name the unavailability of a statistic
|
|
75
|
+
the scorer gates on (`deflation_unavailable`, `bootstrap_unavailable`), and four
|
|
76
|
+
are advisory and never gate (`selection_unavailable`, `high_selection_gap`,
|
|
77
|
+
`is_rediscovery`, `oos_decay`). `selection_unavailable` is advisory because the
|
|
78
|
+
scorer reports `selection_gap` and never consults it in `rank_eligible`, so a
|
|
79
|
+
submission carrying only advisory reasons is still rank-eligible. Read
|
|
80
|
+
`rank_eligible`, not the presence of a reason. See the [error and migration guide](https://github.com/general-liquidity/sharpebench/blob/main/docs/book/src/wasm.md#statistical-unavailability-and-migration).
|
|
81
|
+
|
|
59
82
|
## Why luck-robust?
|
|
60
83
|
|
|
61
84
|
Most agent leaderboards rank a raw Sharpe over a single short window, so they mostly measure noise. SharpeBench gates eligibility on Deflated Sharpe, pass^k reliability across every seed and window, stationary-bootstrap significance, process discipline, and the host drawdown mandate. PSR and the fieldwise multiple-testing family remain visible diagnostics. See the [benchmark repo](https://github.com/general-liquidity/sharpebench) for the full methodology.
|
package/dist/index.js
CHANGED
|
@@ -117,8 +117,22 @@ function honestyConfigJson(opts) {
|
|
|
117
117
|
throw new RangeError("nTrials must be an integer in 1..=4294967295");
|
|
118
118
|
}
|
|
119
119
|
const cfg = { n_trials: opts.nTrials };
|
|
120
|
-
if (opts.trialsSrStd !== undefined)
|
|
120
|
+
if (opts.trialsSrStd !== undefined) {
|
|
121
|
+
// JSON.stringify turns NaN and Infinity into null, which the kernel reads as
|
|
122
|
+
// "omitted" and replaces with the 0.5 prior: refuse it before it crosses.
|
|
123
|
+
if (typeof opts.trialsSrStd !== "number" || !Number.isFinite(opts.trialsSrStd)) {
|
|
124
|
+
throw new RangeError("trialsSrStd must be a finite number (omit it for the 0.5 prior)");
|
|
125
|
+
}
|
|
121
126
|
cfg.trials_sr_std = opts.trialsSrStd;
|
|
127
|
+
}
|
|
128
|
+
if (opts.periodsPerYear !== undefined) {
|
|
129
|
+
// JSON.stringify turns NaN and Infinity into null, which the kernel refuses
|
|
130
|
+
// rather than reading as "omitted"; say which input it was here instead.
|
|
131
|
+
if (typeof opts.periodsPerYear !== "number" || !Number.isFinite(opts.periodsPerYear)) {
|
|
132
|
+
throw new RangeError("periodsPerYear must be a finite number (omit it for the 252 default)");
|
|
133
|
+
}
|
|
134
|
+
cfg.periods_per_year = opts.periodsPerYear;
|
|
135
|
+
}
|
|
122
136
|
if (opts.confidence !== undefined)
|
|
123
137
|
cfg.confidence = opts.confidence;
|
|
124
138
|
if (opts.borderline !== undefined)
|
|
@@ -144,6 +158,7 @@ function toHonestyVerdict(raw) {
|
|
|
144
158
|
verdict: raw.verdict,
|
|
145
159
|
explanation: raw.explanation,
|
|
146
160
|
methodologyVersion: raw.methodology_version,
|
|
161
|
+
...(raw.statistics_error === undefined ? {} : { statisticsError: raw.statistics_error }),
|
|
147
162
|
};
|
|
148
163
|
}
|
|
149
164
|
/**
|
|
@@ -166,6 +181,7 @@ function isMySharpeReal(returns, opts) {
|
|
|
166
181
|
*/
|
|
167
182
|
function isMySharpeRealFull(field, winnerIdx, opts) {
|
|
168
183
|
const raw = parse(kernel.is_my_sharpe_real_full(JSON.stringify(field), winnerIdx, honestyConfigJson(opts)));
|
|
184
|
+
const hlz = raw.hlz;
|
|
169
185
|
return {
|
|
170
186
|
honesty: toHonestyVerdict(raw.honesty),
|
|
171
187
|
realityCheckP: raw.reality_check_p,
|
|
@@ -173,6 +189,14 @@ function isMySharpeRealFull(field, winnerIdx, opts) {
|
|
|
173
189
|
spaConsistentP: raw.spa_consistent_p,
|
|
174
190
|
stepDown: raw.step_down,
|
|
175
191
|
pbo: raw.pbo,
|
|
192
|
+
hlz: {
|
|
193
|
+
tStat: hlz.t_stat,
|
|
194
|
+
tThreshold: hlz.t_threshold,
|
|
195
|
+
passed: hlz.passed,
|
|
196
|
+
explanation: hlz.explanation,
|
|
197
|
+
},
|
|
198
|
+
...(raw.snooping_error === undefined ? {} : { snoopingError: raw.snooping_error }),
|
|
199
|
+
...(raw.pbo_error === undefined ? {} : { pboError: raw.pbo_error }),
|
|
176
200
|
};
|
|
177
201
|
}
|
|
178
202
|
/**
|
package/dist/types.d.ts
CHANGED
|
@@ -153,33 +153,53 @@ export type Verdict = "Pass" | "Borderline" | "Fail";
|
|
|
153
153
|
export interface HonestyOpts {
|
|
154
154
|
/** Number of strategy trials behind this result, integer 1..=4294967295. REQUIRED. */
|
|
155
155
|
nTrials: number;
|
|
156
|
-
/**
|
|
156
|
+
/**
|
|
157
|
+
* **Annualized** cross-trial Sharpe dispersion, divided by
|
|
158
|
+
* `sqrt(periodsPerYear)` before use. Omit → estimated at 0.5 and flagged. A
|
|
159
|
+
* negative value yields a Fail verdict with `statisticsError`; NaN and
|
|
160
|
+
* infinity cannot cross JSON (they would arrive as the 0.5 prior) and throw a
|
|
161
|
+
* `RangeError`.
|
|
162
|
+
*/
|
|
157
163
|
trialsSrStd?: number;
|
|
164
|
+
/**
|
|
165
|
+
* Return periods per year for these returns (daily equities 252, daily crypto
|
|
166
|
+
* 365, hourly 8760, weekly 52). Omit → 252, flagged in the explanation. A
|
|
167
|
+
* non-positive value yields a Fail verdict with `statisticsError`; NaN and
|
|
168
|
+
* infinity cannot cross JSON and throw a `RangeError`.
|
|
169
|
+
*/
|
|
170
|
+
periodsPerYear?: number;
|
|
158
171
|
/** Deflated-Sharpe threshold for a Pass. Default 0.95. */
|
|
159
172
|
confidence?: number;
|
|
160
173
|
/** Deflated-Sharpe threshold for Borderline. Default 0.90. */
|
|
161
174
|
borderline?: number;
|
|
162
|
-
/** PSR / MinTRL benchmark Sharpe to beat. Default 0.0. */
|
|
175
|
+
/** **Per-period** PSR / MinTRL benchmark Sharpe to beat (not converted). Default 0.0. */
|
|
163
176
|
srBenchmark?: number;
|
|
164
177
|
}
|
|
165
|
-
/** The LITE verdict
|
|
178
|
+
/** The LITE verdict. Nonfinite numeric diagnostics serialize as null, not zero. */
|
|
166
179
|
export interface HonestyVerdict {
|
|
167
|
-
sharpe: number;
|
|
180
|
+
sharpe: number | null;
|
|
168
181
|
nObs: number;
|
|
169
|
-
skew: number;
|
|
170
|
-
kurtosis: number;
|
|
182
|
+
skew: number | null;
|
|
183
|
+
kurtosis: number | null;
|
|
171
184
|
nTrials: number;
|
|
172
|
-
expectedMaxSharpe: number;
|
|
173
|
-
deflatedSharpe: number;
|
|
174
|
-
probabilisticSharpe: number;
|
|
175
|
-
/**
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
185
|
+
expectedMaxSharpe: number | null;
|
|
186
|
+
deflatedSharpe: number | null;
|
|
187
|
+
probabilisticSharpe: number | null;
|
|
188
|
+
/**
|
|
189
|
+
* `1 - deflatedSharpe`: the p-value of the test whose null is that this Sharpe
|
|
190
|
+
* is the best of `nTrials` zero-skill trials. Not the probability that the
|
|
191
|
+
* edge is a search artifact.
|
|
192
|
+
*/
|
|
193
|
+
haircut: number | null;
|
|
194
|
+
/** `sharpe * deflatedSharpe`: the Sharpe scaled down by the deflated Sharpe. */
|
|
195
|
+
haircutSharpe: number | null;
|
|
196
|
+
/** Null when no finite track length is returned by the kernel. */
|
|
197
|
+
minTrackRecordLen: number | null;
|
|
180
198
|
verdict: Verdict;
|
|
181
199
|
explanation: string;
|
|
182
200
|
methodologyVersion: string;
|
|
201
|
+
/** When present, deflation was withheld; its numeric sentinels are not estimates. */
|
|
202
|
+
statisticsError?: string;
|
|
183
203
|
[k: string]: unknown;
|
|
184
204
|
}
|
|
185
205
|
/** What a candidate is scored on inside {@link percentileSelection}. */
|
|
@@ -218,14 +238,18 @@ export interface CandidateUtility {
|
|
|
218
238
|
export interface PercentileSelectionResult {
|
|
219
239
|
/** The percentile actually used, clamped to [0, 1]. */
|
|
220
240
|
alpha: number;
|
|
221
|
-
/** True when alpha sits below the recommended floor of 0.3.
|
|
241
|
+
/** True when alpha sits below the recommended floor of 0.3. Computed from
|
|
242
|
+
* `alpha` alone, so a refusal that had nothing to do with alpha (too few
|
|
243
|
+
* observations, a rejected block probability) does not raise it. */
|
|
222
244
|
alpha_warning: boolean;
|
|
223
245
|
/** Every candidate, in input order. */
|
|
224
246
|
candidates: CandidateUtility[];
|
|
225
|
-
/** Index of the
|
|
247
|
+
/** Index of the robust pick, or null for empty or refused input. */
|
|
226
248
|
selected: number | null;
|
|
227
|
-
/** Index of the
|
|
249
|
+
/** Index of the point winner, or null for empty or refused input. */
|
|
228
250
|
point_argmax: number | null;
|
|
251
|
+
/** Null on accepted input; a refusal withholds the entire candidate field. */
|
|
252
|
+
input_error: string | null;
|
|
229
253
|
/** Whether the two picks agree. Disagreement is the interesting case. */
|
|
230
254
|
agrees_with_point_argmax: boolean;
|
|
231
255
|
/** Optimism gap of the point winner: report this next to any headline utility. */
|
|
@@ -293,10 +317,13 @@ export interface CrowdingDecayPrior {
|
|
|
293
317
|
}
|
|
294
318
|
/**
|
|
295
319
|
* A reason an agent was (or should be) demoted. The first five mirror the hard
|
|
296
|
-
* eligibility gates in the scorer;
|
|
297
|
-
*
|
|
320
|
+
* eligibility gates in the scorer; two more name the unavailability of a
|
|
321
|
+
* statistic the scorer does gate on. The last four are advisory quality flags
|
|
322
|
+
* that never gate. `selection_unavailable` is advisory because the scorer
|
|
323
|
+
* reports the selection axis and never consults it in `rank_eligible`, so an
|
|
324
|
+
* agent carrying it alone is still ranked.
|
|
298
325
|
*/
|
|
299
|
-
export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
|
|
326
|
+
export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "deflation_unavailable" | "bootstrap_unavailable" | "selection_unavailable" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
|
|
300
327
|
/** Every disqualification/quality signal that fired for one scored agent. */
|
|
301
328
|
export interface DisqualificationReport {
|
|
302
329
|
agent_id: string;
|
|
@@ -305,7 +332,14 @@ export interface DisqualificationReport {
|
|
|
305
332
|
reasons: FailReason[];
|
|
306
333
|
[k: string]: unknown;
|
|
307
334
|
}
|
|
308
|
-
/**
|
|
335
|
+
/** Harvey-Liu-Zhu hurdle and its interpretation from the kernel. */
|
|
336
|
+
export interface HlzGate {
|
|
337
|
+
tStat: number | null;
|
|
338
|
+
tThreshold: number;
|
|
339
|
+
passed: boolean;
|
|
340
|
+
explanation: string;
|
|
341
|
+
}
|
|
342
|
+
/** The FULL verdict: LITE on the winner plus multiple testing, PBO and HLZ. */
|
|
309
343
|
export interface FullVerdict {
|
|
310
344
|
honesty: HonestyVerdict;
|
|
311
345
|
/** White's Reality Check p-value over the field. */
|
|
@@ -316,8 +350,12 @@ export interface FullVerdict {
|
|
|
316
350
|
spaConsistentP: number;
|
|
317
351
|
/** Romano-Wolf step-down: which field members are significant at α. */
|
|
318
352
|
stepDown: boolean[];
|
|
319
|
-
/** CSCV
|
|
320
|
-
pbo: number;
|
|
353
|
+
/** CSCV estimate; null when unavailable, with a reason in pboError. */
|
|
354
|
+
pbo: number | null;
|
|
355
|
+
hlz: HlzGate;
|
|
356
|
+
/** When present, p-values of 1 and all-false stepDown are refusal sentinels. */
|
|
357
|
+
snoopingError?: string;
|
|
358
|
+
pboError?: string;
|
|
321
359
|
[k: string]: unknown;
|
|
322
360
|
}
|
|
323
361
|
/** Options for a regime-conditional comparison. Regime labels are caller inputs. */
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@general-liquidity/sharpebench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.21.0",
|
|
4
4
|
"description": "Luck-robust quantitative evaluation for trading agents: deflated Sharpe, pass^k reliability, process discipline, and risk gates, backed by the identical Rust kernel compiled to WebAssembly.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"trading",
|
package/pkg/package.json
CHANGED
package/pkg/sharpebench_bg.wasm
CHANGED
|
Binary file
|