@general-liquidity/sharpebench 0.18.4 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -27,7 +27,7 @@ console.log(board[0].agent_id, board[0].deflated_sharpe, board[0].rank_eligible)
27
27
  // Run the scorer's built-in checks against its catalogued gaming attacks.
28
28
  console.log(selfAudit().all_defended); // true
29
29
 
30
- // Options tail-risk: a short-gamma position a linear Sharpe can't see.
30
+ // Price and local sensitivities for one long European option.
31
31
  console.log(greeks({ spot: 100, strike: 100, t_years: 1, rate: 0.05, vol: 0.2, is_call: true }).price);
32
32
  ```
33
33
 
@@ -37,17 +37,17 @@ console.log(greeks({ spot: 100, strike: 100, t_years: 1, rate: 0.05, vol: 0.2, i
37
37
  |---|---|
38
38
  | `score(submissions, config?)` | ranked `CompositeScore[]` |
39
39
  | `scoreAgent(submission, config?)` | one `CompositeScore` (deflated Sharpe, pass^k, process, rolling worst-case Sharpe) |
40
- | `selfAudit()` | `SelfAuditReport`, the benchmark's anti-gaming proof |
40
+ | `selfAudit()` | `SelfAuditReport`, results of the named anti-gaming regressions |
41
41
  | `auditBriefing(briefing, policy?)` | `BriefingAudit`, an input-side salience-bias audit |
42
42
  | `scoreAllocation(trajectory, policy?)` | `AllocationReport`, weight-vector validity plus L1 turnover |
43
- | `greeks(params)` | `GreeksResult`, Black-Scholes price, Greeks, and tail-selling risk |
43
+ | `greeks(params)` | `GreeksResult`, Black-Scholes price, Greeks, and local exposure flags |
44
44
  | `canary(seed)` | `Canary`, a do-not-train contamination tripwire |
45
- | `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict |
46
- | `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, and PBO alongside the one-series verdict |
45
+ | `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict. `trialsSrStd` is annualized; `periodsPerYear` (default 252) says what a period is |
46
+ | `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, PBO and HLZ diagnostics alongside the one-series verdict |
47
47
  | `percentileSelection(candidates, opts?)` | Point winner versus bootstrap-percentile winner and optimism gaps |
48
48
  | `decomposeUncertainty(input)` | Aleatoric, epistemic, and distributional diagnostic legs |
49
49
  | `crowdingHalfLife(adoption, params)` | Caller-calibrated crowding-decay prior, reported but never gating |
50
- | `classifyDisqualification(submissions, config?)` | Named hard-gate and advisory reasons |
50
+ | `classifyDisqualification(submissions, config?)` | Named hard-gate, unavailability, and advisory reasons |
51
51
  | `regimeCompare(a, b, regimes, opts?)` | Regime-conditional distribution comparison and pooled-sign reversal |
52
52
 
53
53
  All inputs and outputs are fully typed (TypeScript declarations ship with the
@@ -56,6 +56,29 @@ committed golden on the Ubuntu CI host. The Rust CI separately pins the two
56
56
  committed golden fields on Linux, macOS, and Windows; this is not a claim about
57
57
  every possible input or platform.
58
58
 
59
+ ## Unavailable results
60
+
61
+ Check error fields before interpreting numeric diagnostics. `statisticsError`
62
+ marks withheld deflation; `snoopingError` marks withheld fieldwise tests, whose
63
+ compatibility sentinels are p-values of 1 and all-false `stepDown`. These are
64
+ not measured results. Unavailable PBO is `null` with `pboError`.
65
+
66
+ Nonfinite numeric diagnostics serialize as `null`, including a minimum track
67
+ record length with no finite solution. TypeScript callers must handle nulls.
68
+ `percentileSelection` reports `input_error` and null winner indices when any
69
+ candidate is unsupported. Its `alpha_warning` reports where `alpha` sits and
70
+ nothing else: a refusal that had nothing to do with alpha, such as too few
71
+ observations or a rejected block probability, leaves the flag false.
72
+
73
+ `classifyDisqualification` returns eleven reasons in three groups: five mirror
74
+ the scorer's hard eligibility gates, two name the unavailability of a statistic
75
+ the scorer gates on (`deflation_unavailable`, `bootstrap_unavailable`), and four
76
+ are advisory and never gate (`selection_unavailable`, `high_selection_gap`,
77
+ `is_rediscovery`, `oos_decay`). `selection_unavailable` is advisory because the
78
+ scorer reports `selection_gap` and never consults it in `rank_eligible`, so a
79
+ submission carrying only advisory reasons is still rank-eligible. Read
80
+ `rank_eligible`, not the presence of a reason. See the [error and migration guide](https://github.com/general-liquidity/sharpebench/blob/main/docs/book/src/wasm.md#statistical-unavailability-and-migration).
81
+
59
82
  ## Why luck-robust?
60
83
 
61
84
  Most agent leaderboards rank a raw Sharpe over a single short window, so they mostly measure noise. SharpeBench gates eligibility on Deflated Sharpe, pass^k reliability across every seed and window, stationary-bootstrap significance, process discipline, and the host drawdown mandate. PSR and the fieldwise multiple-testing family remain visible diagnostics. See the [benchmark repo](https://github.com/general-liquidity/sharpebench) for the full methodology.
package/dist/index.d.ts CHANGED
@@ -17,7 +17,8 @@ export declare function selfAudit(): SelfAuditReport;
17
17
  export declare function auditBriefing(briefing: Briefing, policy?: BriefingPolicy): BriefingAudit;
18
18
  /** Score a target-allocation trajectory: weight validity + L1 turnover. */
19
19
  export declare function scoreAllocation(trajectory: AllocationTrajectory, policy?: AllocationPolicy): AllocationReport;
20
- /** Black-Scholes price + Greeks + tail-selling (short-gamma/vega) classification. */
20
+ /** Price and local Greeks for one long European option. Throws for invalid inputs
21
+ * or undefined Greeks at a payoff kink. Does not infer payoff-loss boundedness. */
21
22
  export declare function greeks(params: GreeksParams): GreeksResult;
22
23
  /** Derive a deterministic do-not-train contamination tripwire from seed material. */
23
24
  export declare function canary(seed: string): Canary;
package/dist/index.js CHANGED
@@ -102,7 +102,8 @@ function auditBriefing(briefing, policy) {
102
102
  function scoreAllocation(trajectory, policy) {
103
103
  return parse(kernel.score_allocation(JSON.stringify(trajectory), optJson(policy)));
104
104
  }
105
- /** Black-Scholes price + Greeks + tail-selling (short-gamma/vega) classification. */
105
+ /** Price and local Greeks for one long European option. Throws for invalid inputs
106
+ * or undefined Greeks at a payoff kink. Does not infer payoff-loss boundedness. */
106
107
  function greeks(params) {
107
108
  return parse(kernel.greeks(JSON.stringify(params)));
108
109
  }
@@ -112,9 +113,26 @@ function canary(seed) {
112
113
  }
113
114
  /** Map camelCase {@link HonestyOpts} → the snake_case `HonestyConfig` JSON the kernel reads. */
114
115
  function honestyConfigJson(opts) {
116
+ if (!Number.isSafeInteger(opts.nTrials) || opts.nTrials < 1 || opts.nTrials > 0xffffffff) {
117
+ throw new RangeError("nTrials must be an integer in 1..=4294967295");
118
+ }
115
119
  const cfg = { n_trials: opts.nTrials };
116
- if (opts.trialsSrStd !== undefined)
120
+ if (opts.trialsSrStd !== undefined) {
121
+ // JSON.stringify turns NaN and Infinity into null, which the kernel reads as
122
+ // "omitted" and replaces with the 0.5 prior: refuse it before it crosses.
123
+ if (typeof opts.trialsSrStd !== "number" || !Number.isFinite(opts.trialsSrStd)) {
124
+ throw new RangeError("trialsSrStd must be a finite number (omit it for the 0.5 prior)");
125
+ }
117
126
  cfg.trials_sr_std = opts.trialsSrStd;
127
+ }
128
+ if (opts.periodsPerYear !== undefined) {
129
+ // JSON.stringify turns NaN and Infinity into null, which the kernel refuses
130
+ // rather than reading as "omitted"; say which input it was here instead.
131
+ if (typeof opts.periodsPerYear !== "number" || !Number.isFinite(opts.periodsPerYear)) {
132
+ throw new RangeError("periodsPerYear must be a finite number (omit it for the 252 default)");
133
+ }
134
+ cfg.periods_per_year = opts.periodsPerYear;
135
+ }
118
136
  if (opts.confidence !== undefined)
119
137
  cfg.confidence = opts.confidence;
120
138
  if (opts.borderline !== undefined)
@@ -140,6 +158,7 @@ function toHonestyVerdict(raw) {
140
158
  verdict: raw.verdict,
141
159
  explanation: raw.explanation,
142
160
  methodologyVersion: raw.methodology_version,
161
+ ...(raw.statistics_error === undefined ? {} : { statisticsError: raw.statistics_error }),
143
162
  };
144
163
  }
145
164
  /**
@@ -162,6 +181,7 @@ function isMySharpeReal(returns, opts) {
162
181
  */
163
182
  function isMySharpeRealFull(field, winnerIdx, opts) {
164
183
  const raw = parse(kernel.is_my_sharpe_real_full(JSON.stringify(field), winnerIdx, honestyConfigJson(opts)));
184
+ const hlz = raw.hlz;
165
185
  return {
166
186
  honesty: toHonestyVerdict(raw.honesty),
167
187
  realityCheckP: raw.reality_check_p,
@@ -169,6 +189,14 @@ function isMySharpeRealFull(field, winnerIdx, opts) {
169
189
  spaConsistentP: raw.spa_consistent_p,
170
190
  stepDown: raw.step_down,
171
191
  pbo: raw.pbo,
192
+ hlz: {
193
+ tStat: hlz.t_stat,
194
+ tThreshold: hlz.t_threshold,
195
+ passed: hlz.passed,
196
+ explanation: hlz.explanation,
197
+ },
198
+ ...(raw.snooping_error === undefined ? {} : { snoopingError: raw.snooping_error }),
199
+ ...(raw.pbo_error === undefined ? {} : { pboError: raw.pbo_error }),
172
200
  };
173
201
  }
174
202
  /**
package/dist/types.d.ts CHANGED
@@ -3,7 +3,7 @@ export interface Run {
3
3
  returns: number[];
4
4
  cost?: number;
5
5
  confidences?: number[];
6
- outcomes?: number[];
6
+ outcomes?: boolean[];
7
7
  trace?: {
8
8
  events: unknown[];
9
9
  };
@@ -16,7 +16,20 @@ export interface AgentSubmission {
16
16
  in_sample_trials?: number;
17
17
  /** Candidate return series from the agent's own selection search. */
18
18
  candidates?: number[][];
19
- }
19
+ /** A separately reported verdict, never a replacement for host eligibility. */
20
+ declared_mandate?: DeclaredMandate | null;
21
+ }
22
+ export type DeclaredMandate = {
23
+ kind: "absolute_return";
24
+ } | {
25
+ kind: "relative_to";
26
+ benchmark_id: string;
27
+ } | {
28
+ kind: "outperform_buy_and_hold";
29
+ } | {
30
+ kind: "drawdown_capped";
31
+ max_per_run_drawdown: number;
32
+ };
20
33
  /** Scoring configuration. Omit (or pass `{}`) to use the luck-robust defaults. */
21
34
  export interface ScoreConfig {
22
35
  n_trials?: number;
@@ -31,6 +44,10 @@ export interface CompositeScore {
31
44
  process_ok: boolean;
32
45
  rank_eligible: boolean;
33
46
  raw_mean_return: number;
47
+ declared_mandate?: DeclaredMandate;
48
+ declared_passed_k?: boolean;
49
+ declared_mandate_eligible?: boolean;
50
+ declared_mandate_ordinal?: number;
34
51
  [k: string]: unknown;
35
52
  }
36
53
  export interface SelfAuditReport {
@@ -111,8 +128,8 @@ export interface Greeks {
111
128
  rho: number;
112
129
  }
113
130
  export interface GreeksRisk {
114
- naked_short_gamma: boolean;
115
- unbounded_tail: boolean;
131
+ /** Local convexity flag, not nakedness or payoff-loss boundedness. */
132
+ net_short_gamma: boolean;
116
133
  short_vega: boolean;
117
134
  net_gamma: number;
118
135
  net_vega: number;
@@ -134,35 +151,55 @@ export type Verdict = "Pass" | "Borderline" | "Fail";
134
151
  * the caller must think about — `nTrials = 1` is almost always a lie.
135
152
  */
136
153
  export interface HonestyOpts {
137
- /** Number of strategy trials behind this result. REQUIRED. */
154
+ /** Number of strategy trials behind this result, integer 1..=4294967295. REQUIRED. */
138
155
  nTrials: number;
139
- /** Cross-trial Sharpe dispersion. Omit → estimated at 0.5 and flagged. */
156
+ /**
157
+ * **Annualized** cross-trial Sharpe dispersion, divided by
158
+ * `sqrt(periodsPerYear)` before use. Omit → estimated at 0.5 and flagged. A
159
+ * negative value yields a Fail verdict with `statisticsError`; NaN and
160
+ * infinity cannot cross JSON (they would arrive as the 0.5 prior) and throw a
161
+ * `RangeError`.
162
+ */
140
163
  trialsSrStd?: number;
164
+ /**
165
+ * Return periods per year for these returns (daily equities 252, daily crypto
166
+ * 365, hourly 8760, weekly 52). Omit → 252, flagged in the explanation. A
167
+ * non-positive value yields a Fail verdict with `statisticsError`; NaN and
168
+ * infinity cannot cross JSON and throw a `RangeError`.
169
+ */
170
+ periodsPerYear?: number;
141
171
  /** Deflated-Sharpe threshold for a Pass. Default 0.95. */
142
172
  confidence?: number;
143
173
  /** Deflated-Sharpe threshold for Borderline. Default 0.90. */
144
174
  borderline?: number;
145
- /** PSR / MinTRL benchmark Sharpe to beat. Default 0.0. */
175
+ /** **Per-period** PSR / MinTRL benchmark Sharpe to beat (not converted). Default 0.0. */
146
176
  srBenchmark?: number;
147
177
  }
148
- /** The LITE verdict: everything derivable from one return series. */
178
+ /** The LITE verdict. Nonfinite numeric diagnostics serialize as null, not zero. */
149
179
  export interface HonestyVerdict {
150
- sharpe: number;
180
+ sharpe: number | null;
151
181
  nObs: number;
152
- skew: number;
153
- kurtosis: number;
182
+ skew: number | null;
183
+ kurtosis: number | null;
154
184
  nTrials: number;
155
- expectedMaxSharpe: number;
156
- deflatedSharpe: number;
157
- probabilisticSharpe: number;
158
- /** `1 - deflatedSharpe`: probability the edge is a search artifact. */
159
- haircut: number;
160
- /** `sharpe * deflatedSharpe`: Sharpe discounted by survival probability. */
161
- haircutSharpe: number;
162
- minTrackRecordLen: number;
185
+ expectedMaxSharpe: number | null;
186
+ deflatedSharpe: number | null;
187
+ probabilisticSharpe: number | null;
188
+ /**
189
+ * `1 - deflatedSharpe`: the p-value of the test whose null is that this Sharpe
190
+ * is the best of `nTrials` zero-skill trials. Not the probability that the
191
+ * edge is a search artifact.
192
+ */
193
+ haircut: number | null;
194
+ /** `sharpe * deflatedSharpe`: the Sharpe scaled down by the deflated Sharpe. */
195
+ haircutSharpe: number | null;
196
+ /** Null when no finite track length is returned by the kernel. */
197
+ minTrackRecordLen: number | null;
163
198
  verdict: Verdict;
164
199
  explanation: string;
165
200
  methodologyVersion: string;
201
+ /** When present, deflation was withheld; its numeric sentinels are not estimates. */
202
+ statisticsError?: string;
166
203
  [k: string]: unknown;
167
204
  }
168
205
  /** What a candidate is scored on inside {@link percentileSelection}. */
@@ -201,14 +238,18 @@ export interface CandidateUtility {
201
238
  export interface PercentileSelectionResult {
202
239
  /** The percentile actually used, clamped to [0, 1]. */
203
240
  alpha: number;
204
- /** True when alpha sits below the recommended floor of 0.3. */
241
+ /** True when alpha sits below the recommended floor of 0.3. Computed from
242
+ * `alpha` alone, so a refusal that had nothing to do with alpha (too few
243
+ * observations, a rejected block probability) does not raise it. */
205
244
  alpha_warning: boolean;
206
245
  /** Every candidate, in input order. */
207
246
  candidates: CandidateUtility[];
208
- /** Index of the candidate with the best percentile utility (the robust pick), or null for empty input. */
247
+ /** Index of the robust pick, or null for empty or refused input. */
209
248
  selected: number | null;
210
- /** Index of the candidate with the best point utility (the naive pick), or null for empty input. */
249
+ /** Index of the point winner, or null for empty or refused input. */
211
250
  point_argmax: number | null;
251
+ /** Null on accepted input; a refusal withholds the entire candidate field. */
252
+ input_error: string | null;
212
253
  /** Whether the two picks agree. Disagreement is the interesting case. */
213
254
  agrees_with_point_argmax: boolean;
214
255
  /** Optimism gap of the point winner: report this next to any headline utility. */
@@ -276,10 +317,13 @@ export interface CrowdingDecayPrior {
276
317
  }
277
318
  /**
278
319
  * A reason an agent was (or should be) demoted. The first five mirror the hard
279
- * eligibility gates in the scorer; the last three are advisory quality flags
280
- * that never gate.
320
+ * eligibility gates in the scorer; two more name the unavailability of a
321
+ * statistic the scorer does gate on. The last four are advisory quality flags
322
+ * that never gate. `selection_unavailable` is advisory because the scorer
323
+ * reports the selection axis and never consults it in `rank_eligible`, so an
324
+ * agent carrying it alone is still ranked.
281
325
  */
282
- export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
326
+ export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "deflation_unavailable" | "bootstrap_unavailable" | "selection_unavailable" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
283
327
  /** Every disqualification/quality signal that fired for one scored agent. */
284
328
  export interface DisqualificationReport {
285
329
  agent_id: string;
@@ -288,7 +332,14 @@ export interface DisqualificationReport {
288
332
  reasons: FailReason[];
289
333
  [k: string]: unknown;
290
334
  }
291
- /** The FULL verdict: LITE on the winner plus the multiple-testing family + PBO. */
335
+ /** Harvey-Liu-Zhu hurdle and its interpretation from the kernel. */
336
+ export interface HlzGate {
337
+ tStat: number | null;
338
+ tThreshold: number;
339
+ passed: boolean;
340
+ explanation: string;
341
+ }
342
+ /** The FULL verdict: LITE on the winner plus multiple testing, PBO and HLZ. */
292
343
  export interface FullVerdict {
293
344
  honesty: HonestyVerdict;
294
345
  /** White's Reality Check p-value over the field. */
@@ -299,8 +350,12 @@ export interface FullVerdict {
299
350
  spaConsistentP: number;
300
351
  /** Romano-Wolf step-down: which field members are significant at α. */
301
352
  stepDown: boolean[];
302
- /** CSCV Probability of Backtest Overfitting over the field. */
303
- pbo: number;
353
+ /** CSCV estimate; null when unavailable, with a reason in pboError. */
354
+ pbo: number | null;
355
+ hlz: HlzGate;
356
+ /** When present, p-values of 1 and all-false stepDown are refusal sentinels. */
357
+ snoopingError?: string;
358
+ pboError?: string;
304
359
  [k: string]: unknown;
305
360
  }
306
361
  /** Options for a regime-conditional comparison. Regime labels are caller inputs. */
@@ -311,7 +366,7 @@ export interface RegimeCompareOpts {
311
366
  }
312
367
  export interface ZagaSplit {
313
368
  n: number;
314
- zero_mass: number;
369
+ near_zero_return_mass: number;
315
370
  n_nonzero: number;
316
371
  positive_share: number;
317
372
  cont_mean: number;
@@ -326,7 +381,7 @@ export interface RegimeComparison {
326
381
  n_periods: number;
327
382
  a: ZagaSplit;
328
383
  b: ZagaSplit;
329
- zero_mass_gap: number;
384
+ near_zero_return_mass_gap: number;
330
385
  mean_gap: number;
331
386
  cont_mean_gap: number;
332
387
  ks_statistic: number;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@general-liquidity/sharpebench",
3
- "version": "0.18.4",
3
+ "version": "0.20.0",
4
4
  "description": "Luck-robust quantitative evaluation for trading agents: deflated Sharpe, pass^k reliability, process discipline, and risk gates, backed by the identical Rust kernel compiled to WebAssembly.",
5
5
  "keywords": [
6
6
  "trading",
package/pkg/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "sharpebench-wasm",
3
3
  "description": "WASM bindings for SharpeBench's deterministic, luck-robust trading-agent scoring kernel.",
4
- "version": "0.18.4",
4
+ "version": "0.20.0",
5
5
  "license": "MIT OR Apache-2.0",
6
6
  "repository": {
7
7
  "type": "git",
Binary file