@general-liquidity/sharpebench 0.19.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -37,17 +37,17 @@ console.log(greeks({ spot: 100, strike: 100, t_years: 1, rate: 0.05, vol: 0.2, i
37
37
  |---|---|
38
38
  | `score(submissions, config?)` | ranked `CompositeScore[]` |
39
39
  | `scoreAgent(submission, config?)` | one `CompositeScore` (deflated Sharpe, pass^k, process, rolling worst-case Sharpe) |
40
- | `selfAudit()` | `SelfAuditReport`, the benchmark's anti-gaming proof |
40
+ | `selfAudit()` | `SelfAuditReport`, results of the named anti-gaming regressions |
41
41
  | `auditBriefing(briefing, policy?)` | `BriefingAudit`, an input-side salience-bias audit |
42
42
  | `scoreAllocation(trajectory, policy?)` | `AllocationReport`, weight-vector validity plus L1 turnover |
43
43
  | `greeks(params)` | `GreeksResult`, Black-Scholes price, Greeks, and local exposure flags |
44
44
  | `canary(seed)` | `Canary`, a do-not-train contamination tripwire |
45
- | `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict |
46
- | `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, and PBO alongside the one-series verdict |
45
+ | `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict. `trialsSrStd` is annualized; `periodsPerYear` (default 252) says what a period is |
46
+ | `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, PBO and HLZ diagnostics alongside the one-series verdict |
47
47
  | `percentileSelection(candidates, opts?)` | Point winner versus bootstrap-percentile winner and optimism gaps |
48
48
  | `decomposeUncertainty(input)` | Aleatoric, epistemic, and distributional diagnostic legs |
49
49
  | `crowdingHalfLife(adoption, params)` | Caller-calibrated crowding-decay prior, reported but never gating |
50
- | `classifyDisqualification(submissions, config?)` | Named hard-gate and advisory reasons |
50
+ | `classifyDisqualification(submissions, config?)` | Named hard-gate, unavailability, and advisory reasons |
51
51
  | `regimeCompare(a, b, regimes, opts?)` | Regime-conditional distribution comparison and pooled-sign reversal |
52
52
 
53
53
  All inputs and outputs are fully typed (TypeScript declarations ship with the
@@ -56,6 +56,29 @@ committed golden on the Ubuntu CI host. The Rust CI separately pins the two
56
56
  committed golden fields on Linux, macOS, and Windows; this is not a claim about
57
57
  every possible input or platform.
58
58
 
59
+ ## Unavailable results
60
+
61
+ Check error fields before interpreting numeric diagnostics. `statisticsError`
62
+ marks withheld deflation; `snoopingError` marks withheld fieldwise tests, whose
63
+ compatibility sentinels are p-values of 1 and all-false `stepDown`. These are
64
+ not measured results. Unavailable PBO is `null` with `pboError`.
65
+
66
+ Nonfinite numeric diagnostics serialize as `null`, including a minimum track
67
+ record length with no finite solution. TypeScript callers must handle nulls.
68
+ `percentileSelection` reports `input_error` and null winner indices when any
69
+ candidate is unsupported. Its `alpha_warning` reports where `alpha` sits and
70
+ nothing else: a refusal that had nothing to do with alpha, such as too few
71
+ observations or a rejected block probability, leaves the flag false.
72
+
73
+ `classifyDisqualification` returns eleven reasons in three groups: five mirror
74
+ the scorer's hard eligibility gates, two name the unavailability of a statistic
75
+ the scorer gates on (`deflation_unavailable`, `bootstrap_unavailable`), and four
76
+ are advisory and never gate (`selection_unavailable`, `high_selection_gap`,
77
+ `is_rediscovery`, `oos_decay`). `selection_unavailable` is advisory because the
78
+ scorer reports `selection_gap` and never consults it in `rank_eligible`, so a
79
+ submission carrying only advisory reasons is still rank-eligible. Read
80
+ `rank_eligible`, not the presence of a reason. See the [error and migration guide](https://github.com/general-liquidity/sharpebench/blob/main/docs/book/src/wasm.md#statistical-unavailability-and-migration).
81
+
59
82
  ## Why luck-robust?
60
83
 
61
84
  Most agent leaderboards rank a raw Sharpe over a single short window, so they mostly measure noise. SharpeBench gates eligibility on Deflated Sharpe, pass^k reliability across every seed and window, stationary-bootstrap significance, process discipline, and the host drawdown mandate. PSR and the fieldwise multiple-testing family remain visible diagnostics. See the [benchmark repo](https://github.com/general-liquidity/sharpebench) for the full methodology.
package/dist/index.js CHANGED
@@ -117,8 +117,22 @@ function honestyConfigJson(opts) {
117
117
  throw new RangeError("nTrials must be an integer in 1..=4294967295");
118
118
  }
119
119
  const cfg = { n_trials: opts.nTrials };
120
- if (opts.trialsSrStd !== undefined)
120
+ if (opts.trialsSrStd !== undefined) {
121
+ // JSON.stringify turns NaN and Infinity into null, which the kernel reads as
122
+ // "omitted" and replaces with the 0.5 prior: refuse it before it crosses.
123
+ if (typeof opts.trialsSrStd !== "number" || !Number.isFinite(opts.trialsSrStd)) {
124
+ throw new RangeError("trialsSrStd must be a finite number (omit it for the 0.5 prior)");
125
+ }
121
126
  cfg.trials_sr_std = opts.trialsSrStd;
127
+ }
128
+ if (opts.periodsPerYear !== undefined) {
129
+ // JSON.stringify turns NaN and Infinity into null, which the kernel refuses
130
+ // rather than reading as "omitted"; say which input it was here instead.
131
+ if (typeof opts.periodsPerYear !== "number" || !Number.isFinite(opts.periodsPerYear)) {
132
+ throw new RangeError("periodsPerYear must be a finite number (omit it for the 252 default)");
133
+ }
134
+ cfg.periods_per_year = opts.periodsPerYear;
135
+ }
122
136
  if (opts.confidence !== undefined)
123
137
  cfg.confidence = opts.confidence;
124
138
  if (opts.borderline !== undefined)
@@ -144,6 +158,7 @@ function toHonestyVerdict(raw) {
144
158
  verdict: raw.verdict,
145
159
  explanation: raw.explanation,
146
160
  methodologyVersion: raw.methodology_version,
161
+ ...(raw.statistics_error === undefined ? {} : { statisticsError: raw.statistics_error }),
147
162
  };
148
163
  }
149
164
  /**
@@ -166,6 +181,7 @@ function isMySharpeReal(returns, opts) {
166
181
  */
167
182
  function isMySharpeRealFull(field, winnerIdx, opts) {
168
183
  const raw = parse(kernel.is_my_sharpe_real_full(JSON.stringify(field), winnerIdx, honestyConfigJson(opts)));
184
+ const hlz = raw.hlz;
169
185
  return {
170
186
  honesty: toHonestyVerdict(raw.honesty),
171
187
  realityCheckP: raw.reality_check_p,
@@ -173,6 +189,14 @@ function isMySharpeRealFull(field, winnerIdx, opts) {
173
189
  spaConsistentP: raw.spa_consistent_p,
174
190
  stepDown: raw.step_down,
175
191
  pbo: raw.pbo,
192
+ hlz: {
193
+ tStat: hlz.t_stat,
194
+ tThreshold: hlz.t_threshold,
195
+ passed: hlz.passed,
196
+ explanation: hlz.explanation,
197
+ },
198
+ ...(raw.snooping_error === undefined ? {} : { snoopingError: raw.snooping_error }),
199
+ ...(raw.pbo_error === undefined ? {} : { pboError: raw.pbo_error }),
176
200
  };
177
201
  }
178
202
  /**
package/dist/types.d.ts CHANGED
@@ -153,33 +153,53 @@ export type Verdict = "Pass" | "Borderline" | "Fail";
153
153
  export interface HonestyOpts {
154
154
  /** Number of strategy trials behind this result, integer 1..=4294967295. REQUIRED. */
155
155
  nTrials: number;
156
- /** Cross-trial Sharpe dispersion. Omit → estimated at 0.5 and flagged. */
156
+ /**
157
+ * **Annualized** cross-trial Sharpe dispersion, divided by
158
+ * `sqrt(periodsPerYear)` before use. Omit → estimated at 0.5 and flagged. A
159
+ * negative value yields a Fail verdict with `statisticsError`; NaN and
160
+ * infinity cannot cross JSON (they would arrive as the 0.5 prior) and throw a
161
+ * `RangeError`.
162
+ */
157
163
  trialsSrStd?: number;
164
+ /**
165
+ * Return periods per year for these returns (daily equities 252, daily crypto
166
+ * 365, hourly 8760, weekly 52). Omit → 252, flagged in the explanation. A
167
+ * non-positive value yields a Fail verdict with `statisticsError`; NaN and
168
+ * infinity cannot cross JSON and throw a `RangeError`.
169
+ */
170
+ periodsPerYear?: number;
158
171
  /** Deflated-Sharpe threshold for a Pass. Default 0.95. */
159
172
  confidence?: number;
160
173
  /** Deflated-Sharpe threshold for Borderline. Default 0.90. */
161
174
  borderline?: number;
162
- /** PSR / MinTRL benchmark Sharpe to beat. Default 0.0. */
175
+ /** **Per-period** PSR / MinTRL benchmark Sharpe to beat (not converted). Default 0.0. */
163
176
  srBenchmark?: number;
164
177
  }
165
- /** The LITE verdict: everything derivable from one return series. */
178
+ /** The LITE verdict. Nonfinite numeric diagnostics serialize as null, not zero. */
166
179
  export interface HonestyVerdict {
167
- sharpe: number;
180
+ sharpe: number | null;
168
181
  nObs: number;
169
- skew: number;
170
- kurtosis: number;
182
+ skew: number | null;
183
+ kurtosis: number | null;
171
184
  nTrials: number;
172
- expectedMaxSharpe: number;
173
- deflatedSharpe: number;
174
- probabilisticSharpe: number;
175
- /** `1 - deflatedSharpe`: probability the edge is a search artifact. */
176
- haircut: number;
177
- /** `sharpe * deflatedSharpe`: Sharpe discounted by survival probability. */
178
- haircutSharpe: number;
179
- minTrackRecordLen: number;
185
+ expectedMaxSharpe: number | null;
186
+ deflatedSharpe: number | null;
187
+ probabilisticSharpe: number | null;
188
+ /**
189
+ * `1 - deflatedSharpe`: the p-value of the test whose null is that this Sharpe
190
+ * is the best of `nTrials` zero-skill trials. Not the probability that the
191
+ * edge is a search artifact.
192
+ */
193
+ haircut: number | null;
194
+ /** `sharpe * deflatedSharpe`: the Sharpe scaled down by the deflated Sharpe. */
195
+ haircutSharpe: number | null;
196
+ /** Null when no finite track length is returned by the kernel. */
197
+ minTrackRecordLen: number | null;
180
198
  verdict: Verdict;
181
199
  explanation: string;
182
200
  methodologyVersion: string;
201
+ /** When present, deflation was withheld; its numeric sentinels are not estimates. */
202
+ statisticsError?: string;
183
203
  [k: string]: unknown;
184
204
  }
185
205
  /** What a candidate is scored on inside {@link percentileSelection}. */
@@ -218,14 +238,18 @@ export interface CandidateUtility {
218
238
  export interface PercentileSelectionResult {
219
239
  /** The percentile actually used, clamped to [0, 1]. */
220
240
  alpha: number;
221
- /** True when alpha sits below the recommended floor of 0.3. */
241
+ /** True when alpha sits below the recommended floor of 0.3. Computed from
242
+ * `alpha` alone, so a refusal that had nothing to do with alpha (too few
243
+ * observations, a rejected block probability) does not raise it. */
222
244
  alpha_warning: boolean;
223
245
  /** Every candidate, in input order. */
224
246
  candidates: CandidateUtility[];
225
- /** Index of the candidate with the best percentile utility (the robust pick), or null for empty input. */
247
+ /** Index of the robust pick, or null for empty or refused input. */
226
248
  selected: number | null;
227
- /** Index of the candidate with the best point utility (the naive pick), or null for empty input. */
249
+ /** Index of the point winner, or null for empty or refused input. */
228
250
  point_argmax: number | null;
251
+ /** Null on accepted input; a refusal withholds the entire candidate field. */
252
+ input_error: string | null;
229
253
  /** Whether the two picks agree. Disagreement is the interesting case. */
230
254
  agrees_with_point_argmax: boolean;
231
255
  /** Optimism gap of the point winner: report this next to any headline utility. */
@@ -293,10 +317,13 @@ export interface CrowdingDecayPrior {
293
317
  }
294
318
  /**
295
319
  * A reason an agent was (or should be) demoted. The first five mirror the hard
296
- * eligibility gates in the scorer; the last three are advisory quality flags
297
- * that never gate.
320
+ * eligibility gates in the scorer; two more name the unavailability of a
321
+ * statistic the scorer does gate on. The last four are advisory quality flags
322
+ * that never gate. `selection_unavailable` is advisory because the scorer
323
+ * reports the selection axis and never consults it in `rank_eligible`, so an
324
+ * agent carrying it alone is still ranked.
298
325
  */
299
- export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
326
+ export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "deflation_unavailable" | "bootstrap_unavailable" | "selection_unavailable" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
300
327
  /** Every disqualification/quality signal that fired for one scored agent. */
301
328
  export interface DisqualificationReport {
302
329
  agent_id: string;
@@ -305,7 +332,14 @@ export interface DisqualificationReport {
305
332
  reasons: FailReason[];
306
333
  [k: string]: unknown;
307
334
  }
308
- /** The FULL verdict: LITE on the winner plus the multiple-testing family + PBO. */
335
+ /** Harvey-Liu-Zhu hurdle and its interpretation from the kernel. */
336
+ export interface HlzGate {
337
+ tStat: number | null;
338
+ tThreshold: number;
339
+ passed: boolean;
340
+ explanation: string;
341
+ }
342
+ /** The FULL verdict: LITE on the winner plus multiple testing, PBO and HLZ. */
309
343
  export interface FullVerdict {
310
344
  honesty: HonestyVerdict;
311
345
  /** White's Reality Check p-value over the field. */
@@ -316,8 +350,12 @@ export interface FullVerdict {
316
350
  spaConsistentP: number;
317
351
  /** Romano-Wolf step-down: which field members are significant at α. */
318
352
  stepDown: boolean[];
319
- /** CSCV Probability of Backtest Overfitting over the field. */
320
- pbo: number;
353
+ /** CSCV estimate; null when unavailable, with a reason in pboError. */
354
+ pbo: number | null;
355
+ hlz: HlzGate;
356
+ /** When present, p-values of 1 and all-false stepDown are refusal sentinels. */
357
+ snoopingError?: string;
358
+ pboError?: string;
321
359
  [k: string]: unknown;
322
360
  }
323
361
  /** Options for a regime-conditional comparison. Regime labels are caller inputs. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@general-liquidity/sharpebench",
3
- "version": "0.19.0",
3
+ "version": "0.20.0",
4
4
  "description": "Luck-robust quantitative evaluation for trading agents: deflated Sharpe, pass^k reliability, process discipline, and risk gates, backed by the identical Rust kernel compiled to WebAssembly.",
5
5
  "keywords": [
6
6
  "trading",
package/pkg/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "sharpebench-wasm",
3
3
  "description": "WASM bindings for SharpeBench's deterministic, luck-robust trading-agent scoring kernel.",
4
- "version": "0.19.0",
4
+ "version": "0.20.0",
5
5
  "license": "MIT OR Apache-2.0",
6
6
  "repository": {
7
7
  "type": "git",
Binary file