@general-liquidity/sharpebench 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- import type { AgentSubmission, AllocationPolicy, AllocationReport, AllocationTrajectory, Briefing, BriefingAudit, BriefingPolicy, Canary, CompositeScore, FullVerdict, GreeksParams, GreeksResult, HonestyOpts, HonestyVerdict, ScoreConfig, SelfAuditReport } from "./types.js";
1
+ import type { AgentSubmission, AllocationPolicy, AllocationReport, AllocationTrajectory, Briefing, BriefingAudit, BriefingPolicy, Canary, CompositeScore, CrowdingDecayPrior, CrowdingParams, DisqualificationReport, FullVerdict, GreeksParams, GreeksResult, HonestyOpts, HonestyVerdict, PercentileSelectionOpts, PercentileSelectionResult, ScoreConfig, SelfAuditReport, UncertaintyInput, UncertaintySplit } from "./types.js";
2
2
  export * from "./types.js";
3
3
  /**
4
4
  * Score and rank a field of submissions on the luck-robust composite. Returns the
@@ -37,3 +37,44 @@ export declare function isMySharpeReal(returns: number[], opts: HonestyOpts): Ho
37
37
  * whose LITE verdict is reported.
38
38
  */
39
39
  export declare function isMySharpeRealFull(field: number[][], winnerIdx: number, opts: HonestyOpts): FullVerdict;
40
+ /**
41
+ * Rank candidate return streams on a percentile of their bootstrapped utility
42
+ * instead of the point-estimate argmax, so the winner has to be good on most
43
+ * resampled histories rather than on the one that happened to be observed.
44
+ *
45
+ * `candidates` is one per-period return array per candidate. The result names
46
+ * both the point winner and the percentile winner; disagreement between them is
47
+ * the interesting case, and the point winner's `optimism_gap` is the number to
48
+ * report next to any headline utility. An `alpha` below 0.3 still computes but
49
+ * sets `alpha_warning` (the extreme lower tail is decided by a handful of
50
+ * unlucky resamples). Deterministic given (candidates, seed).
51
+ */
52
+ export declare function percentileSelection(candidates: number[][], opts?: PercentileSelectionOpts): PercentileSelectionResult;
53
+ /**
54
+ * Decompose the uncertainty behind one scored case into three legs, reported
55
+ * side by side and never summed: aleatoric (irreducible outcome noise; stop
56
+ * looking), epistemic (reducible ignorance; keep looking), and distributional
57
+ * (the reference cannot vouch for this case). Any input may be omitted; the
58
+ * matching leg then reports what it honestly can on no evidence.
59
+ *
60
+ * The result's `epistemic_caveat` is load-bearing: the epistemic leg is a lower
61
+ * bound, never an upper one, because unanimous or correlated signals understate
62
+ * it. Treat only high readings as informative.
63
+ */
64
+ export declare function decomposeUncertainty(input: UncertaintyInput): UncertaintySplit;
65
+ /**
66
+ * Expected edge half-life under a crowding decay model:
67
+ * `ln2 / (theta + deltaMax * adoption^curvature)`, in periods of the caller's
68
+ * IC series. The output is a model prior, reported never gating: it comes out
69
+ * of a model, not out of a dataset, and nothing should rank on it. There is
70
+ * deliberately no default calibration; the caller owns every rate.
71
+ */
72
+ export declare function crowdingHalfLife(adoption: number, params: CrowdingParams): CrowdingDecayPrior;
73
+ /**
74
+ * Name every disqualification/quality signal that fired for each agent in a
75
+ * field of submissions (the same input {@link score} takes). Pure legibility
76
+ * over the composite score: the first five reasons mirror the scorer's hard
77
+ * eligibility gates, the advisory ones (high_selection_gap, is_rediscovery,
78
+ * oos_decay) never gate, and nothing here changes eligibility semantics.
79
+ */
80
+ export declare function classifyDisqualification(submissions: AgentSubmission[], config?: ScoreConfig): DisqualificationReport[];
package/dist/index.js CHANGED
@@ -45,6 +45,10 @@ exports.greeks = greeks;
45
45
  exports.canary = canary;
46
46
  exports.isMySharpeReal = isMySharpeReal;
47
47
  exports.isMySharpeRealFull = isMySharpeRealFull;
48
+ exports.percentileSelection = percentileSelection;
49
+ exports.decomposeUncertainty = decomposeUncertainty;
50
+ exports.crowdingHalfLife = crowdingHalfLife;
51
+ exports.classifyDisqualification = classifyDisqualification;
48
52
  /**
49
53
  * `@general-liquidity/sharpebench` — the luck-robust scoring kernel for AI trading
50
54
  * agents, as a typed JS API over the *identical* Rust kernel that powers the
@@ -166,3 +170,75 @@ function isMySharpeRealFull(field, winnerIdx, opts) {
166
170
  pbo: raw.pbo,
167
171
  };
168
172
  }
173
+ /**
174
+ * Rank candidate return streams on a percentile of their bootstrapped utility
175
+ * instead of the point-estimate argmax, so the winner has to be good on most
176
+ * resampled histories rather than on the one that happened to be observed.
177
+ *
178
+ * `candidates` is one per-period return array per candidate. The result names
179
+ * both the point winner and the percentile winner; disagreement between them is
180
+ * the interesting case, and the point winner's `optimism_gap` is the number to
181
+ * report next to any headline utility. An `alpha` below 0.3 still computes but
182
+ * sets `alpha_warning` (the extreme lower tail is decided by a handful of
183
+ * unlucky resamples). Deterministic given (candidates, seed).
184
+ */
185
+ function percentileSelection(candidates, opts) {
186
+ const params = {};
187
+ if (opts?.utility !== undefined)
188
+ params.utility = opts.utility;
189
+ if (opts?.alpha !== undefined)
190
+ params.alpha = opts.alpha;
191
+ if (opts?.seed !== undefined)
192
+ params.seed = opts.seed;
193
+ if (opts?.nBoot !== undefined)
194
+ params.n_boot = opts.nBoot;
195
+ if (opts?.blockProb !== undefined)
196
+ params.block_prob = opts.blockProb;
197
+ return parse(kernel.percentile_selection(JSON.stringify(candidates), optJson(params)));
198
+ }
199
+ /**
200
+ * Decompose the uncertainty behind one scored case into three legs, reported
201
+ * side by side and never summed: aleatoric (irreducible outcome noise; stop
202
+ * looking), epistemic (reducible ignorance; keep looking), and distributional
203
+ * (the reference cannot vouch for this case). Any input may be omitted; the
204
+ * matching leg then reports what it honestly can on no evidence.
205
+ *
206
+ * The result's `epistemic_caveat` is load-bearing: the epistemic leg is a lower
207
+ * bound, never an upper one, because unanimous or correlated signals understate
208
+ * it. Treat only high readings as informative.
209
+ */
210
+ function decomposeUncertainty(input) {
211
+ return parse(kernel.decompose_uncertainty(JSON.stringify({
212
+ outcomes: input.outcomes,
213
+ signals: input.signals,
214
+ case_returns: input.caseReturns,
215
+ reference_returns: input.referenceReturns,
216
+ })));
217
+ }
218
+ /**
219
+ * Expected edge half-life under a crowding decay model:
220
+ * `ln2 / (theta + deltaMax * adoption^curvature)`, in periods of the caller's
221
+ * IC series. The output is a model prior, reported never gating: it comes out
222
+ * of a model, not out of a dataset, and nothing should rank on it. There is
223
+ * deliberately no default calibration; the caller owns every rate.
224
+ */
225
+ function crowdingHalfLife(adoption, params) {
226
+ const cfg = {
227
+ adoption,
228
+ theta: params.theta,
229
+ delta_max: params.deltaMax,
230
+ };
231
+ if (params.curvature !== undefined)
232
+ cfg.curvature = params.curvature;
233
+ return parse(kernel.crowding_half_life(JSON.stringify(cfg)));
234
+ }
235
+ /**
236
+ * Name every disqualification/quality signal that fired for each agent in a
237
+ * field of submissions (the same input {@link score} takes). Pure legibility
238
+ * over the composite score: the first five reasons mirror the scorer's hard
239
+ * eligibility gates, the advisory ones (high_selection_gap, is_rediscovery,
240
+ * oos_decay) never gate, and nothing here changes eligibility semantics.
241
+ */
242
+ function classifyDisqualification(submissions, config) {
243
+ return parse(kernel.classify_disqualification(JSON.stringify(submissions), optJson(config)));
244
+ }
package/dist/types.d.ts CHANGED
@@ -165,6 +165,129 @@ export interface HonestyVerdict {
165
165
  methodologyVersion: string;
166
166
  [k: string]: unknown;
167
167
  }
168
+ /** What a candidate is scored on inside {@link percentileSelection}. */
169
+ export type SelectionUtility = "mean_return" | "sharpe";
170
+ /** Options for {@link percentileSelection}. Omit for the recommended defaults. */
171
+ export interface PercentileSelectionOpts {
172
+ /** Utility each candidate is scored on. Default `"mean_return"`. */
173
+ utility?: SelectionUtility;
174
+ /**
175
+ * Percentile of the bootstrapped utility distribution to rank on, in [0, 1].
176
+ * Default 0.5 (the middle of the band). Below 0.3 the result carries
177
+ * `alphaWarning: true`: the extreme lower tail is decided by a handful of
178
+ * unlucky resamples. The result is still computed; the warning flags a
179
+ * choice, it does not veto one.
180
+ */
181
+ alpha?: number;
182
+ /** PRNG seed; the result is deterministic given (data, seed). Default 0. */
183
+ seed?: number;
184
+ /** Bootstrap resamples per candidate. Default 2000. */
185
+ nBoot?: number;
186
+ /** Stationary-bootstrap block-restart probability. Default 0.1. */
187
+ blockProb?: number;
188
+ }
189
+ /** Per-candidate result inside a {@link PercentileSelectionResult}. */
190
+ export interface CandidateUtility {
191
+ /** Position of this candidate in the input array. */
192
+ index: number;
193
+ /** Utility on the observed path: the number a naive argmax would rank on. */
194
+ point_utility: number;
195
+ /** The alpha percentile of the bootstrapped utility distribution. */
196
+ percentile_utility: number;
197
+ /** `point_utility - percentile_utility`: how much of the headline number fails to survive resampling. */
198
+ optimism_gap: number;
199
+ }
200
+ /** Selection on a percentile of a bootstrapped utility distribution. */
201
+ export interface PercentileSelectionResult {
202
+ /** The percentile actually used, clamped to [0, 1]. */
203
+ alpha: number;
204
+ /** True when alpha sits below the recommended floor of 0.3. */
205
+ alpha_warning: boolean;
206
+ /** Every candidate, in input order. */
207
+ candidates: CandidateUtility[];
208
+ /** Index of the candidate with the best percentile utility (the robust pick), or null for empty input. */
209
+ selected: number | null;
210
+ /** Index of the candidate with the best point utility (the naive pick), or null for empty input. */
211
+ point_argmax: number | null;
212
+ /** Whether the two picks agree. Disagreement is the interesting case. */
213
+ agrees_with_point_argmax: boolean;
214
+ /** Optimism gap of the point winner: report this next to any headline utility. */
215
+ point_winner_optimism: number;
216
+ [k: string]: unknown;
217
+ }
218
+ /** Inputs for {@link decomposeUncertainty}. Every field is optional; a missing leg's input reads as empty. */
219
+ export interface UncertaintyInput {
220
+ /** Realized binary outcomes (true or 1 = the call was right). Drives the aleatoric leg. */
221
+ outcomes?: Array<boolean | number>;
222
+ /** Independent per-decision confidence streams for the same decisions. Drives the epistemic leg. */
223
+ signals?: number[][];
224
+ /** The case's per-period returns. Drives the distributional leg (with the reference). */
225
+ caseReturns?: number[];
226
+ /** The reference per-period returns the case is compared against. */
227
+ referenceReturns?: number[];
228
+ }
229
+ /**
230
+ * The three legs of uncertainty behind one scored case, each on [0, 1] and
231
+ * reported side by side, never summed. High aleatoric says stop looking, high
232
+ * epistemic says keep looking, high distributional says the case is outside
233
+ * what the reference can speak to.
234
+ */
235
+ export interface UncertaintySplit {
236
+ /** Irreducible outcome noise (base-rate variance; 1 = a fair coin). */
237
+ aleatoric: number;
238
+ /** Reducible ignorance, from signal disagreement plus evidence thinness. */
239
+ epistemic: number;
240
+ /** Unlikeness to the reference series (location or dispersion shift). */
241
+ distributional: number;
242
+ /**
243
+ * Load-bearing limitation, spelled out by the kernel: the epistemic leg is a
244
+ * lower bound, never an upper one. Unanimous or correlated signals understate
245
+ * it, so treat only high readings as informative.
246
+ */
247
+ epistemic_caveat: string;
248
+ [k: string]: unknown;
249
+ }
250
+ /**
251
+ * Parameters of the crowding decay model. All rates are per period of the
252
+ * caller's IC series; there is deliberately no default calibration, because a
253
+ * stock calibration would smuggle a modelled number in as if it were measured.
254
+ */
255
+ export interface CrowdingParams {
256
+ /** Natural mean-reversion rate of the edge at zero adoption. */
257
+ theta: number;
258
+ /** Crowding decay rate at full adoption. */
259
+ deltaMax: number;
260
+ /** Exponent on adoption (1 = linear). Default 1. */
261
+ curvature?: number;
262
+ }
263
+ /** The expected half-life implied by the crowding model: a prior, not a measurement. */
264
+ export interface CrowdingDecayPrior {
265
+ /** Adoption used, clamped to [0, 1]. */
266
+ adoption: number;
267
+ /** theta, echoed back. */
268
+ natural_reversion: number;
269
+ /** The adoption-driven decay component delta(phi). */
270
+ crowding_decay: number;
271
+ /** `ln2 / (theta + delta(phi))` in periods, or null when the model says the edge never decays. */
272
+ expected_half_life: number | null;
273
+ /** Names what this is: a model prior, reported never gating. */
274
+ note: string;
275
+ [k: string]: unknown;
276
+ }
277
+ /**
278
+ * A reason an agent was (or should be) demoted. The first five mirror the hard
279
+ * eligibility gates in the scorer; the last three are advisory quality flags
280
+ * that never gate.
281
+ */
282
+ export type FailReason = "failed_pass_k" | "dsr_below_bar" | "process_violation" | "bootstrap_insignificant" | "mandate_breached" | "high_selection_gap" | "is_rediscovery" | "oos_decay";
283
+ /** Every disqualification/quality signal that fired for one scored agent. */
284
+ export interface DisqualificationReport {
285
+ agent_id: string;
286
+ rank_eligible: boolean;
287
+ /** Reasons in stable order; empty means no signal fired. */
288
+ reasons: FailReason[];
289
+ [k: string]: unknown;
290
+ }
168
291
  /** The FULL verdict: LITE on the winner plus the multiple-testing family + PBO. */
169
292
  export interface FullVerdict {
170
293
  honesty: HonestyVerdict;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@general-liquidity/sharpebench",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "description": "The luck-robust scoring kernel for AI trading agents — deflated Sharpe, pass^k reliability, and process-discipline gates. The identical Rust kernel as the SharpeBench benchmark, compiled to WASM.",
5
5
  "keywords": [
6
6
  "trading",
package/pkg/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "sharpebench-wasm",
3
3
  "description": "WASM bindings for the SharpeBench scoring kernel (consumed by Gordon/Bun via the identical kernel).",
4
- "version": "0.4.0",
4
+ "version": "0.5.0",
5
5
  "license": "MIT OR Apache-2.0",
6
6
  "repository": {
7
7
  "type": "git",
@@ -5,12 +5,20 @@ export function audit_briefing(briefing_json: string, policy_json: string): stri
5
5
 
6
6
  export function canary(seed: string): string;
7
7
 
8
+ export function classify_disqualification(submissions_json: string, config_json: string): string;
9
+
10
+ export function crowding_half_life(params_json: string): string;
11
+
12
+ export function decompose_uncertainty(input_json: string): string;
13
+
8
14
  export function greeks(params_json: string): string;
9
15
 
10
16
  export function is_my_sharpe_real(returns_json: string, config_json: string): string;
11
17
 
12
18
  export function is_my_sharpe_real_full(field_json: string, winner_idx: number, config_json: string): string;
13
19
 
20
+ export function percentile_selection(candidates_json: string, params_json: string): string;
21
+
14
22
  export function score(submissions_json: string, config_json: string): string;
15
23
 
16
24
  export function score_agent(submission_json: string, config_json: string): string;
@@ -43,6 +43,69 @@ function canary(seed) {
43
43
  }
44
44
  exports.canary = canary;
45
45
 
46
+ /**
47
+ * @param {string} submissions_json
48
+ * @param {string} config_json
49
+ * @returns {string}
50
+ */
51
+ function classify_disqualification(submissions_json, config_json) {
52
+ let deferred3_0;
53
+ let deferred3_1;
54
+ try {
55
+ const ptr0 = passStringToWasm0(submissions_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
56
+ const len0 = WASM_VECTOR_LEN;
57
+ const ptr1 = passStringToWasm0(config_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
58
+ const len1 = WASM_VECTOR_LEN;
59
+ const ret = wasm.classify_disqualification(ptr0, len0, ptr1, len1);
60
+ deferred3_0 = ret[0];
61
+ deferred3_1 = ret[1];
62
+ return getStringFromWasm0(ret[0], ret[1]);
63
+ } finally {
64
+ wasm.__wbindgen_free(deferred3_0, deferred3_1, 1);
65
+ }
66
+ }
67
+ exports.classify_disqualification = classify_disqualification;
68
+
69
+ /**
70
+ * @param {string} params_json
71
+ * @returns {string}
72
+ */
73
+ function crowding_half_life(params_json) {
74
+ let deferred2_0;
75
+ let deferred2_1;
76
+ try {
77
+ const ptr0 = passStringToWasm0(params_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
78
+ const len0 = WASM_VECTOR_LEN;
79
+ const ret = wasm.crowding_half_life(ptr0, len0);
80
+ deferred2_0 = ret[0];
81
+ deferred2_1 = ret[1];
82
+ return getStringFromWasm0(ret[0], ret[1]);
83
+ } finally {
84
+ wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
85
+ }
86
+ }
87
+ exports.crowding_half_life = crowding_half_life;
88
+
89
+ /**
90
+ * @param {string} input_json
91
+ * @returns {string}
92
+ */
93
+ function decompose_uncertainty(input_json) {
94
+ let deferred2_0;
95
+ let deferred2_1;
96
+ try {
97
+ const ptr0 = passStringToWasm0(input_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
98
+ const len0 = WASM_VECTOR_LEN;
99
+ const ret = wasm.decompose_uncertainty(ptr0, len0);
100
+ deferred2_0 = ret[0];
101
+ deferred2_1 = ret[1];
102
+ return getStringFromWasm0(ret[0], ret[1]);
103
+ } finally {
104
+ wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
105
+ }
106
+ }
107
+ exports.decompose_uncertainty = decompose_uncertainty;
108
+
46
109
  /**
47
110
  * @param {string} params_json
48
111
  * @returns {string}
@@ -110,6 +173,29 @@ function is_my_sharpe_real_full(field_json, winner_idx, config_json) {
110
173
  }
111
174
  exports.is_my_sharpe_real_full = is_my_sharpe_real_full;
112
175
 
176
+ /**
177
+ * @param {string} candidates_json
178
+ * @param {string} params_json
179
+ * @returns {string}
180
+ */
181
+ function percentile_selection(candidates_json, params_json) {
182
+ let deferred3_0;
183
+ let deferred3_1;
184
+ try {
185
+ const ptr0 = passStringToWasm0(candidates_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
186
+ const len0 = WASM_VECTOR_LEN;
187
+ const ptr1 = passStringToWasm0(params_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
188
+ const len1 = WASM_VECTOR_LEN;
189
+ const ret = wasm.percentile_selection(ptr0, len0, ptr1, len1);
190
+ deferred3_0 = ret[0];
191
+ deferred3_1 = ret[1];
192
+ return getStringFromWasm0(ret[0], ret[1]);
193
+ } finally {
194
+ wasm.__wbindgen_free(deferred3_0, deferred3_1, 1);
195
+ }
196
+ }
197
+ exports.percentile_selection = percentile_selection;
198
+
113
199
  /**
114
200
  * @param {string} submissions_json
115
201
  * @param {string} config_json
Binary file
@@ -3,9 +3,13 @@
3
3
  export const memory: WebAssembly.Memory;
4
4
  export const audit_briefing: (a: number, b: number, c: number, d: number) => [number, number];
5
5
  export const canary: (a: number, b: number) => [number, number];
6
+ export const classify_disqualification: (a: number, b: number, c: number, d: number) => [number, number];
7
+ export const crowding_half_life: (a: number, b: number) => [number, number];
8
+ export const decompose_uncertainty: (a: number, b: number) => [number, number];
6
9
  export const greeks: (a: number, b: number) => [number, number];
7
10
  export const is_my_sharpe_real: (a: number, b: number, c: number, d: number) => [number, number];
8
11
  export const is_my_sharpe_real_full: (a: number, b: number, c: number, d: number, e: number) => [number, number];
12
+ export const percentile_selection: (a: number, b: number, c: number, d: number) => [number, number];
9
13
  export const score: (a: number, b: number, c: number, d: number) => [number, number];
10
14
  export const score_agent: (a: number, b: number, c: number, d: number) => [number, number];
11
15
  export const score_allocation: (a: number, b: number, c: number, d: number) => [number, number];