@general-liquidity/sharpebench 0.14.1 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -11
- package/dist/index.d.ts +7 -1
- package/dist/index.js +16 -0
- package/dist/types.d.ts +38 -0
- package/package.json +2 -1
- package/pkg/package.json +2 -2
- package/pkg/sharpebench.d.ts +2 -0
- package/pkg/sharpebench.js +29 -0
- package/pkg/sharpebench_bg.wasm +0 -0
- package/pkg/sharpebench_bg.wasm.d.ts +1 -0
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
**The luck-robust scoring kernel for AI trading agents.**
|
|
4
4
|
|
|
5
|
-
Rank agents on risk-adjusted *skill that survives deflation
|
|
5
|
+
Rank agents on risk-adjusted *skill that survives deflation*, not the luckiest run over one quarter. This is the **identical Rust kernel** that powers the [SharpeBench](https://github.com/general-liquidity/sharpebench) benchmark, compiled to WebAssembly, with a typed JS/TS API. No native add-on and no network: pure deterministic scoring.
|
|
6
6
|
|
|
7
7
|
```bash
|
|
8
8
|
npm install @general-liquidity/sharpebench
|
|
@@ -13,15 +13,18 @@ npm install @general-liquidity/sharpebench
|
|
|
13
13
|
```ts
|
|
14
14
|
import { score, greeks, selfAudit } from "@general-liquidity/sharpebench";
|
|
15
15
|
|
|
16
|
-
// Rank a field. Raw return is reported but is NEVER the rank key
|
|
17
|
-
// only if its edge survives deflation, pass^k reliability,
|
|
16
|
+
// Rank a field. Raw return is reported but is NEVER the rank key. An agent ranks
|
|
17
|
+
// only if its edge survives deflation, pass^k reliability, process discipline,
|
|
18
|
+
// the stationary-bootstrap null, and the configured mandate. This tiny input
|
|
19
|
+
// demonstrates the API shape; use the benchmark's full run geometry for a
|
|
20
|
+
// meaningful eligibility verdict.
|
|
18
21
|
const board = score([
|
|
19
22
|
{ agent_id: "skilled", runs: [{ returns: [0.002, 0.0021, 0.0019, 0.002] }] },
|
|
20
23
|
{ agent_id: "lucky", runs: [{ returns: [0.05, 0, 0, 0] }] },
|
|
21
24
|
]);
|
|
22
25
|
console.log(board[0].agent_id, board[0].deflated_sharpe, board[0].rank_eligible);
|
|
23
26
|
|
|
24
|
-
//
|
|
27
|
+
// Run the scorer's built-in checks against its catalogued gaming attacks.
|
|
25
28
|
console.log(selfAudit().all_defended); // true
|
|
26
29
|
|
|
27
30
|
// Options tail-risk: a short-gamma position a linear Sharpe can't see.
|
|
@@ -34,17 +37,28 @@ console.log(greeks({ spot: 100, strike: 100, t_years: 1, rate: 0.05, vol: 0.2, i
|
|
|
34
37
|
|---|---|
|
|
35
38
|
| `score(submissions, config?)` | ranked `CompositeScore[]` |
|
|
36
39
|
| `scoreAgent(submission, config?)` | one `CompositeScore` (deflated Sharpe, pass^k, process, rolling worst-case Sharpe) |
|
|
37
|
-
| `selfAudit()` | `SelfAuditReport
|
|
38
|
-
| `auditBriefing(briefing, policy?)` | `BriefingAudit
|
|
39
|
-
| `scoreAllocation(trajectory, policy?)` | `AllocationReport
|
|
40
|
-
| `greeks(params)` | `GreeksResult
|
|
41
|
-
| `canary(seed)` | `Canary
|
|
40
|
+
| `selfAudit()` | `SelfAuditReport`, the benchmark's anti-gaming proof |
|
|
41
|
+
| `auditBriefing(briefing, policy?)` | `BriefingAudit`, an input-side salience-bias audit |
|
|
42
|
+
| `scoreAllocation(trajectory, policy?)` | `AllocationReport`, weight-vector validity plus L1 turnover |
|
|
43
|
+
| `greeks(params)` | `GreeksResult`, Black-Scholes price, Greeks, and tail-selling risk |
|
|
44
|
+
| `canary(seed)` | `Canary`, a do-not-train contamination tripwire |
|
|
45
|
+
| `isMySharpeReal(returns, opts)` | One-series deflation, PSR, haircut, MinTRL, and verdict |
|
|
46
|
+
| `isMySharpeRealFull(field, winner, opts)` | Fieldwise Reality Check, SPA, step-down, and PBO alongside the one-series verdict |
|
|
47
|
+
| `percentileSelection(candidates, opts?)` | Point winner versus bootstrap-percentile winner and optimism gaps |
|
|
48
|
+
| `decomposeUncertainty(input)` | Aleatoric, epistemic, and distributional diagnostic legs |
|
|
49
|
+
| `crowdingHalfLife(adoption, params)` | Caller-calibrated crowding-decay prior, reported but never gating |
|
|
50
|
+
| `classifyDisqualification(submissions, config?)` | Named hard-gate and advisory reasons |
|
|
51
|
+
| `regimeCompare(a, b, regimes, opts?)` | Regime-conditional distribution comparison and pooled-sign reversal |
|
|
42
52
|
|
|
43
|
-
All inputs and outputs are fully typed (TypeScript declarations ship with the
|
|
53
|
+
All inputs and outputs are fully typed (TypeScript declarations ship with the
|
|
54
|
+
package). The npm tests compare the WASM package with the native kernel and
|
|
55
|
+
committed golden on the Ubuntu CI host. The Rust CI separately pins the two
|
|
56
|
+
committed golden fields on Linux, macOS, and Windows; this is not a claim about
|
|
57
|
+
every possible input or platform.
|
|
44
58
|
|
|
45
59
|
## Why luck-robust?
|
|
46
60
|
|
|
47
|
-
Most agent leaderboards rank a raw Sharpe over a single short window
|
|
61
|
+
Most agent leaderboards rank a raw Sharpe over a single short window, so they mostly measure noise. SharpeBench gates eligibility on Deflated Sharpe, pass^k reliability across every seed and window, stationary-bootstrap significance, process discipline, and the host drawdown mandate. PSR and the fieldwise multiple-testing family remain visible diagnostics. See the [benchmark repo](https://github.com/general-liquidity/sharpebench) for the full methodology.
|
|
48
62
|
|
|
49
63
|
## License
|
|
50
64
|
|
package/dist/index.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { AgentSubmission, AllocationPolicy, AllocationReport, AllocationTrajectory, Briefing, BriefingAudit, BriefingPolicy, Canary, CompositeScore, CrowdingDecayPrior, CrowdingParams, DisqualificationReport, FullVerdict, GreeksParams, GreeksResult, HonestyOpts, HonestyVerdict, PercentileSelectionOpts, PercentileSelectionResult, ScoreConfig, SelfAuditReport, UncertaintyInput, UncertaintySplit } from "./types.js";
|
|
1
|
+
import type { AgentSubmission, AllocationPolicy, AllocationReport, AllocationTrajectory, Briefing, BriefingAudit, BriefingPolicy, Canary, CompositeScore, CrowdingDecayPrior, CrowdingParams, DisqualificationReport, FullVerdict, GreeksParams, GreeksResult, HonestyOpts, HonestyVerdict, PercentileSelectionOpts, PercentileSelectionResult, RegimeCompareOpts, RegimeDistributionReport, ScoreConfig, SelfAuditReport, UncertaintyInput, UncertaintySplit } from "./types.js";
|
|
2
2
|
export * from "./types.js";
|
|
3
3
|
/**
|
|
4
4
|
* Score and rank a field of submissions on the luck-robust composite. Returns the
|
|
@@ -78,3 +78,9 @@ export declare function crowdingHalfLife(adoption: number, params: CrowdingParam
|
|
|
78
78
|
* oos_decay) never gate, and nothing here changes eligibility semantics.
|
|
79
79
|
*/
|
|
80
80
|
export declare function classifyDisqualification(submissions: AgentSubmission[], config?: ScoreConfig): DisqualificationReport[];
|
|
81
|
+
/**
|
|
82
|
+
* Compare two aligned return streams inside caller-supplied market regimes.
|
|
83
|
+
* The report separates zero/no-trade mass from continuous returns and names a
|
|
84
|
+
* pooled edge whose sign reverses in a sufficiently supported regime.
|
|
85
|
+
*/
|
|
86
|
+
export declare function regimeCompare(returnsA: number[], returnsB: number[], regimes: string[], opts?: RegimeCompareOpts): RegimeDistributionReport;
|
package/dist/index.js
CHANGED
|
@@ -49,6 +49,7 @@ exports.percentileSelection = percentileSelection;
|
|
|
49
49
|
exports.decomposeUncertainty = decomposeUncertainty;
|
|
50
50
|
exports.crowdingHalfLife = crowdingHalfLife;
|
|
51
51
|
exports.classifyDisqualification = classifyDisqualification;
|
|
52
|
+
exports.regimeCompare = regimeCompare;
|
|
52
53
|
/**
|
|
53
54
|
* `@general-liquidity/sharpebench` — the luck-robust scoring kernel for AI trading
|
|
54
55
|
* agents, as a typed JS API over the *identical* Rust kernel that powers the
|
|
@@ -242,3 +243,18 @@ function crowdingHalfLife(adoption, params) {
|
|
|
242
243
|
function classifyDisqualification(submissions, config) {
|
|
243
244
|
return parse(kernel.classify_disqualification(JSON.stringify(submissions), optJson(config)));
|
|
244
245
|
}
|
|
246
|
+
/**
|
|
247
|
+
* Compare two aligned return streams inside caller-supplied market regimes.
|
|
248
|
+
* The report separates zero/no-trade mass from continuous returns and names a
|
|
249
|
+
* pooled edge whose sign reverses in a sufficiently supported regime.
|
|
250
|
+
*/
|
|
251
|
+
function regimeCompare(returnsA, returnsB, regimes, opts) {
|
|
252
|
+
const options = {};
|
|
253
|
+
if (opts?.zeroTol !== undefined)
|
|
254
|
+
options.zero_tol = opts.zeroTol;
|
|
255
|
+
if (opts?.minPeriods !== undefined)
|
|
256
|
+
options.min_periods = opts.minPeriods;
|
|
257
|
+
if (opts?.tieTol !== undefined)
|
|
258
|
+
options.tie_tol = opts.tieTol;
|
|
259
|
+
return parse(kernel.regime_compare(JSON.stringify(returnsA), JSON.stringify(returnsB), JSON.stringify(regimes), optJson(options)));
|
|
260
|
+
}
|
package/dist/types.d.ts
CHANGED
|
@@ -303,3 +303,41 @@ export interface FullVerdict {
|
|
|
303
303
|
pbo: number;
|
|
304
304
|
[k: string]: unknown;
|
|
305
305
|
}
|
|
306
|
+
/** Options for a regime-conditional comparison. Regime labels are caller inputs. */
|
|
307
|
+
export interface RegimeCompareOpts {
|
|
308
|
+
zeroTol?: number;
|
|
309
|
+
minPeriods?: number;
|
|
310
|
+
tieTol?: number;
|
|
311
|
+
}
|
|
312
|
+
export interface ZagaSplit {
|
|
313
|
+
n: number;
|
|
314
|
+
zero_mass: number;
|
|
315
|
+
n_nonzero: number;
|
|
316
|
+
positive_share: number;
|
|
317
|
+
cont_mean: number;
|
|
318
|
+
cont_sd: number;
|
|
319
|
+
cont_median: number;
|
|
320
|
+
gamma_shape: number;
|
|
321
|
+
gamma_rate: number;
|
|
322
|
+
pooled_mean: number;
|
|
323
|
+
}
|
|
324
|
+
export interface RegimeComparison {
|
|
325
|
+
regime: string;
|
|
326
|
+
n_periods: number;
|
|
327
|
+
a: ZagaSplit;
|
|
328
|
+
b: ZagaSplit;
|
|
329
|
+
zero_mass_gap: number;
|
|
330
|
+
mean_gap: number;
|
|
331
|
+
cont_mean_gap: number;
|
|
332
|
+
ks_statistic: number;
|
|
333
|
+
edge_sign: -1 | 0 | 1;
|
|
334
|
+
counted: boolean;
|
|
335
|
+
}
|
|
336
|
+
export interface RegimeDistributionReport {
|
|
337
|
+
regimes: RegimeComparison[];
|
|
338
|
+
pooled_mean_gap: number;
|
|
339
|
+
pooled_edge_sign: -1 | 0 | 1;
|
|
340
|
+
reversal_regimes: string[];
|
|
341
|
+
pooled_hides_reversal: boolean;
|
|
342
|
+
edge_dispersion: number;
|
|
343
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@general-liquidity/sharpebench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.16.0",
|
|
4
4
|
"description": "The luck-robust scoring kernel for AI trading agents — deflated Sharpe, pass^k reliability, and process-discipline gates. The identical Rust kernel as the SharpeBench benchmark, compiled to WASM.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"trading",
|
|
@@ -29,6 +29,7 @@
|
|
|
29
29
|
"scripts": {
|
|
30
30
|
"build": "tsc",
|
|
31
31
|
"test": "node --test \"test/**/*.test.js\"",
|
|
32
|
+
"smoke-install": "node test/smoke-install.mjs",
|
|
32
33
|
"prepublishOnly": "tsc && node -e \"require('fs').rmSync('pkg/.gitignore',{force:true})\""
|
|
33
34
|
},
|
|
34
35
|
"devDependencies": {
|
package/pkg/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sharpebench-wasm",
|
|
3
3
|
"description": "WASM bindings for the SharpeBench scoring kernel (consumed by Gordon/Bun via the identical kernel).",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.16.0",
|
|
5
5
|
"license": "MIT OR Apache-2.0",
|
|
6
6
|
"repository": {
|
|
7
7
|
"type": "git",
|
|
@@ -14,4 +14,4 @@
|
|
|
14
14
|
],
|
|
15
15
|
"main": "sharpebench.js",
|
|
16
16
|
"types": "sharpebench.d.ts"
|
|
17
|
-
}
|
|
17
|
+
}
|
package/pkg/sharpebench.d.ts
CHANGED
|
@@ -19,6 +19,8 @@ export function is_my_sharpe_real_full(field_json: string, winner_idx: number, c
|
|
|
19
19
|
|
|
20
20
|
export function percentile_selection(candidates_json: string, params_json: string): string;
|
|
21
21
|
|
|
22
|
+
export function regime_compare(returns_a_json: string, returns_b_json: string, regimes_json: string, options_json: string): string;
|
|
23
|
+
|
|
22
24
|
export function score(submissions_json: string, config_json: string): string;
|
|
23
25
|
|
|
24
26
|
export function score_agent(submission_json: string, config_json: string): string;
|
package/pkg/sharpebench.js
CHANGED
|
@@ -196,6 +196,35 @@ function percentile_selection(candidates_json, params_json) {
|
|
|
196
196
|
}
|
|
197
197
|
exports.percentile_selection = percentile_selection;
|
|
198
198
|
|
|
199
|
+
/**
|
|
200
|
+
* @param {string} returns_a_json
|
|
201
|
+
* @param {string} returns_b_json
|
|
202
|
+
* @param {string} regimes_json
|
|
203
|
+
* @param {string} options_json
|
|
204
|
+
* @returns {string}
|
|
205
|
+
*/
|
|
206
|
+
function regime_compare(returns_a_json, returns_b_json, regimes_json, options_json) {
|
|
207
|
+
let deferred5_0;
|
|
208
|
+
let deferred5_1;
|
|
209
|
+
try {
|
|
210
|
+
const ptr0 = passStringToWasm0(returns_a_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
|
211
|
+
const len0 = WASM_VECTOR_LEN;
|
|
212
|
+
const ptr1 = passStringToWasm0(returns_b_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
|
213
|
+
const len1 = WASM_VECTOR_LEN;
|
|
214
|
+
const ptr2 = passStringToWasm0(regimes_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
|
215
|
+
const len2 = WASM_VECTOR_LEN;
|
|
216
|
+
const ptr3 = passStringToWasm0(options_json, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
|
217
|
+
const len3 = WASM_VECTOR_LEN;
|
|
218
|
+
const ret = wasm.regime_compare(ptr0, len0, ptr1, len1, ptr2, len2, ptr3, len3);
|
|
219
|
+
deferred5_0 = ret[0];
|
|
220
|
+
deferred5_1 = ret[1];
|
|
221
|
+
return getStringFromWasm0(ret[0], ret[1]);
|
|
222
|
+
} finally {
|
|
223
|
+
wasm.__wbindgen_free(deferred5_0, deferred5_1, 1);
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
exports.regime_compare = regime_compare;
|
|
227
|
+
|
|
199
228
|
/**
|
|
200
229
|
* @param {string} submissions_json
|
|
201
230
|
* @param {string} config_json
|
package/pkg/sharpebench_bg.wasm
CHANGED
|
Binary file
|
|
@@ -10,6 +10,7 @@ export const greeks: (a: number, b: number) => [number, number];
|
|
|
10
10
|
export const is_my_sharpe_real: (a: number, b: number, c: number, d: number) => [number, number];
|
|
11
11
|
export const is_my_sharpe_real_full: (a: number, b: number, c: number, d: number, e: number) => [number, number];
|
|
12
12
|
export const percentile_selection: (a: number, b: number, c: number, d: number) => [number, number];
|
|
13
|
+
export const regime_compare: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number) => [number, number];
|
|
13
14
|
export const score: (a: number, b: number, c: number, d: number) => [number, number];
|
|
14
15
|
export const score_agent: (a: number, b: number, c: number, d: number) => [number, number];
|
|
15
16
|
export const score_allocation: (a: number, b: number, c: number, d: number) => [number, number];
|