@tangle-network/agent-eval 0.117.1 → 0.118.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
- package/dist/chunk-FTUMG2U7.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10992 -1288
- package/dist/index.js +101 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1978 -697
- package/dist/traces.js +53 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
|
@@ -1,139 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Always-valid sequential evaluation.
|
|
3
|
-
*
|
|
4
|
-
* `researchReport` assumes a single pre-specified analysis. Real
|
|
5
|
-
* consumers run campaigns weekly / nightly / per-PR; each new run silently
|
|
6
|
-
* inflates the false-discovery rate, because the BH-FDR guarantee is for
|
|
7
|
-
* the *first* look, not the 47th. Without time-uniform inference,
|
|
8
|
-
* launch-decision teams either (a) don't peek, which forfeits the cost
|
|
9
|
-
* advantage of stop-when-decisive, or (b) peek and pretend they didn't,
|
|
10
|
-
* which forfeits scientific validity.
|
|
11
|
-
*
|
|
12
|
-
* This module ships **e-value-based confidence sequences** for paired
|
|
13
|
-
* bounded outcomes. The methodology is the predictable plug-in betting
|
|
14
|
-
* martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*
|
|
15
|
-
* stopping time. Concretely:
|
|
16
|
-
*
|
|
17
|
-
* For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,
|
|
18
|
-
* a betting fraction λ_i is chosen using only D_{1..i-1} (predictable
|
|
19
|
-
* plug-in), and the running e-value is
|
|
20
|
-
*
|
|
21
|
-
* E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)
|
|
22
|
-
*
|
|
23
|
-
* E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by
|
|
24
|
-
* Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null
|
|
25
|
-
* at any time without inflating the type-I error.
|
|
26
|
-
*
|
|
27
|
-
* Combined with `runEvalCampaign`, every consumer running rolling
|
|
28
|
-
* campaigns gains the ability to ship the moment evidence is decisive,
|
|
29
|
-
* stop-early on dead-on-arrival variants, and accumulate evidence across
|
|
30
|
-
* partial runs without spending the FDR budget. No new sweep is wasted.
|
|
31
|
-
*
|
|
32
|
-
* References:
|
|
33
|
-
* - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).
|
|
34
|
-
* Time-uniform, nonparametric, nonasymptotic confidence sequences.
|
|
35
|
-
* Annals of Statistics, 49(2), 1055–1080.
|
|
36
|
-
* - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded
|
|
37
|
-
* random variables by betting. JRSS B, 86(1), 1–27.
|
|
38
|
-
*/
|
|
39
|
-
type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent';
|
|
40
|
-
interface PairedEvalueOptions {
|
|
41
|
-
/**
|
|
42
|
-
* Bound on |delta|. Default 1 (matching most score scales). Must satisfy
|
|
43
|
-
* c > 0; deltas outside [-c, c] are clipped with a warning attached to
|
|
44
|
-
* the return value.
|
|
45
|
-
*/
|
|
46
|
-
bound?: number;
|
|
47
|
-
/** Target Type-I error. Default 0.05. */
|
|
48
|
-
alpha?: number;
|
|
49
|
-
/**
|
|
50
|
-
* Region of Practical Equivalence on the *mean* paired delta. When
|
|
51
|
-
* supplied, the verdict can return `'equivalent'` once the running
|
|
52
|
-
* confidence sequence on the mean is fully contained in [low, high].
|
|
53
|
-
*/
|
|
54
|
-
rope?: {
|
|
55
|
-
low: number;
|
|
56
|
-
high: number;
|
|
57
|
-
};
|
|
58
|
-
/** Initial bet shrinkage (0 < scale ≤ 1). Default 0.5 — empirically robust. */
|
|
59
|
-
initialBetShrinkage?: number;
|
|
60
|
-
}
|
|
61
|
-
interface PairedEvalueStep {
|
|
62
|
-
/** 1-indexed observation count. */
|
|
63
|
-
t: number;
|
|
64
|
-
delta: number;
|
|
65
|
-
/** Running e-value E_t = ∏ (1 + λ_i · D_i). */
|
|
66
|
-
evalue: number;
|
|
67
|
-
/** Time-uniform p-value at stopping time t. */
|
|
68
|
-
pValue: number;
|
|
69
|
-
/** Lower bound of the empirical Bernstein confidence sequence at level 1-α. */
|
|
70
|
-
csLow: number;
|
|
71
|
-
csHigh: number;
|
|
72
|
-
/** Verdict at this stopping time. */
|
|
73
|
-
decision: SequentialDecision;
|
|
74
|
-
}
|
|
75
|
-
interface PairedEvalueSequence {
|
|
76
|
-
steps: PairedEvalueStep[];
|
|
77
|
-
/** The decision at the final step. */
|
|
78
|
-
finalDecision: SequentialDecision;
|
|
79
|
-
/** Index (1-based) at which a non-`continue` decision first fired, or null. */
|
|
80
|
-
decisionFiredAt: number | null;
|
|
81
|
-
/** True if any deltas were clipped to [-bound, bound]. */
|
|
82
|
-
clipped: boolean;
|
|
83
|
-
}
|
|
84
|
-
/**
|
|
85
|
-
* Run the paired e-value sequence over an in-order delta stream.
|
|
86
|
-
*
|
|
87
|
-
* Use for *streaming* / interim analyses: pass the deltas you have so
|
|
88
|
-
* far, get the verdict at every prefix length. The decision is
|
|
89
|
-
* monotone-stable in the sense that once `'reject_now'` or `'promote_now'`
|
|
90
|
-
* fires, the verdict at later steps remains decisive (the e-value is a
|
|
91
|
-
* non-negative martingale; once it crosses the threshold, it's crossed).
|
|
92
|
-
*/
|
|
93
|
-
declare function pairedEvalueSequence(deltas: number[], opts?: PairedEvalueOptions): PairedEvalueSequence;
|
|
94
|
-
interface InterimReleaseConfidenceInput {
|
|
95
|
-
/**
|
|
96
|
-
* One delta series per candidate (paired deltas vs comparator). Order
|
|
97
|
-
* within a series is the order the campaigns were run.
|
|
98
|
-
*/
|
|
99
|
-
deltaSeries: Array<{
|
|
100
|
-
candidateId: string;
|
|
101
|
-
deltas: number[];
|
|
102
|
-
}>;
|
|
103
|
-
alpha?: number;
|
|
104
|
-
bound?: number;
|
|
105
|
-
rope?: {
|
|
106
|
-
low: number;
|
|
107
|
-
high: number;
|
|
108
|
-
};
|
|
109
|
-
}
|
|
110
|
-
interface InterimReleaseConfidence {
|
|
111
|
-
candidates: Array<{
|
|
112
|
-
candidateId: string;
|
|
113
|
-
decision: SequentialDecision;
|
|
114
|
-
decisionFiredAt: number | null;
|
|
115
|
-
finalEvalue: number;
|
|
116
|
-
finalPValue: number;
|
|
117
|
-
pairs: number;
|
|
118
|
-
csLow: number;
|
|
119
|
-
csHigh: number;
|
|
120
|
-
}>;
|
|
121
|
-
/**
|
|
122
|
-
* Campaign-level recommendation: pick the strongest 'promote_now', else
|
|
123
|
-
* 'continue' if any candidate is still live, else 'reject_now' if every
|
|
124
|
-
* candidate is dead, else 'equivalent'.
|
|
125
|
-
*/
|
|
126
|
-
recommendation: {
|
|
127
|
-
decision: SequentialDecision;
|
|
128
|
-
candidateId: string | null;
|
|
129
|
-
};
|
|
130
|
-
}
|
|
131
|
-
/**
|
|
132
|
-
* Run interim sequential analyses across many candidates at once,
|
|
133
|
-
* preserving the time-uniform α guarantee for each candidate's series and
|
|
134
|
-
* synthesising a campaign-level recommendation. Designed to be called on
|
|
135
|
-
* every campaign tick — the recommendation is anytime-valid.
|
|
136
|
-
*/
|
|
137
|
-
declare function evaluateInterimReleaseConfidence(input: InterimReleaseConfidenceInput): InterimReleaseConfidence;
|
|
138
|
-
|
|
139
|
-
export { type InterimReleaseConfidence as I, type PairedEvalueOptions as P, type SequentialDecision as S, type InterimReleaseConfidenceInput as a, type PairedEvalueSequence as b, type PairedEvalueStep as c, evaluateInterimReleaseConfidence as e, pairedEvalueSequence as p };
|
|
@@ -1,33 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Series convergence — detects whether a sequence of scalar measurements
|
|
3
|
-
* is stabilizing, drifting, or noisy.
|
|
4
|
-
*
|
|
5
|
-
* Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
|
|
6
|
-
* about progress *within* a single run; this module is about drift
|
|
7
|
-
* *across* runs (e.g. "are my nightly eval scores stabilizing?").
|
|
8
|
-
*
|
|
9
|
-
* Three signals:
|
|
10
|
-
* - stabilized: last K values have low variance (< epsilon) — done
|
|
11
|
-
* - drifting: recent trend is monotonic and beyond noise — regressing or improving
|
|
12
|
-
* - noisy: neither — keep iterating, but flag as untrustworthy for gating
|
|
13
|
-
*/
|
|
14
|
-
interface SeriesConvergenceOptions {
|
|
15
|
-
/** Window size for "recent" analysis (default 5). */
|
|
16
|
-
window?: number;
|
|
17
|
-
/** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
|
|
18
|
-
stableCv?: number;
|
|
19
|
-
/** Minimum monotone run length to call drift (default 3). */
|
|
20
|
-
driftRun?: number;
|
|
21
|
-
}
|
|
22
|
-
interface SeriesConvergenceResult {
|
|
23
|
-
state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
|
|
24
|
-
windowMean: number;
|
|
25
|
-
windowCv: number;
|
|
26
|
-
/** Longest monotonic run at the tail of the series (positive for up, negative for down). */
|
|
27
|
-
tailRun: number;
|
|
28
|
-
/** True when n ≥ window AND windowCv ≤ stableCv. */
|
|
29
|
-
stable: boolean;
|
|
30
|
-
}
|
|
31
|
-
declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
|
|
32
|
-
|
|
33
|
-
export { type SeriesConvergenceOptions as S, type SeriesConvergenceResult as a, analyzeSeries as b };
|
|
@@ -1,494 +0,0 @@
|
|
|
1
|
-
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
2
|
-
import { J as JudgeScore } from './types-BkfcQnxV.js';
|
|
3
|
-
|
|
4
|
-
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
|
-
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
6
|
-
declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
|
|
7
|
-
/** Weighted mean — falls back to uniform weights when omitted */
|
|
8
|
-
declare function weightedMean(scores: {
|
|
9
|
-
score: number;
|
|
10
|
-
weight?: number;
|
|
11
|
-
}[]): number;
|
|
12
|
-
/** Bootstrap confidence interval */
|
|
13
|
-
declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
|
|
14
|
-
seed?: number;
|
|
15
|
-
resamples?: number;
|
|
16
|
-
}): {
|
|
17
|
-
mean: number;
|
|
18
|
-
lower: number;
|
|
19
|
-
upper: number;
|
|
20
|
-
};
|
|
21
|
-
/**
|
|
22
|
-
* Inter-rater reliability — simplified Krippendorff's alpha.
|
|
23
|
-
*
|
|
24
|
-
* Each inner array is one judge's scores for all items.
|
|
25
|
-
* All arrays must have the same length (same items scored).
|
|
26
|
-
*/
|
|
27
|
-
declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
|
|
28
|
-
/**
|
|
29
|
-
* Mann-Whitney U test for comparing two independent groups.
|
|
30
|
-
* Returns U statistic and approximate p-value (normal approximation).
|
|
31
|
-
*/
|
|
32
|
-
declare function mannWhitneyU(a: number[], b: number[]): {
|
|
33
|
-
u: number;
|
|
34
|
-
p: number;
|
|
35
|
-
};
|
|
36
|
-
/** Partial credit: returns 0-1 ratio of current toward target */
|
|
37
|
-
declare function partialCredit(current: number, target: number): number;
|
|
38
|
-
/**
|
|
39
|
-
* Paired t-test — before/after measurements on the SAME items.
|
|
40
|
-
* Pairing removes inter-item variance, giving tighter significance than
|
|
41
|
-
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
42
|
-
* scenarios.
|
|
43
|
-
*/
|
|
44
|
-
declare function pairedTTest(before: number[], after: number[]): {
|
|
45
|
-
t: number;
|
|
46
|
-
df: number;
|
|
47
|
-
p: number;
|
|
48
|
-
};
|
|
49
|
-
/**
|
|
50
|
-
* Wilcoxon signed-rank test — paired non-parametric alternative.
|
|
51
|
-
* Use when the differences aren't normally distributed.
|
|
52
|
-
*/
|
|
53
|
-
declare function wilcoxonSignedRank(before: number[], after: number[]): {
|
|
54
|
-
w: number;
|
|
55
|
-
p: number;
|
|
56
|
-
};
|
|
57
|
-
/**
|
|
58
|
-
* Cohen's d — standardized effect size for two independent groups.
|
|
59
|
-
* Positive d means group b has higher mean than group a.
|
|
60
|
-
* Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
|
|
61
|
-
*/
|
|
62
|
-
declare function cohensD(a: number[], b: number[]): number;
|
|
63
|
-
type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
|
|
64
|
-
/**
|
|
65
|
-
* Cliff's delta — a non-parametric effect size for two independent samples.
|
|
66
|
-
* `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
|
|
67
|
-
* ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
|
|
68
|
-
*
|
|
69
|
-
* Distribution-free counterpart to Cohen's d: no normality assumption, robust
|
|
70
|
-
* to the bounded/skewed score distributions judges produce. Pairs with
|
|
71
|
-
* `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
|
|
72
|
-
* path. Returns 0 when either sample is empty.
|
|
73
|
-
*/
|
|
74
|
-
declare function cliffsDelta(before: number[], after: number[]): number;
|
|
75
|
-
/**
|
|
76
|
-
* Map a Cliff's delta to a qualitative magnitude using the standard
|
|
77
|
-
* Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
|
|
78
|
-
* <0.474 medium, else large.
|
|
79
|
-
*/
|
|
80
|
-
declare function interpretCliffs(delta: number): CliffsMagnitude;
|
|
81
|
-
/**
|
|
82
|
-
* Average-rank-with-ties transform (1-indexed). Tied values receive the mean
|
|
83
|
-
* of the ranks they span, the standard correction for Spearman's ρ.
|
|
84
|
-
*/
|
|
85
|
-
declare function ranks(xs: number[]): number[];
|
|
86
|
-
/**
|
|
87
|
-
* Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
|
|
88
|
-
* equal-length series. See the edge-case contract above: NaN for n < 2 or
|
|
89
|
-
* unequal lengths, 1 when both series are constant, 0 when exactly one is.
|
|
90
|
-
*/
|
|
91
|
-
declare function pearsonR(a: number[], b: number[]): number;
|
|
92
|
-
/**
|
|
93
|
-
* Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
|
|
94
|
-
* transform of each series. Same edge-case contract as {@link pearsonR}.
|
|
95
|
-
*/
|
|
96
|
-
declare function spearmanR(a: number[], b: number[]): number;
|
|
97
|
-
interface WeightedCompositeInput {
|
|
98
|
-
/** Per-dimension scores (typically 0..1). */
|
|
99
|
-
dims: Record<string, number>;
|
|
100
|
-
/** Weight per dimension. Every weighted dimension MUST be present in
|
|
101
|
-
* `dims` — a weight for an absent dimension is a config error and throws,
|
|
102
|
-
* because silently dropping it would renormalise the composite onto a
|
|
103
|
-
* different denominator than intended. */
|
|
104
|
-
weights: Record<string, number>;
|
|
105
|
-
/** Optional pass threshold; when set, the result reports `pass`. */
|
|
106
|
-
threshold?: number;
|
|
107
|
-
}
|
|
108
|
-
interface WeightedCompositeResult {
|
|
109
|
-
composite: number;
|
|
110
|
-
pass?: boolean;
|
|
111
|
-
}
|
|
112
|
-
/**
|
|
113
|
-
* Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
|
|
114
|
-
* the weighted dimensions. The canonical replacement for the per-consumer
|
|
115
|
-
* hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
|
|
116
|
-
*
|
|
117
|
-
* Fail-loud: throws if a weighted dimension is missing from `dims`, if any
|
|
118
|
-
* weight is negative, or if the weights sum to 0 — none of which can produce
|
|
119
|
-
* a meaningful composite.
|
|
120
|
-
*/
|
|
121
|
-
declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
|
|
122
|
-
interface CorpusScoreRecord {
|
|
123
|
-
/** Stable identifier for the rated item (scenario, span, turn, …). */
|
|
124
|
-
itemId: string;
|
|
125
|
-
/** Identifier for the judge that produced this score. */
|
|
126
|
-
judgeName: string;
|
|
127
|
-
/** Dimension name (matches `JudgeScore.dimension`). */
|
|
128
|
-
dimension: string;
|
|
129
|
-
/** Numeric score; must be finite. */
|
|
130
|
-
score: number;
|
|
131
|
-
}
|
|
132
|
-
interface CorpusAgreementPerDimension extends ContinuousAgreement {
|
|
133
|
-
dimension: string;
|
|
134
|
-
/** Item IDs that contributed to this dimension's matrix (every judge scored them). */
|
|
135
|
-
itemIds: string[];
|
|
136
|
-
/** Judge IDs that contributed to this dimension's matrix. */
|
|
137
|
-
judgeIds: string[];
|
|
138
|
-
}
|
|
139
|
-
interface CorpusAgreementReport {
|
|
140
|
-
/** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
|
|
141
|
-
perDimension: CorpusAgreementPerDimension[];
|
|
142
|
-
/** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
|
|
143
|
-
overallIcc: number;
|
|
144
|
-
/** Mean weighted κ across dimensions (NaN if none finite). */
|
|
145
|
-
overallWeightedKappa: number;
|
|
146
|
-
/** Dimensions evaluated (sorted). */
|
|
147
|
-
dimensions: string[];
|
|
148
|
-
/** Judges seen across the corpus (sorted). */
|
|
149
|
-
judgeIds: string[];
|
|
150
|
-
}
|
|
151
|
-
interface CorpusAgreementOptions extends ContinuousAgreementOptions {
|
|
152
|
-
/**
|
|
153
|
-
* Restrict the audit to these dimensions. Default = every dimension
|
|
154
|
-
* that appears in the input. A dimension named here but absent from
|
|
155
|
-
* the input throws — silent omission would corrupt the overall metric.
|
|
156
|
-
*/
|
|
157
|
-
dimensions?: string[];
|
|
158
|
-
/**
|
|
159
|
-
* Restrict the audit to these judges. Default = every judge that
|
|
160
|
-
* appears in the input. A judge named here but absent from a
|
|
161
|
-
* dimension throws (see "fail loud" below).
|
|
162
|
-
*/
|
|
163
|
-
judges?: string[];
|
|
164
|
-
}
|
|
165
|
-
/**
|
|
166
|
-
* Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
|
|
167
|
-
*
|
|
168
|
-
* For each dimension, builds the [n_items][n_judges] matrix of scores
|
|
169
|
-
* (keeping only items every judge rated on that dimension), then runs
|
|
170
|
-
* `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
|
|
171
|
-
* bootstrap CIs. Reports a pooled mean across dimensions as a single
|
|
172
|
-
* "is this judge panel reliable on this corpus?" number.
|
|
173
|
-
*
|
|
174
|
-
* Fail-loud contract:
|
|
175
|
-
* - Empty input throws.
|
|
176
|
-
* - Fewer than 2 judges or fewer than 2 items per dimension throws.
|
|
177
|
-
* - A judge present in some dimensions but with zero scored items on
|
|
178
|
-
* another dimension throws (would silently shrink the matrix).
|
|
179
|
-
* - Duplicate (itemId, judgeName, dimension) records throw.
|
|
180
|
-
*/
|
|
181
|
-
declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
182
|
-
/**
|
|
183
|
-
* Convenience adapter for `JudgeScore[]` data keyed externally by item.
|
|
184
|
-
*
|
|
185
|
-
* Use when you have per-item arrays of `JudgeScore[]` (e.g. one
|
|
186
|
-
* `ScenarioResult.judgeScores` per scenario) and want corpus-wide
|
|
187
|
-
* agreement without manually flattening. `itemId` must be unique per
|
|
188
|
-
* row of `itemsScores`.
|
|
189
|
-
*/
|
|
190
|
-
declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
|
|
191
|
-
itemId: string;
|
|
192
|
-
scores: JudgeScore[];
|
|
193
|
-
}>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
|
|
194
|
-
/**
|
|
195
|
-
* Required N per arm for a two-sample comparison at target effect size,
|
|
196
|
-
* alpha, and power. Normal-approximation formula:
|
|
197
|
-
* n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
|
|
198
|
-
* where d is Cohen's d. Returns Infinity for effect ≤ 0.
|
|
199
|
-
*/
|
|
200
|
-
declare function requiredSampleSize(opts: {
|
|
201
|
-
effect: number;
|
|
202
|
-
alpha?: number;
|
|
203
|
-
power?: number;
|
|
204
|
-
twoSided?: boolean;
|
|
205
|
-
}): number;
|
|
206
|
-
/**
|
|
207
|
-
* Minimum detectable paired effect (standardised units) for a target paired
|
|
208
|
-
* sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
|
|
209
|
-
* sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
|
|
210
|
-
* have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
|
|
211
|
-
*/
|
|
212
|
-
declare function pairedMde(opts: {
|
|
213
|
-
nPaired: number;
|
|
214
|
-
alpha?: number;
|
|
215
|
-
power?: number;
|
|
216
|
-
twoSided?: boolean;
|
|
217
|
-
}): number;
|
|
218
|
-
/**
|
|
219
|
-
* Number of paired observations needed for a McNemar test to reach a target
|
|
220
|
-
* power — the pre-registration companion to {@link mcnemar}. Parametrised by the
|
|
221
|
-
* expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
|
|
222
|
-
* `p01` (P[control wins]); concordant pairs carry no information, so the count
|
|
223
|
-
* is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
|
|
224
|
-
* approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
|
|
225
|
-
* `δ = p10 − p01`,
|
|
226
|
-
* n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
|
|
227
|
-
* Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
|
|
228
|
-
* tiny discordant counts where the exact {@link mcnemar} differs from the normal
|
|
229
|
-
* approximation, treat the result as a lower bound and prefer the discordant-pair
|
|
230
|
-
* floor.
|
|
231
|
-
*/
|
|
232
|
-
declare function mcnemarRequiredN(opts: {
|
|
233
|
-
p10: number;
|
|
234
|
-
p01: number;
|
|
235
|
-
alpha?: number;
|
|
236
|
-
power?: number;
|
|
237
|
-
twoSided?: boolean;
|
|
238
|
-
}): number;
|
|
239
|
-
/**
|
|
240
|
-
* Power of a McNemar test at a given number of paired observations, the inverse
|
|
241
|
-
* of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
|
|
242
|
-
* Returns a value in [0, 1]; equals `alpha` when there is no effect.
|
|
243
|
-
*/
|
|
244
|
-
declare function mcnemarPower(opts: {
|
|
245
|
-
p10: number;
|
|
246
|
-
p01: number;
|
|
247
|
-
nPairs: number;
|
|
248
|
-
alpha?: number;
|
|
249
|
-
twoSided?: boolean;
|
|
250
|
-
}): number;
|
|
251
|
-
/** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
|
|
252
|
-
declare function bonferroni(pValues: number[], alpha?: number): {
|
|
253
|
-
adjusted: number[];
|
|
254
|
-
significant: boolean[];
|
|
255
|
-
};
|
|
256
|
-
/**
|
|
257
|
-
* Holm step-down family-wise error adjustment.
|
|
258
|
-
*
|
|
259
|
-
* P-values are sorted from smallest to largest, multiplied by their remaining
|
|
260
|
-
* hypothesis count, and made monotonically non-decreasing before being mapped
|
|
261
|
-
* back to input order. This uniformly dominates plain Bonferroni while keeping
|
|
262
|
-
* strong family-wise error control under arbitrary dependence.
|
|
263
|
-
*/
|
|
264
|
-
declare function holm(pValues: readonly number[], alpha?: number): {
|
|
265
|
-
adjusted: number[];
|
|
266
|
-
significant: boolean[];
|
|
267
|
-
};
|
|
268
|
-
/**
|
|
269
|
-
* Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
|
|
270
|
-
* significance at the target FDR; handles ties and preserves q monotonicity.
|
|
271
|
-
*/
|
|
272
|
-
declare function benjaminiHochberg(pValues: number[], fdr?: number): {
|
|
273
|
-
qValues: number[];
|
|
274
|
-
significant: boolean[];
|
|
275
|
-
};
|
|
276
|
-
interface PairedBootstrapResult {
|
|
277
|
-
/** Number of paired observations. */
|
|
278
|
-
n: number;
|
|
279
|
-
/** Median of paired deltas (after − before). */
|
|
280
|
-
median: number;
|
|
281
|
-
/** Mean of paired deltas. */
|
|
282
|
-
mean: number;
|
|
283
|
-
/** Lower bound of the bootstrap CI on the chosen statistic. */
|
|
284
|
-
low: number;
|
|
285
|
-
/** Upper bound of the bootstrap CI on the chosen statistic. */
|
|
286
|
-
high: number;
|
|
287
|
-
/** Confidence level used (e.g. 0.95). */
|
|
288
|
-
confidence: number;
|
|
289
|
-
/** Number of bootstrap resamples used. */
|
|
290
|
-
resamples: number;
|
|
291
|
-
}
|
|
292
|
-
interface PairedBootstrapOptions {
|
|
293
|
-
/** Confidence level. Default 0.95. */
|
|
294
|
-
confidence?: number;
|
|
295
|
-
/** Bootstrap resample count. Default 2000. */
|
|
296
|
-
resamples?: number;
|
|
297
|
-
/** Statistic to bootstrap. Default 'median'. */
|
|
298
|
-
statistic?: 'median' | 'mean';
|
|
299
|
-
/** Deterministic seed. If omitted, uses Math.random(). */
|
|
300
|
-
seed?: number;
|
|
301
|
-
}
|
|
302
|
-
/**
|
|
303
|
-
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
304
|
-
* statistic (median by default); pairs are resampled with replacement. The
|
|
305
|
-
* lower bound is what the promotion gate checks — `low > threshold` means the
|
|
306
|
-
* gain is real at the confidence level. Throws on unequal sample sizes.
|
|
307
|
-
*/
|
|
308
|
-
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
309
|
-
/** Pre-registered direction for a one-sided paired sign test. */
|
|
310
|
-
type SignTestAlternative = 'greater' | 'less';
|
|
311
|
-
/** Exact one-sided sign-test result for paired numeric differences. */
|
|
312
|
-
interface PairedSignTestResult {
|
|
313
|
-
/** Total supplied differences, including zero ties. */
|
|
314
|
-
n: number;
|
|
315
|
-
/** Strictly positive differences. */
|
|
316
|
-
positive: number;
|
|
317
|
-
/** Strictly negative differences. */
|
|
318
|
-
negative: number;
|
|
319
|
-
/** Zero differences excluded from the binomial test. */
|
|
320
|
-
ties: number;
|
|
321
|
-
/** Non-zero differences used by the binomial test. */
|
|
322
|
-
nNonTies: number;
|
|
323
|
-
/** Direction of the pre-registered alternative hypothesis. */
|
|
324
|
-
alternative: SignTestAlternative;
|
|
325
|
-
/** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
|
|
326
|
-
pValue: number;
|
|
327
|
-
}
|
|
328
|
-
/**
|
|
329
|
-
* Exact one-sided sign test over paired differences.
|
|
330
|
-
*
|
|
331
|
-
* Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
|
|
332
|
-
* tests whether positive signs are more likely than negative signs and returns
|
|
333
|
-
* `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
|
|
334
|
-
* negative signs as successes instead. With a continuous difference
|
|
335
|
-
* distribution this is the usual directional median test. Exact zero
|
|
336
|
-
* differences are ties and do not enter the binomial denominator. All-tie and
|
|
337
|
-
* empty inputs return p = 1. Every input difference must be finite, and the
|
|
338
|
-
* direction must be chosen explicitly so a caller cannot select it after
|
|
339
|
-
* seeing the signs.
|
|
340
|
-
*/
|
|
341
|
-
declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
|
|
342
|
-
/** A binomial proportion estimate with a confidence interval. */
|
|
343
|
-
interface ProportionInterval {
|
|
344
|
-
/** Point estimate successes / n (0 when n = 0). */
|
|
345
|
-
estimate: number;
|
|
346
|
-
/** Lower bound, clamped to [0, 1]. */
|
|
347
|
-
lower: number;
|
|
348
|
-
/** Upper bound, clamped to [0, 1]. */
|
|
349
|
-
upper: number;
|
|
350
|
-
}
|
|
351
|
-
/**
|
|
352
|
-
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
353
|
-
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
354
|
-
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
355
|
-
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
356
|
-
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
357
|
-
*/
|
|
358
|
-
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
359
|
-
/** Result of a McNemar paired-binary significance test. */
|
|
360
|
-
interface McNemarResult {
|
|
361
|
-
/** Total paired observations. */
|
|
362
|
-
n: number;
|
|
363
|
-
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
364
|
-
nDiscordant: number;
|
|
365
|
-
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
366
|
-
b: number;
|
|
367
|
-
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
368
|
-
c: number;
|
|
369
|
-
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
370
|
-
statistic: number;
|
|
371
|
-
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
372
|
-
pValue: number;
|
|
373
|
-
}
|
|
374
|
-
/**
|
|
375
|
-
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
376
|
-
* "does treatment change the success rate vs control on the SAME items". Only
|
|
377
|
-
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
378
|
-
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
379
|
-
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
380
|
-
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
381
|
-
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
382
|
-
* chi-square approximation needed, though it is returned as `statistic` for
|
|
383
|
-
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
384
|
-
* the module's (before, after) convention. Throws on unequal lengths.
|
|
385
|
-
*/
|
|
386
|
-
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
387
|
-
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
388
|
-
interface RiskDifferenceResult {
|
|
389
|
-
/** Total paired observations. */
|
|
390
|
-
n: number;
|
|
391
|
-
/** Discordant pairs: treatment-win count. */
|
|
392
|
-
b: number;
|
|
393
|
-
/** Discordant pairs: control-win count. */
|
|
394
|
-
c: number;
|
|
395
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
396
|
-
riskDifference: number;
|
|
397
|
-
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
398
|
-
lower: number;
|
|
399
|
-
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
400
|
-
upper: number;
|
|
401
|
-
/** Confidence level used. */
|
|
402
|
-
confidence: number;
|
|
403
|
-
}
|
|
404
|
-
/**
|
|
405
|
-
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
406
|
-
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
407
|
-
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
408
|
-
* the discordant counts, not the independent-samples formula (which overstates
|
|
409
|
-
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
410
|
-
* arrays, control first. Throws on unequal lengths.
|
|
411
|
-
*/
|
|
412
|
-
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
413
|
-
/**
|
|
414
|
-
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
415
|
-
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
416
|
-
* problem of which `c` pass, the probability that at least one of a random k of
|
|
417
|
-
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
418
|
-
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
419
|
-
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
420
|
-
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
421
|
-
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
422
|
-
*/
|
|
423
|
-
declare function passAtK(n: number, c: number, k: number): number;
|
|
424
|
-
interface EProcessOptions {
|
|
425
|
-
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
426
|
-
* (Ville's inequality). Default 0.05. */
|
|
427
|
-
alpha?: number;
|
|
428
|
-
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
429
|
-
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
430
|
-
* Default 0.5. */
|
|
431
|
-
maxBet?: number;
|
|
432
|
-
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
433
|
-
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
434
|
-
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
435
|
-
nullMean?: number;
|
|
436
|
-
}
|
|
437
|
-
interface EProcessStep {
|
|
438
|
-
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
439
|
-
wealth: number;
|
|
440
|
-
/** Observations consumed so far. */
|
|
441
|
-
n: number;
|
|
442
|
-
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
443
|
-
decided: boolean;
|
|
444
|
-
}
|
|
445
|
-
interface EProcessState extends EProcessStep {
|
|
446
|
-
alpha: number;
|
|
447
|
-
maxBet: number;
|
|
448
|
-
nullMean: number;
|
|
449
|
-
/** The decision boundary 1/alpha. */
|
|
450
|
-
threshold: number;
|
|
451
|
-
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
452
|
-
decidedAtN?: number;
|
|
453
|
-
}
|
|
454
|
-
interface EProcess {
|
|
455
|
-
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
456
|
-
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
457
|
-
update(x: number): EProcessStep;
|
|
458
|
-
state(): EProcessState;
|
|
459
|
-
}
|
|
460
|
-
/**
|
|
461
|
-
* Betting test-martingale for bounded observations — the e-process core of
|
|
462
|
-
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
463
|
-
* of bounded random variables by betting", JRSS-B 2024).
|
|
464
|
-
*
|
|
465
|
-
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
466
|
-
*
|
|
467
|
-
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
468
|
-
*
|
|
469
|
-
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
470
|
-
*
|
|
471
|
-
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
472
|
-
*
|
|
473
|
-
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
474
|
-
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
475
|
-
*
|
|
476
|
-
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
477
|
-
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
478
|
-
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
479
|
-
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
480
|
-
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
481
|
-
* (no prior evidence), so the first observation never moves wealth.
|
|
482
|
-
*
|
|
483
|
-
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
484
|
-
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
485
|
-
* the decision time is the first crossing.
|
|
486
|
-
*/
|
|
487
|
-
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
488
|
-
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
489
|
-
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
490
|
-
* share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
|
|
491
|
-
* gate verdicts is non-reproducible by construction). */
|
|
492
|
-
declare function mulberry32(seed: number): () => number;
|
|
493
|
-
|
|
494
|
-
export { mcnemarPower as A, mcnemarRequiredN as B, type CorpusAgreementReport as C, mulberry32 as D, type EProcessState as E, normalizeScores as F, pairedMde as G, pairedRiskDifference as H, pairedSignTest as I, pairedTTest as J, partialCredit as K, passAtK as L, type McNemarResult as M, pearsonR as N, ranks as O, type PairedBootstrapOptions as P, requiredSampleSize as Q, type RiskDifferenceResult as R, type SignTestAlternative as S, spearmanR as T, weightedComposite as U, weightedMean as V, type WeightedCompositeInput as W, wilson as X, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type PairedSignTestResult as j, type ProportionInterval as k, type WeightedCompositeResult as l, bonferroni as m, cliffsDelta as n, cohensD as o, pairedBootstrap as p, confidenceInterval as q, corpusInterRaterAgreement as r, corpusInterRaterAgreementFromJudgeScores as s, eProcess as t, holm as u, interRaterReliability as v, wilcoxonSignedRank as w, interpretCliffs as x, mannWhitneyU as y, mcnemar as z };
|