@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,139 +0,0 @@
1
- /**
2
- * Always-valid sequential evaluation.
3
- *
4
- * `researchReport` assumes a single pre-specified analysis. Real
5
- * consumers run campaigns weekly / nightly / per-PR; each new run silently
6
- * inflates the false-discovery rate, because the BH-FDR guarantee is for
7
- * the *first* look, not the 47th. Without time-uniform inference,
8
- * launch-decision teams either (a) don't peek, which forfeits the cost
9
- * advantage of stop-when-decisive, or (b) peek and pretend they didn't,
10
- * which forfeits scientific validity.
11
- *
12
- * This module ships **e-value-based confidence sequences** for paired
13
- * bounded outcomes. The methodology is the predictable plug-in betting
14
- * martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*
15
- * stopping time. Concretely:
16
- *
17
- * For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,
18
- * a betting fraction λ_i is chosen using only D_{1..i-1} (predictable
19
- * plug-in), and the running e-value is
20
- *
21
- * E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)
22
- *
23
- * E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by
24
- * Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null
25
- * at any time without inflating the type-I error.
26
- *
27
- * Combined with `runEvalCampaign`, every consumer running rolling
28
- * campaigns gains the ability to ship the moment evidence is decisive,
29
- * stop-early on dead-on-arrival variants, and accumulate evidence across
30
- * partial runs without spending the FDR budget. No new sweep is wasted.
31
- *
32
- * References:
33
- * - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).
34
- * Time-uniform, nonparametric, nonasymptotic confidence sequences.
35
- * Annals of Statistics, 49(2), 1055–1080.
36
- * - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded
37
- * random variables by betting. JRSS B, 86(1), 1–27.
38
- */
39
- type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent';
40
- interface PairedEvalueOptions {
41
- /**
42
- * Bound on |delta|. Default 1 (matching most score scales). Must satisfy
43
- * c > 0; deltas outside [-c, c] are clipped with a warning attached to
44
- * the return value.
45
- */
46
- bound?: number;
47
- /** Target Type-I error. Default 0.05. */
48
- alpha?: number;
49
- /**
50
- * Region of Practical Equivalence on the *mean* paired delta. When
51
- * supplied, the verdict can return `'equivalent'` once the running
52
- * confidence sequence on the mean is fully contained in [low, high].
53
- */
54
- rope?: {
55
- low: number;
56
- high: number;
57
- };
58
- /** Initial bet shrinkage (0 < scale ≤ 1). Default 0.5 — empirically robust. */
59
- initialBetShrinkage?: number;
60
- }
61
- interface PairedEvalueStep {
62
- /** 1-indexed observation count. */
63
- t: number;
64
- delta: number;
65
- /** Running e-value E_t = ∏ (1 + λ_i · D_i). */
66
- evalue: number;
67
- /** Time-uniform p-value at stopping time t. */
68
- pValue: number;
69
- /** Lower bound of the empirical Bernstein confidence sequence at level 1-α. */
70
- csLow: number;
71
- csHigh: number;
72
- /** Verdict at this stopping time. */
73
- decision: SequentialDecision;
74
- }
75
- interface PairedEvalueSequence {
76
- steps: PairedEvalueStep[];
77
- /** The decision at the final step. */
78
- finalDecision: SequentialDecision;
79
- /** Index (1-based) at which a non-`continue` decision first fired, or null. */
80
- decisionFiredAt: number | null;
81
- /** True if any deltas were clipped to [-bound, bound]. */
82
- clipped: boolean;
83
- }
84
- /**
85
- * Run the paired e-value sequence over an in-order delta stream.
86
- *
87
- * Use for *streaming* / interim analyses: pass the deltas you have so
88
- * far, get the verdict at every prefix length. The decision is
89
- * monotone-stable in the sense that once `'reject_now'` or `'promote_now'`
90
- * fires, the verdict at later steps remains decisive (the e-value is a
91
- * non-negative martingale; once it crosses the threshold, it's crossed).
92
- */
93
- declare function pairedEvalueSequence(deltas: number[], opts?: PairedEvalueOptions): PairedEvalueSequence;
94
- interface InterimReleaseConfidenceInput {
95
- /**
96
- * One delta series per candidate (paired deltas vs comparator). Order
97
- * within a series is the order the campaigns were run.
98
- */
99
- deltaSeries: Array<{
100
- candidateId: string;
101
- deltas: number[];
102
- }>;
103
- alpha?: number;
104
- bound?: number;
105
- rope?: {
106
- low: number;
107
- high: number;
108
- };
109
- }
110
- interface InterimReleaseConfidence {
111
- candidates: Array<{
112
- candidateId: string;
113
- decision: SequentialDecision;
114
- decisionFiredAt: number | null;
115
- finalEvalue: number;
116
- finalPValue: number;
117
- pairs: number;
118
- csLow: number;
119
- csHigh: number;
120
- }>;
121
- /**
122
- * Campaign-level recommendation: pick the strongest 'promote_now', else
123
- * 'continue' if any candidate is still live, else 'reject_now' if every
124
- * candidate is dead, else 'equivalent'.
125
- */
126
- recommendation: {
127
- decision: SequentialDecision;
128
- candidateId: string | null;
129
- };
130
- }
131
- /**
132
- * Run interim sequential analyses across many candidates at once,
133
- * preserving the time-uniform α guarantee for each candidate's series and
134
- * synthesising a campaign-level recommendation. Designed to be called on
135
- * every campaign tick — the recommendation is anytime-valid.
136
- */
137
- declare function evaluateInterimReleaseConfidence(input: InterimReleaseConfidenceInput): InterimReleaseConfidence;
138
-
139
- export { type InterimReleaseConfidence as I, type PairedEvalueOptions as P, type SequentialDecision as S, type InterimReleaseConfidenceInput as a, type PairedEvalueSequence as b, type PairedEvalueStep as c, evaluateInterimReleaseConfidence as e, pairedEvalueSequence as p };
@@ -1,33 +0,0 @@
1
- /**
2
- * Series convergence — detects whether a sequence of scalar measurements
3
- * is stabilizing, drifting, or noisy.
4
- *
5
- * Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
6
- * about progress *within* a single run; this module is about drift
7
- * *across* runs (e.g. "are my nightly eval scores stabilizing?").
8
- *
9
- * Three signals:
10
- * - stabilized: last K values have low variance (< epsilon) — done
11
- * - drifting: recent trend is monotonic and beyond noise — regressing or improving
12
- * - noisy: neither — keep iterating, but flag as untrustworthy for gating
13
- */
14
- interface SeriesConvergenceOptions {
15
- /** Window size for "recent" analysis (default 5). */
16
- window?: number;
17
- /** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
18
- stableCv?: number;
19
- /** Minimum monotone run length to call drift (default 3). */
20
- driftRun?: number;
21
- }
22
- interface SeriesConvergenceResult {
23
- state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
24
- windowMean: number;
25
- windowCv: number;
26
- /** Longest monotonic run at the tail of the series (positive for up, negative for down). */
27
- tailRun: number;
28
- /** True when n ≥ window AND windowCv ≤ stableCv. */
29
- stable: boolean;
30
- }
31
- declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
32
-
33
- export { type SeriesConvergenceOptions as S, type SeriesConvergenceResult as a, analyzeSeries as b };
@@ -1,494 +0,0 @@
1
- import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
2
- import { J as JudgeScore } from './types-BkfcQnxV.js';
3
-
4
- /** Identity: dimensions already follow "higher = better" by prompt convention
5
- * (inverted dims like hallucination are scored 10 = best at the source). */
6
- declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
7
- /** Weighted mean — falls back to uniform weights when omitted */
8
- declare function weightedMean(scores: {
9
- score: number;
10
- weight?: number;
11
- }[]): number;
12
- /** Bootstrap confidence interval */
13
- declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
14
- seed?: number;
15
- resamples?: number;
16
- }): {
17
- mean: number;
18
- lower: number;
19
- upper: number;
20
- };
21
- /**
22
- * Inter-rater reliability — simplified Krippendorff's alpha.
23
- *
24
- * Each inner array is one judge's scores for all items.
25
- * All arrays must have the same length (same items scored).
26
- */
27
- declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
28
- /**
29
- * Mann-Whitney U test for comparing two independent groups.
30
- * Returns U statistic and approximate p-value (normal approximation).
31
- */
32
- declare function mannWhitneyU(a: number[], b: number[]): {
33
- u: number;
34
- p: number;
35
- };
36
- /** Partial credit: returns 0-1 ratio of current toward target */
37
- declare function partialCredit(current: number, target: number): number;
38
- /**
39
- * Paired t-test — before/after measurements on the SAME items.
40
- * Pairing removes inter-item variance, giving tighter significance than
41
- * an unpaired test when comparing prompt v1 vs prompt v2 on identical
42
- * scenarios.
43
- */
44
- declare function pairedTTest(before: number[], after: number[]): {
45
- t: number;
46
- df: number;
47
- p: number;
48
- };
49
- /**
50
- * Wilcoxon signed-rank test — paired non-parametric alternative.
51
- * Use when the differences aren't normally distributed.
52
- */
53
- declare function wilcoxonSignedRank(before: number[], after: number[]): {
54
- w: number;
55
- p: number;
56
- };
57
- /**
58
- * Cohen's d — standardized effect size for two independent groups.
59
- * Positive d means group b has higher mean than group a.
60
- * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
61
- */
62
- declare function cohensD(a: number[], b: number[]): number;
63
- type CliffsMagnitude = 'negligible' | 'small' | 'medium' | 'large';
64
- /**
65
- * Cliff's delta — a non-parametric effect size for two independent samples.
66
- * `δ = (#(after > before) − #(after < before)) / (n_before · n_after)`,
67
- * ranging [-1, 1]. Positive ⇒ `after` tends to exceed `before` (improvement).
68
- *
69
- * Distribution-free counterpart to Cohen's d: no normality assumption, robust
70
- * to the bounded/skewed score distributions judges produce. Pairs with
71
- * `pairedBootstrap` / `wilcoxonSignedRank` for the non-parametric reporting
72
- * path. Returns 0 when either sample is empty.
73
- */
74
- declare function cliffsDelta(before: number[], after: number[]): number;
75
- /**
76
- * Map a Cliff's delta to a qualitative magnitude using the standard
77
- * Romano et al. thresholds (|δ|): <0.147 negligible, <0.33 small,
78
- * <0.474 medium, else large.
79
- */
80
- declare function interpretCliffs(delta: number): CliffsMagnitude;
81
- /**
82
- * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
83
- * of the ranks they span, the standard correction for Spearman's ρ.
84
- */
85
- declare function ranks(xs: number[]): number[];
86
- /**
87
- * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
88
- * equal-length series. See the edge-case contract above: NaN for n < 2 or
89
- * unequal lengths, 1 when both series are constant, 0 when exactly one is.
90
- */
91
- declare function pearsonR(a: number[], b: number[]): number;
92
- /**
93
- * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
94
- * transform of each series. Same edge-case contract as {@link pearsonR}.
95
- */
96
- declare function spearmanR(a: number[], b: number[]): number;
97
- interface WeightedCompositeInput {
98
- /** Per-dimension scores (typically 0..1). */
99
- dims: Record<string, number>;
100
- /** Weight per dimension. Every weighted dimension MUST be present in
101
- * `dims` — a weight for an absent dimension is a config error and throws,
102
- * because silently dropping it would renormalise the composite onto a
103
- * different denominator than intended. */
104
- weights: Record<string, number>;
105
- /** Optional pass threshold; when set, the result reports `pass`. */
106
- threshold?: number;
107
- }
108
- interface WeightedCompositeResult {
109
- composite: number;
110
- pass?: boolean;
111
- }
112
- /**
113
- * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
114
- * the weighted dimensions. The canonical replacement for the per-consumer
115
- * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
116
- *
117
- * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
118
- * weight is negative, or if the weights sum to 0 — none of which can produce
119
- * a meaningful composite.
120
- */
121
- declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
122
- interface CorpusScoreRecord {
123
- /** Stable identifier for the rated item (scenario, span, turn, …). */
124
- itemId: string;
125
- /** Identifier for the judge that produced this score. */
126
- judgeName: string;
127
- /** Dimension name (matches `JudgeScore.dimension`). */
128
- dimension: string;
129
- /** Numeric score; must be finite. */
130
- score: number;
131
- }
132
- interface CorpusAgreementPerDimension extends ContinuousAgreement {
133
- dimension: string;
134
- /** Item IDs that contributed to this dimension's matrix (every judge scored them). */
135
- itemIds: string[];
136
- /** Judge IDs that contributed to this dimension's matrix. */
137
- judgeIds: string[];
138
- }
139
- interface CorpusAgreementReport {
140
- /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */
141
- perDimension: CorpusAgreementPerDimension[];
142
- /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */
143
- overallIcc: number;
144
- /** Mean weighted κ across dimensions (NaN if none finite). */
145
- overallWeightedKappa: number;
146
- /** Dimensions evaluated (sorted). */
147
- dimensions: string[];
148
- /** Judges seen across the corpus (sorted). */
149
- judgeIds: string[];
150
- }
151
- interface CorpusAgreementOptions extends ContinuousAgreementOptions {
152
- /**
153
- * Restrict the audit to these dimensions. Default = every dimension
154
- * that appears in the input. A dimension named here but absent from
155
- * the input throws — silent omission would corrupt the overall metric.
156
- */
157
- dimensions?: string[];
158
- /**
159
- * Restrict the audit to these judges. Default = every judge that
160
- * appears in the input. A judge named here but absent from a
161
- * dimension throws (see "fail loud" below).
162
- */
163
- judges?: string[];
164
- }
165
- /**
166
- * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.
167
- *
168
- * For each dimension, builds the [n_items][n_judges] matrix of scores
169
- * (keeping only items every judge rated on that dimension), then runs
170
- * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and
171
- * bootstrap CIs. Reports a pooled mean across dimensions as a single
172
- * "is this judge panel reliable on this corpus?" number.
173
- *
174
- * Fail-loud contract:
175
- * - Empty input throws.
176
- * - Fewer than 2 judges or fewer than 2 items per dimension throws.
177
- * - A judge present in some dimensions but with zero scored items on
178
- * another dimension throws (would silently shrink the matrix).
179
- * - Duplicate (itemId, judgeName, dimension) records throw.
180
- */
181
- declare function corpusInterRaterAgreement(records: CorpusScoreRecord[], opts?: CorpusAgreementOptions): CorpusAgreementReport;
182
- /**
183
- * Convenience adapter for `JudgeScore[]` data keyed externally by item.
184
- *
185
- * Use when you have per-item arrays of `JudgeScore[]` (e.g. one
186
- * `ScenarioResult.judgeScores` per scenario) and want corpus-wide
187
- * agreement without manually flattening. `itemId` must be unique per
188
- * row of `itemsScores`.
189
- */
190
- declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
191
- itemId: string;
192
- scores: JudgeScore[];
193
- }>, opts?: CorpusAgreementOptions): CorpusAgreementReport;
194
- /**
195
- * Required N per arm for a two-sample comparison at target effect size,
196
- * alpha, and power. Normal-approximation formula:
197
- * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2
198
- * where d is Cohen's d. Returns Infinity for effect ≤ 0.
199
- */
200
- declare function requiredSampleSize(opts: {
201
- effect: number;
202
- alpha?: number;
203
- power?: number;
204
- twoSided?: boolean;
205
- }): number;
206
- /**
207
- * Minimum detectable paired effect (standardised units) for a target paired
208
- * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by
209
- * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap
210
- * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.
211
- */
212
- declare function pairedMde(opts: {
213
- nPaired: number;
214
- alpha?: number;
215
- power?: number;
216
- twoSided?: boolean;
217
- }): number;
218
- /**
219
- * Number of paired observations needed for a McNemar test to reach a target
220
- * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
221
- * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
222
- * `p01` (P[control wins]); concordant pairs carry no information, so the count
223
- * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
224
- * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
225
- * `δ = p10 − p01`,
226
- * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
227
- * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
228
- * tiny discordant counts where the exact {@link mcnemar} differs from the normal
229
- * approximation, treat the result as a lower bound and prefer the discordant-pair
230
- * floor.
231
- */
232
- declare function mcnemarRequiredN(opts: {
233
- p10: number;
234
- p01: number;
235
- alpha?: number;
236
- power?: number;
237
- twoSided?: boolean;
238
- }): number;
239
- /**
240
- * Power of a McNemar test at a given number of paired observations, the inverse
241
- * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
242
- * Returns a value in [0, 1]; equals `alpha` when there is no effect.
243
- */
244
- declare function mcnemarPower(opts: {
245
- p10: number;
246
- p01: number;
247
- nPairs: number;
248
- alpha?: number;
249
- twoSided?: boolean;
250
- }): number;
251
- /** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
252
- declare function bonferroni(pValues: number[], alpha?: number): {
253
- adjusted: number[];
254
- significant: boolean[];
255
- };
256
- /**
257
- * Holm step-down family-wise error adjustment.
258
- *
259
- * P-values are sorted from smallest to largest, multiplied by their remaining
260
- * hypothesis count, and made monotonically non-decreasing before being mapped
261
- * back to input order. This uniformly dominates plain Bonferroni while keeping
262
- * strong family-wise error control under arbitrary dependence.
263
- */
264
- declare function holm(pValues: readonly number[], alpha?: number): {
265
- adjusted: number[];
266
- significant: boolean[];
267
- };
268
- /**
269
- * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
270
- * significance at the target FDR; handles ties and preserves q monotonicity.
271
- */
272
- declare function benjaminiHochberg(pValues: number[], fdr?: number): {
273
- qValues: number[];
274
- significant: boolean[];
275
- };
276
- interface PairedBootstrapResult {
277
- /** Number of paired observations. */
278
- n: number;
279
- /** Median of paired deltas (after − before). */
280
- median: number;
281
- /** Mean of paired deltas. */
282
- mean: number;
283
- /** Lower bound of the bootstrap CI on the chosen statistic. */
284
- low: number;
285
- /** Upper bound of the bootstrap CI on the chosen statistic. */
286
- high: number;
287
- /** Confidence level used (e.g. 0.95). */
288
- confidence: number;
289
- /** Number of bootstrap resamples used. */
290
- resamples: number;
291
- }
292
- interface PairedBootstrapOptions {
293
- /** Confidence level. Default 0.95. */
294
- confidence?: number;
295
- /** Bootstrap resample count. Default 2000. */
296
- resamples?: number;
297
- /** Statistic to bootstrap. Default 'median'. */
298
- statistic?: 'median' | 'mean';
299
- /** Deterministic seed. If omitted, uses Math.random(). */
300
- seed?: number;
301
- }
302
- /**
303
- * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
304
- * statistic (median by default); pairs are resampled with replacement. The
305
- * lower bound is what the promotion gate checks — `low > threshold` means the
306
- * gain is real at the confidence level. Throws on unequal sample sizes.
307
- */
308
- declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
309
- /** Pre-registered direction for a one-sided paired sign test. */
310
- type SignTestAlternative = 'greater' | 'less';
311
- /** Exact one-sided sign-test result for paired numeric differences. */
312
- interface PairedSignTestResult {
313
- /** Total supplied differences, including zero ties. */
314
- n: number;
315
- /** Strictly positive differences. */
316
- positive: number;
317
- /** Strictly negative differences. */
318
- negative: number;
319
- /** Zero differences excluded from the binomial test. */
320
- ties: number;
321
- /** Non-zero differences used by the binomial test. */
322
- nNonTies: number;
323
- /** Direction of the pre-registered alternative hypothesis. */
324
- alternative: SignTestAlternative;
325
- /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
326
- pValue: number;
327
- }
328
- /**
329
- * Exact one-sided sign test over paired differences.
330
- *
331
- * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
332
- * tests whether positive signs are more likely than negative signs and returns
333
- * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
334
- * negative signs as successes instead. With a continuous difference
335
- * distribution this is the usual directional median test. Exact zero
336
- * differences are ties and do not enter the binomial denominator. All-tie and
337
- * empty inputs return p = 1. Every input difference must be finite, and the
338
- * direction must be chosen explicitly so a caller cannot select it after
339
- * seeing the signs.
340
- */
341
- declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
342
- /** A binomial proportion estimate with a confidence interval. */
343
- interface ProportionInterval {
344
- /** Point estimate successes / n (0 when n = 0). */
345
- estimate: number;
346
- /** Lower bound, clamped to [0, 1]. */
347
- lower: number;
348
- /** Upper bound, clamped to [0, 1]. */
349
- upper: number;
350
- }
351
- /**
352
- * Wilson score interval for a binomial proportion. Correct at small n and near
353
- * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
354
- * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
355
- * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
356
- * proportion. `n = 0 ⇒ {0, 0, 0}`.
357
- */
358
- declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
359
- /** Result of a McNemar paired-binary significance test. */
360
- interface McNemarResult {
361
- /** Total paired observations. */
362
- n: number;
363
- /** Discordant pairs (b + c) — the only ones that carry signal. */
364
- nDiscordant: number;
365
- /** Pairs where treatment succeeded and control failed ("newly correct"). */
366
- b: number;
367
- /** Pairs where control succeeded and treatment failed ("newly wrong"). */
368
- c: number;
369
- /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
370
- statistic: number;
371
- /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
372
- pValue: number;
373
- }
374
- /**
375
- * McNemar's test for paired binary outcomes — the correct significance test for
376
- * "does treatment change the success rate vs control on the SAME items". Only
377
- * discordant pairs (one arm right, the other wrong) carry information; concordant
378
- * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
379
- * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
380
- * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
381
- * at the small discordant counts typical of eval runs (no continuity-corrected
382
- * chi-square approximation needed, though it is returned as `statistic` for
383
- * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
384
- * the module's (before, after) convention. Throws on unequal lengths.
385
- */
386
- declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
387
- /** A paired binary effect size (treatment rate − control rate) with a CI. */
388
- interface RiskDifferenceResult {
389
- /** Total paired observations. */
390
- n: number;
391
- /** Discordant pairs: treatment-win count. */
392
- b: number;
393
- /** Discordant pairs: control-win count. */
394
- c: number;
395
- /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
396
- riskDifference: number;
397
- /** Lower bound of the CI, clamped to [-1, 1]. */
398
- lower: number;
399
- /** Upper bound of the CI, clamped to [-1, 1]. */
400
- upper: number;
401
- /** Confidence level used. */
402
- confidence: number;
403
- }
404
- /**
405
- * Paired risk difference (the effect-size companion to {@link mcnemar}): the
406
- * change in success rate p(treatment) − p(control) on matched items, which for
407
- * paired binary data equals (b − c) / n. The CI uses the paired variance from
408
- * the discordant counts, not the independent-samples formula (which overstates
409
- * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
410
- * arrays, control first. Throws on unequal lengths.
411
- */
412
- declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
413
- /**
414
- * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
415
- * Language Models Trained on Code"). Given `n` independent samples for one
416
- * problem of which `c` pass, the probability that at least one of a random k of
417
- * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
418
- * first k pass" is biased high at small n; this is the variance-reduced estimator
419
- * averaged implicitly over all k-subsets. Average the per-problem values across
420
- * the suite for the corpus pass@k. Computed in the numerically stable product
421
- * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
422
- */
423
- declare function passAtK(n: number, c: number, k: number): number;
424
- interface EProcessOptions {
425
- /** Type-I error budget. The process decides when wealth ≥ 1/alpha
426
- * (Ville's inequality). Default 0.05. */
427
- alpha?: number;
428
- /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
429
- * maxBet < 1/nullMean so every wealth factor stays strictly positive.
430
- * Default 0.5. */
431
- maxBet?: number;
432
- /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
433
- * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
434
- * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
435
- nullMean?: number;
436
- }
437
- interface EProcessStep {
438
- /** Current wealth W_n — the e-value against H0 after n observations. */
439
- wealth: number;
440
- /** Observations consumed so far. */
441
- n: number;
442
- /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
443
- decided: boolean;
444
- }
445
- interface EProcessState extends EProcessStep {
446
- alpha: number;
447
- maxBet: number;
448
- nullMean: number;
449
- /** The decision boundary 1/alpha. */
450
- threshold: number;
451
- /** Observation count at the first threshold crossing; undefined until decided. */
452
- decidedAtN?: number;
453
- }
454
- interface EProcess {
455
- /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
456
- * input — a silent clamp would corrupt the type-I guarantee. */
457
- update(x: number): EProcessStep;
458
- state(): EProcessState;
459
- }
460
- /**
461
- * Betting test-martingale for bounded observations — the e-process core of
462
- * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
463
- * of bounded random variables by betting", JRSS-B 2024).
464
- *
465
- * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
466
- *
467
- * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
468
- *
469
- * with the truncated GROW-style plug-in bet computed from PRIOR observations:
470
- *
471
- * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
472
- *
473
- * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
474
- * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
475
- *
476
- * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
477
- * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
478
- * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
479
- * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
480
- * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
481
- * (no prior evidence), so the first observation never moves wealth.
482
- *
483
- * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
484
- * wealth keeps updating after the crossing (the e-process remains valid), but
485
- * the decision time is the first crossing.
486
- */
487
- declare function eProcess(opts?: EProcessOptions): EProcess;
488
- /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
489
- * cryptographic. Exported so e-process shuffles and bootstrap resampling
490
- * share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
491
- * gate verdicts is non-reproducible by construction). */
492
- declare function mulberry32(seed: number): () => number;
493
-
494
- export { mcnemarPower as A, mcnemarRequiredN as B, type CorpusAgreementReport as C, mulberry32 as D, type EProcessState as E, normalizeScores as F, pairedMde as G, pairedRiskDifference as H, pairedSignTest as I, pairedTTest as J, partialCredit as K, passAtK as L, type McNemarResult as M, pearsonR as N, ranks as O, type PairedBootstrapOptions as P, requiredSampleSize as Q, type RiskDifferenceResult as R, type SignTestAlternative as S, spearmanR as T, weightedComposite as U, weightedMean as V, type WeightedCompositeInput as W, wilson as X, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type PairedSignTestResult as j, type ProportionInterval as k, type WeightedCompositeResult as l, bonferroni as m, cliffsDelta as n, cohensD as o, pairedBootstrap as p, confidenceInterval as q, corpusInterRaterAgreement as r, corpusInterRaterAgreementFromJudgeScores as s, eProcess as t, holm as u, interRaterReliability as v, wilcoxonSignedRank as w, interpretCliffs as x, mannWhitneyU as y, mcnemar as z };