@tangle-network/agent-eval 0.133.1 → 0.133.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +165 -0
- package/dist/{analyze-runs-DZr7JW-m.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
- package/dist/{analyze-runs-DZr7JW-m.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BU7P6PCW.js → benchmarks-BP9sgMia.js} +3 -3
- package/dist/{benchmarks-BU7P6PCW.js.map → benchmarks-BP9sgMia.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-CnzHQndg.js → campaign--V4ffEKR.js} +12 -6
- package/dist/{campaign-CnzHQndg.js.map → campaign--V4ffEKR.js.map} +1 -1
- package/dist/{client-D4F9hdzR.d.ts → client-Du7B81wW.d.ts} +28 -14
- package/dist/client-Du7B81wW.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +9 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/index-3cdlURSk.d.ts.map +1 -1
- package/dist/{index-Wek5mU0y.d.ts → index-B5MNN1f1.d.ts} +3 -3
- package/dist/{index-Wek5mU0y.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
- package/dist/{index-Ba636PKl.d.ts → index-DOqvIJ8I.d.ts} +27 -10
- package/dist/index-DOqvIJ8I.d.ts.map +1 -0
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +56 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +39 -22
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-BGrHeDu3.js → opencode-sqlite-8r6WUfHc.js} +2 -3
- package/dist/opencode-sqlite-8r6WUfHc.js.map +1 -0
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-DfmKSIEE.d.ts → release-report-DKBtegGt.d.ts} +2 -2
- package/dist/{release-report-DfmKSIEE.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-DMimgHtN.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
- package/dist/{researcher-DMimgHtN.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.js +4 -4
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-CeTlDrf6.js → rollout-CreDz__7.js} +2 -2
- package/dist/{rollout-CeTlDrf6.js.map → rollout-CreDz__7.js.map} +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CAASpcS3.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
- package/dist/{skillopt-optimization-method-CAASpcS3.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-BoIzh7Dl.js → skillopt-optimization-method-vvJ4bMNI.js} +123 -24
- package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-DnUcjVpV.d.ts → summary-report-DyOhItws.d.ts} +4 -3
- package/dist/summary-report-DyOhItws.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-_lnTLM3z.js → supervisor-run-B7lUGoyZ.js} +2 -2
- package/dist/{supervisor-run-_lnTLM3z.js.map → supervisor-run-B7lUGoyZ.js.map} +1 -1
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/client-D4F9hdzR.d.ts.map +0 -1
- package/dist/index-Ba636PKl.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/opencode-sqlite-BGrHeDu3.js.map +0 -1
- package/dist/skillopt-optimization-method-BoIzh7Dl.js.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-DnUcjVpV.d.ts.map +0 -1
|
@@ -9,7 +9,14 @@ declare function weightedMean(scores: {
|
|
|
9
9
|
score: number;
|
|
10
10
|
weight?: number;
|
|
11
11
|
}[]): number;
|
|
12
|
-
/**
|
|
12
|
+
/**
|
|
13
|
+
* Percentile bootstrap confidence interval on the mean of `scores`.
|
|
14
|
+
*
|
|
15
|
+
* Descriptive spread. It is not a significance test, and at small n its bounds
|
|
16
|
+
* are anti-conservative in the same way {@link pairedBootstrap}'s are — see
|
|
17
|
+
* {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
|
|
18
|
+
* the scores themselves, so the interval is reproducible either way.
|
|
19
|
+
*/
|
|
13
20
|
declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
|
|
14
21
|
seed?: number;
|
|
15
22
|
resamples?: number;
|
|
@@ -19,47 +26,136 @@ declare function confidenceInterval(scores: number[], confidence?: number, opts?
|
|
|
19
26
|
upper: number;
|
|
20
27
|
};
|
|
21
28
|
/**
|
|
22
|
-
* Inter-rater reliability —
|
|
29
|
+
* Inter-rater reliability — Krippendorff's α under the squared-difference
|
|
30
|
+
* metric, pooled across dimensions.
|
|
23
31
|
*
|
|
24
|
-
* Each inner array is one judge's scores
|
|
25
|
-
*
|
|
32
|
+
* Each inner array is one judge's scores. Items are matched by position
|
|
33
|
+
* WITHIN a dimension: the k-th score a judge supplies carrying dimension
|
|
34
|
+
* `d` is item k of `d`, and the ratings compared against each other are
|
|
35
|
+
* the ones different judges gave to the same item. Every judge that scores
|
|
36
|
+
* a dimension at all must supply the same number of scores for it —
|
|
37
|
+
* ragged input cannot be aligned into items and throws rather than
|
|
38
|
+
* comparing mismatched items.
|
|
39
|
+
*
|
|
40
|
+
* α = 1 − D_observed / D_expected: D_observed averages the squared
|
|
41
|
+
* difference over within-item judge pairs, D_expected over every pair of
|
|
42
|
+
* ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
|
|
43
|
+
* negative is systematic disagreement.
|
|
26
44
|
*/
|
|
27
45
|
declare function interRaterReliability(judgeScores: JudgeScore[][]): number;
|
|
46
|
+
/** How a rank test's p-value was actually computed. */
|
|
47
|
+
type RankTestMethod = 'exact' | 'permutation' | 'asymptotic';
|
|
28
48
|
/**
|
|
29
|
-
*
|
|
30
|
-
*
|
|
49
|
+
* What the caller asks for. `'auto'` selects `'exact'` inside the enumeration
|
|
50
|
+
* threshold and `'permutation'` above it, and never selects `'asymptotic'`.
|
|
31
51
|
*/
|
|
32
|
-
|
|
52
|
+
type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic';
|
|
53
|
+
interface RankTestOptions {
|
|
54
|
+
/** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */
|
|
55
|
+
method?: RankTestMethodRequest;
|
|
56
|
+
/** Resamples on the Monte Carlo permutation path. Default 100000. */
|
|
57
|
+
permutations?: number;
|
|
58
|
+
/** Seed for the permutation path. Omitted ⇒ derived from the data itself, so
|
|
59
|
+
* the result is reproducible either way. */
|
|
60
|
+
seed?: number;
|
|
61
|
+
}
|
|
62
|
+
/** Maximum dynamic-programming cells used by an exact two-sample rank test. */
|
|
63
|
+
declare const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
|
|
64
|
+
/** Maximum inner-loop transitions used by an exact two-sample rank test. */
|
|
65
|
+
declare const MANN_WHITNEY_EXACT_MAX_WORK = 250000;
|
|
66
|
+
/** Non-zero differences up to which the signed-rank null is enumerated exactly. */
|
|
67
|
+
declare const WILCOXON_EXACT_MAX_N = 20;
|
|
68
|
+
/** Resamples used when a rank test falls back to Monte Carlo permutation. */
|
|
69
|
+
declare const DEFAULT_PERMUTATIONS = 100000;
|
|
70
|
+
interface MannWhitneyResult {
|
|
71
|
+
/** `min(U_a, U_b)` — the conventional reported statistic. */
|
|
33
72
|
u: number;
|
|
73
|
+
/** U for sample `a`. Carries the direction of the effect, which `u` discards. */
|
|
74
|
+
uA: number;
|
|
75
|
+
/** Two-sided p-value. */
|
|
34
76
|
p: number;
|
|
35
|
-
|
|
77
|
+
/** How `p` was computed. */
|
|
78
|
+
method: RankTestMethod;
|
|
79
|
+
/** Smallest two-sided p this design can produce. `p` can never be below it. */
|
|
80
|
+
pFloor: number;
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* Mann-Whitney U — two independent samples, no distributional assumption.
|
|
84
|
+
*
|
|
85
|
+
* Exact conditional (permutation) p by default when the dynamic program fits
|
|
86
|
+
* {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
|
|
87
|
+
* {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
|
|
88
|
+
* permutation above those limits. This keeps imbalanced designs such as 1+24
|
|
89
|
+
* exact without admitting expensive balanced designs merely because they have
|
|
90
|
+
* the same total size. Throws on non-finite input and on `method:
|
|
91
|
+
* 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
|
|
92
|
+
* pFloor = 1` — no design, no attainable evidence.
|
|
93
|
+
*/
|
|
94
|
+
declare function mannWhitneyU(a: number[], b: number[], opts?: RankTestOptions): MannWhitneyResult;
|
|
36
95
|
/** Partial credit: returns 0-1 ratio of current toward target */
|
|
37
96
|
declare function partialCredit(current: number, target: number): number;
|
|
97
|
+
interface PairedTTestResult {
|
|
98
|
+
/** Null when the statistic is undefined — see {@link pairedTTest}. */
|
|
99
|
+
t: number | null;
|
|
100
|
+
df: number;
|
|
101
|
+
/** Null exactly when `t` is null. */
|
|
102
|
+
p: number | null;
|
|
103
|
+
}
|
|
38
104
|
/**
|
|
39
105
|
* Paired t-test — before/after measurements on the SAME items.
|
|
40
106
|
* Pairing removes inter-item variance, giving tighter significance than
|
|
41
107
|
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
42
108
|
* scenarios.
|
|
109
|
+
*
|
|
110
|
+
* Returns `t = p = null` where the statistic is undefined: fewer than two
|
|
111
|
+
* pairs, or a non-zero constant delta whose observed variance is zero. A
|
|
112
|
+
* constant shift carries no information about the variance it would have to
|
|
113
|
+
* be compared against, so the honest answer is "undefined", not `p = 0` —
|
|
114
|
+
* three observations cannot buy absolute certainty. This is the same contract
|
|
115
|
+
* {@link pairedCohensDz} states for the same condition. An all-zero delta is
|
|
116
|
+
* different: it is a measured null, and returns `t = 0, p = 1`.
|
|
43
117
|
*/
|
|
44
|
-
declare function pairedTTest(before: number[], after: number[]):
|
|
45
|
-
|
|
46
|
-
|
|
118
|
+
declare function pairedTTest(before: number[], after: number[]): PairedTTestResult;
|
|
119
|
+
interface WilcoxonSignedRankResult {
|
|
120
|
+
/** W⁺, the rank sum of the positive differences. (scipy reports
|
|
121
|
+
* `min(W⁺, W⁻)`; compare statistics only after converting.) */
|
|
122
|
+
w: number;
|
|
123
|
+
/** Two-sided p-value. */
|
|
47
124
|
p: number;
|
|
48
|
-
|
|
125
|
+
/** How `p` was computed. */
|
|
126
|
+
method: RankTestMethod;
|
|
127
|
+
/** Smallest two-sided p this design can produce. */
|
|
128
|
+
pFloor: number;
|
|
129
|
+
/** Non-zero differences — zero differences are dropped and carry no rank. */
|
|
130
|
+
nNonZero: number;
|
|
131
|
+
}
|
|
49
132
|
/**
|
|
50
|
-
* Wilcoxon signed-rank
|
|
51
|
-
*
|
|
133
|
+
* Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
|
|
134
|
+
*
|
|
135
|
+
* Exact conditional (sign-flip) p by default at `n ≤
|
|
136
|
+
* {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
|
|
137
|
+
* permutation above it. Throws on non-finite input and on `method:
|
|
138
|
+
* 'asymptotic'` where an exact answer is available.
|
|
139
|
+
*
|
|
140
|
+
* `n` is the count of NON-ZERO differences: exact ties are dropped before
|
|
141
|
+
* ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
|
|
142
|
+
* All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
|
|
143
|
+
* `pFloor` states rather than leaving `p = 1` to be read as a measured null.
|
|
52
144
|
*/
|
|
53
|
-
declare function wilcoxonSignedRank(before: number[], after: number[]):
|
|
54
|
-
w: number;
|
|
55
|
-
p: number;
|
|
56
|
-
};
|
|
145
|
+
declare function wilcoxonSignedRank(before: number[], after: number[], opts?: RankTestOptions): WilcoxonSignedRankResult;
|
|
57
146
|
/**
|
|
58
147
|
* Cohen's d — standardized effect size for two independent groups.
|
|
59
148
|
* Positive d means group b has higher mean than group a.
|
|
60
149
|
* Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
|
|
150
|
+
*
|
|
151
|
+
* Returns null where the standardized effect is undefined: fewer than two
|
|
152
|
+
* observations in either group, or a zero pooled standard deviation with
|
|
153
|
+
* unequal means. Null is NOT "no effect" — zero within-group spread across a
|
|
154
|
+
* real mean gap is an unbounded effect, the opposite of negligible. Equal
|
|
155
|
+
* means with zero spread is a genuine 0. Same contract as
|
|
156
|
+
* {@link pairedCohensDz}.
|
|
61
157
|
*/
|
|
62
|
-
declare function cohensD(a: number[], b: number[]): number;
|
|
158
|
+
declare function cohensD(a: number[], b: number[]): number | null;
|
|
63
159
|
/**
|
|
64
160
|
* Cohen's dz for paired observations: mean(after - before) divided by the
|
|
65
161
|
* sample standard deviation of those within-pair deltas.
|
|
@@ -215,6 +311,11 @@ declare function requiredSampleSize(opts: {
|
|
|
215
311
|
/**
|
|
216
312
|
* Required number of paired observations for a target Cohen's dz.
|
|
217
313
|
* Unlike the independent-groups formula, this has no two-arm factor of two.
|
|
314
|
+
*
|
|
315
|
+
* Normal quantiles with no t correction, so treat the result as a LOWER bound:
|
|
316
|
+
* it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
|
|
317
|
+
* it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
|
|
318
|
+
* consults to decide whether 3–10 repetitions suffice.
|
|
218
319
|
*/
|
|
219
320
|
declare function requiredPairedSampleSize(opts: {
|
|
220
321
|
effect: number;
|
|
@@ -267,8 +368,14 @@ declare function mcnemarPower(opts: {
|
|
|
267
368
|
alpha?: number;
|
|
268
369
|
twoSided?: boolean;
|
|
269
370
|
}): number;
|
|
270
|
-
/**
|
|
271
|
-
|
|
371
|
+
/**
|
|
372
|
+
* Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
|
|
373
|
+
*
|
|
374
|
+
* Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
|
|
375
|
+
* {@link holm}, which uniformly dominates this correction and must therefore
|
|
376
|
+
* never reject less. Validates its inputs on the same terms.
|
|
377
|
+
*/
|
|
378
|
+
declare function bonferroni(pValues: readonly number[], alpha?: number): {
|
|
272
379
|
adjusted: number[];
|
|
273
380
|
significant: boolean[];
|
|
274
381
|
};
|
|
@@ -287,8 +394,11 @@ declare function holm(pValues: readonly number[], alpha?: number): {
|
|
|
287
394
|
/**
|
|
288
395
|
* Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
|
|
289
396
|
* significance at the target FDR; handles ties and preserves q monotonicity.
|
|
397
|
+
*
|
|
398
|
+
* Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
|
|
399
|
+
* exactly-`fdr` q-value is a discovery.
|
|
290
400
|
*/
|
|
291
|
-
declare function benjaminiHochberg(pValues: number[], fdr?: number): {
|
|
401
|
+
declare function benjaminiHochberg(pValues: readonly number[], fdr?: number): {
|
|
292
402
|
qValues: number[];
|
|
293
403
|
significant: boolean[];
|
|
294
404
|
};
|
|
@@ -307,7 +417,20 @@ interface PairedBootstrapResult {
|
|
|
307
417
|
confidence: number;
|
|
308
418
|
/** Number of bootstrap resamples used. */
|
|
309
419
|
resamples: number;
|
|
420
|
+
/** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */
|
|
421
|
+
gateEligible: boolean;
|
|
310
422
|
}
|
|
423
|
+
/**
|
|
424
|
+
* Pairs below which a percentile bootstrap interval is descriptive spread only.
|
|
425
|
+
*
|
|
426
|
+
* `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
|
|
427
|
+
* seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
|
|
428
|
+
* median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
|
|
429
|
+
* three points, not an implementation error — scipy's BCa gives 16.0 % on the
|
|
430
|
+
* same n = 3 data — so no change to the estimator moves it. Below this floor
|
|
431
|
+
* the decision belongs to the exact sign test or exact signed-rank test.
|
|
432
|
+
*/
|
|
433
|
+
declare const BOOTSTRAP_GATE_MIN_N = 20;
|
|
311
434
|
interface PairedBootstrapOptions {
|
|
312
435
|
/** Confidence level. Default 0.95. */
|
|
313
436
|
confidence?: number;
|
|
@@ -315,14 +438,19 @@ interface PairedBootstrapOptions {
|
|
|
315
438
|
resamples?: number;
|
|
316
439
|
/** Statistic to bootstrap. Default 'median'. */
|
|
317
440
|
statistic?: 'median' | 'mean';
|
|
318
|
-
/** Deterministic seed. If omitted,
|
|
441
|
+
/** Deterministic seed. If omitted, derived from the deltas so the interval
|
|
442
|
+
* is reproducible regardless. */
|
|
319
443
|
seed?: number;
|
|
320
444
|
}
|
|
321
445
|
/**
|
|
322
446
|
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
323
|
-
* statistic (median by default); pairs are resampled with replacement.
|
|
324
|
-
*
|
|
325
|
-
*
|
|
447
|
+
* statistic (median by default); pairs are resampled with replacement. Throws
|
|
448
|
+
* on unequal sample sizes.
|
|
449
|
+
*
|
|
450
|
+
* `low > threshold` carries the stated confidence ONLY at `n ≥
|
|
451
|
+
* {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
|
|
452
|
+
* check fires under a true null several times more often than nominal, so the
|
|
453
|
+
* interval is descriptive spread and a promotion must not turn on it.
|
|
326
454
|
*/
|
|
327
455
|
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
328
456
|
/** Pre-registered direction for a one-sided paired sign test. */
|
|
@@ -506,9 +634,9 @@ interface EProcess {
|
|
|
506
634
|
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
507
635
|
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
508
636
|
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
509
|
-
* share ONE PRNG implementation
|
|
510
|
-
*
|
|
637
|
+
* share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
|
|
638
|
+
* stream, including 0. */
|
|
511
639
|
declare function mulberry32(seed: number): () => number;
|
|
512
640
|
//#endregion
|
|
513
|
-
export {
|
|
514
|
-
//# sourceMappingURL=statistics-
|
|
641
|
+
export { partialCredit as $, benjaminiHochberg as A, interpretCliffs as B, RankTestOptions as C, WeightedCompositeInput as D, WILCOXON_EXACT_MAX_N as E, corpusInterRaterAgreement as F, mulberry32 as G, mcnemar as H, corpusInterRaterAgreementFromJudgeScores as I, pairedCohensDz as J, normalizeScores as K, eProcess as L, cliffsDelta as M, cohensD as N, WeightedCompositeResult as O, confidenceInterval as P, pairedTTest as Q, holm as R, RankTestMethodRequest as S, SignTestAlternative as T, mcnemarPower as U, mannWhitneyU as V, mcnemarRequiredN as W, pairedRiskDifference as X, pairedMde as Y, pairedSignTest as Z, PairedBootstrapResult as _, CorpusAgreementReport as a, spearmanR as at, ProportionInterval as b, EProcess as c, wilcoxonSignedRank as ct, EProcessStep as d, passAtK as et, MANN_WHITNEY_EXACT_MAX_STATES as f, PairedBootstrapOptions as g, McNemarResult as h, CorpusAgreementPerDimension as i, requiredSampleSize as it, bonferroni as j, WilcoxonSignedRankResult as k, EProcessOptions as l, wilson as lt, MannWhitneyResult as m, CliffsMagnitude as n, ranks as nt, CorpusScoreRecord as o, weightedComposite as ot, MANN_WHITNEY_EXACT_MAX_WORK as p, pairedBootstrap as q, CorpusAgreementOptions as r, requiredPairedSampleSize as rt, DEFAULT_PERMUTATIONS as s, weightedMean as st, BOOTSTRAP_GATE_MIN_N as t, pearsonR as tt, EProcessState as u, PairedSignTestResult as v, RiskDifferenceResult as w, RankTestMethod as x, PairedTTestResult as y, interRaterReliability as z };
|
|
642
|
+
//# sourceMappingURL=statistics-D_4Snl-5.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"statistics-D_4Snl-5.d.ts","names":[],"sources":["../src/statistics.ts"],"mappings":";;;;;cAaa,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;;;;;;;;;;;;iBAiDlB,sBAAsB,aAAa;;KAkFvC;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;;iBAwFa,cAAc,iBAAiB;UAK9B;;EAEf;EACA;;EAEA;;;;;;;;;;;;;;;;iBAiBc,YAAY,kBAAkB,kBAAkB;UAyB/C;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;iBAmFa,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA8Ba,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;;iBAwBc,WACd,4BACA;EACG;EAAoB;;;;;;;;;;iBAgBT,KACd,4BACA;EACG;EAAoB;;;;;;;;;iBA4BT,kBACd,4BACA;EACG;EAAmB;;UAuCP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;cAaW;UAEI;;EAEf;;EAEA;;EAEA;;;EAGA;;;;;;;;;;;;iBAac,gBACd,kBACA,iBACA,OAAM,yBACL;;KA0DS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;UAmBxD;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;iBAWc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;iBAyCa,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsatC,WAAW"}
|