@tangle-network/agent-eval 0.133.1 → 0.133.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +165 -0
- package/dist/{analyze-runs-DZr7JW-m.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
- package/dist/{analyze-runs-DZr7JW-m.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BU7P6PCW.js → benchmarks-BP9sgMia.js} +3 -3
- package/dist/{benchmarks-BU7P6PCW.js.map → benchmarks-BP9sgMia.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-CnzHQndg.js → campaign--V4ffEKR.js} +12 -6
- package/dist/{campaign-CnzHQndg.js.map → campaign--V4ffEKR.js.map} +1 -1
- package/dist/{client-D4F9hdzR.d.ts → client-Du7B81wW.d.ts} +28 -14
- package/dist/client-Du7B81wW.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +9 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/index-3cdlURSk.d.ts.map +1 -1
- package/dist/{index-Wek5mU0y.d.ts → index-B5MNN1f1.d.ts} +3 -3
- package/dist/{index-Wek5mU0y.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
- package/dist/{index-Ba636PKl.d.ts → index-DOqvIJ8I.d.ts} +27 -10
- package/dist/index-DOqvIJ8I.d.ts.map +1 -0
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +56 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +39 -22
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-BGrHeDu3.js → opencode-sqlite-8r6WUfHc.js} +2 -3
- package/dist/opencode-sqlite-8r6WUfHc.js.map +1 -0
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-DfmKSIEE.d.ts → release-report-DKBtegGt.d.ts} +2 -2
- package/dist/{release-report-DfmKSIEE.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-DMimgHtN.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
- package/dist/{researcher-DMimgHtN.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.js +4 -4
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-CeTlDrf6.js → rollout-CreDz__7.js} +2 -2
- package/dist/{rollout-CeTlDrf6.js.map → rollout-CreDz__7.js.map} +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CAASpcS3.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
- package/dist/{skillopt-optimization-method-CAASpcS3.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-BoIzh7Dl.js → skillopt-optimization-method-vvJ4bMNI.js} +123 -24
- package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-DnUcjVpV.d.ts → summary-report-DyOhItws.d.ts} +4 -3
- package/dist/summary-report-DyOhItws.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-_lnTLM3z.js → supervisor-run-B7lUGoyZ.js} +2 -2
- package/dist/{supervisor-run-_lnTLM3z.js.map → supervisor-run-B7lUGoyZ.js.map} +1 -1
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/client-D4F9hdzR.d.ts.map +0 -1
- package/dist/index-Ba636PKl.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/opencode-sqlite-BGrHeDu3.js.map +0 -1
- package/dist/skillopt-optimization-method-BoIzh7Dl.js.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-DnUcjVpV.d.ts.map +0 -1
|
@@ -402,14 +402,26 @@ function lnGamma(z) {
|
|
|
402
402
|
const t = z + g + .5;
|
|
403
403
|
return .5 * Math.log(2 * Math.PI) + (z + .5) * Math.log(t) - t + Math.log(x);
|
|
404
404
|
}
|
|
405
|
-
/**
|
|
405
|
+
/**
|
|
406
|
+
* Regularized incomplete beta function I_x(a, b).
|
|
407
|
+
*
|
|
408
|
+
* The Lentz continued fraction converges only for `x < (a+1)/(a+b+2)`; outside
|
|
409
|
+
* that domain it must be reached through the symmetry `I_x(a,b) = 1 −
|
|
410
|
+
* I_{1−x}(b,a)`. `studentTCdf` drives `x → 1` as `|t| → 0`, so the mirrored
|
|
411
|
+
* branch is the one every near-null t-statistic takes.
|
|
412
|
+
*/
|
|
406
413
|
function regularizedIncompleteBeta(x, a, b) {
|
|
407
414
|
if (x <= 0) return 0;
|
|
408
415
|
if (x >= 1) return 1;
|
|
409
416
|
const logBeta = lnGamma(a) + lnGamma(b) - lnGamma(a + b);
|
|
410
|
-
const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - logBeta)
|
|
411
|
-
|
|
412
|
-
|
|
417
|
+
const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - logBeta);
|
|
418
|
+
if (x < (a + 1) / (a + b + 2)) return front * betaContinuedFraction(x, a, b) / a;
|
|
419
|
+
return 1 - front * betaContinuedFraction(1 - x, b, a) / b;
|
|
420
|
+
}
|
|
421
|
+
/** Modified Lentz evaluation of the beta continued fraction at `x`. */
|
|
422
|
+
function betaContinuedFraction(x, a, b) {
|
|
423
|
+
const maxIterations = 300;
|
|
424
|
+
const epsilon = 3e-15;
|
|
413
425
|
let c = 1;
|
|
414
426
|
let d = 1 - (a + b) * x / (a + 1);
|
|
415
427
|
if (Math.abs(d) < 1e-30) d = 1e-30;
|
|
@@ -434,17 +446,42 @@ function regularizedIncompleteBeta(x, a, b) {
|
|
|
434
446
|
fraction *= delta;
|
|
435
447
|
if (Math.abs(delta - 1) < epsilon) break;
|
|
436
448
|
}
|
|
437
|
-
return
|
|
449
|
+
return fraction;
|
|
438
450
|
}
|
|
439
451
|
//#endregion
|
|
440
452
|
//#region src/math/student-t.ts
|
|
441
|
-
/**
|
|
453
|
+
/**
|
|
454
|
+
* Student-t CDF via the regularized incomplete beta function.
|
|
455
|
+
*/
|
|
442
456
|
function studentTCdf(t, degreesOfFreedom) {
|
|
443
457
|
if (degreesOfFreedom <= 0) return .5;
|
|
444
|
-
if (degreesOfFreedom > 100) return normalCdf(t);
|
|
445
458
|
const beta = regularizedIncompleteBeta(degreesOfFreedom / (degreesOfFreedom + t * t), degreesOfFreedom / 2, .5);
|
|
446
459
|
return t >= 0 ? 1 - .5 * beta : .5 * beta;
|
|
447
460
|
}
|
|
461
|
+
/**
|
|
462
|
+
* Inverse Student-t CDF, solved against {@link studentTCdf}.
|
|
463
|
+
*
|
|
464
|
+
* The CDF is monotone, so bracket expansion followed by bisection is stable
|
|
465
|
+
* across fractional degrees of freedom and does not need a separate
|
|
466
|
+
* approximation with a different error profile.
|
|
467
|
+
*/
|
|
468
|
+
function studentTQuantile(probability, degreesOfFreedom) {
|
|
469
|
+
if (!Number.isFinite(probability) || probability < 0 || probability > 1) throw new RangeError(`studentTQuantile: probability must be in [0,1], got ${probability}`);
|
|
470
|
+
if (!Number.isFinite(degreesOfFreedom) || degreesOfFreedom <= 0) throw new RangeError(`studentTQuantile: degreesOfFreedom must be positive and finite, got ${degreesOfFreedom}`);
|
|
471
|
+
if (probability === 0) return Number.NEGATIVE_INFINITY;
|
|
472
|
+
if (probability === 1) return Number.POSITIVE_INFINITY;
|
|
473
|
+
if (probability === .5) return 0;
|
|
474
|
+
if (probability < .5) return -studentTQuantile(1 - probability, degreesOfFreedom);
|
|
475
|
+
let low = 0;
|
|
476
|
+
let high = 1;
|
|
477
|
+
while (studentTCdf(high, degreesOfFreedom) < probability) high *= 2;
|
|
478
|
+
for (let iteration = 0; iteration < 64; iteration++) {
|
|
479
|
+
const middle = (low + high) / 2;
|
|
480
|
+
if (studentTCdf(middle, degreesOfFreedom) < probability) low = middle;
|
|
481
|
+
else high = middle;
|
|
482
|
+
}
|
|
483
|
+
return (low + high) / 2;
|
|
484
|
+
}
|
|
448
485
|
//#endregion
|
|
449
486
|
//#region src/statistics.ts
|
|
450
487
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
@@ -462,7 +499,14 @@ function weightedMean(scores) {
|
|
|
462
499
|
}
|
|
463
500
|
return totalWeight > 0 ? weightedSum / totalWeight : 0;
|
|
464
501
|
}
|
|
465
|
-
/**
|
|
502
|
+
/**
|
|
503
|
+
* Percentile bootstrap confidence interval on the mean of `scores`.
|
|
504
|
+
*
|
|
505
|
+
* Descriptive spread. It is not a significance test, and at small n its bounds
|
|
506
|
+
* are anti-conservative in the same way {@link pairedBootstrap}'s are — see
|
|
507
|
+
* {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
|
|
508
|
+
* the scores themselves, so the interval is reproducible either way.
|
|
509
|
+
*/
|
|
466
510
|
function confidenceInterval(scores, confidence = .95, opts = {}) {
|
|
467
511
|
if (scores.length === 0) return {
|
|
468
512
|
mean: 0,
|
|
@@ -477,7 +521,7 @@ function confidenceInterval(scores, confidence = .95, opts = {}) {
|
|
|
477
521
|
const n = scores.length;
|
|
478
522
|
const mean = scores.reduce((a, b) => a + b, 0) / n;
|
|
479
523
|
const B = opts.resamples ?? 1e3;
|
|
480
|
-
const rng = makeRng(opts.seed);
|
|
524
|
+
const rng = makeRng(opts.seed, scores);
|
|
481
525
|
const bootstrapMeans = [];
|
|
482
526
|
for (let i = 0; i < B; i++) {
|
|
483
527
|
let sum = 0;
|
|
@@ -495,26 +539,45 @@ function confidenceInterval(scores, confidence = .95, opts = {}) {
|
|
|
495
539
|
};
|
|
496
540
|
}
|
|
497
541
|
/**
|
|
498
|
-
* Inter-rater reliability —
|
|
542
|
+
* Inter-rater reliability — Krippendorff's α under the squared-difference
|
|
543
|
+
* metric, pooled across dimensions.
|
|
544
|
+
*
|
|
545
|
+
* Each inner array is one judge's scores. Items are matched by position
|
|
546
|
+
* WITHIN a dimension: the k-th score a judge supplies carrying dimension
|
|
547
|
+
* `d` is item k of `d`, and the ratings compared against each other are
|
|
548
|
+
* the ones different judges gave to the same item. Every judge that scores
|
|
549
|
+
* a dimension at all must supply the same number of scores for it —
|
|
550
|
+
* ragged input cannot be aligned into items and throws rather than
|
|
551
|
+
* comparing mismatched items.
|
|
499
552
|
*
|
|
500
|
-
*
|
|
501
|
-
*
|
|
553
|
+
* α = 1 − D_observed / D_expected: D_observed averages the squared
|
|
554
|
+
* difference over within-item judge pairs, D_expected over every pair of
|
|
555
|
+
* ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
|
|
556
|
+
* negative is systematic disagreement.
|
|
502
557
|
*/
|
|
503
558
|
function interRaterReliability(judgeScores) {
|
|
504
559
|
if (judgeScores.length < 2) return 1;
|
|
505
|
-
const
|
|
506
|
-
for (
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
560
|
+
const perDimension = /* @__PURE__ */ new Map();
|
|
561
|
+
for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) for (const s of judgeScores[judgeIndex]) {
|
|
562
|
+
let byJudge = perDimension.get(s.dimension);
|
|
563
|
+
if (byJudge === void 0) {
|
|
564
|
+
byJudge = Array.from({ length: judgeScores.length }, () => []);
|
|
565
|
+
perDimension.set(s.dimension, byJudge);
|
|
566
|
+
}
|
|
567
|
+
byJudge[judgeIndex].push(s.score);
|
|
511
568
|
}
|
|
512
569
|
const allValues = [];
|
|
513
570
|
const pairDiffs = [];
|
|
514
|
-
for (const
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
571
|
+
for (const [dimension, byJudge] of perDimension) {
|
|
572
|
+
const scoring = byJudge.filter((scores) => scores.length > 0);
|
|
573
|
+
if (scoring.length < 2) continue;
|
|
574
|
+
const itemCount = scoring[0].length;
|
|
575
|
+
if (scoring.some((scores) => scores.length !== itemCount)) throw new ValidationError(`interRaterReliability: dimension '${dimension}' has judges supplying ${scoring.map((scores) => scores.length).join("/")} scores — items cannot be aligned`);
|
|
576
|
+
for (let item = 0; item < itemCount; item++) {
|
|
577
|
+
const ratings = scoring.map((scores) => scores[item]);
|
|
578
|
+
for (const v of ratings) allValues.push(v);
|
|
579
|
+
for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
|
|
580
|
+
}
|
|
518
581
|
}
|
|
519
582
|
if (pairDiffs.length === 0 || allValues.length < 2) return 1;
|
|
520
583
|
const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length;
|
|
@@ -528,47 +591,97 @@ function interRaterReliability(judgeScores) {
|
|
|
528
591
|
if (expectedDisagreement === 0) return 1;
|
|
529
592
|
return 1 - observedDisagreement / expectedDisagreement;
|
|
530
593
|
}
|
|
594
|
+
/** Maximum dynamic-programming cells used by an exact two-sample rank test. */
|
|
595
|
+
const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
|
|
596
|
+
/** Maximum inner-loop transitions used by an exact two-sample rank test. */
|
|
597
|
+
const MANN_WHITNEY_EXACT_MAX_WORK = 25e4;
|
|
598
|
+
/** Non-zero differences up to which the signed-rank null is enumerated exactly. */
|
|
599
|
+
const WILCOXON_EXACT_MAX_N = 20;
|
|
600
|
+
/** Resamples used when a rank test falls back to Monte Carlo permutation. */
|
|
601
|
+
const DEFAULT_PERMUTATIONS = 1e5;
|
|
531
602
|
/**
|
|
532
|
-
* Mann-Whitney U
|
|
533
|
-
*
|
|
603
|
+
* Mann-Whitney U — two independent samples, no distributional assumption.
|
|
604
|
+
*
|
|
605
|
+
* Exact conditional (permutation) p by default when the dynamic program fits
|
|
606
|
+
* {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
|
|
607
|
+
* {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
|
|
608
|
+
* permutation above those limits. This keeps imbalanced designs such as 1+24
|
|
609
|
+
* exact without admitting expensive balanced designs merely because they have
|
|
610
|
+
* the same total size. Throws on non-finite input and on `method:
|
|
611
|
+
* 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
|
|
612
|
+
* pFloor = 1` — no design, no attainable evidence.
|
|
534
613
|
*/
|
|
535
|
-
function mannWhitneyU(a, b) {
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
p: 1
|
|
539
|
-
};
|
|
614
|
+
function mannWhitneyU(a, b, opts = {}) {
|
|
615
|
+
assertFiniteSample("mannWhitneyU", "a", a);
|
|
616
|
+
assertFiniteSample("mannWhitneyU", "b", b);
|
|
540
617
|
const n1 = a.length;
|
|
541
618
|
const n2 = b.length;
|
|
619
|
+
if (n1 === 0 || n2 === 0) return {
|
|
620
|
+
u: 0,
|
|
621
|
+
uA: 0,
|
|
622
|
+
p: 1,
|
|
623
|
+
method: "exact",
|
|
624
|
+
pFloor: 1
|
|
625
|
+
};
|
|
626
|
+
const total = n1 + n2;
|
|
542
627
|
const combined = [...a.map((v) => ({
|
|
543
628
|
v,
|
|
544
|
-
|
|
629
|
+
fromA: true
|
|
545
630
|
})), ...b.map((v) => ({
|
|
546
631
|
v,
|
|
547
|
-
|
|
632
|
+
fromA: false
|
|
548
633
|
}))].sort((x, y) => x.v - y.v);
|
|
549
|
-
const
|
|
550
|
-
let
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
634
|
+
const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v));
|
|
635
|
+
let rankSumA = 0;
|
|
636
|
+
for (let k = 0; k < total; k++) if (combined[k].fromA) rankSumA += midranks[k];
|
|
637
|
+
const uA = rankSumA - n1 * (n1 + 1) / 2;
|
|
638
|
+
const u = Math.min(uA, n1 * n2 - uA);
|
|
639
|
+
const doubled = midranks.map((rank) => Math.round(rank * 2));
|
|
640
|
+
const doubledDeviation = Math.abs(2 * uA - n1 * n2);
|
|
641
|
+
const selectedN = Math.min(n1, n2);
|
|
642
|
+
const otherN = total - selectedN;
|
|
643
|
+
const exactCost = exactTwoSampleCost(doubled, selectedN);
|
|
644
|
+
const designFloor = exactTwoSampleFloor(doubled, selectedN);
|
|
645
|
+
const method = selectRankTestMethod("mannWhitneyU", opts.method ?? "auto", `n1=${n1}, n2=${n2}`, exactCost.states <= 8192 && exactCost.work <= 25e4, designFloor, `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString("en-US")} states and ${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString("en-US")} transitions`);
|
|
646
|
+
if (method === "exact") {
|
|
647
|
+
const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation);
|
|
648
|
+
return {
|
|
649
|
+
u,
|
|
650
|
+
uA,
|
|
651
|
+
p,
|
|
652
|
+
method,
|
|
653
|
+
pFloor
|
|
654
|
+
};
|
|
557
655
|
}
|
|
558
|
-
|
|
559
|
-
for (let k = 0; k < combined.length; k++) if (combined[k].group === "a") r1 += ranks[k];
|
|
560
|
-
const u1 = r1 - n1 * (n1 + 1) / 2;
|
|
561
|
-
const u2 = n1 * n2 - u1;
|
|
562
|
-
const u = Math.min(u1, u2);
|
|
563
|
-
const mu = n1 * n2 / 2;
|
|
564
|
-
const sigma = Math.sqrt(n1 * n2 * (n1 + n2 + 1) / 12);
|
|
565
|
-
if (sigma === 0) return {
|
|
656
|
+
if (method === "asymptotic") return {
|
|
566
657
|
u,
|
|
567
|
-
|
|
658
|
+
uA,
|
|
659
|
+
p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),
|
|
660
|
+
method,
|
|
661
|
+
pFloor: designFloor
|
|
568
662
|
};
|
|
663
|
+
const permutations = resolvePermutations("mannWhitneyU", opts.permutations);
|
|
664
|
+
const rng = opts.seed === void 0 ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed);
|
|
665
|
+
let atLeastAsExtreme = 0;
|
|
666
|
+
const pool = [...doubled];
|
|
667
|
+
for (let iteration = 0; iteration < permutations; iteration++) {
|
|
668
|
+
let doubledRankSum = 0;
|
|
669
|
+
for (let k = 0; k < selectedN; k++) {
|
|
670
|
+
const pick = k + Math.floor(rng() * (total - k));
|
|
671
|
+
const swapped = pool[pick];
|
|
672
|
+
pool[pick] = pool[k];
|
|
673
|
+
pool[k] = swapped;
|
|
674
|
+
doubledRankSum += swapped;
|
|
675
|
+
}
|
|
676
|
+
if (Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >= doubledDeviation) atLeastAsExtreme++;
|
|
677
|
+
}
|
|
678
|
+
const pFloor = Math.max(1 / (permutations + 1), designFloor);
|
|
569
679
|
return {
|
|
570
680
|
u,
|
|
571
|
-
|
|
681
|
+
uA,
|
|
682
|
+
p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),
|
|
683
|
+
method,
|
|
684
|
+
pFloor
|
|
572
685
|
};
|
|
573
686
|
}
|
|
574
687
|
/** Partial credit: returns 0-1 ratio of current toward target */
|
|
@@ -581,23 +694,37 @@ function partialCredit(current, target) {
|
|
|
581
694
|
* Pairing removes inter-item variance, giving tighter significance than
|
|
582
695
|
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
583
696
|
* scenarios.
|
|
697
|
+
*
|
|
698
|
+
* Returns `t = p = null` where the statistic is undefined: fewer than two
|
|
699
|
+
* pairs, or a non-zero constant delta whose observed variance is zero. A
|
|
700
|
+
* constant shift carries no information about the variance it would have to
|
|
701
|
+
* be compared against, so the honest answer is "undefined", not `p = 0` —
|
|
702
|
+
* three observations cannot buy absolute certainty. This is the same contract
|
|
703
|
+
* {@link pairedCohensDz} states for the same condition. An all-zero delta is
|
|
704
|
+
* different: it is a measured null, and returns `t = 0, p = 1`.
|
|
584
705
|
*/
|
|
585
706
|
function pairedTTest(before, after) {
|
|
586
707
|
if (before.length !== after.length) throw new ValidationError(`pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
708
|
+
assertFiniteSample("pairedTTest", "before", before);
|
|
709
|
+
assertFiniteSample("pairedTTest", "after", after);
|
|
587
710
|
const n = before.length;
|
|
588
711
|
if (n < 2) return {
|
|
589
|
-
t:
|
|
712
|
+
t: null,
|
|
590
713
|
df: 0,
|
|
591
|
-
p:
|
|
714
|
+
p: null
|
|
592
715
|
};
|
|
593
716
|
const diffs = before.map((b, i) => after[i] - b);
|
|
594
717
|
const mean = diffs.reduce((a, b) => a + b, 0) / n;
|
|
595
718
|
const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1);
|
|
596
719
|
const se = Math.sqrt(variance / n);
|
|
597
|
-
if (se === 0) return {
|
|
598
|
-
t:
|
|
720
|
+
if (se === 0) return mean === 0 ? {
|
|
721
|
+
t: 0,
|
|
722
|
+
df: n - 1,
|
|
723
|
+
p: 1
|
|
724
|
+
} : {
|
|
725
|
+
t: null,
|
|
599
726
|
df: n - 1,
|
|
600
|
-
p:
|
|
727
|
+
p: null
|
|
601
728
|
};
|
|
602
729
|
const t = mean / se;
|
|
603
730
|
const df = n - 1;
|
|
@@ -608,55 +735,101 @@ function pairedTTest(before, after) {
|
|
|
608
735
|
};
|
|
609
736
|
}
|
|
610
737
|
/**
|
|
611
|
-
* Wilcoxon signed-rank
|
|
612
|
-
*
|
|
738
|
+
* Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
|
|
739
|
+
*
|
|
740
|
+
* Exact conditional (sign-flip) p by default at `n ≤
|
|
741
|
+
* {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
|
|
742
|
+
* permutation above it. Throws on non-finite input and on `method:
|
|
743
|
+
* 'asymptotic'` where an exact answer is available.
|
|
744
|
+
*
|
|
745
|
+
* `n` is the count of NON-ZERO differences: exact ties are dropped before
|
|
746
|
+
* ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
|
|
747
|
+
* All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
|
|
748
|
+
* `pFloor` states rather than leaving `p = 1` to be read as a measured null.
|
|
613
749
|
*/
|
|
614
|
-
function wilcoxonSignedRank(before, after) {
|
|
750
|
+
function wilcoxonSignedRank(before, after, opts = {}) {
|
|
615
751
|
if (before.length !== after.length) throw new ValidationError(`wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
752
|
+
assertFiniteSample("wilcoxonSignedRank", "before", before);
|
|
753
|
+
assertFiniteSample("wilcoxonSignedRank", "after", after);
|
|
616
754
|
const diffs = before.map((b, i) => after[i] - b).filter((d) => d !== 0);
|
|
617
755
|
const n = diffs.length;
|
|
618
|
-
if (n
|
|
756
|
+
if (n === 0) return {
|
|
619
757
|
w: 0,
|
|
620
|
-
p: 1
|
|
758
|
+
p: 1,
|
|
759
|
+
method: "exact",
|
|
760
|
+
pFloor: 1,
|
|
761
|
+
nNonZero: 0
|
|
621
762
|
};
|
|
622
|
-
const
|
|
763
|
+
const order = diffs.map((d, i) => ({
|
|
623
764
|
abs: Math.abs(d),
|
|
624
|
-
sign: Math.sign(d),
|
|
625
765
|
i
|
|
626
|
-
})).sort((
|
|
766
|
+
})).sort((x, y) => x.abs - y.abs);
|
|
767
|
+
const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs));
|
|
627
768
|
const ranks = new Array(n);
|
|
628
|
-
let
|
|
629
|
-
while (i < n) {
|
|
630
|
-
let j = i;
|
|
631
|
-
while (j < n && absRanks[j].abs === absRanks[i].abs) j++;
|
|
632
|
-
const avg = (i + 1 + j) / 2;
|
|
633
|
-
for (let k = i; k < j; k++) ranks[absRanks[k].i] = avg;
|
|
634
|
-
i = j;
|
|
635
|
-
}
|
|
769
|
+
for (let k = 0; k < n; k++) ranks[order[k].i] = midranks[k];
|
|
636
770
|
let wPlus = 0;
|
|
637
771
|
for (let k = 0; k < n; k++) if (diffs[k] > 0) wPlus += ranks[k];
|
|
638
|
-
const
|
|
639
|
-
const
|
|
640
|
-
const
|
|
641
|
-
const
|
|
772
|
+
const doubled = midranks.map((rank) => Math.round(rank * 2));
|
|
773
|
+
const doubledDeviation = Math.abs(2 * wPlus - n * (n + 1) / 2);
|
|
774
|
+
const designFloor = Math.min(1, 2 ** (1 - n));
|
|
775
|
+
const method = selectRankTestMethod("wilcoxonSignedRank", opts.method ?? "auto", `n=${n} non-zero differences`, n <= 20, designFloor, `20 non-zero differences`);
|
|
776
|
+
if (method === "exact") {
|
|
777
|
+
const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation);
|
|
778
|
+
return {
|
|
779
|
+
w: wPlus,
|
|
780
|
+
p,
|
|
781
|
+
method,
|
|
782
|
+
pFloor,
|
|
783
|
+
nNonZero: n
|
|
784
|
+
};
|
|
785
|
+
}
|
|
786
|
+
if (method === "asymptotic") {
|
|
787
|
+
const variance = n * (n + 1) * (2 * n + 1) / 24 - tieTerm / 48;
|
|
788
|
+
return {
|
|
789
|
+
w: wPlus,
|
|
790
|
+
p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),
|
|
791
|
+
method,
|
|
792
|
+
pFloor: designFloor,
|
|
793
|
+
nNonZero: n
|
|
794
|
+
};
|
|
795
|
+
}
|
|
796
|
+
const permutations = resolvePermutations("wilcoxonSignedRank", opts.permutations);
|
|
797
|
+
const rng = makeRng(opts.seed, before, after);
|
|
798
|
+
const doubledCentre = n * (n + 1) / 2;
|
|
799
|
+
let atLeastAsExtreme = 0;
|
|
800
|
+
for (let iteration = 0; iteration < permutations; iteration++) {
|
|
801
|
+
let doubledWPlus = 0;
|
|
802
|
+
for (let k = 0; k < n; k++) if (rng() < .5) doubledWPlus += doubled[k];
|
|
803
|
+
if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++;
|
|
804
|
+
}
|
|
642
805
|
return {
|
|
643
806
|
w: wPlus,
|
|
644
|
-
p
|
|
807
|
+
p: (1 + atLeastAsExtreme) / (permutations + 1),
|
|
808
|
+
method,
|
|
809
|
+
pFloor: Math.max(1 / (permutations + 1), designFloor),
|
|
810
|
+
nNonZero: n
|
|
645
811
|
};
|
|
646
812
|
}
|
|
647
813
|
/**
|
|
648
814
|
* Cohen's d — standardized effect size for two independent groups.
|
|
649
815
|
* Positive d means group b has higher mean than group a.
|
|
650
816
|
* Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
|
|
817
|
+
*
|
|
818
|
+
* Returns null where the standardized effect is undefined: fewer than two
|
|
819
|
+
* observations in either group, or a zero pooled standard deviation with
|
|
820
|
+
* unequal means. Null is NOT "no effect" — zero within-group spread across a
|
|
821
|
+
* real mean gap is an unbounded effect, the opposite of negligible. Equal
|
|
822
|
+
* means with zero spread is a genuine 0. Same contract as
|
|
823
|
+
* {@link pairedCohensDz}.
|
|
651
824
|
*/
|
|
652
825
|
function cohensD(a, b) {
|
|
653
|
-
if (a.length < 2 || b.length < 2) return
|
|
826
|
+
if (a.length < 2 || b.length < 2) return null;
|
|
654
827
|
const meanA = a.reduce((x, y) => x + y, 0) / a.length;
|
|
655
828
|
const meanB = b.reduce((x, y) => x + y, 0) / b.length;
|
|
656
829
|
const varA = a.reduce((acc, x) => acc + (x - meanA) ** 2, 0) / (a.length - 1);
|
|
657
830
|
const varB = b.reduce((acc, x) => acc + (x - meanB) ** 2, 0) / (b.length - 1);
|
|
658
831
|
const pooled = Math.sqrt(((a.length - 1) * varA + (b.length - 1) * varB) / (a.length + b.length - 2));
|
|
659
|
-
if (pooled === 0) return 0;
|
|
832
|
+
if (pooled === 0) return meanB === meanA ? 0 : null;
|
|
660
833
|
return (meanB - meanA) / pooled;
|
|
661
834
|
}
|
|
662
835
|
/**
|
|
@@ -900,6 +1073,11 @@ function requiredSampleSize(opts) {
|
|
|
900
1073
|
/**
|
|
901
1074
|
* Required number of paired observations for a target Cohen's dz.
|
|
902
1075
|
* Unlike the independent-groups formula, this has no two-arm factor of two.
|
|
1076
|
+
*
|
|
1077
|
+
* Normal quantiles with no t correction, so treat the result as a LOWER bound:
|
|
1078
|
+
* it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
|
|
1079
|
+
* it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
|
|
1080
|
+
* consults to decide whether 3–10 repetitions suffice.
|
|
903
1081
|
*/
|
|
904
1082
|
function requiredPairedSampleSize(opts) {
|
|
905
1083
|
const effect = opts.effect;
|
|
@@ -968,13 +1146,21 @@ function mcnemarPower(opts) {
|
|
|
968
1146
|
const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom;
|
|
969
1147
|
return Math.min(1, Math.max(0, normalCdf(zBeta)));
|
|
970
1148
|
}
|
|
971
|
-
/**
|
|
1149
|
+
/**
|
|
1150
|
+
* Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
|
|
1151
|
+
*
|
|
1152
|
+
* Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
|
|
1153
|
+
* {@link holm}, which uniformly dominates this correction and must therefore
|
|
1154
|
+
* never reject less. Validates its inputs on the same terms.
|
|
1155
|
+
*/
|
|
972
1156
|
function bonferroni(pValues, alpha = .05) {
|
|
1157
|
+
assertAlpha("bonferroni", "alpha", alpha);
|
|
1158
|
+
assertPValues("bonferroni", pValues);
|
|
973
1159
|
const k = pValues.length;
|
|
974
1160
|
const adjusted = pValues.map((p) => Math.min(1, p * k));
|
|
975
1161
|
return {
|
|
976
1162
|
adjusted,
|
|
977
|
-
significant: adjusted.map((p) => p
|
|
1163
|
+
significant: adjusted.map((p) => p <= alpha)
|
|
978
1164
|
};
|
|
979
1165
|
}
|
|
980
1166
|
/**
|
|
@@ -986,8 +1172,8 @@ function bonferroni(pValues, alpha = .05) {
|
|
|
986
1172
|
* strong family-wise error control under arbitrary dependence.
|
|
987
1173
|
*/
|
|
988
1174
|
function holm(pValues, alpha = .05) {
|
|
989
|
-
|
|
990
|
-
|
|
1175
|
+
assertAlpha("holm", "alpha", alpha);
|
|
1176
|
+
assertPValues("holm", pValues);
|
|
991
1177
|
const count = pValues.length;
|
|
992
1178
|
if (count === 0) return {
|
|
993
1179
|
adjusted: [],
|
|
@@ -1013,8 +1199,13 @@ function holm(pValues, alpha = .05) {
|
|
|
1013
1199
|
/**
|
|
1014
1200
|
* Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
|
|
1015
1201
|
* significance at the target FDR; handles ties and preserves q monotonicity.
|
|
1202
|
+
*
|
|
1203
|
+
* Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
|
|
1204
|
+
* exactly-`fdr` q-value is a discovery.
|
|
1016
1205
|
*/
|
|
1017
1206
|
function benjaminiHochberg(pValues, fdr = .05) {
|
|
1207
|
+
assertAlpha("benjaminiHochberg", "fdr", fdr);
|
|
1208
|
+
assertPValues("benjaminiHochberg", pValues);
|
|
1018
1209
|
const n = pValues.length;
|
|
1019
1210
|
if (n === 0) return {
|
|
1020
1211
|
qValues: [],
|
|
@@ -1029,21 +1220,42 @@ function benjaminiHochberg(pValues, fdr = .05) {
|
|
|
1029
1220
|
for (let k = n - 1; k >= 0; k--) {
|
|
1030
1221
|
const rank = k + 1;
|
|
1031
1222
|
const entry = indexed[k];
|
|
1032
|
-
const raw =
|
|
1223
|
+
const raw = n / rank * entry.p;
|
|
1033
1224
|
const bounded = Math.min(minRight, raw);
|
|
1034
1225
|
minRight = bounded;
|
|
1035
1226
|
q[entry.i] = Math.min(1, bounded);
|
|
1036
1227
|
}
|
|
1037
1228
|
return {
|
|
1038
1229
|
qValues: q,
|
|
1039
|
-
significant: q.map((v) => v
|
|
1230
|
+
significant: q.map((v) => v <= fdr)
|
|
1040
1231
|
};
|
|
1041
1232
|
}
|
|
1233
|
+
function assertAlpha(fn, label, value) {
|
|
1234
|
+
if (!Number.isFinite(value) || value <= 0 || value >= 1) throw new ValidationError(`${fn}: ${label} must be in (0,1), got ${value}`);
|
|
1235
|
+
}
|
|
1236
|
+
function assertPValues(fn, pValues) {
|
|
1237
|
+
for (const [index, pValue] of pValues.entries()) if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) throw new ValidationError(`${fn}: pValues[${index}] must be in [0,1], got ${pValue}`);
|
|
1238
|
+
}
|
|
1239
|
+
/**
|
|
1240
|
+
* Pairs below which a percentile bootstrap interval is descriptive spread only.
|
|
1241
|
+
*
|
|
1242
|
+
* `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
|
|
1243
|
+
* seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
|
|
1244
|
+
* median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
|
|
1245
|
+
* three points, not an implementation error — scipy's BCa gives 16.0 % on the
|
|
1246
|
+
* same n = 3 data — so no change to the estimator moves it. Below this floor
|
|
1247
|
+
* the decision belongs to the exact sign test or exact signed-rank test.
|
|
1248
|
+
*/
|
|
1249
|
+
const BOOTSTRAP_GATE_MIN_N = 20;
|
|
1042
1250
|
/**
|
|
1043
1251
|
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
1044
|
-
* statistic (median by default); pairs are resampled with replacement.
|
|
1045
|
-
*
|
|
1046
|
-
*
|
|
1252
|
+
* statistic (median by default); pairs are resampled with replacement. Throws
|
|
1253
|
+
* on unequal sample sizes.
|
|
1254
|
+
*
|
|
1255
|
+
* `low > threshold` carries the stated confidence ONLY at `n ≥
|
|
1256
|
+
* {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
|
|
1257
|
+
* check fires under a true null several times more often than nominal, so the
|
|
1258
|
+
* interval is descriptive spread and a promotion must not turn on it.
|
|
1047
1259
|
*/
|
|
1048
1260
|
function pairedBootstrap(before, after, opts = {}) {
|
|
1049
1261
|
if (before.length !== after.length) throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
@@ -1053,6 +1265,7 @@ function pairedBootstrap(before, after, opts = {}) {
|
|
|
1053
1265
|
if (confidence <= 0 || confidence >= 1) throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`);
|
|
1054
1266
|
const n = before.length;
|
|
1055
1267
|
const deltas = before.map((b, i) => after[i] - b);
|
|
1268
|
+
const gateEligible = n >= 20;
|
|
1056
1269
|
if (n === 0) return {
|
|
1057
1270
|
n: 0,
|
|
1058
1271
|
median: 0,
|
|
@@ -1060,7 +1273,8 @@ function pairedBootstrap(before, after, opts = {}) {
|
|
|
1060
1273
|
low: 0,
|
|
1061
1274
|
high: 0,
|
|
1062
1275
|
confidence,
|
|
1063
|
-
resamples
|
|
1276
|
+
resamples,
|
|
1277
|
+
gateEligible
|
|
1064
1278
|
};
|
|
1065
1279
|
if (n === 1) {
|
|
1066
1280
|
const d = deltas[0];
|
|
@@ -1071,10 +1285,11 @@ function pairedBootstrap(before, after, opts = {}) {
|
|
|
1071
1285
|
low: d,
|
|
1072
1286
|
high: d,
|
|
1073
1287
|
confidence,
|
|
1074
|
-
resamples
|
|
1288
|
+
resamples,
|
|
1289
|
+
gateEligible
|
|
1075
1290
|
};
|
|
1076
1291
|
}
|
|
1077
|
-
const rng = makeRng(opts.seed);
|
|
1292
|
+
const rng = makeRng(opts.seed, deltas);
|
|
1078
1293
|
const samples = new Array(resamples);
|
|
1079
1294
|
for (let b = 0; b < resamples; b++) if (statistic === "mean") {
|
|
1080
1295
|
let sum = 0;
|
|
@@ -1096,7 +1311,8 @@ function pairedBootstrap(before, after, opts = {}) {
|
|
|
1096
1311
|
low: samples[lowIdx],
|
|
1097
1312
|
high: samples[Math.max(highIdx, lowIdx)],
|
|
1098
1313
|
confidence,
|
|
1099
|
-
resamples
|
|
1314
|
+
resamples,
|
|
1315
|
+
gateEligible
|
|
1100
1316
|
};
|
|
1101
1317
|
}
|
|
1102
1318
|
/**
|
|
@@ -1369,6 +1585,209 @@ function eProcess(opts = {}) {
|
|
|
1369
1585
|
}
|
|
1370
1586
|
};
|
|
1371
1587
|
}
|
|
1588
|
+
/** Every rank test refuses non-finite input. Beyond the arithmetic being
|
|
1589
|
+
* undefined, the tie-grouping scan compares values with `===`, and
|
|
1590
|
+
* `NaN === NaN` is false, so a NaN would leave the group boundary unable to
|
|
1591
|
+
* advance and spin the loop forever. */
|
|
1592
|
+
function assertFiniteSample(fn, label, xs) {
|
|
1593
|
+
for (let i = 0; i < xs.length; i++) if (!Number.isFinite(xs[i])) throw new ValidationError(`${fn}: ${label}[${i}] must be finite, got ${xs[i]}`);
|
|
1594
|
+
}
|
|
1595
|
+
/**
|
|
1596
|
+
* Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie
|
|
1597
|
+
* groups of size `t` — the correction term both asymptotic rank-test variances
|
|
1598
|
+
* need.
|
|
1599
|
+
*/
|
|
1600
|
+
function midranksWithTieTerm(sorted) {
|
|
1601
|
+
const midranks = new Array(sorted.length);
|
|
1602
|
+
let tieTerm = 0;
|
|
1603
|
+
let i = 0;
|
|
1604
|
+
while (i < sorted.length) {
|
|
1605
|
+
let j = i;
|
|
1606
|
+
while (j < sorted.length && sorted[j] === sorted[i]) j++;
|
|
1607
|
+
const average = (i + 1 + j) / 2;
|
|
1608
|
+
for (let k = i; k < j; k++) midranks[k] = average;
|
|
1609
|
+
const groupSize = j - i;
|
|
1610
|
+
if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize;
|
|
1611
|
+
i = j;
|
|
1612
|
+
}
|
|
1613
|
+
return {
|
|
1614
|
+
midranks,
|
|
1615
|
+
tieTerm
|
|
1616
|
+
};
|
|
1617
|
+
}
|
|
1618
|
+
function selectRankTestMethod(fn, request, design, exactFeasible, designFloor, threshold) {
|
|
1619
|
+
if (request === "auto") return exactFeasible ? "exact" : "permutation";
|
|
1620
|
+
if (request === "exact") {
|
|
1621
|
+
if (exactFeasible) return "exact";
|
|
1622
|
+
throw new ValidationError(`${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges to the same answer.`);
|
|
1623
|
+
}
|
|
1624
|
+
if (exactFeasible) throw new ValidationError(`${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no attainable outcome. Use method 'exact' (the default) or add repetitions past ${threshold}.`);
|
|
1625
|
+
return "asymptotic";
|
|
1626
|
+
}
|
|
1627
|
+
function resolvePermutations(fn, permutations) {
|
|
1628
|
+
if (permutations === void 0) return DEFAULT_PERMUTATIONS;
|
|
1629
|
+
if (!Number.isInteger(permutations) || permutations < 1) throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`);
|
|
1630
|
+
return permutations;
|
|
1631
|
+
}
|
|
1632
|
+
/** Two-sided normal-approximation tail with the continuity correction. */
|
|
1633
|
+
function asymptoticTwoSidedP(deviation, sigma) {
|
|
1634
|
+
if (!(sigma > 0)) return 1;
|
|
1635
|
+
return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - .5) / sigma)));
|
|
1636
|
+
}
|
|
1637
|
+
/** SD of U under the permutation null, corrected for the realised ties. The
|
|
1638
|
+
* tie term reduces (N+1) and reaches it exactly when every value is tied, so
|
|
1639
|
+
* the variance floors at 0 rather than going negative. */
|
|
1640
|
+
function twoSampleSigma(n1, n2, total, tieTerm) {
|
|
1641
|
+
if (total < 2) return 0;
|
|
1642
|
+
const variance = n1 * n2 / 12 * (total + 1 - tieTerm / (total * (total - 1)));
|
|
1643
|
+
return Math.sqrt(Math.max(0, variance));
|
|
1644
|
+
}
|
|
1645
|
+
function logChoose(n, k) {
|
|
1646
|
+
return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1);
|
|
1647
|
+
}
|
|
1648
|
+
function formatProbability(value) {
|
|
1649
|
+
return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3);
|
|
1650
|
+
}
|
|
1651
|
+
/**
|
|
1652
|
+
* Exact DP allocation and loop count for this observed rank vector.
|
|
1653
|
+
*
|
|
1654
|
+
* The smaller arm is sufficient because selecting its complement produces the
|
|
1655
|
+
* same two-sided U deviation while using fewer rows in the state table.
|
|
1656
|
+
*/
|
|
1657
|
+
function exactTwoSampleCost(doubledRanks, selectedN) {
|
|
1658
|
+
const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
|
|
1659
|
+
let work = 0;
|
|
1660
|
+
for (let placed = 0; placed < doubledRanks.length; placed++) work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed] + 1);
|
|
1661
|
+
return {
|
|
1662
|
+
states: (selectedN + 1) * (maxSum + 1),
|
|
1663
|
+
work
|
|
1664
|
+
};
|
|
1665
|
+
}
|
|
1666
|
+
/**
|
|
1667
|
+
* Smallest attainable two-sided p under the observed ties.
|
|
1668
|
+
*
|
|
1669
|
+
* Only subsets with the minimum or maximum rank sum can attain the largest
|
|
1670
|
+
* deviation. Their multiplicity is the number of ways to choose within the
|
|
1671
|
+
* tie group at each boundary, so this calculation is exact without allocating
|
|
1672
|
+
* the full null distribution.
|
|
1673
|
+
*/
|
|
1674
|
+
function exactTwoSampleFloor(doubledRanks, selectedN) {
|
|
1675
|
+
const total = doubledRanks.length;
|
|
1676
|
+
const otherN = total - selectedN;
|
|
1677
|
+
const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0);
|
|
1678
|
+
const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0);
|
|
1679
|
+
if (minimumSum === maximumSum) return 1;
|
|
1680
|
+
const centre = selectedN * (selectedN + 1) + selectedN * otherN;
|
|
1681
|
+
const minimumDeviation = Math.abs(minimumSum - centre);
|
|
1682
|
+
const maximumDeviation = Math.abs(maximumSum - centre);
|
|
1683
|
+
const totalLogWays = logChoose(total, selectedN);
|
|
1684
|
+
const minimumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "minimum") - totalLogWays);
|
|
1685
|
+
const maximumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "maximum") - totalLogWays);
|
|
1686
|
+
if (minimumDeviation > maximumDeviation) return minimumMass;
|
|
1687
|
+
if (maximumDeviation > minimumDeviation) return maximumMass;
|
|
1688
|
+
return Math.min(1, minimumMass + maximumMass);
|
|
1689
|
+
}
|
|
1690
|
+
function logExtremeSubsetWays(sortedRanks, selectedN, side) {
|
|
1691
|
+
const boundaryIndex = side === "minimum" ? selectedN - 1 : sortedRanks.length - selectedN;
|
|
1692
|
+
const boundary = sortedRanks[boundaryIndex];
|
|
1693
|
+
let first = boundaryIndex;
|
|
1694
|
+
let afterLast = boundaryIndex + 1;
|
|
1695
|
+
while (first > 0 && sortedRanks[first - 1] === boundary) first--;
|
|
1696
|
+
while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++;
|
|
1697
|
+
return logChoose(afterLast - first, selectedN - (side === "minimum" ? first : sortedRanks.length - afterLast));
|
|
1698
|
+
}
|
|
1699
|
+
/**
|
|
1700
|
+
* Exact conditional two-sided p for the two-sample rank test.
|
|
1701
|
+
*
|
|
1702
|
+
* Convolves the observed doubled midranks into the null distribution of group
|
|
1703
|
+
* a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the
|
|
1704
|
+
* splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the
|
|
1705
|
+
* realised multiset makes the tie handling exact rather than a correction.
|
|
1706
|
+
*
|
|
1707
|
+
* The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −
|
|
1708
|
+
* U` and permutes the split set onto itself), so the two-sided p is the mass
|
|
1709
|
+
* at least as far from the centre as the observation.
|
|
1710
|
+
*/
|
|
1711
|
+
function exactTwoSampleP(doubledRanks, n1, n2, doubledDeviation) {
|
|
1712
|
+
const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
|
|
1713
|
+
const width = maxSum + 1;
|
|
1714
|
+
const ways = Array.from({ length: n1 + 1 }, () => new Float64Array(width));
|
|
1715
|
+
ways[0][0] = 1;
|
|
1716
|
+
let placed = 0;
|
|
1717
|
+
for (const rank of doubledRanks) {
|
|
1718
|
+
for (let k = Math.min(n1, placed + 1); k >= 1; k--) {
|
|
1719
|
+
const from = ways[k - 1];
|
|
1720
|
+
const into = ways[k];
|
|
1721
|
+
for (let sum = maxSum - rank; sum >= 0; sum--) {
|
|
1722
|
+
const count = from[sum];
|
|
1723
|
+
if (count !== 0) into[sum + rank] += count;
|
|
1724
|
+
}
|
|
1725
|
+
}
|
|
1726
|
+
placed++;
|
|
1727
|
+
}
|
|
1728
|
+
const shift = n1 * (n1 + 1) + n1 * n2;
|
|
1729
|
+
const chosen = ways[n1];
|
|
1730
|
+
let totalWays = 0;
|
|
1731
|
+
let extremeWays = 0;
|
|
1732
|
+
let tailWays = 0;
|
|
1733
|
+
let maxDeviation = -1;
|
|
1734
|
+
for (let sum = 0; sum < width; sum++) {
|
|
1735
|
+
const count = chosen[sum];
|
|
1736
|
+
if (count === 0) continue;
|
|
1737
|
+
totalWays += count;
|
|
1738
|
+
const deviation = Math.abs(sum - shift);
|
|
1739
|
+
if (deviation >= doubledDeviation) tailWays += count;
|
|
1740
|
+
if (deviation > maxDeviation) {
|
|
1741
|
+
maxDeviation = deviation;
|
|
1742
|
+
extremeWays = count;
|
|
1743
|
+
} else if (deviation === maxDeviation) extremeWays += count;
|
|
1744
|
+
}
|
|
1745
|
+
return {
|
|
1746
|
+
p: tailWays / totalWays,
|
|
1747
|
+
pFloor: extremeWays / totalWays
|
|
1748
|
+
};
|
|
1749
|
+
}
|
|
1750
|
+
/**
|
|
1751
|
+
* Exact conditional two-sided p for the paired signed-rank test.
|
|
1752
|
+
*
|
|
1753
|
+
* Convolves the observed doubled absolute midranks over all `2ⁿ` sign
|
|
1754
|
+
* assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the
|
|
1755
|
+
* arithmetic. The null is symmetric about `n(n+1)/4`.
|
|
1756
|
+
*/
|
|
1757
|
+
function exactSignedRankP(doubledRanks, doubledDeviation) {
|
|
1758
|
+
const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
|
|
1759
|
+
const width = maxSum + 1;
|
|
1760
|
+
let mass = new Float64Array(width);
|
|
1761
|
+
mass[0] = 1;
|
|
1762
|
+
for (const rank of doubledRanks) {
|
|
1763
|
+
const next = new Float64Array(width);
|
|
1764
|
+
for (let sum = 0; sum < width; sum++) {
|
|
1765
|
+
const probability = mass[sum];
|
|
1766
|
+
if (probability === 0) continue;
|
|
1767
|
+
next[sum] += probability * .5;
|
|
1768
|
+
next[sum + rank] += probability * .5;
|
|
1769
|
+
}
|
|
1770
|
+
mass = next;
|
|
1771
|
+
}
|
|
1772
|
+
const centre = maxSum / 2;
|
|
1773
|
+
let tail = 0;
|
|
1774
|
+
let extreme = 0;
|
|
1775
|
+
let maxDeviation = -1;
|
|
1776
|
+
for (let sum = 0; sum < width; sum++) {
|
|
1777
|
+
const probability = mass[sum];
|
|
1778
|
+
if (probability === 0) continue;
|
|
1779
|
+
const deviation = Math.abs(sum - centre);
|
|
1780
|
+
if (deviation >= doubledDeviation) tail += probability;
|
|
1781
|
+
if (deviation > maxDeviation) {
|
|
1782
|
+
maxDeviation = deviation;
|
|
1783
|
+
extreme = probability;
|
|
1784
|
+
} else if (deviation === maxDeviation) extreme += probability;
|
|
1785
|
+
}
|
|
1786
|
+
return {
|
|
1787
|
+
p: Math.min(1, tail),
|
|
1788
|
+
pFloor: Math.min(1, extreme)
|
|
1789
|
+
};
|
|
1790
|
+
}
|
|
1372
1791
|
/** Standard-normal inverse CDF (Acklam approximation). */
|
|
1373
1792
|
function zQuantile(p) {
|
|
1374
1793
|
if (p <= 0 || p >= 1) {
|
|
@@ -1427,16 +1846,45 @@ function medianInPlace(xs) {
|
|
|
1427
1846
|
const mid = Math.floor(xs.length / 2);
|
|
1428
1847
|
return xs.length % 2 === 0 ? (xs[mid - 1] + xs[mid]) / 2 : xs[mid];
|
|
1429
1848
|
}
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1849
|
+
/**
|
|
1850
|
+
* PRNG for every resampling path in this module.
|
|
1851
|
+
*
|
|
1852
|
+
* With no caller seed the seed is DERIVED FROM THE DATA rather than taken from
|
|
1853
|
+
* `Math.random`, so re-running the same input reproduces the same interval —
|
|
1854
|
+
* a gate verdict that cannot be re-derived is not evidence. Distinct data
|
|
1855
|
+
* still gets a distinct stream. Same pattern as `promotion-gate.ts`.
|
|
1856
|
+
*/
|
|
1857
|
+
function makeRng(seed, ...series) {
|
|
1858
|
+
return mulberry32(seed ?? seedFromData(series));
|
|
1859
|
+
}
|
|
1860
|
+
/** FNV-1a over the IEEE-754 bytes of every observation. */
|
|
1861
|
+
function seedFromData(series) {
|
|
1862
|
+
const view = /* @__PURE__ */ new DataView(/* @__PURE__ */ new ArrayBuffer(8));
|
|
1863
|
+
let hash = 2166136261;
|
|
1864
|
+
for (const xs of series) {
|
|
1865
|
+
for (const x of xs) {
|
|
1866
|
+
view.setFloat64(0, x);
|
|
1867
|
+
for (let byte = 0; byte < 8; byte++) hash = Math.imul(hash ^ view.getUint8(byte), 16777619);
|
|
1868
|
+
}
|
|
1869
|
+
hash = Math.imul(hash ^ 255, 16777619);
|
|
1870
|
+
}
|
|
1871
|
+
return hash | 0;
|
|
1872
|
+
}
|
|
1873
|
+
/** Order-independent seed for a symmetric two-sample statistic. */
|
|
1874
|
+
function symmetricTwoSampleSeed(a, b) {
|
|
1875
|
+
const sortedA = [...a].sort((left, right) => left - right);
|
|
1876
|
+
const sortedB = [...b].sort((left, right) => left - right);
|
|
1877
|
+
const forward = seedFromData([sortedA, sortedB]) >>> 0;
|
|
1878
|
+
const reversed = seedFromData([sortedB, sortedA]) >>> 0;
|
|
1879
|
+
return Math.min(forward, reversed);
|
|
1433
1880
|
}
|
|
1434
1881
|
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
1435
1882
|
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
1436
|
-
* share ONE PRNG implementation
|
|
1437
|
-
*
|
|
1883
|
+
* share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
|
|
1884
|
+
* stream, including 0. */
|
|
1438
1885
|
function mulberry32(seed) {
|
|
1439
|
-
|
|
1886
|
+
if (!Number.isFinite(seed)) throw new ValidationError(`mulberry32: seed must be a finite number, got ${seed}`);
|
|
1887
|
+
let s = seed | 0;
|
|
1440
1888
|
return () => {
|
|
1441
1889
|
s = s + 1831565813 | 0;
|
|
1442
1890
|
let t = s;
|
|
@@ -1446,6 +1894,6 @@ function mulberry32(seed) {
|
|
|
1446
1894
|
};
|
|
1447
1895
|
}
|
|
1448
1896
|
//#endregion
|
|
1449
|
-
export {
|
|
1897
|
+
export { passAtK as A, studentTCdf as B, pairedBootstrap as C, pairedSignTest as D, pairedRiskDifference as E, spearmanR as F, continuousAgreement as G, normalCdf as H, weightedComposite as I, verbosityBias as J, positionalBias as K, weightedMean as L, ranks as M, requiredPairedSampleSize as N, pairedTTest as O, requiredSampleSize as P, wilcoxonSignedRank as R, normalizeScores as S, pairedMde as T, calibrateJudge as U, studentTQuantile as V, calibrateJudgeContinuous as W, mannWhitneyU as _, WILCOXON_EXACT_MAX_N as a, mcnemarRequiredN as b, cliffsDelta as c, corpusInterRaterAgreement as d, corpusInterRaterAgreementFromJudgeScores as f, interpretCliffs as g, interRaterReliability as h, MANN_WHITNEY_EXACT_MAX_WORK as i, pearsonR as j, partialCredit as k, cohensD as l, holm as m, DEFAULT_PERMUTATIONS as n, benjaminiHochberg as o, eProcess as p, selfPreference as q, MANN_WHITNEY_EXACT_MAX_STATES as r, bonferroni as s, BOOTSTRAP_GATE_MIN_N as t, confidenceInterval as u, mcnemar as v, pairedCohensDz as w, mulberry32 as x, mcnemarPower as y, wilson as z };
|
|
1450
1898
|
|
|
1451
|
-
//# sourceMappingURL=statistics-
|
|
1899
|
+
//# sourceMappingURL=statistics-RwRNu2__.js.map
|