@tangle-network/agent-eval 0.133.1 → 0.133.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/CHANGELOG.md +165 -0
  2. package/dist/{analyze-runs-DZr7JW-m.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
  3. package/dist/{analyze-runs-DZr7JW-m.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  5. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  6. package/dist/baseline-BaPxoROc.js +149 -0
  7. package/dist/baseline-BaPxoROc.js.map +1 -0
  8. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  9. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +1 -1
  12. package/dist/{benchmarks-BU7P6PCW.js → benchmarks-BP9sgMia.js} +3 -3
  13. package/dist/{benchmarks-BU7P6PCW.js.map → benchmarks-BP9sgMia.js.map} +1 -1
  14. package/dist/builder-eval/index.js +1 -1
  15. package/dist/campaign/index.d.ts +2 -2
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-CnzHQndg.js → campaign--V4ffEKR.js} +12 -6
  18. package/dist/{campaign-CnzHQndg.js.map → campaign--V4ffEKR.js.map} +1 -1
  19. package/dist/{client-D4F9hdzR.d.ts → client-Du7B81wW.d.ts} +28 -14
  20. package/dist/client-Du7B81wW.d.ts.map +1 -0
  21. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  22. package/dist/client-LIuo-KPv.js.map +1 -0
  23. package/dist/contract/index.d.ts +3 -3
  24. package/dist/contract/index.d.ts.map +1 -1
  25. package/dist/contract/index.js +9 -8
  26. package/dist/contract/index.js.map +1 -1
  27. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  28. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/hosted/index.d.ts.map +1 -1
  31. package/dist/hosted/index.js +1 -1
  32. package/dist/index-3cdlURSk.d.ts.map +1 -1
  33. package/dist/{index-Wek5mU0y.d.ts → index-B5MNN1f1.d.ts} +3 -3
  34. package/dist/{index-Wek5mU0y.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
  35. package/dist/{index-Ba636PKl.d.ts → index-DOqvIJ8I.d.ts} +27 -10
  36. package/dist/index-DOqvIJ8I.d.ts.map +1 -0
  37. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  38. package/dist/index-DSC51roc2.d.ts.map +1 -0
  39. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  40. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  41. package/dist/index.d.ts +56 -10
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +39 -22
  44. package/dist/index.js.map +1 -1
  45. package/dist/ledger-core/index.d.ts +2 -2
  46. package/dist/ledger-core/index.js +2 -2
  47. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  48. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  49. package/dist/matrix/index.d.ts +1 -1
  50. package/dist/meta-eval/index.d.ts +1 -1
  51. package/dist/meta-eval/index.js +2 -2
  52. package/dist/multishot/index.d.ts +1 -1
  53. package/dist/openapi.json +1 -1
  54. package/dist/{opencode-sqlite-BGrHeDu3.js → opencode-sqlite-8r6WUfHc.js} +2 -3
  55. package/dist/opencode-sqlite-8r6WUfHc.js.map +1 -0
  56. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  57. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  58. package/dist/pipelines/index.d.ts +1 -1
  59. package/dist/pipelines/index.js +3 -2
  60. package/dist/pipelines/index.js.map +1 -1
  61. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  62. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  63. package/dist/{release-report-DfmKSIEE.d.ts → release-report-DKBtegGt.d.ts} +2 -2
  64. package/dist/{release-report-DfmKSIEE.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
  65. package/dist/reporting.d.ts +3 -3
  66. package/dist/reporting.js +4 -4
  67. package/dist/{researcher-DMimgHtN.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
  68. package/dist/{researcher-DMimgHtN.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
  69. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  70. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  71. package/dist/rl.d.ts +1 -1
  72. package/dist/rl.js +4 -4
  73. package/dist/rollout/index.js +2 -2
  74. package/dist/{rollout-CeTlDrf6.js → rollout-CreDz__7.js} +2 -2
  75. package/dist/{rollout-CeTlDrf6.js.map → rollout-CreDz__7.js.map} +1 -1
  76. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  77. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  78. package/dist/{skillopt-optimization-method-CAASpcS3.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
  79. package/dist/{skillopt-optimization-method-CAASpcS3.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
  80. package/dist/{skillopt-optimization-method-BoIzh7Dl.js → skillopt-optimization-method-vvJ4bMNI.js} +123 -24
  81. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
  82. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  83. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  84. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  85. package/dist/statistics-RwRNu2__.js.map +1 -0
  86. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  87. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  88. package/dist/{summary-report-DnUcjVpV.d.ts → summary-report-DyOhItws.d.ts} +4 -3
  89. package/dist/summary-report-DyOhItws.d.ts.map +1 -0
  90. package/dist/supervisor-run/index.js +1 -1
  91. package/dist/{supervisor-run-_lnTLM3z.js → supervisor-run-B7lUGoyZ.js} +2 -2
  92. package/dist/{supervisor-run-_lnTLM3z.js.map → supervisor-run-B7lUGoyZ.js.map} +1 -1
  93. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  94. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  95. package/docs/design/statistics-decisions.md +271 -0
  96. package/docs/design.md +1 -0
  97. package/docs/insight-report.md +1 -1
  98. package/docs/research-report-methodology.md +4 -1
  99. package/package.json +2 -1
  100. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  101. package/dist/baseline-DcX5hQDv.js.map +0 -1
  102. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  103. package/dist/client-CYzbdJOZ.js.map +0 -1
  104. package/dist/client-D4F9hdzR.d.ts.map +0 -1
  105. package/dist/index-Ba636PKl.d.ts.map +0 -1
  106. package/dist/index-DSC51roc.d.ts.map +0 -1
  107. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  108. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  109. package/dist/opencode-sqlite-BGrHeDu3.js.map +0 -1
  110. package/dist/skillopt-optimization-method-BoIzh7Dl.js.map +0 -1
  111. package/dist/statistics-DWM_AyLe.js.map +0 -1
  112. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  113. package/dist/summary-report-DnUcjVpV.d.ts.map +0 -1
@@ -402,14 +402,26 @@ function lnGamma(z) {
402
402
  const t = z + g + .5;
403
403
  return .5 * Math.log(2 * Math.PI) + (z + .5) * Math.log(t) - t + Math.log(x);
404
404
  }
405
- /** Regularized incomplete beta function via a Lentz continued fraction. */
405
+ /**
406
+ * Regularized incomplete beta function I_x(a, b).
407
+ *
408
+ * The Lentz continued fraction converges only for `x < (a+1)/(a+b+2)`; outside
409
+ * that domain it must be reached through the symmetry `I_x(a,b) = 1 −
410
+ * I_{1−x}(b,a)`. `studentTCdf` drives `x → 1` as `|t| → 0`, so the mirrored
411
+ * branch is the one every near-null t-statistic takes.
412
+ */
406
413
  function regularizedIncompleteBeta(x, a, b) {
407
414
  if (x <= 0) return 0;
408
415
  if (x >= 1) return 1;
409
416
  const logBeta = lnGamma(a) + lnGamma(b) - lnGamma(a + b);
410
- const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - logBeta) / a;
411
- const maxIterations = 200;
412
- const epsilon = 3e-7;
417
+ const front = Math.exp(Math.log(x) * a + Math.log(1 - x) * b - logBeta);
418
+ if (x < (a + 1) / (a + b + 2)) return front * betaContinuedFraction(x, a, b) / a;
419
+ return 1 - front * betaContinuedFraction(1 - x, b, a) / b;
420
+ }
421
+ /** Modified Lentz evaluation of the beta continued fraction at `x`. */
422
+ function betaContinuedFraction(x, a, b) {
423
+ const maxIterations = 300;
424
+ const epsilon = 3e-15;
413
425
  let c = 1;
414
426
  let d = 1 - (a + b) * x / (a + 1);
415
427
  if (Math.abs(d) < 1e-30) d = 1e-30;
@@ -434,17 +446,42 @@ function regularizedIncompleteBeta(x, a, b) {
434
446
  fraction *= delta;
435
447
  if (Math.abs(delta - 1) < epsilon) break;
436
448
  }
437
- return front * fraction;
449
+ return fraction;
438
450
  }
439
451
  //#endregion
440
452
  //#region src/math/student-t.ts
441
- /** Student-t CDF via the regularized incomplete beta function. */
453
+ /**
454
+ * Student-t CDF via the regularized incomplete beta function.
455
+ */
442
456
  function studentTCdf(t, degreesOfFreedom) {
443
457
  if (degreesOfFreedom <= 0) return .5;
444
- if (degreesOfFreedom > 100) return normalCdf(t);
445
458
  const beta = regularizedIncompleteBeta(degreesOfFreedom / (degreesOfFreedom + t * t), degreesOfFreedom / 2, .5);
446
459
  return t >= 0 ? 1 - .5 * beta : .5 * beta;
447
460
  }
461
+ /**
462
+ * Inverse Student-t CDF, solved against {@link studentTCdf}.
463
+ *
464
+ * The CDF is monotone, so bracket expansion followed by bisection is stable
465
+ * across fractional degrees of freedom and does not need a separate
466
+ * approximation with a different error profile.
467
+ */
468
+ function studentTQuantile(probability, degreesOfFreedom) {
469
+ if (!Number.isFinite(probability) || probability < 0 || probability > 1) throw new RangeError(`studentTQuantile: probability must be in [0,1], got ${probability}`);
470
+ if (!Number.isFinite(degreesOfFreedom) || degreesOfFreedom <= 0) throw new RangeError(`studentTQuantile: degreesOfFreedom must be positive and finite, got ${degreesOfFreedom}`);
471
+ if (probability === 0) return Number.NEGATIVE_INFINITY;
472
+ if (probability === 1) return Number.POSITIVE_INFINITY;
473
+ if (probability === .5) return 0;
474
+ if (probability < .5) return -studentTQuantile(1 - probability, degreesOfFreedom);
475
+ let low = 0;
476
+ let high = 1;
477
+ while (studentTCdf(high, degreesOfFreedom) < probability) high *= 2;
478
+ for (let iteration = 0; iteration < 64; iteration++) {
479
+ const middle = (low + high) / 2;
480
+ if (studentTCdf(middle, degreesOfFreedom) < probability) low = middle;
481
+ else high = middle;
482
+ }
483
+ return (low + high) / 2;
484
+ }
448
485
  //#endregion
449
486
  //#region src/statistics.ts
450
487
  /** Identity: dimensions already follow "higher = better" by prompt convention
@@ -462,7 +499,14 @@ function weightedMean(scores) {
462
499
  }
463
500
  return totalWeight > 0 ? weightedSum / totalWeight : 0;
464
501
  }
465
- /** Bootstrap confidence interval */
502
+ /**
503
+ * Percentile bootstrap confidence interval on the mean of `scores`.
504
+ *
505
+ * Descriptive spread. It is not a significance test, and at small n its bounds
506
+ * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
507
+ * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
508
+ * the scores themselves, so the interval is reproducible either way.
509
+ */
466
510
  function confidenceInterval(scores, confidence = .95, opts = {}) {
467
511
  if (scores.length === 0) return {
468
512
  mean: 0,
@@ -477,7 +521,7 @@ function confidenceInterval(scores, confidence = .95, opts = {}) {
477
521
  const n = scores.length;
478
522
  const mean = scores.reduce((a, b) => a + b, 0) / n;
479
523
  const B = opts.resamples ?? 1e3;
480
- const rng = makeRng(opts.seed);
524
+ const rng = makeRng(opts.seed, scores);
481
525
  const bootstrapMeans = [];
482
526
  for (let i = 0; i < B; i++) {
483
527
  let sum = 0;
@@ -495,26 +539,45 @@ function confidenceInterval(scores, confidence = .95, opts = {}) {
495
539
  };
496
540
  }
497
541
  /**
498
- * Inter-rater reliability — simplified Krippendorff's alpha.
542
+ * Inter-rater reliability — Krippendorff's α under the squared-difference
543
+ * metric, pooled across dimensions.
544
+ *
545
+ * Each inner array is one judge's scores. Items are matched by position
546
+ * WITHIN a dimension: the k-th score a judge supplies carrying dimension
547
+ * `d` is item k of `d`, and the ratings compared against each other are
548
+ * the ones different judges gave to the same item. Every judge that scores
549
+ * a dimension at all must supply the same number of scores for it —
550
+ * ragged input cannot be aligned into items and throws rather than
551
+ * comparing mismatched items.
499
552
  *
500
- * Each inner array is one judge's scores for all items.
501
- * All arrays must have the same length (same items scored).
553
+ * α = 1 D_observed / D_expected: D_observed averages the squared
554
+ * difference over within-item judge pairs, D_expected over every pair of
555
+ * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,
556
+ * negative is systematic disagreement.
502
557
  */
503
558
  function interRaterReliability(judgeScores) {
504
559
  if (judgeScores.length < 2) return 1;
505
- const dimensionMap = /* @__PURE__ */ new Map();
506
- for (const judgeSet of judgeScores) for (const s of judgeSet) {
507
- if (!dimensionMap.has(s.dimension)) dimensionMap.set(s.dimension, []);
508
- const arr = dimensionMap.get(s.dimension);
509
- if (arr.length === 0 || arr[arr.length - 1].length >= judgeScores.length) arr.push([s.score]);
510
- else arr[arr.length - 1].push(s.score);
560
+ const perDimension = /* @__PURE__ */ new Map();
561
+ for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) for (const s of judgeScores[judgeIndex]) {
562
+ let byJudge = perDimension.get(s.dimension);
563
+ if (byJudge === void 0) {
564
+ byJudge = Array.from({ length: judgeScores.length }, () => []);
565
+ perDimension.set(s.dimension, byJudge);
566
+ }
567
+ byJudge[judgeIndex].push(s.score);
511
568
  }
512
569
  const allValues = [];
513
570
  const pairDiffs = [];
514
- for (const items of dimensionMap.values()) for (const ratings of items) {
515
- if (ratings.length < 2) continue;
516
- for (const v of ratings) allValues.push(v);
517
- for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
571
+ for (const [dimension, byJudge] of perDimension) {
572
+ const scoring = byJudge.filter((scores) => scores.length > 0);
573
+ if (scoring.length < 2) continue;
574
+ const itemCount = scoring[0].length;
575
+ if (scoring.some((scores) => scores.length !== itemCount)) throw new ValidationError(`interRaterReliability: dimension '${dimension}' has judges supplying ${scoring.map((scores) => scores.length).join("/")} scores — items cannot be aligned`);
576
+ for (let item = 0; item < itemCount; item++) {
577
+ const ratings = scoring.map((scores) => scores[item]);
578
+ for (const v of ratings) allValues.push(v);
579
+ for (let i = 0; i < ratings.length; i++) for (let j = i + 1; j < ratings.length; j++) pairDiffs.push((ratings[i] - ratings[j]) ** 2);
580
+ }
518
581
  }
519
582
  if (pairDiffs.length === 0 || allValues.length < 2) return 1;
520
583
  const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length;
@@ -528,47 +591,97 @@ function interRaterReliability(judgeScores) {
528
591
  if (expectedDisagreement === 0) return 1;
529
592
  return 1 - observedDisagreement / expectedDisagreement;
530
593
  }
594
+ /** Maximum dynamic-programming cells used by an exact two-sample rank test. */
595
+ const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
596
+ /** Maximum inner-loop transitions used by an exact two-sample rank test. */
597
+ const MANN_WHITNEY_EXACT_MAX_WORK = 25e4;
598
+ /** Non-zero differences up to which the signed-rank null is enumerated exactly. */
599
+ const WILCOXON_EXACT_MAX_N = 20;
600
+ /** Resamples used when a rank test falls back to Monte Carlo permutation. */
601
+ const DEFAULT_PERMUTATIONS = 1e5;
531
602
  /**
532
- * Mann-Whitney U test for comparing two independent groups.
533
- * Returns U statistic and approximate p-value (normal approximation).
603
+ * Mann-Whitney U two independent samples, no distributional assumption.
604
+ *
605
+ * Exact conditional (permutation) p by default when the dynamic program fits
606
+ * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
607
+ * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
608
+ * permutation above those limits. This keeps imbalanced designs such as 1+24
609
+ * exact without admitting expensive balanced designs merely because they have
610
+ * the same total size. Throws on non-finite input and on `method:
611
+ * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
612
+ * pFloor = 1` — no design, no attainable evidence.
534
613
  */
535
- function mannWhitneyU(a, b) {
536
- if (a.length === 0 || b.length === 0) return {
537
- u: 0,
538
- p: 1
539
- };
614
+ function mannWhitneyU(a, b, opts = {}) {
615
+ assertFiniteSample("mannWhitneyU", "a", a);
616
+ assertFiniteSample("mannWhitneyU", "b", b);
540
617
  const n1 = a.length;
541
618
  const n2 = b.length;
619
+ if (n1 === 0 || n2 === 0) return {
620
+ u: 0,
621
+ uA: 0,
622
+ p: 1,
623
+ method: "exact",
624
+ pFloor: 1
625
+ };
626
+ const total = n1 + n2;
542
627
  const combined = [...a.map((v) => ({
543
628
  v,
544
- group: "a"
629
+ fromA: true
545
630
  })), ...b.map((v) => ({
546
631
  v,
547
- group: "b"
632
+ fromA: false
548
633
  }))].sort((x, y) => x.v - y.v);
549
- const ranks = new Array(combined.length);
550
- let i = 0;
551
- while (i < combined.length) {
552
- let j = i;
553
- while (j < combined.length && combined[j].v === combined[i].v) j++;
554
- const avgRank = (i + 1 + j) / 2;
555
- for (let k = i; k < j; k++) ranks[k] = avgRank;
556
- i = j;
634
+ const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v));
635
+ let rankSumA = 0;
636
+ for (let k = 0; k < total; k++) if (combined[k].fromA) rankSumA += midranks[k];
637
+ const uA = rankSumA - n1 * (n1 + 1) / 2;
638
+ const u = Math.min(uA, n1 * n2 - uA);
639
+ const doubled = midranks.map((rank) => Math.round(rank * 2));
640
+ const doubledDeviation = Math.abs(2 * uA - n1 * n2);
641
+ const selectedN = Math.min(n1, n2);
642
+ const otherN = total - selectedN;
643
+ const exactCost = exactTwoSampleCost(doubled, selectedN);
644
+ const designFloor = exactTwoSampleFloor(doubled, selectedN);
645
+ const method = selectRankTestMethod("mannWhitneyU", opts.method ?? "auto", `n1=${n1}, n2=${n2}`, exactCost.states <= 8192 && exactCost.work <= 25e4, designFloor, `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString("en-US")} states and ${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString("en-US")} transitions`);
646
+ if (method === "exact") {
647
+ const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation);
648
+ return {
649
+ u,
650
+ uA,
651
+ p,
652
+ method,
653
+ pFloor
654
+ };
557
655
  }
558
- let r1 = 0;
559
- for (let k = 0; k < combined.length; k++) if (combined[k].group === "a") r1 += ranks[k];
560
- const u1 = r1 - n1 * (n1 + 1) / 2;
561
- const u2 = n1 * n2 - u1;
562
- const u = Math.min(u1, u2);
563
- const mu = n1 * n2 / 2;
564
- const sigma = Math.sqrt(n1 * n2 * (n1 + n2 + 1) / 12);
565
- if (sigma === 0) return {
656
+ if (method === "asymptotic") return {
566
657
  u,
567
- p: 1
658
+ uA,
659
+ p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),
660
+ method,
661
+ pFloor: designFloor
568
662
  };
663
+ const permutations = resolvePermutations("mannWhitneyU", opts.permutations);
664
+ const rng = opts.seed === void 0 ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed);
665
+ let atLeastAsExtreme = 0;
666
+ const pool = [...doubled];
667
+ for (let iteration = 0; iteration < permutations; iteration++) {
668
+ let doubledRankSum = 0;
669
+ for (let k = 0; k < selectedN; k++) {
670
+ const pick = k + Math.floor(rng() * (total - k));
671
+ const swapped = pool[pick];
672
+ pool[pick] = pool[k];
673
+ pool[k] = swapped;
674
+ doubledRankSum += swapped;
675
+ }
676
+ if (Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >= doubledDeviation) atLeastAsExtreme++;
677
+ }
678
+ const pFloor = Math.max(1 / (permutations + 1), designFloor);
569
679
  return {
570
680
  u,
571
- p: 2 * (1 - normalCdf(Math.abs(u - mu) / sigma))
681
+ uA,
682
+ p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),
683
+ method,
684
+ pFloor
572
685
  };
573
686
  }
574
687
  /** Partial credit: returns 0-1 ratio of current toward target */
@@ -581,23 +694,37 @@ function partialCredit(current, target) {
581
694
  * Pairing removes inter-item variance, giving tighter significance than
582
695
  * an unpaired test when comparing prompt v1 vs prompt v2 on identical
583
696
  * scenarios.
697
+ *
698
+ * Returns `t = p = null` where the statistic is undefined: fewer than two
699
+ * pairs, or a non-zero constant delta whose observed variance is zero. A
700
+ * constant shift carries no information about the variance it would have to
701
+ * be compared against, so the honest answer is "undefined", not `p = 0` —
702
+ * three observations cannot buy absolute certainty. This is the same contract
703
+ * {@link pairedCohensDz} states for the same condition. An all-zero delta is
704
+ * different: it is a measured null, and returns `t = 0, p = 1`.
584
705
  */
585
706
  function pairedTTest(before, after) {
586
707
  if (before.length !== after.length) throw new ValidationError(`pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`);
708
+ assertFiniteSample("pairedTTest", "before", before);
709
+ assertFiniteSample("pairedTTest", "after", after);
587
710
  const n = before.length;
588
711
  if (n < 2) return {
589
- t: 0,
712
+ t: null,
590
713
  df: 0,
591
- p: 1
714
+ p: null
592
715
  };
593
716
  const diffs = before.map((b, i) => after[i] - b);
594
717
  const mean = diffs.reduce((a, b) => a + b, 0) / n;
595
718
  const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1);
596
719
  const se = Math.sqrt(variance / n);
597
- if (se === 0) return {
598
- t: mean === 0 ? 0 : Infinity,
720
+ if (se === 0) return mean === 0 ? {
721
+ t: 0,
722
+ df: n - 1,
723
+ p: 1
724
+ } : {
725
+ t: null,
599
726
  df: n - 1,
600
- p: mean === 0 ? 1 : 0
727
+ p: null
601
728
  };
602
729
  const t = mean / se;
603
730
  const df = n - 1;
@@ -608,55 +735,101 @@ function pairedTTest(before, after) {
608
735
  };
609
736
  }
610
737
  /**
611
- * Wilcoxon signed-rank test — paired non-parametric alternative.
612
- * Use when the differences aren't normally distributed.
738
+ * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
739
+ *
740
+ * Exact conditional (sign-flip) p by default at `n ≤
741
+ * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
742
+ * permutation above it. Throws on non-finite input and on `method:
743
+ * 'asymptotic'` where an exact answer is available.
744
+ *
745
+ * `n` is the count of NON-ZERO differences: exact ties are dropped before
746
+ * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
747
+ * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
748
+ * `pFloor` states rather than leaving `p = 1` to be read as a measured null.
613
749
  */
614
- function wilcoxonSignedRank(before, after) {
750
+ function wilcoxonSignedRank(before, after, opts = {}) {
615
751
  if (before.length !== after.length) throw new ValidationError(`wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`);
752
+ assertFiniteSample("wilcoxonSignedRank", "before", before);
753
+ assertFiniteSample("wilcoxonSignedRank", "after", after);
616
754
  const diffs = before.map((b, i) => after[i] - b).filter((d) => d !== 0);
617
755
  const n = diffs.length;
618
- if (n < 6) return {
756
+ if (n === 0) return {
619
757
  w: 0,
620
- p: 1
758
+ p: 1,
759
+ method: "exact",
760
+ pFloor: 1,
761
+ nNonZero: 0
621
762
  };
622
- const absRanks = diffs.map((d, i) => ({
763
+ const order = diffs.map((d, i) => ({
623
764
  abs: Math.abs(d),
624
- sign: Math.sign(d),
625
765
  i
626
- })).sort((a, b) => a.abs - b.abs);
766
+ })).sort((x, y) => x.abs - y.abs);
767
+ const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs));
627
768
  const ranks = new Array(n);
628
- let i = 0;
629
- while (i < n) {
630
- let j = i;
631
- while (j < n && absRanks[j].abs === absRanks[i].abs) j++;
632
- const avg = (i + 1 + j) / 2;
633
- for (let k = i; k < j; k++) ranks[absRanks[k].i] = avg;
634
- i = j;
635
- }
769
+ for (let k = 0; k < n; k++) ranks[order[k].i] = midranks[k];
636
770
  let wPlus = 0;
637
771
  for (let k = 0; k < n; k++) if (diffs[k] > 0) wPlus += ranks[k];
638
- const mean = n * (n + 1) / 4;
639
- const variance = n * (n + 1) * (2 * n + 1) / 24;
640
- const z = (wPlus - mean) / Math.sqrt(variance);
641
- const p = 2 * (1 - normalCdf(Math.abs(z)));
772
+ const doubled = midranks.map((rank) => Math.round(rank * 2));
773
+ const doubledDeviation = Math.abs(2 * wPlus - n * (n + 1) / 2);
774
+ const designFloor = Math.min(1, 2 ** (1 - n));
775
+ const method = selectRankTestMethod("wilcoxonSignedRank", opts.method ?? "auto", `n=${n} non-zero differences`, n <= 20, designFloor, `20 non-zero differences`);
776
+ if (method === "exact") {
777
+ const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation);
778
+ return {
779
+ w: wPlus,
780
+ p,
781
+ method,
782
+ pFloor,
783
+ nNonZero: n
784
+ };
785
+ }
786
+ if (method === "asymptotic") {
787
+ const variance = n * (n + 1) * (2 * n + 1) / 24 - tieTerm / 48;
788
+ return {
789
+ w: wPlus,
790
+ p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),
791
+ method,
792
+ pFloor: designFloor,
793
+ nNonZero: n
794
+ };
795
+ }
796
+ const permutations = resolvePermutations("wilcoxonSignedRank", opts.permutations);
797
+ const rng = makeRng(opts.seed, before, after);
798
+ const doubledCentre = n * (n + 1) / 2;
799
+ let atLeastAsExtreme = 0;
800
+ for (let iteration = 0; iteration < permutations; iteration++) {
801
+ let doubledWPlus = 0;
802
+ for (let k = 0; k < n; k++) if (rng() < .5) doubledWPlus += doubled[k];
803
+ if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++;
804
+ }
642
805
  return {
643
806
  w: wPlus,
644
- p
807
+ p: (1 + atLeastAsExtreme) / (permutations + 1),
808
+ method,
809
+ pFloor: Math.max(1 / (permutations + 1), designFloor),
810
+ nNonZero: n
645
811
  };
646
812
  }
647
813
  /**
648
814
  * Cohen's d — standardized effect size for two independent groups.
649
815
  * Positive d means group b has higher mean than group a.
650
816
  * Rule of thumb: |d| < 0.2 negligible, 0.2–0.5 small, 0.5–0.8 medium, > 0.8 large.
817
+ *
818
+ * Returns null where the standardized effect is undefined: fewer than two
819
+ * observations in either group, or a zero pooled standard deviation with
820
+ * unequal means. Null is NOT "no effect" — zero within-group spread across a
821
+ * real mean gap is an unbounded effect, the opposite of negligible. Equal
822
+ * means with zero spread is a genuine 0. Same contract as
823
+ * {@link pairedCohensDz}.
651
824
  */
652
825
  function cohensD(a, b) {
653
- if (a.length < 2 || b.length < 2) return 0;
826
+ if (a.length < 2 || b.length < 2) return null;
654
827
  const meanA = a.reduce((x, y) => x + y, 0) / a.length;
655
828
  const meanB = b.reduce((x, y) => x + y, 0) / b.length;
656
829
  const varA = a.reduce((acc, x) => acc + (x - meanA) ** 2, 0) / (a.length - 1);
657
830
  const varB = b.reduce((acc, x) => acc + (x - meanB) ** 2, 0) / (b.length - 1);
658
831
  const pooled = Math.sqrt(((a.length - 1) * varA + (b.length - 1) * varB) / (a.length + b.length - 2));
659
- if (pooled === 0) return 0;
832
+ if (pooled === 0) return meanB === meanA ? 0 : null;
660
833
  return (meanB - meanA) / pooled;
661
834
  }
662
835
  /**
@@ -900,6 +1073,11 @@ function requiredSampleSize(opts) {
900
1073
  /**
901
1074
  * Required number of paired observations for a target Cohen's dz.
902
1075
  * Unlike the independent-groups formula, this has no two-arm factor of two.
1076
+ *
1077
+ * Normal quantiles with no t correction, so treat the result as a LOWER bound:
1078
+ * it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where
1079
+ * it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller
1080
+ * consults to decide whether 3–10 repetitions suffice.
903
1081
  */
904
1082
  function requiredPairedSampleSize(opts) {
905
1083
  const effect = opts.effect;
@@ -968,13 +1146,21 @@ function mcnemarPower(opts) {
968
1146
  const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom;
969
1147
  return Math.min(1, Math.max(0, normalCdf(zBeta)));
970
1148
  }
971
- /** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
1149
+ /**
1150
+ * Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.
1151
+ *
1152
+ * Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching
1153
+ * {@link holm}, which uniformly dominates this correction and must therefore
1154
+ * never reject less. Validates its inputs on the same terms.
1155
+ */
972
1156
  function bonferroni(pValues, alpha = .05) {
1157
+ assertAlpha("bonferroni", "alpha", alpha);
1158
+ assertPValues("bonferroni", pValues);
973
1159
  const k = pValues.length;
974
1160
  const adjusted = pValues.map((p) => Math.min(1, p * k));
975
1161
  return {
976
1162
  adjusted,
977
- significant: adjusted.map((p) => p < alpha)
1163
+ significant: adjusted.map((p) => p <= alpha)
978
1164
  };
979
1165
  }
980
1166
  /**
@@ -986,8 +1172,8 @@ function bonferroni(pValues, alpha = .05) {
986
1172
  * strong family-wise error control under arbitrary dependence.
987
1173
  */
988
1174
  function holm(pValues, alpha = .05) {
989
- if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError(`holm: alpha must be in (0,1), got ${alpha}`);
990
- for (const [index, pValue] of pValues.entries()) if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) throw new ValidationError(`holm: pValues[${index}] must be in [0,1], got ${pValue}`);
1175
+ assertAlpha("holm", "alpha", alpha);
1176
+ assertPValues("holm", pValues);
991
1177
  const count = pValues.length;
992
1178
  if (count === 0) return {
993
1179
  adjusted: [],
@@ -1013,8 +1199,13 @@ function holm(pValues, alpha = .05) {
1013
1199
  /**
1014
1200
  * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
1015
1201
  * significance at the target FDR; handles ties and preserves q monotonicity.
1202
+ *
1203
+ * Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an
1204
+ * exactly-`fdr` q-value is a discovery.
1016
1205
  */
1017
1206
  function benjaminiHochberg(pValues, fdr = .05) {
1207
+ assertAlpha("benjaminiHochberg", "fdr", fdr);
1208
+ assertPValues("benjaminiHochberg", pValues);
1018
1209
  const n = pValues.length;
1019
1210
  if (n === 0) return {
1020
1211
  qValues: [],
@@ -1029,21 +1220,42 @@ function benjaminiHochberg(pValues, fdr = .05) {
1029
1220
  for (let k = n - 1; k >= 0; k--) {
1030
1221
  const rank = k + 1;
1031
1222
  const entry = indexed[k];
1032
- const raw = entry.p * n / rank;
1223
+ const raw = n / rank * entry.p;
1033
1224
  const bounded = Math.min(minRight, raw);
1034
1225
  minRight = bounded;
1035
1226
  q[entry.i] = Math.min(1, bounded);
1036
1227
  }
1037
1228
  return {
1038
1229
  qValues: q,
1039
- significant: q.map((v) => v < fdr)
1230
+ significant: q.map((v) => v <= fdr)
1040
1231
  };
1041
1232
  }
1233
+ function assertAlpha(fn, label, value) {
1234
+ if (!Number.isFinite(value) || value <= 0 || value >= 1) throw new ValidationError(`${fn}: ${label} must be in (0,1), got ${value}`);
1235
+ }
1236
+ function assertPValues(fn, pValues) {
1237
+ for (const [index, pValue] of pValues.entries()) if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) throw new ValidationError(`${fn}: pValues[${index}] must be in [0,1], got ${pValue}`);
1238
+ }
1239
+ /**
1240
+ * Pairs below which a percentile bootstrap interval is descriptive spread only.
1241
+ *
1242
+ * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
1243
+ * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
1244
+ * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
1245
+ * three points, not an implementation error — scipy's BCa gives 16.0 % on the
1246
+ * same n = 3 data — so no change to the estimator moves it. Below this floor
1247
+ * the decision belongs to the exact sign test or exact signed-rank test.
1248
+ */
1249
+ const BOOTSTRAP_GATE_MIN_N = 20;
1042
1250
  /**
1043
1251
  * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
1044
- * statistic (median by default); pairs are resampled with replacement. The
1045
- * lower bound is what the promotion gate checks — `low > threshold` means the
1046
- * gain is real at the confidence level. Throws on unequal sample sizes.
1252
+ * statistic (median by default); pairs are resampled with replacement. Throws
1253
+ * on unequal sample sizes.
1254
+ *
1255
+ * `low > threshold` carries the stated confidence ONLY at `n ≥
1256
+ * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
1257
+ * check fires under a true null several times more often than nominal, so the
1258
+ * interval is descriptive spread and a promotion must not turn on it.
1047
1259
  */
1048
1260
  function pairedBootstrap(before, after, opts = {}) {
1049
1261
  if (before.length !== after.length) throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`);
@@ -1053,6 +1265,7 @@ function pairedBootstrap(before, after, opts = {}) {
1053
1265
  if (confidence <= 0 || confidence >= 1) throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`);
1054
1266
  const n = before.length;
1055
1267
  const deltas = before.map((b, i) => after[i] - b);
1268
+ const gateEligible = n >= 20;
1056
1269
  if (n === 0) return {
1057
1270
  n: 0,
1058
1271
  median: 0,
@@ -1060,7 +1273,8 @@ function pairedBootstrap(before, after, opts = {}) {
1060
1273
  low: 0,
1061
1274
  high: 0,
1062
1275
  confidence,
1063
- resamples
1276
+ resamples,
1277
+ gateEligible
1064
1278
  };
1065
1279
  if (n === 1) {
1066
1280
  const d = deltas[0];
@@ -1071,10 +1285,11 @@ function pairedBootstrap(before, after, opts = {}) {
1071
1285
  low: d,
1072
1286
  high: d,
1073
1287
  confidence,
1074
- resamples
1288
+ resamples,
1289
+ gateEligible
1075
1290
  };
1076
1291
  }
1077
- const rng = makeRng(opts.seed);
1292
+ const rng = makeRng(opts.seed, deltas);
1078
1293
  const samples = new Array(resamples);
1079
1294
  for (let b = 0; b < resamples; b++) if (statistic === "mean") {
1080
1295
  let sum = 0;
@@ -1096,7 +1311,8 @@ function pairedBootstrap(before, after, opts = {}) {
1096
1311
  low: samples[lowIdx],
1097
1312
  high: samples[Math.max(highIdx, lowIdx)],
1098
1313
  confidence,
1099
- resamples
1314
+ resamples,
1315
+ gateEligible
1100
1316
  };
1101
1317
  }
1102
1318
  /**
@@ -1369,6 +1585,209 @@ function eProcess(opts = {}) {
1369
1585
  }
1370
1586
  };
1371
1587
  }
1588
+ /** Every rank test refuses non-finite input. Beyond the arithmetic being
1589
+ * undefined, the tie-grouping scan compares values with `===`, and
1590
+ * `NaN === NaN` is false, so a NaN would leave the group boundary unable to
1591
+ * advance and spin the loop forever. */
1592
+ function assertFiniteSample(fn, label, xs) {
1593
+ for (let i = 0; i < xs.length; i++) if (!Number.isFinite(xs[i])) throw new ValidationError(`${fn}: ${label}[${i}] must be finite, got ${xs[i]}`);
1594
+ }
1595
+ /**
1596
+ * Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie
1597
+ * groups of size `t` — the correction term both asymptotic rank-test variances
1598
+ * need.
1599
+ */
1600
+ function midranksWithTieTerm(sorted) {
1601
+ const midranks = new Array(sorted.length);
1602
+ let tieTerm = 0;
1603
+ let i = 0;
1604
+ while (i < sorted.length) {
1605
+ let j = i;
1606
+ while (j < sorted.length && sorted[j] === sorted[i]) j++;
1607
+ const average = (i + 1 + j) / 2;
1608
+ for (let k = i; k < j; k++) midranks[k] = average;
1609
+ const groupSize = j - i;
1610
+ if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize;
1611
+ i = j;
1612
+ }
1613
+ return {
1614
+ midranks,
1615
+ tieTerm
1616
+ };
1617
+ }
1618
+ function selectRankTestMethod(fn, request, design, exactFeasible, designFloor, threshold) {
1619
+ if (request === "auto") return exactFeasible ? "exact" : "permutation";
1620
+ if (request === "exact") {
1621
+ if (exactFeasible) return "exact";
1622
+ throw new ValidationError(`${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges to the same answer.`);
1623
+ }
1624
+ if (exactFeasible) throw new ValidationError(`${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no attainable outcome. Use method 'exact' (the default) or add repetitions past ${threshold}.`);
1625
+ return "asymptotic";
1626
+ }
1627
+ function resolvePermutations(fn, permutations) {
1628
+ if (permutations === void 0) return DEFAULT_PERMUTATIONS;
1629
+ if (!Number.isInteger(permutations) || permutations < 1) throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`);
1630
+ return permutations;
1631
+ }
1632
+ /** Two-sided normal-approximation tail with the continuity correction. */
1633
+ function asymptoticTwoSidedP(deviation, sigma) {
1634
+ if (!(sigma > 0)) return 1;
1635
+ return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - .5) / sigma)));
1636
+ }
1637
+ /** SD of U under the permutation null, corrected for the realised ties. The
1638
+ * tie term reduces (N+1) and reaches it exactly when every value is tied, so
1639
+ * the variance floors at 0 rather than going negative. */
1640
+ function twoSampleSigma(n1, n2, total, tieTerm) {
1641
+ if (total < 2) return 0;
1642
+ const variance = n1 * n2 / 12 * (total + 1 - tieTerm / (total * (total - 1)));
1643
+ return Math.sqrt(Math.max(0, variance));
1644
+ }
1645
+ function logChoose(n, k) {
1646
+ return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1);
1647
+ }
1648
+ function formatProbability(value) {
1649
+ return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3);
1650
+ }
1651
+ /**
1652
+ * Exact DP allocation and loop count for this observed rank vector.
1653
+ *
1654
+ * The smaller arm is sufficient because selecting its complement produces the
1655
+ * same two-sided U deviation while using fewer rows in the state table.
1656
+ */
1657
+ function exactTwoSampleCost(doubledRanks, selectedN) {
1658
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
1659
+ let work = 0;
1660
+ for (let placed = 0; placed < doubledRanks.length; placed++) work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed] + 1);
1661
+ return {
1662
+ states: (selectedN + 1) * (maxSum + 1),
1663
+ work
1664
+ };
1665
+ }
1666
+ /**
1667
+ * Smallest attainable two-sided p under the observed ties.
1668
+ *
1669
+ * Only subsets with the minimum or maximum rank sum can attain the largest
1670
+ * deviation. Their multiplicity is the number of ways to choose within the
1671
+ * tie group at each boundary, so this calculation is exact without allocating
1672
+ * the full null distribution.
1673
+ */
1674
+ function exactTwoSampleFloor(doubledRanks, selectedN) {
1675
+ const total = doubledRanks.length;
1676
+ const otherN = total - selectedN;
1677
+ const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0);
1678
+ const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0);
1679
+ if (minimumSum === maximumSum) return 1;
1680
+ const centre = selectedN * (selectedN + 1) + selectedN * otherN;
1681
+ const minimumDeviation = Math.abs(minimumSum - centre);
1682
+ const maximumDeviation = Math.abs(maximumSum - centre);
1683
+ const totalLogWays = logChoose(total, selectedN);
1684
+ const minimumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "minimum") - totalLogWays);
1685
+ const maximumMass = Math.exp(logExtremeSubsetWays(doubledRanks, selectedN, "maximum") - totalLogWays);
1686
+ if (minimumDeviation > maximumDeviation) return minimumMass;
1687
+ if (maximumDeviation > minimumDeviation) return maximumMass;
1688
+ return Math.min(1, minimumMass + maximumMass);
1689
+ }
1690
+ function logExtremeSubsetWays(sortedRanks, selectedN, side) {
1691
+ const boundaryIndex = side === "minimum" ? selectedN - 1 : sortedRanks.length - selectedN;
1692
+ const boundary = sortedRanks[boundaryIndex];
1693
+ let first = boundaryIndex;
1694
+ let afterLast = boundaryIndex + 1;
1695
+ while (first > 0 && sortedRanks[first - 1] === boundary) first--;
1696
+ while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++;
1697
+ return logChoose(afterLast - first, selectedN - (side === "minimum" ? first : sortedRanks.length - afterLast));
1698
+ }
1699
+ /**
1700
+ * Exact conditional two-sided p for the two-sample rank test.
1701
+ *
1702
+ * Convolves the observed doubled midranks into the null distribution of group
1703
+ * a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the
1704
+ * splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the
1705
+ * realised multiset makes the tie handling exact rather than a correction.
1706
+ *
1707
+ * The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −
1708
+ * U` and permutes the split set onto itself), so the two-sided p is the mass
1709
+ * at least as far from the centre as the observation.
1710
+ */
1711
+ function exactTwoSampleP(doubledRanks, n1, n2, doubledDeviation) {
1712
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
1713
+ const width = maxSum + 1;
1714
+ const ways = Array.from({ length: n1 + 1 }, () => new Float64Array(width));
1715
+ ways[0][0] = 1;
1716
+ let placed = 0;
1717
+ for (const rank of doubledRanks) {
1718
+ for (let k = Math.min(n1, placed + 1); k >= 1; k--) {
1719
+ const from = ways[k - 1];
1720
+ const into = ways[k];
1721
+ for (let sum = maxSum - rank; sum >= 0; sum--) {
1722
+ const count = from[sum];
1723
+ if (count !== 0) into[sum + rank] += count;
1724
+ }
1725
+ }
1726
+ placed++;
1727
+ }
1728
+ const shift = n1 * (n1 + 1) + n1 * n2;
1729
+ const chosen = ways[n1];
1730
+ let totalWays = 0;
1731
+ let extremeWays = 0;
1732
+ let tailWays = 0;
1733
+ let maxDeviation = -1;
1734
+ for (let sum = 0; sum < width; sum++) {
1735
+ const count = chosen[sum];
1736
+ if (count === 0) continue;
1737
+ totalWays += count;
1738
+ const deviation = Math.abs(sum - shift);
1739
+ if (deviation >= doubledDeviation) tailWays += count;
1740
+ if (deviation > maxDeviation) {
1741
+ maxDeviation = deviation;
1742
+ extremeWays = count;
1743
+ } else if (deviation === maxDeviation) extremeWays += count;
1744
+ }
1745
+ return {
1746
+ p: tailWays / totalWays,
1747
+ pFloor: extremeWays / totalWays
1748
+ };
1749
+ }
1750
+ /**
1751
+ * Exact conditional two-sided p for the paired signed-rank test.
1752
+ *
1753
+ * Convolves the observed doubled absolute midranks over all `2ⁿ` sign
1754
+ * assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the
1755
+ * arithmetic. The null is symmetric about `n(n+1)/4`.
1756
+ */
1757
+ function exactSignedRankP(doubledRanks, doubledDeviation) {
1758
+ const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0);
1759
+ const width = maxSum + 1;
1760
+ let mass = new Float64Array(width);
1761
+ mass[0] = 1;
1762
+ for (const rank of doubledRanks) {
1763
+ const next = new Float64Array(width);
1764
+ for (let sum = 0; sum < width; sum++) {
1765
+ const probability = mass[sum];
1766
+ if (probability === 0) continue;
1767
+ next[sum] += probability * .5;
1768
+ next[sum + rank] += probability * .5;
1769
+ }
1770
+ mass = next;
1771
+ }
1772
+ const centre = maxSum / 2;
1773
+ let tail = 0;
1774
+ let extreme = 0;
1775
+ let maxDeviation = -1;
1776
+ for (let sum = 0; sum < width; sum++) {
1777
+ const probability = mass[sum];
1778
+ if (probability === 0) continue;
1779
+ const deviation = Math.abs(sum - centre);
1780
+ if (deviation >= doubledDeviation) tail += probability;
1781
+ if (deviation > maxDeviation) {
1782
+ maxDeviation = deviation;
1783
+ extreme = probability;
1784
+ } else if (deviation === maxDeviation) extreme += probability;
1785
+ }
1786
+ return {
1787
+ p: Math.min(1, tail),
1788
+ pFloor: Math.min(1, extreme)
1789
+ };
1790
+ }
1372
1791
  /** Standard-normal inverse CDF (Acklam approximation). */
1373
1792
  function zQuantile(p) {
1374
1793
  if (p <= 0 || p >= 1) {
@@ -1427,16 +1846,45 @@ function medianInPlace(xs) {
1427
1846
  const mid = Math.floor(xs.length / 2);
1428
1847
  return xs.length % 2 === 0 ? (xs[mid - 1] + xs[mid]) / 2 : xs[mid];
1429
1848
  }
1430
- function makeRng(seed) {
1431
- if (seed === void 0) return Math.random;
1432
- return mulberry32(seed);
1849
+ /**
1850
+ * PRNG for every resampling path in this module.
1851
+ *
1852
+ * With no caller seed the seed is DERIVED FROM THE DATA rather than taken from
1853
+ * `Math.random`, so re-running the same input reproduces the same interval —
1854
+ * a gate verdict that cannot be re-derived is not evidence. Distinct data
1855
+ * still gets a distinct stream. Same pattern as `promotion-gate.ts`.
1856
+ */
1857
+ function makeRng(seed, ...series) {
1858
+ return mulberry32(seed ?? seedFromData(series));
1859
+ }
1860
+ /** FNV-1a over the IEEE-754 bytes of every observation. */
1861
+ function seedFromData(series) {
1862
+ const view = /* @__PURE__ */ new DataView(/* @__PURE__ */ new ArrayBuffer(8));
1863
+ let hash = 2166136261;
1864
+ for (const xs of series) {
1865
+ for (const x of xs) {
1866
+ view.setFloat64(0, x);
1867
+ for (let byte = 0; byte < 8; byte++) hash = Math.imul(hash ^ view.getUint8(byte), 16777619);
1868
+ }
1869
+ hash = Math.imul(hash ^ 255, 16777619);
1870
+ }
1871
+ return hash | 0;
1872
+ }
1873
+ /** Order-independent seed for a symmetric two-sample statistic. */
1874
+ function symmetricTwoSampleSeed(a, b) {
1875
+ const sortedA = [...a].sort((left, right) => left - right);
1876
+ const sortedB = [...b].sort((left, right) => left - right);
1877
+ const forward = seedFromData([sortedA, sortedB]) >>> 0;
1878
+ const reversed = seedFromData([sortedB, sortedA]) >>> 0;
1879
+ return Math.min(forward, reversed);
1433
1880
  }
1434
1881
  /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
1435
1882
  * cryptographic. Exported so e-process shuffles and bootstrap resampling
1436
- * share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
1437
- * gate verdicts is non-reproducible by construction). */
1883
+ * share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
1884
+ * stream, including 0. */
1438
1885
  function mulberry32(seed) {
1439
- let s = seed | 0 || 2654435769;
1886
+ if (!Number.isFinite(seed)) throw new ValidationError(`mulberry32: seed must be a finite number, got ${seed}`);
1887
+ let s = seed | 0;
1440
1888
  return () => {
1441
1889
  s = s + 1831565813 | 0;
1442
1890
  let t = s;
@@ -1446,6 +1894,6 @@ function mulberry32(seed) {
1446
1894
  };
1447
1895
  }
1448
1896
  //#endregion
1449
- export { spearmanR as A, positionalBias as B, pairedTTest as C, ranks as D, pearsonR as E, studentTCdf as F, verbosityBias as H, normalCdf as I, calibrateJudge as L, weightedMean as M, wilcoxonSignedRank as N, requiredPairedSampleSize as O, wilson as P, calibrateJudgeContinuous as R, pairedSignTest as S, passAtK as T, selfPreference as V, normalizeScores as _, confidenceInterval as a, pairedMde as b, eProcess as c, interpretCliffs as d, mannWhitneyU as f, mulberry32 as g, mcnemarRequiredN as h, cohensD as i, weightedComposite as j, requiredSampleSize as k, holm as l, mcnemarPower as m, bonferroni as n, corpusInterRaterAgreement as o, mcnemar as p, cliffsDelta as r, corpusInterRaterAgreementFromJudgeScores as s, benjaminiHochberg as t, interRaterReliability as u, pairedBootstrap as v, partialCredit as w, pairedRiskDifference as x, pairedCohensDz as y, continuousAgreement as z };
1897
+ export { passAtK as A, studentTCdf as B, pairedBootstrap as C, pairedSignTest as D, pairedRiskDifference as E, spearmanR as F, continuousAgreement as G, normalCdf as H, weightedComposite as I, verbosityBias as J, positionalBias as K, weightedMean as L, ranks as M, requiredPairedSampleSize as N, pairedTTest as O, requiredSampleSize as P, wilcoxonSignedRank as R, normalizeScores as S, pairedMde as T, calibrateJudge as U, studentTQuantile as V, calibrateJudgeContinuous as W, mannWhitneyU as _, WILCOXON_EXACT_MAX_N as a, mcnemarRequiredN as b, cliffsDelta as c, corpusInterRaterAgreement as d, corpusInterRaterAgreementFromJudgeScores as f, interpretCliffs as g, interRaterReliability as h, MANN_WHITNEY_EXACT_MAX_WORK as i, pearsonR as j, partialCredit as k, cohensD as l, holm as m, DEFAULT_PERMUTATIONS as n, benjaminiHochberg as o, eProcess as p, selfPreference as q, MANN_WHITNEY_EXACT_MAX_STATES as r, bonferroni as s, BOOTSTRAP_GATE_MIN_N as t, confidenceInterval as u, mcnemar as v, pairedCohensDz as w, mulberry32 as x, mcnemarPower as y, wilson as z };
1450
1898
 
1451
- //# sourceMappingURL=statistics-DWM_AyLe.js.map
1899
+ //# sourceMappingURL=statistics-RwRNu2__.js.map