@tangle-network/agent-eval 0.161.0 → 0.163.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
- package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
- package/dist/analyst/index.d.ts +2 -2
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +3 -3
- package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
- package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
- package/dist/{benchmark-command-BVtaq_ve.js → benchmark-command-CF-4GEWZ.js} +9 -8
- package/dist/benchmark-command-CF-4GEWZ.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +22 -8
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BSmOwskD.js → campaign-DQZmc2Dq.js} +12 -12
- package/dist/{campaign-BSmOwskD.js.map → campaign-DQZmc2Dq.js.map} +1 -1
- package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
- package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
- package/dist/cli.js +3 -3
- package/dist/contract/index.js +7 -7
- package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
- package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
- package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-D08pWIJb.js} +13 -13
- package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-D08pWIJb.js.map} +1 -1
- package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
- package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-DhA9qKIm.js} +2 -2
- package/dist/{dspy-rlm-engine-DptEII26.js.map → dspy-rlm-engine-DhA9qKIm.js.map} +1 -1
- package/dist/emitter-D_jYSGRd.d.ts.map +1 -1
- package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
- package/dist/emitter-DeQHiDMm.js.map +1 -0
- package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-BfohKmzx.js} +3 -3
- package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-BfohKmzx.js.map} +1 -1
- package/dist/experiment/index.js +8 -8
- package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
- package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
- package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-BFmh36vW.js} +2 -2
- package/dist/{external-optimizer-process-WosTBChy.js.map → external-optimizer-process-BFmh36vW.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-CqLMW3nh.js} +2 -2
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js.map → external-optimizer-subprocess-CqLMW3nh.js.map} +1 -1
- package/dist/fuzz.d.ts +1 -2
- package/dist/fuzz.d.ts.map +1 -1
- package/dist/fuzz.js +4 -10
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.js +26 -26
- package/dist/index.js.map +1 -1
- package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
- package/dist/internal-BMFSR8Ns.js.map +1 -0
- package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
- package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
- package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
- package/dist/llm-client-BFMRpmqb.js.map +1 -0
- package/dist/{llm-judge-BhasIPFT.js → llm-judge-Du7WQPh7.js} +7 -7
- package/dist/{llm-judge-BhasIPFT.js.map → llm-judge-Du7WQPh7.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -1
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +14 -22
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
- package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
- package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
- package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
- package/dist/pipelines/index.d.ts +3 -2
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -19
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
- package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
- package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
- package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
- package/dist/{produced-state-DZ89riy5.js → produced-state-Be0BK3RN.js} +3 -3
- package/dist/{produced-state-DZ89riy5.js.map → produced-state-Be0BK3RN.js.map} +1 -1
- package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
- package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
- package/dist/{query-DxPYqpmT.d.ts → query-CwnHlu5p.d.ts} +19 -2
- package/dist/query-CwnHlu5p.d.ts.map +1 -0
- package/dist/{query-CHmMP42p.js → query-_5g6re3_.js} +44 -2
- package/dist/query-_5g6re3_.js.map +1 -0
- package/dist/random-Dn5fPWkt.js +21 -0
- package/dist/random-Dn5fPWkt.js.map +1 -0
- package/dist/record-id-DUgsK5qp.js +17 -0
- package/dist/record-id-DUgsK5qp.js.map +1 -0
- package/dist/{release-confidence-DKfD2RYU.js → release-confidence-nGDJiiwc.js} +2 -2
- package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-nGDJiiwc.js.map} +1 -1
- package/dist/reporting.js +6 -6
- package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-O5zKDANP.js} +2 -2
- package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-O5zKDANP.js.map} +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +9 -15
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
- package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
- package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-BsDMOwJr.js} +2 -2
- package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-BsDMOwJr.js.map} +1 -1
- package/dist/{sequential-rYW-Ophm.js → sequential-BLMbdrD7.js} +3 -3
- package/dist/{sequential-rYW-Ophm.js.map → sequential-BLMbdrD7.js.map} +1 -1
- package/dist/{server-BtFd4uzB.js → server-BjYiJHoJ.js} +2 -2
- package/dist/{server-BtFd4uzB.js.map → server-BjYiJHoJ.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-UArRo-nr.js} +6 -6
- package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-UArRo-nr.js.map} +1 -1
- package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-BVga3c37.js} +21 -17
- package/dist/store-tool-spans-BVga3c37.js.map +1 -0
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +1 -1
- package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
- package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
- package/dist/{summary-report-BI5hUtvK.js → summary-report-BXeQ5Ues.js} +6 -6
- package/dist/{summary-report-BI5hUtvK.js.map → summary-report-BXeQ5Ues.js.map} +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{tool-waste-BqzmVdJk.js → tool-waste-8BQiUc8K.js} +4 -4
- package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-8BQiUc8K.js.map} +1 -1
- package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-CKc7bYIg.d.ts} +2 -2
- package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-CKc7bYIg.d.ts.map} +1 -1
- package/dist/trace-repair/index.js +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +8 -17
- package/dist/traces.js.map +1 -1
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +3 -13
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/{types-BPb2Kf_C2.d.ts → types-BPb2Kf_C.d.ts} +1 -1
- package/dist/types-BPb2Kf_C.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/package.json +2 -2
- package/dist/benchmark-command-BVtaq_ve.js.map +0 -1
- package/dist/emitter-BpYFQPj4.js.map +0 -1
- package/dist/internal-BDHPCnjk.js.map +0 -1
- package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
- package/dist/llm-client-hgDieDNN.js.map +0 -1
- package/dist/query-CHmMP42p.js.map +0 -1
- package/dist/query-DxPYqpmT.d.ts.map +0 -1
- package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
- package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"paired-arms-D-XRF_fy.js","names":[],"sources":["../src/statistics/paired-binary.ts","../src/math/normal.ts","../src/statistics/rank-tests.ts","../src/paired-arms.ts"],"sourcesContent":["import { regularizedIncompleteBeta } from '../math/special-functions'\nimport { binomialSignTwoSided, zQuantile } from './internal'\n\n// ── Binomial proportion + paired-binary + coding-eval estimators ─────\n//\n// The paired family above (pairedBootstrap/pairedTTest/wilcoxonSignedRank)\n// operates on continuous scores. Pass/fail A/B comparisons — \"does treatment\n// X raise the success RATE vs control\" — are binary and paired, so they need\n// their own correct estimators: McNemar for significance (only the discordant\n// pairs carry signal), the paired risk difference for effect size, Wilson for\n// a single-arm proportion CI, and pass@k for the standard k-sample coding-eval\n// metric. The normal approximation is wrong for proportions near 0/1 and for\n// the small discordant counts typical of eval runs, so these are exact /\n// Wilson-based, not Wald.\n\n/** A binomial proportion estimate with a confidence interval. */\nexport interface ProportionInterval {\n /** Point estimate successes / n (0 when n = 0). */\n estimate: number\n /** Lower bound, clamped to [0, 1]. */\n lower: number\n /** Upper bound, clamped to [0, 1]. */\n upper: number\n}\n\n/**\n * Wilson score interval for a binomial proportion. Correct at small n and near\n * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and\n * understates coverage. Use this for any pass-rate / hit-rate / realness-rate\n * CI — the continuous `confidenceInterval` assumes the wrong distribution for a\n * proportion. `n = 0 ⇒ {0, 0, 0}`.\n */\nexport function wilson(successes: number, n: number, confidence = 0.95): ProportionInterval {\n if (n <= 0) return { estimate: 0, lower: 0, upper: 0 }\n if (successes < 0 || successes > n) {\n throw new Error(`wilson: successes (${successes}) must be in [0, ${n}]`)\n }\n const z = zQuantile(1 - (1 - confidence) / 2)\n const p = successes / n\n const z2 = z * z\n const denom = 1 + z2 / n\n const center = (p + z2 / (2 * n)) / denom\n const half = (z * Math.sqrt((p * (1 - p) + z2 / (4 * n)) / n)) / denom\n return {\n estimate: p,\n lower: Math.max(0, center - half),\n upper: Math.min(1, center + half),\n }\n}\n\n/**\n * Are these per-item outcomes binary (every value exactly 0 or 1)?\n *\n * The discriminator a promotion gate needs before choosing a paired statistic.\n * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is\n * normally dominated by zeros (both arms solve, or both arms miss, most items),\n * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in\n * success rate is — and a bootstrap CI on that median collapses to [0, 0].\n * A gate keying on `ci.low > threshold` is then structurally unable to see\n * either a gain or a regression. Detect this shape and switch to the\n * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})\n * instead of silently answering \"no\" forever.\n *\n * Empty input is NOT binary: there is no evidence of the outcome's shape, and\n * defaulting an empty vector into the binary branch would pick a statistic on\n * no data at all.\n *\n * NOT the right discriminator for a gate. It recognises the literal {0, 1}\n * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which\n * judges in this codebase do routinely — reads as non-binary, and a single\n * partial-credit score in an otherwise pass/fail vector flips it to false while\n * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any\n * two-point encoding). This predicate remains for callers that specifically\n * mean \"literally 0/1\".\n */\nexport function isBinaryOutcomeVector(values: ArrayLike<number>): boolean {\n if (values.length === 0) return false\n for (let i = 0; i < values.length; i++) {\n const v = values[i]!\n if (v !== 0 && v !== 1) return false\n }\n return true\n}\n\n/** Result of a McNemar paired-binary significance test. */\nexport interface McNemarResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs (b + c) — the only ones that carry signal. */\n nDiscordant: number\n /** Pairs where treatment succeeded and control failed (\"newly correct\"). */\n b: number\n /** Pairs where control succeeded and treatment failed (\"newly wrong\"). */\n c: number\n /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */\n statistic: number\n /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */\n pValue: number\n}\n\n/**\n * McNemar's test for paired binary outcomes — the correct significance test for\n * \"does treatment change the success rate vs control on the SAME items\". Only\n * discordant pairs (one arm right, the other wrong) carry information; concordant\n * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw\n * rates is wrong here. The p-value is exact: under H0 the b \"treatment-wins\" are\n * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct\n * at the small discordant counts typical of eval runs (no continuity-corrected\n * chi-square approximation needed, though it is returned as `statistic` for\n * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match\n * the module's (before, after) convention. Throws on unequal lengths.\n */\nexport function mcnemar(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n): McNemarResult {\n if (control.length !== treatment.length) {\n throw new Error(`mcnemar: unequal sample sizes (${control.length} vs ${treatment.length})`)\n }\n const n = control.length\n let b = 0 // treatment 1, control 0\n let c = 0 // treatment 0, control 1\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const nDiscordant = b + c\n const statistic = nDiscordant === 0 ? 0 : (Math.abs(b - c) - 1) ** 2 / nDiscordant\n return { n, nDiscordant, b, c, statistic, pValue: binomialSignTwoSided(b, c) }\n}\n\n/** A paired binary effect size (treatment rate − control rate) with a CI. */\nexport interface RiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Lower bound of the CI, clamped to [-1, 1]. */\n lower: number\n /** Upper bound of the CI, clamped to [-1, 1]. */\n upper: number\n /** Confidence level used. */\n confidence: number\n}\n\n/**\n * Paired risk difference (the effect-size companion to {@link mcnemar}): the\n * change in success rate p(treatment) − p(control) on matched items, which for\n * paired binary data equals (b − c) / n. The CI uses the paired variance from\n * the discordant counts, not the independent-samples formula (which overstates\n * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)\n * arrays, control first. Throws on unequal lengths.\n *\n * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald\n * normal approximation, which badly UNDERCOVERS when only a handful of pairs are\n * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,\n * while McNemar's exact test on the same data gives p = 0.50. A gate keying on\n * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose\n * interval is dual to the exact test by construction, for any decision.\n */\nexport function pairedRiskDifference(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): RiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n const n = control.length\n if (n === 0) return { n: 0, b: 0, c: 0, riskDifference: 0, lower: 0, upper: 0, confidence }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const rd = (b - c) / n\n const variance = (b + c - (b - c) ** 2 / n) / (n * n)\n const z = zQuantile(1 - (1 - confidence) / 2)\n const half = z * Math.sqrt(Math.max(0, variance))\n return {\n n,\n b,\n c,\n riskDifference: rd,\n lower: Math.max(-1, rd - half),\n upper: Math.min(1, rd + half),\n confidence,\n }\n}\n\n/** A paired binary effect size with an EXACT interval and the exact test that\n * bounds it — one object so a caller cannot read the estimate without the\n * significance it is entitled to. */\nexport interface ExactRiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Discordant pairs (b + c) — the only ones carrying information. */\n nDiscordant: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */\n lower: number\n /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */\n upper: number\n /** Confidence level used. */\n confidence: number\n /** McNemar's exact two-sided p-value on the same discordant counts. */\n pValue: number\n}\n\n/**\n * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a\n * promotion gate may decide on.\n *\n * Conditional on the number of discordant pairs m = b + c, the treatment-win\n * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the\n * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a\n * Clopper-Pearson exact interval for π maps straight onto RD. This buys the\n * property the Wald interval in {@link pairedRiskDifference} does not have:\n *\n * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**\n *\n * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial\n * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the\n * interval and the test can never disagree, and a gate keyed on `lower` cannot\n * promote what the exact test refuses. The exact p is returned in the same\n * object so the two are impossible to compute apart.\n *\n * The interval is conservative (exact intervals over-cover; conditioning on m\n * discards the concordant pairs' information about m itself). That is the\n * correct direction for a promotion gate: it refuses more often, never less.\n *\n * With m = 0 there are no discordant pairs and π is not identified: the result\n * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —\n * callers must treat a zero-width interval as \"cannot decide\", not as \"no\n * difference\". Inputs are paired 0/1 (or boolean) arrays, control first.\n * Throws on unequal lengths.\n */\nexport function pairedRiskDifferenceExact(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): ExactRiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifferenceExact: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedRiskDifferenceExact: confidence must be in (0,1), got ${confidence}`)\n }\n const n = control.length\n if (n === 0) {\n return {\n n: 0,\n b: 0,\n c: 0,\n nDiscordant: 0,\n riskDifference: 0,\n lower: 0,\n upper: 0,\n confidence,\n pValue: 1,\n }\n }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const m = b + c\n const riskDifference = (b - c) / n\n const pValue = binomialSignTwoSided(b, c)\n if (m === 0) {\n return { n, b, c, nDiscordant: 0, riskDifference: 0, lower: 0, upper: 0, confidence, pValue }\n }\n const alpha = 1 - confidence\n const piLow = b === 0 ? 0 : betaQuantile(alpha / 2, b, m - b + 1)\n const piHigh = b === m ? 1 : betaQuantile(1 - alpha / 2, b + 1, m - b)\n const scale = m / n\n return {\n n,\n b,\n c,\n nDiscordant: m,\n riskDifference,\n lower: Math.max(-1, (2 * piLow - 1) * scale),\n upper: Math.min(1, (2 * piHigh - 1) * scale),\n confidence,\n pValue,\n }\n}\n\n/** Inverse regularized incomplete beta by bisection on\n * {@link regularizedIncompleteBeta}, which is monotone increasing in x. 80\n * halvings of [0,1] resolve to ~8e-25, far past the continued fraction's own\n * 3e-15 tolerance, so the quantile is as exact as the CDF it inverts. */\nfunction betaQuantile(p: number, a: number, b: number): number {\n if (p <= 0) return 0\n if (p >= 1) return 1\n let lo = 0\n let hi = 1\n for (let i = 0; i < 80; i++) {\n const mid = (lo + hi) / 2\n if (regularizedIncompleteBeta(mid, a, b) < p) lo = mid\n else hi = mid\n }\n return (lo + hi) / 2\n}\n\n/** A paired binary effect size with an interval that is valid at a NONZERO\n * margin — the estimator a noninferiority decision may be made on. */\nexport interface ScoreRiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Discordant pairs (b + c). */\n nDiscordant: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Score-interval lower bound on the population risk difference. */\n lower: number\n /** Score-interval upper bound on the population risk difference. */\n upper: number\n /** Confidence level used. */\n confidence: number\n}\n\n/**\n * Constrained MLE of q = P(treatment loses) under the hypothesis RD = `delta`.\n *\n * Profiling the two concordant cells out of the multinomial leaves\n * `L(q) = b·log(q+delta) + c·log(q) + e·log(1 − 2q − delta)` with `e = n − b − c`,\n * whose stationary point is the positive root of\n * `2n·q² − [(b + c) − delta·(b + 3c + 2e)]·q − c·delta·(1 − delta) = 0`.\n * At `delta = 0` this returns `(b + c) / 2n`, the familiar null.\n */\nfunction constrainedLossRate(b: number, c: number, n: number, delta: number): number {\n const e = n - b - c\n const quadratic = 2 * n\n const linear = -(b + c - delta * (b + 3 * c + 2 * e))\n const constant = -c * delta * (1 - delta)\n const discriminant = linear * linear - 4 * quadratic * constant\n const root = discriminant > 0 ? Math.sqrt(discriminant) : 0\n const q = (-linear + root) / (2 * quadratic)\n // Clamp into the region where all four cell probabilities stay non-negative.\n return Math.min(Math.max(q, Math.max(0, -delta)), Math.max(0, (1 - delta) / 2))\n}\n\n/** Tango's score statistic for H0: RD = `delta`. `Var(b − c) = n·(2q + delta −\n * delta²)` under that hypothesis, evaluated at the constrained MLE of q. */\nfunction tangoScore(b: number, c: number, n: number, delta: number): number {\n const numerator = b - c - n * delta\n const q = constrainedLossRate(b, c, n, delta)\n const variance = n * (2 * q + delta * (1 - delta))\n if (!(variance > 0)) {\n if (numerator === 0) return 0\n return numerator > 0 ? Number.POSITIVE_INFINITY : Number.NEGATIVE_INFINITY\n }\n return numerator / Math.sqrt(variance)\n}\n\n/**\n * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a\n * promotion gate may decide on **at a nonzero margin**.\n *\n * {@link pairedRiskDifferenceExact} conditions on the observed discordant count\n * `m = b + c`, builds a Clopper-Pearson interval for the win share among those\n * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing\n * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the\n * population risk difference at a nonzero margin, because the sampling\n * variability of `m/n` itself is discarded. The gap is not academic: with the\n * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk\n * difference sits exactly on that margin clears a nominal-95 % `lower > margin`\n * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates\n * each) when the conditional interval decides.\n *\n * Tango's interval inverts the score test of RD = delta, which estimates the\n * nuisance loss rate under each hypothesised delta instead of fixing it at the\n * observed value, so `m` contributes its own uncertainty. It is the method\n * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it\n * is not conditional, so it stays valid as the margin moves away from zero.\n *\n * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is\n * monotone decreasing in delta, so each crossing is unique. Inputs are paired\n * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.\n */\nexport function pairedRiskDifferenceScore(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): ScoreRiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifferenceScore: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedRiskDifferenceScore: confidence must be in (0,1), got ${confidence}`)\n }\n const n = control.length\n if (n === 0) {\n return { n: 0, b: 0, c: 0, nDiscordant: 0, riskDifference: 0, lower: -1, upper: 1, confidence }\n }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const riskDifference = (b - c) / n\n const z = zQuantile(1 - (1 - confidence) / 2)\n\n // Lower bound: the smallest delta still inside the interval, i.e. the root of\n // score(delta) = +z on [-1, riskDifference]. score(riskDifference) = 0 < z, so\n // the right endpoint is always inside and the bisection is well posed.\n let lo = -1\n let hi = riskDifference\n for (let i = 0; i < 200; i++) {\n const mid = (lo + hi) / 2\n if (tangoScore(b, c, n, mid) > z) lo = mid\n else hi = mid\n }\n const lower = (lo + hi) / 2\n\n // Upper bound: root of score(delta) = -z on [riskDifference, 1].\n let ulo = riskDifference\n let uhi = 1\n for (let i = 0; i < 200; i++) {\n const mid = (ulo + uhi) / 2\n if (tangoScore(b, c, n, mid) > -z) ulo = mid\n else uhi = mid\n }\n const upper = (ulo + uhi) / 2\n\n return {\n n,\n b,\n c,\n nDiscordant: b + c,\n riskDifference,\n lower: Math.max(-1, lower),\n upper: Math.min(1, upper),\n confidence,\n }\n}\n\n/**\n * The common positive level `s` such that EVERY value across both paired arms is\n * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived\n * in. Returns null when the outcomes are not two-point, when the two arms use\n * different levels, or when no positive value was observed at all (all-zero\n * arms: the level is not identified, and there is nothing to decide anyway).\n *\n * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only\n * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as\n * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),\n * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test\n * silently sends it down the median path that cannot see it. Any positive level\n * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired\n * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by\n * s and rescaling the result back into the caller's native units.\n *\n * Non-finite values ⇒ null: an unusable outcome must not be classified as a\n * clean pass/fail shape.\n */\nexport function pairedBinaryScale(\n before: ArrayLike<number>,\n after: ArrayLike<number>,\n): number | null {\n let level: number | null = null\n for (const arm of [before, after]) {\n for (let i = 0; i < arm.length; i++) {\n const v = arm[i]!\n if (!Number.isFinite(v)) return null\n if (v === 0) continue\n if (v < 0) return null\n if (level === null) level = v\n else if (v !== level) return null\n }\n }\n return level\n}\n\n/**\n * Unbiased pass@k for code generation (Chen et al. 2021, \"Evaluating Large\n * Language Models Trained on Code\"). Given `n` independent samples for one\n * problem of which `c` pass, the probability that at least one of a random k of\n * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as \"did any of the\n * first k pass\" is biased high at small n; this is the variance-reduced estimator\n * averaged implicitly over all k-subsets. Average the per-problem values across\n * the suite for the corpus pass@k. Computed in the numerically stable product\n * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.\n */\nexport function passAtK(n: number, c: number, k: number): number {\n if (!Number.isInteger(n) || !Number.isInteger(c) || !Number.isInteger(k)) {\n throw new Error(`passAtK: n, c, k must be integers (got n=${n}, c=${c}, k=${k})`)\n }\n if (k < 1 || k > n || c < 0 || c > n) {\n throw new Error(`passAtK: require 1 ≤ k ≤ n and 0 ≤ c ≤ n (got n=${n}, c=${c}, k=${k})`)\n }\n if (n - c < k) return 1\n let prob = 1\n for (let i = n - c + 1; i <= n; i++) prob *= 1 - k / i\n return 1 - prob\n}\n","/**\n * Standard normal cumulative distribution using Abramowitz and Stegun 7.1.26.\n *\n * The approximation is evaluated as erf(x / sqrt(2)). Computing the negative\n * tail from the complementary term avoids cancellation when x is far below 0.\n * The maximum absolute CDF error is approximately 7.5e-8.\n */\nexport function normalCdf(x: number): number {\n if (x === 0) return 0.5\n\n const a1 = 0.254829592\n const a2 = -0.284496736\n const a3 = 1.421413741\n const a4 = -1.453152027\n const a5 = 1.061405429\n const p = 0.3275911\n\n const scaled = Math.abs(x) / Math.SQRT2\n const t = 1 / (1 + p * scaled)\n const complement = ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-scaled * scaled)\n\n return x < 0 ? complement / 2 : 1 - complement / 2\n}\n","import { ValidationError } from '../errors'\nimport { normalCdf } from '../math/normal'\nimport { lnGamma } from '../math/special-functions'\nimport { assertFiniteSample, makeRng, symmetricTwoSampleSeed } from './internal'\n\n// ── Rank tests: exact by default ─────────────────────────────────────\n//\n// At 3–10 repetitions per arm the binding constraint on a rank test is\n// combinatorial, not numerical. Three versus three admits only 20 splits, so\n// the attainable two-sided p-grid starts at 0.1000 and α = 0.05 is out of\n// reach at that design; a normal approximation reports 0.0495, a p-value that\n// describes no attainable outcome. No better approximation fixes this — adding\n// the tie correction moves `[0,0,0]` versus `[1,1,1]` from 0.0495 to 0.0469,\n// further from its exact 0.1000, not closer.\n//\n// So both null distributions are enumerated exactly inside bounded state and\n// work budgets, by convolution over the observed midranks (identical to\n// enumerating every split / sign pattern, and cheaper), which conditions on the\n// realised tie pattern for free. Above those budgets the default is a seeded\n// Monte Carlo permutation. The asymptotic path is never chosen automatically,\n// and asking for it inside the exact-feasible range throws.\n//\n// Every result carries `method` and `pFloor` so a downstream gate can SEE the\n// discreteness rather than infer it: a gate handed data whose `pFloor` exceeds\n// its alpha is underpowered by construction, which is a true statement about\n// the experiment, not a false one about the effect.\n\n/** How a rank test's p-value was actually computed. */\nexport type RankTestMethod = 'exact' | 'permutation' | 'asymptotic'\n\n/**\n * What the caller asks for. `'auto'` selects `'exact'` inside the enumeration\n * threshold and `'permutation'` above it, and never selects `'asymptotic'`.\n */\nexport type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic'\n\nexport interface RankTestOptions {\n /** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */\n method?: RankTestMethodRequest\n /** Resamples on the Monte Carlo permutation path. Default 100000. */\n permutations?: number\n /** Seed for the permutation path. Omitted ⇒ derived from the data itself, so\n * the result is reproducible either way. */\n seed?: number\n}\n\n/** Maximum dynamic-programming cells used by an exact two-sample rank test. */\nexport const MANN_WHITNEY_EXACT_MAX_STATES = 8_192\n/** Maximum inner-loop transitions used by an exact two-sample rank test. */\nexport const MANN_WHITNEY_EXACT_MAX_WORK = 250_000\n/** Non-zero differences up to which the signed-rank null is enumerated exactly. */\nexport const WILCOXON_EXACT_MAX_N = 20\n/** Resamples used when a rank test falls back to Monte Carlo permutation. */\nexport const DEFAULT_PERMUTATIONS = 100_000\n\nexport interface MannWhitneyResult {\n /** `min(U_a, U_b)` — the conventional reported statistic. */\n u: number\n /** U for sample `a`. Carries the direction of the effect, which `u` discards. */\n uA: number\n /** Two-sided p-value. */\n p: number\n /** How `p` was computed. */\n method: RankTestMethod\n /** Smallest two-sided p this design can produce. `p` can never be below it. */\n pFloor: number\n}\n\n/**\n * Mann-Whitney U — two independent samples, no distributional assumption.\n *\n * Exact conditional (permutation) p by default when the dynamic program fits\n * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and\n * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo\n * permutation above those limits. This keeps imbalanced designs such as 1+24\n * exact without admitting expensive balanced designs merely because they have\n * the same total size. Throws on non-finite input and on `method:\n * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,\n * pFloor = 1` — no design, no attainable evidence.\n */\nexport function mannWhitneyU(\n a: number[],\n b: number[],\n opts: RankTestOptions = {},\n): MannWhitneyResult {\n assertFiniteSample('mannWhitneyU', 'a', a)\n assertFiniteSample('mannWhitneyU', 'b', b)\n\n const n1 = a.length\n const n2 = b.length\n if (n1 === 0 || n2 === 0) return { u: 0, uA: 0, p: 1, method: 'exact', pFloor: 1 }\n\n const total = n1 + n2\n const combined = [\n ...a.map((v) => ({ v, fromA: true })),\n ...b.map((v) => ({ v, fromA: false })),\n ].sort((x, y) => x.v - y.v)\n\n const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v))\n let rankSumA = 0\n for (let k = 0; k < total; k++) {\n if (combined[k]!.fromA) rankSumA += midranks[k]!\n }\n\n const uA = rankSumA - (n1 * (n1 + 1)) / 2\n const u = Math.min(uA, n1 * n2 - uA)\n // Midranks are integers or halves, so doubling makes the null convolution\n // integral. U is centred on n₁n₂/2, hence a doubled centre of n₁n₂.\n const doubled = midranks.map((rank) => Math.round(rank * 2))\n const doubledDeviation = Math.abs(2 * uA - n1 * n2)\n const selectedN = Math.min(n1, n2)\n const otherN = total - selectedN\n const exactCost = exactTwoSampleCost(doubled, selectedN)\n\n const designFloor = exactTwoSampleFloor(doubled, selectedN)\n const method = selectRankTestMethod(\n 'mannWhitneyU',\n opts.method ?? 'auto',\n `n1=${n1}, n2=${n2}`,\n exactCost.states <= MANN_WHITNEY_EXACT_MAX_STATES &&\n exactCost.work <= MANN_WHITNEY_EXACT_MAX_WORK,\n designFloor,\n `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString('en-US')} states and ` +\n `${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString('en-US')} transitions`,\n )\n\n if (method === 'exact') {\n const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation)\n return { u, uA, p, method, pFloor }\n }\n\n if (method === 'asymptotic') {\n return {\n u,\n uA,\n p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),\n method,\n pFloor: designFloor,\n }\n }\n\n const permutations = resolvePermutations('mannWhitneyU', opts.permutations)\n const rng = opts.seed === undefined ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed)\n let atLeastAsExtreme = 0\n const pool = [...doubled]\n for (let iteration = 0; iteration < permutations; iteration++) {\n let doubledRankSum = 0\n for (let k = 0; k < selectedN; k++) {\n const pick = k + Math.floor(rng() * (total - k))\n const swapped = pool[pick]!\n pool[pick] = pool[k]!\n pool[k] = swapped\n doubledRankSum += swapped\n }\n if (\n Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >=\n doubledDeviation\n ) {\n atLeastAsExtreme++\n }\n }\n const pFloor = Math.max(1 / (permutations + 1), designFloor)\n return {\n u,\n uA,\n p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),\n method,\n pFloor,\n }\n}\n\nexport interface WilcoxonSignedRankResult {\n /** W⁺, the rank sum of the positive differences. (scipy reports\n * `min(W⁺, W⁻)`; compare statistics only after converting.) */\n w: number\n /** Two-sided p-value. */\n p: number\n /** How `p` was computed. */\n method: RankTestMethod\n /** Smallest two-sided p this design can produce. */\n pFloor: number\n /** Non-zero differences — zero differences are dropped and carry no rank. */\n nNonZero: number\n}\n\n/**\n * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.\n *\n * Exact conditional (sign-flip) p by default at `n ≤\n * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo\n * permutation above it. Throws on non-finite input and on `method:\n * 'asymptotic'` where an exact answer is available.\n *\n * `n` is the count of NON-ZERO differences: exact ties are dropped before\n * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.\n * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which\n * `pFloor` states rather than leaving `p = 1` to be read as a measured null.\n */\nexport function wilcoxonSignedRank(\n before: number[],\n after: number[],\n opts: RankTestOptions = {},\n): WilcoxonSignedRankResult {\n if (before.length !== after.length) {\n throw new ValidationError(\n `wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n assertFiniteSample('wilcoxonSignedRank', 'before', before)\n assertFiniteSample('wilcoxonSignedRank', 'after', after)\n\n const diffs = before.map((b, i) => after[i]! - b).filter((d) => d !== 0)\n const n = diffs.length\n if (n === 0) return { w: 0, p: 1, method: 'exact', pFloor: 1, nNonZero: 0 }\n\n const order = diffs.map((d, i) => ({ abs: Math.abs(d), i })).sort((x, y) => x.abs - y.abs)\n const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs))\n const ranks: number[] = new Array(n)\n for (let k = 0; k < n; k++) ranks[order[k]!.i] = midranks[k]!\n\n let wPlus = 0\n for (let k = 0; k < n; k++) if (diffs[k]! > 0) wPlus += ranks[k]!\n\n // Midranks are integers or halves; doubling makes the convolution integral.\n // Σ midranks = n(n+1)/2 whatever the tie pattern, so E[W⁺] = n(n+1)/4 and\n // the doubled centre is n(n+1)/2.\n const doubled = midranks.map((rank) => Math.round(rank * 2))\n const doubledDeviation = Math.abs(2 * wPlus - (n * (n + 1)) / 2)\n\n const designFloor = Math.min(1, 2 ** (1 - n))\n const method = selectRankTestMethod(\n 'wilcoxonSignedRank',\n opts.method ?? 'auto',\n `n=${n} non-zero differences`,\n n <= WILCOXON_EXACT_MAX_N,\n designFloor,\n `${WILCOXON_EXACT_MAX_N} non-zero differences`,\n )\n\n if (method === 'exact') {\n const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation)\n return { w: wPlus, p, method, pFloor, nNonZero: n }\n }\n\n if (method === 'asymptotic') {\n const variance = (n * (n + 1) * (2 * n + 1)) / 24 - tieTerm / 48\n return {\n w: wPlus,\n p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),\n method,\n pFloor: designFloor,\n nNonZero: n,\n }\n }\n\n const permutations = resolvePermutations('wilcoxonSignedRank', opts.permutations)\n const rng = makeRng(opts.seed, before, after)\n const doubledCentre = (n * (n + 1)) / 2\n let atLeastAsExtreme = 0\n for (let iteration = 0; iteration < permutations; iteration++) {\n let doubledWPlus = 0\n for (let k = 0; k < n; k++) if (rng() < 0.5) doubledWPlus += doubled[k]!\n if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++\n }\n return {\n w: wPlus,\n p: (1 + atLeastAsExtreme) / (permutations + 1),\n method,\n pFloor: Math.max(1 / (permutations + 1), designFloor),\n nNonZero: n,\n }\n}\n\n/**\n * Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie\n * groups of size `t` — the correction term both asymptotic rank-test variances\n * need.\n */\nfunction midranksWithTieTerm(sorted: readonly number[]): { midranks: number[]; tieTerm: number } {\n const midranks = new Array<number>(sorted.length)\n let tieTerm = 0\n let i = 0\n while (i < sorted.length) {\n let j = i\n while (j < sorted.length && sorted[j] === sorted[i]) j++\n const average = (i + 1 + j) / 2\n for (let k = i; k < j; k++) midranks[k] = average\n const groupSize = j - i\n if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize\n i = j\n }\n return { midranks, tieTerm }\n}\n\nfunction selectRankTestMethod(\n fn: string,\n request: RankTestMethodRequest,\n design: string,\n exactFeasible: boolean,\n designFloor: number,\n threshold: string,\n): RankTestMethod {\n if (request === 'auto') return exactFeasible ? 'exact' : 'permutation'\n if (request === 'exact') {\n if (exactFeasible) return 'exact'\n throw new ValidationError(\n `${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ` +\n `${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges ` +\n 'to the same answer.',\n )\n }\n if (exactFeasible) {\n throw new ValidationError(\n `${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design ` +\n `starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no ` +\n `attainable outcome. Use method 'exact' (the default) or add repetitions past ` +\n `${threshold}.`,\n )\n }\n return 'asymptotic'\n}\n\nfunction resolvePermutations(fn: string, permutations: number | undefined): number {\n if (permutations === undefined) return DEFAULT_PERMUTATIONS\n if (!Number.isInteger(permutations) || permutations < 1) {\n throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`)\n }\n return permutations\n}\n\n/** Two-sided normal-approximation tail with the continuity correction. */\nfunction asymptoticTwoSidedP(deviation: number, sigma: number): number {\n if (!(sigma > 0)) return 1\n return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - 0.5) / sigma)))\n}\n\n/** SD of U under the permutation null, corrected for the realised ties. The\n * tie term reduces (N+1) and reaches it exactly when every value is tied, so\n * the variance floors at 0 rather than going negative. */\nfunction twoSampleSigma(n1: number, n2: number, total: number, tieTerm: number): number {\n if (total < 2) return 0\n const variance = ((n1 * n2) / 12) * (total + 1 - tieTerm / (total * (total - 1)))\n return Math.sqrt(Math.max(0, variance))\n}\n\nfunction logChoose(n: number, k: number): number {\n return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1)\n}\n\nfunction formatProbability(value: number): string {\n return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3)\n}\n\n/**\n * Exact DP allocation and loop count for this observed rank vector.\n *\n * The smaller arm is sufficient because selecting its complement produces the\n * same two-sided U deviation while using fewer rows in the state table.\n */\nfunction exactTwoSampleCost(\n doubledRanks: readonly number[],\n selectedN: number,\n): { states: number; work: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n let work = 0\n for (let placed = 0; placed < doubledRanks.length; placed++) {\n work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed]! + 1)\n }\n return {\n states: (selectedN + 1) * (maxSum + 1),\n work,\n }\n}\n\n/**\n * Smallest attainable two-sided p under the observed ties.\n *\n * Only subsets with the minimum or maximum rank sum can attain the largest\n * deviation. Their multiplicity is the number of ways to choose within the\n * tie group at each boundary, so this calculation is exact without allocating\n * the full null distribution.\n */\nfunction exactTwoSampleFloor(doubledRanks: readonly number[], selectedN: number): number {\n const total = doubledRanks.length\n const otherN = total - selectedN\n const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0)\n const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0)\n if (minimumSum === maximumSum) return 1\n\n const centre = selectedN * (selectedN + 1) + selectedN * otherN\n const minimumDeviation = Math.abs(minimumSum - centre)\n const maximumDeviation = Math.abs(maximumSum - centre)\n const totalLogWays = logChoose(total, selectedN)\n const minimumMass = Math.exp(\n logExtremeSubsetWays(doubledRanks, selectedN, 'minimum') - totalLogWays,\n )\n const maximumMass = Math.exp(\n logExtremeSubsetWays(doubledRanks, selectedN, 'maximum') - totalLogWays,\n )\n\n if (minimumDeviation > maximumDeviation) return minimumMass\n if (maximumDeviation > minimumDeviation) return maximumMass\n return Math.min(1, minimumMass + maximumMass)\n}\n\nfunction logExtremeSubsetWays(\n sortedRanks: readonly number[],\n selectedN: number,\n side: 'minimum' | 'maximum',\n): number {\n const boundaryIndex = side === 'minimum' ? selectedN - 1 : sortedRanks.length - selectedN\n const boundary = sortedRanks[boundaryIndex]!\n let first = boundaryIndex\n let afterLast = boundaryIndex + 1\n while (first > 0 && sortedRanks[first - 1] === boundary) first--\n while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++\n\n const tieSize = afterLast - first\n const fixed = side === 'minimum' ? first : sortedRanks.length - afterLast\n return logChoose(tieSize, selectedN - fixed)\n}\n\n/**\n * Exact conditional two-sided p for the two-sample rank test.\n *\n * Convolves the observed doubled midranks into the null distribution of group\n * a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the\n * splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the\n * realised multiset makes the tie handling exact rather than a correction.\n *\n * The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −\n * U` and permutes the split set onto itself), so the two-sided p is the mass\n * at least as far from the centre as the observation.\n */\nfunction exactTwoSampleP(\n doubledRanks: readonly number[],\n n1: number,\n n2: number,\n doubledDeviation: number,\n): { p: number; pFloor: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n const width = maxSum + 1\n // ways[k][s] = number of size-k subsets whose doubled rank sum is s.\n const ways: Float64Array[] = Array.from({ length: n1 + 1 }, () => new Float64Array(width))\n ways[0]![0] = 1\n let placed = 0\n for (const rank of doubledRanks) {\n for (let k = Math.min(n1, placed + 1); k >= 1; k--) {\n const from = ways[k - 1]!\n const into = ways[k]!\n for (let sum = maxSum - rank; sum >= 0; sum--) {\n const count = from[sum]!\n if (count !== 0) into[sum + rank]! += count\n }\n }\n placed++\n }\n\n // U = rankSum − n₁(n₁+1)/2, so doubled U = s − n₁(n₁+1), and the doubled\n // centre 2·(n₁n₂/2) is n₁n₂.\n const shift = n1 * (n1 + 1) + n1 * n2\n const chosen = ways[n1]!\n let totalWays = 0\n let extremeWays = 0\n let tailWays = 0\n let maxDeviation = -1\n for (let sum = 0; sum < width; sum++) {\n const count = chosen[sum]!\n if (count === 0) continue\n totalWays += count\n const deviation = Math.abs(sum - shift)\n if (deviation >= doubledDeviation) tailWays += count\n if (deviation > maxDeviation) {\n maxDeviation = deviation\n extremeWays = count\n } else if (deviation === maxDeviation) {\n extremeWays += count\n }\n }\n return { p: tailWays / totalWays, pFloor: extremeWays / totalWays }\n}\n\n/**\n * Exact conditional two-sided p for the paired signed-rank test.\n *\n * Convolves the observed doubled absolute midranks over all `2ⁿ` sign\n * assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the\n * arithmetic. The null is symmetric about `n(n+1)/4`.\n */\nfunction exactSignedRankP(\n doubledRanks: readonly number[],\n doubledDeviation: number,\n): { p: number; pFloor: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n const width = maxSum + 1\n let mass = new Float64Array(width)\n mass[0] = 1\n for (const rank of doubledRanks) {\n const next = new Float64Array(width)\n for (let sum = 0; sum < width; sum++) {\n const probability = mass[sum]!\n if (probability === 0) continue\n next[sum]! += probability * 0.5\n next[sum + rank]! += probability * 0.5\n }\n mass = next\n }\n\n const centre = maxSum / 2\n let tail = 0\n let extreme = 0\n let maxDeviation = -1\n for (let sum = 0; sum < width; sum++) {\n const probability = mass[sum]!\n if (probability === 0) continue\n const deviation = Math.abs(sum - centre)\n if (deviation >= doubledDeviation) tail += probability\n if (deviation > maxDeviation) {\n maxDeviation = deviation\n extreme = probability\n } else if (deviation === maxDeviation) {\n extreme += probability\n }\n }\n return { p: Math.min(1, tail), pFloor: Math.min(1, extreme) }\n}\n","/**\n * Matched-pair arm comparison — \"did the treatment arm beat the baseline arm\n * on the SAME work items?\"\n *\n * An arm A/B over run records is only trustworthy when it is PAIRED: the same\n * task/scenario/seed evaluated under both arms, compared item-by-item, so\n * inter-item difficulty variance cancels instead of masquerading as an arm\n * effect. This module owns the two error-prone steps every consumer otherwise\n * hand-rolls:\n *\n * 1. Pairing — matching rows across arms by `pairKey` (and by `repKey`\n * within multi-rep items), with leftovers REPORTED rather than silently\n * dropped (a silently unbalanced pairing biases every paired statistic\n * downstream). Pairing never keys on outcome content: matching reps by\n * their outcomes deflates discordant-pair counts and makes McNemar\n * anti-conservative, so reps pair only by (`pairKey`, `repKey`) identity.\n * 2. Composition — feeding the matched pairs to the correct paired\n * estimators that already live in `statistics`: `mcnemar` +\n * `pairedRiskDifference` for pass/fail, `pairedBootstrap` +\n * `wilcoxonSignedRank` for continuous metrics. No statistic is\n * re-implemented here.\n *\n * The row shape is deliberately structural — callers project a `RunRecord`\n * (or any record) into `{ pairKey, arm, pass?, metrics? }`. Arm names are\n * caller-supplied parameters; the module ships no domain literal.\n */\n\nimport { ValidationError } from './errors'\nimport type { RunRecord } from './run-record'\nimport type { McNemarResult, PairedBootstrapOptions, PairedBootstrapResult } from './statistics'\nimport {\n mcnemar,\n pairedBootstrap,\n pairedRiskDifference,\n type RiskDifferenceResult,\n wilcoxonSignedRank,\n} from './statistics'\n\n/** One arm observation of one work item. Structural on purpose: callers\n * project their own record type (e.g. a `RunRecord`) into this shape. */\nexport interface PairedArmRow {\n /** Matching key — rows sharing a `pairKey` across both arms form pairs\n * (typically the task/scenario/seed identity). */\n pairKey: string\n /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on\n * every row of a `pairKey` that has more than one rep in either arm; reps\n * then pair only on exact (`pairKey`, `repKey`) match, never on outcome\n * content. Optional when each arm has at most one rep of the item. */\n repKey?: string\n /** Arm label this row was produced under. */\n arm: string\n /** Binary outcome; omit when the comparison has no pass/fail notion. */\n pass?: boolean\n /** Named numeric measurements (score, cost, latency, …). */\n metrics?: Record<string, number>\n}\n\nexport interface PairArmsOptions {\n /** Arm treated as the control side of every pair. */\n baselineArm: string\n /** Arm treated as the treatment side of every pair. */\n treatmentArm: string\n}\n\n/** One matched (baseline, treatment) observation of the same work item. */\nexport interface MatchedPair {\n pairKey: string\n /** 0-based position of this pair within its `pairKey`, ordered by sorted\n * `repKey` (always 0 for a single-rep item). The rep identity itself is on\n * the rows (`baseline.repKey` / `treatment.repKey`). */\n repIndex: number\n baseline: PairedArmRow\n treatment: PairedArmRow\n}\n\nexport interface PairArmsResult {\n /** Matched pairs, ordered by (`pairKey`, `repIndex`). */\n pairs: MatchedPair[]\n /** Baseline rows left without a treatment counterpart — reported, never\n * silently dropped. */\n unpairedBaseline: PairedArmRow[]\n /** Treatment rows left without a baseline counterpart. */\n unpairedTreatment: PairedArmRow[]\n}\n\n/**\n * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.\n *\n * A `pairKey` with at most one row per arm pairs directly, no `repKey`\n * needed. A `pairKey` with multiple reps in either arm requires `repKey` on\n * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)\n * match — pairing is keyed purely on row identity, never on outcome content\n * (outcome-keyed matching deflates discordant counts and biases McNemar), and\n * is therefore independent of input order. Reps whose `repKey` has no\n * counterpart in the other arm, and items present in only one arm, land in\n * the unpaired lists — reported, never truncated.\n *\n * Fail-loud: throws when either named arm has zero rows (an unknown arm\n * name would otherwise read as \"everything unpaired\"), when the two arm\n * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or\n * when a (`pairKey`, arm) group repeats a `repKey` (the match would be\n * ambiguous).\n */\nexport function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult {\n const { baselineArm, treatmentArm } = opts\n if (baselineArm === treatmentArm) {\n throw new ValidationError(\n `pairArms: baselineArm and treatmentArm are both '${baselineArm}' — an arm cannot be compared to itself`,\n )\n }\n\n // arm → pairKey → rows\n const byArm = new Map<string, Map<string, PairedArmRow[]>>()\n const armsSeen = new Set<string>()\n for (const row of rows) {\n armsSeen.add(row.arm)\n if (row.arm !== baselineArm && row.arm !== treatmentArm) continue\n const byKey = byArm.get(row.arm) ?? new Map<string, PairedArmRow[]>()\n const group = byKey.get(row.pairKey) ?? []\n group.push(row)\n byKey.set(row.pairKey, group)\n byArm.set(row.arm, byKey)\n }\n\n for (const arm of [baselineArm, treatmentArm]) {\n if (!byArm.has(arm)) {\n const seen = [...armsSeen].sort().join(', ') || '<none>'\n throw new ValidationError(`pairArms: no rows for arm '${arm}' (arms present: ${seen})`)\n }\n }\n\n const baselineByKey = byArm.get(baselineArm)!\n const treatmentByKey = byArm.get(treatmentArm)!\n\n const allKeys = [...new Set([...baselineByKey.keys(), ...treatmentByKey.keys()])].sort()\n const pairs: MatchedPair[] = []\n const unpairedBaseline: PairedArmRow[] = []\n const unpairedTreatment: PairedArmRow[] = []\n for (const pairKey of allKeys) {\n const b = baselineByKey.get(pairKey) ?? []\n const t = treatmentByKey.get(pairKey) ?? []\n\n if (b.length <= 1 && t.length <= 1) {\n if (b.length === 1 && t.length === 1) {\n const baseline = b[0]!\n const treatment = t[0]!\n if (baseline.repKey !== undefined || treatment.repKey !== undefined) {\n if (\n baseline.repKey === undefined ||\n treatment.repKey === undefined ||\n baseline.repKey !== treatment.repKey\n ) {\n unpairedBaseline.push(baseline)\n unpairedTreatment.push(treatment)\n continue\n }\n }\n pairs.push({ pairKey, repIndex: 0, baseline, treatment })\n } else {\n unpairedBaseline.push(...b)\n unpairedTreatment.push(...t)\n }\n continue\n }\n\n const bByRep = indexByRepKey(b, pairKey, baselineArm)\n const tByRep = indexByRepKey(t, pairKey, treatmentArm)\n const repKeys = [...new Set([...bByRep.keys(), ...tByRep.keys()])].sort()\n let repIndex = 0\n for (const repKey of repKeys) {\n const baseline = bByRep.get(repKey)\n const treatment = tByRep.get(repKey)\n if (baseline !== undefined && treatment !== undefined) {\n pairs.push({ pairKey, repIndex: repIndex++, baseline, treatment })\n } else if (baseline !== undefined) {\n unpairedBaseline.push(baseline)\n } else if (treatment !== undefined) {\n unpairedTreatment.push(treatment)\n }\n }\n }\n\n return { pairs, unpairedBaseline, unpairedTreatment }\n}\n\n/** Index a multi-rep (pairKey, arm) group by `repKey`, enforcing that every\n * row carries one and that no repKey repeats within the group. */\nfunction indexByRepKey(\n group: readonly PairedArmRow[],\n pairKey: string,\n arm: string,\n): Map<string, PairedArmRow> {\n const byRep = new Map<string, PairedArmRow>()\n for (const row of group) {\n if (row.repKey === undefined) {\n throw new ValidationError(\n `pairArms: pairKey '${pairKey}' has multiple reps in an arm, but a row in arm '${arm}' ` +\n `is missing repKey — multi-rep items require an explicit repKey on every row so reps ` +\n `pair by identity (pairing reps by outcome or by index would bias the paired statistics)`,\n )\n }\n if (byRep.has(row.repKey)) {\n throw new ValidationError(\n `pairArms: duplicate repKey '${row.repKey}' for pairKey '${pairKey}' in arm '${arm}' — ` +\n `(pairKey, repKey) must uniquely identify a rep within an arm`,\n )\n }\n byRep.set(row.repKey, row)\n }\n return byRep\n}\n\n/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */\nexport interface PairedCorrectness {\n /** Discordant pairs where the treatment passed and the baseline failed. */\n b10: number\n /** Discordant pairs where the baseline passed and the treatment failed. */\n b01: number\n /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */\n mcnemar: McNemarResult\n /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */\n riskDifference: RiskDifferenceResult\n}\n\n/** Paired delta summary for one named metric (delta = treatment − baseline). */\nexport interface PairedMetricDelta {\n name: string\n /** Pairs where BOTH sides carry a finite value for this metric. */\n n: number\n /** Pairs where at least one side does not carry the metric. */\n nMissing: number\n /** Median paired delta, or null when `n === 0`. */\n medianDelta: number | null\n /** Mean paired delta, or null when `n === 0`. */\n meanDelta: number | null\n /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when\n * `n === 0` — a zero-width [0, 0] interval on no data would read as a\n * measured tight null. */\n bootstrapCi: PairedBootstrapResult | null\n /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */\n wilcoxon: { w: number; p: number } | null\n}\n\nexport interface ComparePairedArmsOptions extends PairArmsOptions {\n /** Metrics to compare. Default: every metric name observed on any matched\n * pair, sorted. A name that appears on no pair is still reported (with\n * `n = 0`) so a misspelled metric is visible instead of vanishing. */\n metricNames?: string[]\n /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */\n bootstrap?: PairedBootstrapOptions\n}\n\nexport interface PairedArmsComparison {\n nPairs: number\n nUnpairedBaseline: number\n nUnpairedTreatment: number\n /** null when no matched pair carries `pass` on both sides — a pass/fail\n * verdict over rows that never measured pass/fail would be fabricated. */\n correctness: PairedCorrectness | null\n metricDeltas: PairedMetricDelta[]\n}\n\n/**\n * Full matched-pair arm comparison: pair via {@link pairArms}, then compose\n * the paired estimators from `statistics` over the matched pairs.\n *\n * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`\n * is that subset's size); each metric uses only the pairs where both sides\n * carry a finite value for it, with the remainder counted in `nMissing`.\n * Deltas are treatment − baseline throughout.\n *\n * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a\n * non-finite metric value — silently treating corrupt telemetry as \"metric\n * absent\" would misreport it as missing coverage.\n */\nexport function comparePairedArms(\n rows: readonly PairedArmRow[],\n opts: ComparePairedArmsOptions,\n): PairedArmsComparison {\n const { pairs, unpairedBaseline, unpairedTreatment } = pairArms(rows, opts)\n\n let correctness: PairedCorrectness | null = null\n const baselinePass: number[] = []\n const treatmentPass: number[] = []\n for (const pair of pairs) {\n if (pair.baseline.pass === undefined || pair.treatment.pass === undefined) continue\n baselinePass.push(pair.baseline.pass ? 1 : 0)\n treatmentPass.push(pair.treatment.pass ? 1 : 0)\n }\n if (baselinePass.length > 0) {\n const mc = mcnemar(baselinePass, treatmentPass)\n correctness = {\n b10: mc.b,\n b01: mc.c,\n mcnemar: mc,\n riskDifference: pairedRiskDifference(baselinePass, treatmentPass),\n }\n }\n\n const metricNames =\n opts.metricNames ??\n [\n ...new Set(\n pairs.flatMap((p) => [\n ...Object.keys(p.baseline.metrics ?? {}),\n ...Object.keys(p.treatment.metrics ?? {}),\n ]),\n ),\n ].sort()\n\n const metricDeltas: PairedMetricDelta[] = metricNames.map((name) => {\n const before: number[] = []\n const after: number[] = []\n let nMissing = 0\n for (const pair of pairs) {\n const b = metricValue(pair.baseline, name)\n const t = metricValue(pair.treatment, name)\n if (b === undefined || t === undefined) {\n nMissing++\n continue\n }\n before.push(b)\n after.push(t)\n }\n const bootstrapCi = before.length === 0 ? null : pairedBootstrap(before, after, opts.bootstrap)\n return {\n name,\n n: before.length,\n nMissing,\n medianDelta: bootstrapCi?.median ?? null,\n meanDelta: bootstrapCi?.mean ?? null,\n bootstrapCi,\n wilcoxon: before.length === 0 ? null : wilcoxonSignedRank(before, after),\n }\n })\n\n return {\n nPairs: pairs.length,\n nUnpairedBaseline: unpairedBaseline.length,\n nUnpairedTreatment: unpairedTreatment.length,\n correctness,\n metricDeltas,\n }\n}\n\nexport interface MatchedRunRecordPair {\n pairKey: string\n repKey: string\n baseline: RunRecord\n treatment: RunRecord\n}\n\nexport interface PairRunRecordsResult {\n pairs: MatchedRunRecordPair[]\n unpairedBaseline: RunRecord[]\n unpairedTreatment: RunRecord[]\n}\n\ninterface RunRecordArmRow extends PairedArmRow {\n run: RunRecord\n repKey: string\n}\n\n/**\n * Pair two RunRecord arms by the identity of the evaluated work:\n * `(experimentId, scenarioId, seed)`.\n *\n * Falling back to array order, candidate id, or experiment id can compare\n * different tasks and fabricate lift. Duplicate identities throw.\n */\nexport function pairRunRecords(\n baselineRuns: readonly RunRecord[],\n treatmentRuns: readonly RunRecord[],\n): PairRunRecordsResult {\n const baselineRows = runRecordArmRows(baselineRuns, 'baseline')\n const treatmentRows = runRecordArmRows(treatmentRuns, 'treatment')\n validateRunRecordArmRows(baselineRows, 'baseline')\n validateRunRecordArmRows(treatmentRows, 'treatment')\n if (baselineRows.length === 0 || treatmentRows.length === 0) {\n return {\n pairs: [],\n unpairedBaseline: baselineRows.map((row) => row.run),\n unpairedTreatment: treatmentRows.map((row) => row.run),\n }\n }\n\n const result = pairArms([...baselineRows, ...treatmentRows], {\n baselineArm: 'baseline',\n treatmentArm: 'treatment',\n })\n return {\n pairs: result.pairs.map((pair) => {\n const baseline = pair.baseline as RunRecordArmRow\n const treatment = pair.treatment as RunRecordArmRow\n return {\n pairKey: pair.pairKey,\n repKey: baseline.repKey,\n baseline: baseline.run,\n treatment: treatment.run,\n }\n }),\n unpairedBaseline: result.unpairedBaseline.map((row) => (row as RunRecordArmRow).run),\n unpairedTreatment: result.unpairedTreatment.map((row) => (row as RunRecordArmRow).run),\n }\n}\n\nfunction runRecordArmRows(runs: readonly RunRecord[], arm: string): RunRecordArmRow[] {\n return runs.map((run) => {\n const scenarioId = run.scenarioId.trim()\n if (!scenarioId) {\n throw new ValidationError(\n `pairRunRecords: run '${run.runId}' is missing scenarioId; paired comparisons require explicit scenario identity`,\n )\n }\n return {\n pairKey: JSON.stringify([run.experimentId, scenarioId]),\n repKey: String(run.seed),\n arm,\n run,\n }\n })\n}\n\nfunction validateRunRecordArmRows(rows: readonly RunRecordArmRow[], arm: string): void {\n const byPairKey = new Map<string, RunRecordArmRow[]>()\n for (const row of rows) {\n const group = byPairKey.get(row.pairKey) ?? []\n group.push(row)\n byPairKey.set(row.pairKey, group)\n }\n for (const [pairKey, group] of byPairKey) {\n if (group.length > 1) indexByRepKey(group, pairKey, arm)\n }\n}\n\nfunction metricValue(row: PairedArmRow, name: string): number | undefined {\n const v = row.metrics?.[name]\n if (v === undefined) return undefined\n if (!Number.isFinite(v)) {\n throw new ValidationError(\n `comparePairedArms: non-finite value for metric '${name}' on pairKey '${row.pairKey}' (arm '${row.arm}'): ${v}`,\n )\n }\n return v\n}\n"],"mappings":";;;;;;;;;;;AAgCA,SAAgB,OAAO,WAAmB,GAAW,aAAa,KAA0B;CAC1F,IAAI,KAAK,GAAG,OAAO;EAAE,UAAU;EAAG,OAAO;EAAG,OAAO;CAAE;CACrD,IAAI,YAAY,KAAK,YAAY,GAC/B,MAAM,IAAI,MAAM,sBAAsB,UAAU,mBAAmB,EAAE,EAAE;CAEzE,MAAM,IAAI,UAAU,KAAK,IAAI,cAAc,CAAC;CAC5C,MAAM,IAAI,YAAY;CACtB,MAAM,KAAK,IAAI;CACf,MAAM,QAAQ,IAAI,KAAK;CACvB,MAAM,UAAU,IAAI,MAAM,IAAI,MAAM;CACpC,MAAM,OAAQ,IAAI,KAAK,MAAM,KAAK,IAAI,KAAK,MAAM,IAAI,MAAM,CAAC,IAAK;CACjE,OAAO;EACL,UAAU;EACV,OAAO,KAAK,IAAI,GAAG,SAAS,IAAI;EAChC,OAAO,KAAK,IAAI,GAAG,SAAS,IAAI;CAClC;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,sBAAsB,QAAoC;CACxE,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;EACtC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,KAAK,MAAM,GAAG,OAAO;CACjC;CACA,OAAO;AACT;;;;;;;;;;;;;AA8BA,SAAgB,QACd,SACA,WACe;CACf,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MAAM,kCAAkC,QAAQ,OAAO,MAAM,UAAU,OAAO,EAAE;CAE5F,MAAM,IAAI,QAAQ;CAClB,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,cAAc,IAAI;CACxB,MAAM,YAAY,gBAAgB,IAAI,KAAK,KAAK,IAAI,IAAI,CAAC,IAAI,MAAM,IAAI;CACvE,OAAO;EAAE;EAAG;EAAa;EAAG;EAAG;EAAW,QAAQ,qBAAqB,GAAG,CAAC;CAAE;AAC/E;;;;;;;;;;;;;;;;AAmCA,SAAgB,qBACd,SACA,WACA,aAAa,KACS;CACtB,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,+CAA+C,QAAQ,OAAO,MAAM,UAAU,OAAO,EACvF;CAEF,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GAAG,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,GAAG;EAAG,gBAAgB;EAAG,OAAO;EAAG,OAAO;EAAG;CAAW;CAC1F,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,MAAM,IAAI,KAAK;CACrB,MAAM,YAAY,IAAI,KAAK,IAAI,MAAM,IAAI,MAAM,IAAI;CAEnD,MAAM,OADI,UAAU,KAAK,IAAI,cAAc,CAC9B,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,CAAC;CAChD,OAAO;EACL;EACA;EACA;EACA,gBAAgB;EAChB,OAAO,KAAK,IAAI,IAAI,KAAK,IAAI;EAC7B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI;EAC5B;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAsDA,SAAgB,0BACd,SACA,WACA,aAAa,KACc;CAC3B,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,oDAAoD,QAAQ,OAAO,MAAM,UAAU,OAAO,EAC5F;CAEF,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,+DAA+D,YAAY;CAE7F,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GACR,OAAO;EACL,GAAG;EACH,GAAG;EACH,GAAG;EACH,aAAa;EACb,gBAAgB;EAChB,OAAO;EACP,OAAO;EACP;EACA,QAAQ;CACV;CAEF,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,IAAI,IAAI;CACd,MAAM,kBAAkB,IAAI,KAAK;CACjC,MAAM,SAAS,qBAAqB,GAAG,CAAC;CACxC,IAAI,MAAM,GACR,OAAO;EAAE;EAAG;EAAG;EAAG,aAAa;EAAG,gBAAgB;EAAG,OAAO;EAAG,OAAO;EAAG;EAAY;CAAO;CAE9F,MAAM,QAAQ,IAAI;CAClB,MAAM,QAAQ,MAAM,IAAI,IAAI,aAAa,QAAQ,GAAG,GAAG,IAAI,IAAI,CAAC;CAChE,MAAM,SAAS,MAAM,IAAI,IAAI,aAAa,IAAI,QAAQ,GAAG,IAAI,GAAG,IAAI,CAAC;CACrE,MAAM,QAAQ,IAAI;CAClB,OAAO;EACL;EACA;EACA;EACA,aAAa;EACb;EACA,OAAO,KAAK,IAAI,KAAK,IAAI,QAAQ,KAAK,KAAK;EAC3C,OAAO,KAAK,IAAI,IAAI,IAAI,SAAS,KAAK,KAAK;EAC3C;EACA;CACF;AACF;;;;;AAMA,SAAS,aAAa,GAAW,GAAW,GAAmB;CAC7D,IAAI,KAAK,GAAG,OAAO;CACnB,IAAI,KAAK,GAAG,OAAO;CACnB,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,KAAK;EAC3B,MAAM,OAAO,KAAK,MAAM;EACxB,IAAI,0BAA0B,KAAK,GAAG,CAAC,IAAI,GAAG,KAAK;OAC9C,KAAK;CACZ;CACA,QAAQ,KAAK,MAAM;AACrB;;;;;;;;;;AAgCA,SAAS,oBAAoB,GAAW,GAAW,GAAW,OAAuB;CACnF,MAAM,IAAI,IAAI,IAAI;CAClB,MAAM,YAAY,IAAI;CACtB,MAAM,SAAS,EAAE,IAAI,IAAI,SAAS,IAAI,IAAI,IAAI,IAAI;CAClD,MAAM,WAAW,CAAC,IAAI,SAAS,IAAI;CACnC,MAAM,eAAe,SAAS,SAAS,IAAI,YAAY;CACvD,MAAM,OAAO,eAAe,IAAI,KAAK,KAAK,YAAY,IAAI;CAC1D,MAAM,KAAK,CAAC,SAAS,SAAS,IAAI;CAElC,OAAO,KAAK,IAAI,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,KAAK,CAAC,GAAG,KAAK,IAAI,IAAI,IAAI,SAAS,CAAC,CAAC;AAChF;;;AAIA,SAAS,WAAW,GAAW,GAAW,GAAW,OAAuB;CAC1E,MAAM,YAAY,IAAI,IAAI,IAAI;CAE9B,MAAM,WAAW,KAAK,IADZ,oBAAoB,GAAG,GAAG,GAAG,KACb,IAAI,SAAS,IAAI;CAC3C,IAAI,EAAE,WAAW,IAAI;EACnB,IAAI,cAAc,GAAG,OAAO;EAC5B,OAAO,YAAY,IAAI,OAAO,oBAAoB,OAAO;CAC3D;CACA,OAAO,YAAY,KAAK,KAAK,QAAQ;AACvC;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,0BACd,SACA,WACA,aAAa,KACc;CAC3B,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,oDAAoD,QAAQ,OAAO,MAAM,UAAU,OAAO,EAC5F;CAEF,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,+DAA+D,YAAY;CAE7F,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GACR,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,GAAG;EAAG,aAAa;EAAG,gBAAgB;EAAG,OAAO;EAAI,OAAO;EAAG;CAAW;CAEhG,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,kBAAkB,IAAI,KAAK;CACjC,MAAM,IAAI,UAAU,KAAK,IAAI,cAAc,CAAC;CAK5C,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,KAAK;EAC5B,MAAM,OAAO,KAAK,MAAM;EACxB,IAAI,WAAW,GAAG,GAAG,GAAG,GAAG,IAAI,GAAG,KAAK;OAClC,KAAK;CACZ;CACA,MAAM,SAAS,KAAK,MAAM;CAG1B,IAAI,MAAM;CACV,IAAI,MAAM;CACV,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,KAAK;EAC5B,MAAM,OAAO,MAAM,OAAO;EAC1B,IAAI,WAAW,GAAG,GAAG,GAAG,GAAG,IAAI,CAAC,GAAG,MAAM;OACpC,MAAM;CACb;CACA,MAAM,SAAS,MAAM,OAAO;CAE5B,OAAO;EACL;EACA;EACA;EACA,aAAa,IAAI;EACjB;EACA,OAAO,KAAK,IAAI,IAAI,KAAK;EACzB,OAAO,KAAK,IAAI,GAAG,KAAK;EACxB;CACF;AACF;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,kBACd,QACA,OACe;CACf,IAAI,QAAuB;CAC3B,KAAK,MAAM,OAAO,CAAC,QAAQ,KAAK,GAC9B,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,MAAM,IAAI,IAAI;EACd,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;EAChC,IAAI,MAAM,GAAG;EACb,IAAI,IAAI,GAAG,OAAO;EAClB,IAAI,UAAU,MAAM,QAAQ;OACvB,IAAI,MAAM,OAAO,OAAO;CAC/B;CAEF,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,QAAQ,GAAW,GAAW,GAAmB;CAC/D,IAAI,CAAC,OAAO,UAAU,CAAC,KAAK,CAAC,OAAO,UAAU,CAAC,KAAK,CAAC,OAAO,UAAU,CAAC,GACrE,MAAM,IAAI,MAAM,4CAA4C,EAAE,MAAM,EAAE,MAAM,EAAE,EAAE;CAElF,IAAI,IAAI,KAAK,IAAI,KAAK,IAAI,KAAK,IAAI,GACjC,MAAM,IAAI,MAAM,mDAAmD,EAAE,MAAM,EAAE,MAAM,EAAE,EAAE;CAEzF,IAAI,IAAI,IAAI,GAAG,OAAO;CACtB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,IAAI,IAAI,GAAG,KAAK,GAAG,KAAK,QAAQ,IAAI,IAAI;CACrD,OAAO,IAAI;AACb;;;;;;;;;;ACzgBA,SAAgB,UAAU,GAAmB;CAC3C,IAAI,MAAM,GAAG,OAAO;CAEpB,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,IAAI;CAEV,MAAM,SAAS,KAAK,IAAI,CAAC,IAAI,KAAK;CAClC,MAAM,IAAI,KAAK,IAAI,IAAI;CACvB,MAAM,iBAAiB,KAAK,IAAI,MAAM,IAAI,MAAM,IAAI,MAAM,IAAI,MAAM,IAAI,KAAK,IAAI,CAAC,SAAS,MAAM;CAEjG,OAAO,IAAI,IAAI,aAAa,IAAI,IAAI,aAAa;AACnD;;;;ACyBA,MAAa,gCAAgC;;AAE7C,MAAa,8BAA8B;;AAE3C,MAAa,uBAAuB;;AAEpC,MAAa,uBAAuB;;;;;;;;;;;;;AA2BpC,SAAgB,aACd,GACA,GACA,OAAwB,CAAC,GACN;CACnB,mBAAmB,gBAAgB,KAAK,CAAC;CACzC,mBAAmB,gBAAgB,KAAK,CAAC;CAEzC,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,EAAE;CACb,IAAI,OAAO,KAAK,OAAO,GAAG,OAAO;EAAE,GAAG;EAAG,IAAI;EAAG,GAAG;EAAG,QAAQ;EAAS,QAAQ;CAAE;CAEjF,MAAM,QAAQ,KAAK;CACnB,MAAM,WAAW,CACf,GAAG,EAAE,KAAK,OAAO;EAAE;EAAG,OAAO;CAAK,EAAE,GACpC,GAAG,EAAE,KAAK,OAAO;EAAE;EAAG,OAAO;CAAM,EAAE,CACvC,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,IAAI,EAAE,CAAC;CAE1B,MAAM,EAAE,UAAU,YAAY,oBAAoB,SAAS,KAAK,UAAU,MAAM,CAAC,CAAC;CAClF,IAAI,WAAW;CACf,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,KACzB,IAAI,SAAS,EAAE,CAAE,OAAO,YAAY,SAAS;CAG/C,MAAM,KAAK,WAAY,MAAM,KAAK,KAAM;CACxC,MAAM,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;CAGnC,MAAM,UAAU,SAAS,KAAK,SAAS,KAAK,MAAM,OAAO,CAAC,CAAC;CAC3D,MAAM,mBAAmB,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;CAClD,MAAM,YAAY,KAAK,IAAI,IAAI,EAAE;CACjC,MAAM,SAAS,QAAQ;CACvB,MAAM,YAAY,mBAAmB,SAAS,SAAS;CAEvD,MAAM,cAAc,oBAAoB,SAAS,SAAS;CAC1D,MAAM,SAAS,qBACb,gBACA,KAAK,UAAU,QACf,MAAM,GAAG,OAAO,MAChB,UAAU,UAAA,QACR,UAAU,QAAA,MACZ,aACA,GAAG,8BAA8B,eAAe,OAAO,EAAE,cACpD,4BAA4B,eAAe,OAAO,EAAE,aAC3D;CAEA,IAAI,WAAW,SAAS;EACtB,MAAM,EAAE,GAAG,WAAW,gBAAgB,SAAS,WAAW,QAAQ,gBAAgB;EAClF,OAAO;GAAE;GAAG;GAAI;GAAG;GAAQ;EAAO;CACpC;CAEA,IAAI,WAAW,cACb,OAAO;EACL;EACA;EACA,GAAG,oBAAoB,mBAAmB,GAAG,eAAe,IAAI,IAAI,OAAO,OAAO,CAAC;EACnF;EACA,QAAQ;CACV;CAGF,MAAM,eAAe,oBAAoB,gBAAgB,KAAK,YAAY;CAC1E,MAAM,MAAM,KAAK,SAAS,KAAA,IAAY,QAAQ,uBAAuB,GAAG,CAAC,CAAC,IAAI,QAAQ,KAAK,IAAI;CAC/F,IAAI,mBAAmB;CACvB,MAAM,OAAO,CAAC,GAAG,OAAO;CACxB,KAAK,IAAI,YAAY,GAAG,YAAY,cAAc,aAAa;EAC7D,IAAI,iBAAiB;EACrB,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAAK;GAClC,MAAM,OAAO,IAAI,KAAK,MAAM,IAAI,KAAK,QAAQ,EAAE;GAC/C,MAAM,UAAU,KAAK;GACrB,KAAK,QAAQ,KAAK;GAClB,KAAK,KAAK;GACV,kBAAkB;EACpB;EACA,IACE,KAAK,IAAI,iBAAiB,aAAa,YAAY,KAAK,YAAY,MAAM,KAC1E,kBAEA;CAEJ;CACA,MAAM,SAAS,KAAK,IAAI,KAAK,eAAe,IAAI,WAAW;CAC3D,OAAO;EACL;EACA;EACA,GAAG,KAAK,KAAK,IAAI,qBAAqB,eAAe,IAAI,MAAM;EAC/D;EACA;CACF;AACF;;;;;;;;;;;;;;AA6BA,SAAgB,mBACd,QACA,OACA,OAAwB,CAAC,GACC;CAC1B,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,gBACR,6CAA6C,OAAO,OAAO,MAAM,MAAM,OAAO,EAChF;CAEF,mBAAmB,sBAAsB,UAAU,MAAM;CACzD,mBAAmB,sBAAsB,SAAS,KAAK;CAEvD,MAAM,QAAQ,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC,CAAC,CAAC,QAAQ,MAAM,MAAM,CAAC;CACvE,MAAM,IAAI,MAAM;CAChB,IAAI,MAAM,GAAG,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,QAAQ;EAAS,QAAQ;EAAG,UAAU;CAAE;CAE1E,MAAM,QAAQ,MAAM,KAAK,GAAG,OAAO;EAAE,KAAK,KAAK,IAAI,CAAC;EAAG;CAAE,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,MAAM,EAAE,GAAG;CACzF,MAAM,EAAE,UAAU,YAAY,oBAAoB,MAAM,KAAK,UAAU,MAAM,GAAG,CAAC;CACjF,MAAM,QAAkB,IAAI,MAAM,CAAC;CACnC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,MAAM,MAAM,EAAE,CAAE,KAAK,SAAS;CAE1D,IAAI,QAAQ;CACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,IAAI,MAAM,KAAM,GAAG,SAAS,MAAM;CAK9D,MAAM,UAAU,SAAS,KAAK,SAAS,KAAK,MAAM,OAAO,CAAC,CAAC;CAC3D,MAAM,mBAAmB,KAAK,IAAI,IAAI,QAAS,KAAK,IAAI,KAAM,CAAC;CAE/D,MAAM,cAAc,KAAK,IAAI,GAAG,MAAM,IAAI,EAAE;CAC5C,MAAM,SAAS,qBACb,sBACA,KAAK,UAAU,QACf,KAAK,EAAE,wBACP,KAAA,IACA,aACA,yBACF;CAEA,IAAI,WAAW,SAAS;EACtB,MAAM,EAAE,GAAG,WAAW,iBAAiB,SAAS,gBAAgB;EAChE,OAAO;GAAE,GAAG;GAAO;GAAG;GAAQ;GAAQ,UAAU;EAAE;CACpD;CAEA,IAAI,WAAW,cAAc;EAC3B,MAAM,WAAY,KAAK,IAAI,MAAM,IAAI,IAAI,KAAM,KAAK,UAAU;EAC9D,OAAO;GACL,GAAG;GACH,GAAG,oBAAoB,mBAAmB,GAAG,KAAK,KAAK,QAAQ,CAAC;GAChE;GACA,QAAQ;GACR,UAAU;EACZ;CACF;CAEA,MAAM,eAAe,oBAAoB,sBAAsB,KAAK,YAAY;CAChF,MAAM,MAAM,QAAQ,KAAK,MAAM,QAAQ,KAAK;CAC5C,MAAM,gBAAiB,KAAK,IAAI,KAAM;CACtC,IAAI,mBAAmB;CACvB,KAAK,IAAI,YAAY,GAAG,YAAY,cAAc,aAAa;EAC7D,IAAI,eAAe;EACnB,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,IAAI,IAAI,IAAI,IAAK,gBAAgB,QAAQ;EACrE,IAAI,KAAK,IAAI,eAAe,aAAa,KAAK,kBAAkB;CAClE;CACA,OAAO;EACL,GAAG;EACH,IAAI,IAAI,qBAAqB,eAAe;EAC5C;EACA,QAAQ,KAAK,IAAI,KAAK,eAAe,IAAI,WAAW;EACpD,UAAU;CACZ;AACF;;;;;;AAOA,SAAS,oBAAoB,QAAoE;CAC/F,MAAM,WAAW,IAAI,MAAc,OAAO,MAAM;CAChD,IAAI,UAAU;CACd,IAAI,IAAI;CACR,OAAO,IAAI,OAAO,QAAQ;EACxB,IAAI,IAAI;EACR,OAAO,IAAI,OAAO,UAAU,OAAO,OAAO,OAAO,IAAI;EACrD,MAAM,WAAW,IAAI,IAAI,KAAK;EAC9B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,SAAS,KAAK;EAC1C,MAAM,YAAY,IAAI;EACtB,IAAI,YAAY,GAAG,WAAW,aAAa,IAAI;EAC/C,IAAI;CACN;CACA,OAAO;EAAE;EAAU;CAAQ;AAC7B;AAEA,SAAS,qBACP,IACA,SACA,QACA,eACA,aACA,WACgB;CAChB,IAAI,YAAY,QAAQ,OAAO,gBAAgB,UAAU;CACzD,IAAI,YAAY,SAAS;EACvB,IAAI,eAAe,OAAO;EAC1B,MAAM,IAAI,gBACR,GAAG,GAAG,sCAAsC,OAAO,+BAC9C,UAAU,yFAEjB;CACF;CACA,IAAI,eACF,MAAM,IAAI,gBACR,GAAG,GAAG,sCAAsC,OAAO,+CACpC,kBAAkB,WAAW,EAAE,0HAEzC,UAAU,EACjB;CAEF,OAAO;AACT;AAEA,SAAS,oBAAoB,IAAY,cAA0C;CACjF,IAAI,iBAAiB,KAAA,GAAW,OAAO;CACvC,IAAI,CAAC,OAAO,UAAU,YAAY,KAAK,eAAe,GACpD,MAAM,IAAI,gBAAgB,GAAG,GAAG,iDAAiD,cAAc;CAEjG,OAAO;AACT;;AAGA,SAAS,oBAAoB,WAAmB,OAAuB;CACrE,IAAI,EAAE,QAAQ,IAAI,OAAO;CACzB,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,UAAU,KAAK,IAAI,GAAG,YAAY,EAAG,IAAI,KAAK,EAAE;AAC9E;;;;AAKA,SAAS,eAAe,IAAY,IAAY,OAAe,SAAyB;CACtF,IAAI,QAAQ,GAAG,OAAO;CACtB,MAAM,WAAa,KAAK,KAAM,MAAO,QAAQ,IAAI,WAAW,SAAS,QAAQ;CAC7E,OAAO,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,CAAC;AACxC;AAEA,SAAS,UAAU,GAAW,GAAmB;CAC/C,OAAO,QAAQ,IAAI,CAAC,IAAI,QAAQ,IAAI,CAAC,IAAI,QAAQ,IAAI,IAAI,CAAC;AAC5D;AAEA,SAAS,kBAAkB,OAAuB;CAChD,OAAO,SAAS,QAAQ,UAAU,IAAI,MAAM,QAAQ,CAAC,IAAI,MAAM,cAAc,CAAC;AAChF;;;;;;;AAQA,SAAS,mBACP,cACA,WACkC;CAClC,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,IAAI,OAAO;CACX,KAAK,IAAI,SAAS,GAAG,SAAS,aAAa,QAAQ,UACjD,QAAQ,KAAK,IAAI,WAAW,SAAS,CAAC,KAAK,SAAS,aAAa,UAAW;CAE9E,OAAO;EACL,SAAS,YAAY,MAAM,SAAS;EACpC;CACF;AACF;;;;;;;;;AAUA,SAAS,oBAAoB,cAAiC,WAA2B;CACvF,MAAM,QAAQ,aAAa;CAC3B,MAAM,SAAS,QAAQ;CACvB,MAAM,aAAa,aAAa,MAAM,GAAG,SAAS,CAAC,CAAC,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CACvF,MAAM,aAAa,aAAa,MAAM,QAAQ,SAAS,CAAC,CAAC,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC5F,IAAI,eAAe,YAAY,OAAO;CAEtC,MAAM,SAAS,aAAa,YAAY,KAAK,YAAY;CACzD,MAAM,mBAAmB,KAAK,IAAI,aAAa,MAAM;CACrD,MAAM,mBAAmB,KAAK,IAAI,aAAa,MAAM;CACrD,MAAM,eAAe,UAAU,OAAO,SAAS;CAC/C,MAAM,cAAc,KAAK,IACvB,qBAAqB,cAAc,WAAW,SAAS,IAAI,YAC7D;CACA,MAAM,cAAc,KAAK,IACvB,qBAAqB,cAAc,WAAW,SAAS,IAAI,YAC7D;CAEA,IAAI,mBAAmB,kBAAkB,OAAO;CAChD,IAAI,mBAAmB,kBAAkB,OAAO;CAChD,OAAO,KAAK,IAAI,GAAG,cAAc,WAAW;AAC9C;AAEA,SAAS,qBACP,aACA,WACA,MACQ;CACR,MAAM,gBAAgB,SAAS,YAAY,YAAY,IAAI,YAAY,SAAS;CAChF,MAAM,WAAW,YAAY;CAC7B,IAAI,QAAQ;CACZ,IAAI,YAAY,gBAAgB;CAChC,OAAO,QAAQ,KAAK,YAAY,QAAQ,OAAO,UAAU;CACzD,OAAO,YAAY,YAAY,UAAU,YAAY,eAAe,UAAU;CAI9E,OAAO,UAFS,YAAY,OAEF,aADZ,SAAS,YAAY,QAAQ,YAAY,SAAS,UACrB;AAC7C;;;;;;;;;;;;;AAcA,SAAS,gBACP,cACA,IACA,IACA,kBAC+B;CAC/B,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,MAAM,QAAQ,SAAS;CAEvB,MAAM,OAAuB,MAAM,KAAK,EAAE,QAAQ,KAAK,EAAE,SAAS,IAAI,aAAa,KAAK,CAAC;CACzF,KAAK,EAAE,CAAE,KAAK;CACd,IAAI,SAAS;CACb,KAAK,MAAM,QAAQ,cAAc;EAC/B,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,SAAS,CAAC,GAAG,KAAK,GAAG,KAAK;GAClD,MAAM,OAAO,KAAK,IAAI;GACtB,MAAM,OAAO,KAAK;GAClB,KAAK,IAAI,MAAM,SAAS,MAAM,OAAO,GAAG,OAAO;IAC7C,MAAM,QAAQ,KAAK;IACnB,IAAI,UAAU,GAAG,KAAK,MAAM,SAAU;GACxC;EACF;EACA;CACF;CAIA,MAAM,QAAQ,MAAM,KAAK,KAAK,KAAK;CACnC,MAAM,SAAS,KAAK;CACpB,IAAI,YAAY;CAChB,IAAI,cAAc;CAClB,IAAI,WAAW;CACf,IAAI,eAAe;CACnB,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;EACpC,MAAM,QAAQ,OAAO;EACrB,IAAI,UAAU,GAAG;EACjB,aAAa;EACb,MAAM,YAAY,KAAK,IAAI,MAAM,KAAK;EACtC,IAAI,aAAa,kBAAkB,YAAY;EAC/C,IAAI,YAAY,cAAc;GAC5B,eAAe;GACf,cAAc;EAChB,OAAO,IAAI,cAAc,cACvB,eAAe;CAEnB;CACA,OAAO;EAAE,GAAG,WAAW;EAAW,QAAQ,cAAc;CAAU;AACpE;;;;;;;;AASA,SAAS,iBACP,cACA,kBAC+B;CAC/B,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,MAAM,QAAQ,SAAS;CACvB,IAAI,OAAO,IAAI,aAAa,KAAK;CACjC,KAAK,KAAK;CACV,KAAK,MAAM,QAAQ,cAAc;EAC/B,MAAM,OAAO,IAAI,aAAa,KAAK;EACnC,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;GACpC,MAAM,cAAc,KAAK;GACzB,IAAI,gBAAgB,GAAG;GACvB,KAAK,QAAS,cAAc;GAC5B,KAAK,MAAM,SAAU,cAAc;EACrC;EACA,OAAO;CACT;CAEA,MAAM,SAAS,SAAS;CACxB,IAAI,OAAO;CACX,IAAI,UAAU;CACd,IAAI,eAAe;CACnB,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;EACpC,MAAM,cAAc,KAAK;EACzB,IAAI,gBAAgB,GAAG;EACvB,MAAM,YAAY,KAAK,IAAI,MAAM,MAAM;EACvC,IAAI,aAAa,kBAAkB,QAAQ;EAC3C,IAAI,YAAY,cAAc;GAC5B,eAAe;GACf,UAAU;EACZ,OAAO,IAAI,cAAc,cACvB,WAAW;CAEf;CACA,OAAO;EAAE,GAAG,KAAK,IAAI,GAAG,IAAI;EAAG,QAAQ,KAAK,IAAI,GAAG,OAAO;CAAE;AAC9D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtaA,SAAgB,SAAS,MAA+B,MAAuC;CAC7F,MAAM,EAAE,aAAa,iBAAiB;CACtC,IAAI,gBAAgB,cAClB,MAAM,IAAI,gBACR,oDAAoD,YAAY,wCAClE;CAIF,MAAM,wBAAQ,IAAI,IAAyC;CAC3D,MAAM,2BAAW,IAAI,IAAY;CACjC,KAAK,MAAM,OAAO,MAAM;EACtB,SAAS,IAAI,IAAI,GAAG;EACpB,IAAI,IAAI,QAAQ,eAAe,IAAI,QAAQ,cAAc;EACzD,MAAM,QAAQ,MAAM,IAAI,IAAI,GAAG,qBAAK,IAAI,IAA4B;EACpE,MAAM,QAAQ,MAAM,IAAI,IAAI,OAAO,KAAK,CAAC;EACzC,MAAM,KAAK,GAAG;EACd,MAAM,IAAI,IAAI,SAAS,KAAK;EAC5B,MAAM,IAAI,IAAI,KAAK,KAAK;CAC1B;CAEA,KAAK,MAAM,OAAO,CAAC,aAAa,YAAY,GAC1C,IAAI,CAAC,MAAM,IAAI,GAAG,GAEhB,MAAM,IAAI,gBAAgB,8BAA8B,IAAI,mBAD/C,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,KAAK,SACoC,EAAE;CAI1F,MAAM,gBAAgB,MAAM,IAAI,WAAW;CAC3C,MAAM,iBAAiB,MAAM,IAAI,YAAY;CAE7C,MAAM,UAAU,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,cAAc,KAAK,GAAG,GAAG,eAAe,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;CACvF,MAAM,QAAuB,CAAC;CAC9B,MAAM,mBAAmC,CAAC;CAC1C,MAAM,oBAAoC,CAAC;CAC3C,KAAK,MAAM,WAAW,SAAS;EAC7B,MAAM,IAAI,cAAc,IAAI,OAAO,KAAK,CAAC;EACzC,MAAM,IAAI,eAAe,IAAI,OAAO,KAAK,CAAC;EAE1C,IAAI,EAAE,UAAU,KAAK,EAAE,UAAU,GAAG;GAClC,IAAI,EAAE,WAAW,KAAK,EAAE,WAAW,GAAG;IACpC,MAAM,WAAW,EAAE;IACnB,MAAM,YAAY,EAAE;IACpB,IAAI,SAAS,WAAW,KAAA,KAAa,UAAU,WAAW,KAAA,GAEtD;SAAA,SAAS,WAAW,KAAA,KACpB,UAAU,WAAW,KAAA,KACrB,SAAS,WAAW,UAAU,QAC9B;MACA,iBAAiB,KAAK,QAAQ;MAC9B,kBAAkB,KAAK,SAAS;MAChC;KACF;;IAEF,MAAM,KAAK;KAAE;KAAS,UAAU;KAAG;KAAU;IAAU,CAAC;GAC1D,OAAO;IACL,iBAAiB,KAAK,GAAG,CAAC;IAC1B,kBAAkB,KAAK,GAAG,CAAC;GAC7B;GACA;EACF;EAEA,MAAM,SAAS,cAAc,GAAG,SAAS,WAAW;EACpD,MAAM,SAAS,cAAc,GAAG,SAAS,YAAY;EACrD,MAAM,UAAU,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,GAAG,GAAG,OAAO,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;EACxE,IAAI,WAAW;EACf,KAAK,MAAM,UAAU,SAAS;GAC5B,MAAM,WAAW,OAAO,IAAI,MAAM;GAClC,MAAM,YAAY,OAAO,IAAI,MAAM;GACnC,IAAI,aAAa,KAAA,KAAa,cAAc,KAAA,GAC1C,MAAM,KAAK;IAAE;IAAS,UAAU;IAAY;IAAU;GAAU,CAAC;QAC5D,IAAI,aAAa,KAAA,GACtB,iBAAiB,KAAK,QAAQ;QACzB,IAAI,cAAc,KAAA,GACvB,kBAAkB,KAAK,SAAS;EAEpC;CACF;CAEA,OAAO;EAAE;EAAO;EAAkB;CAAkB;AACtD;;;AAIA,SAAS,cACP,OACA,SACA,KAC2B;CAC3B,MAAM,wBAAQ,IAAI,IAA0B;CAC5C,KAAK,MAAM,OAAO,OAAO;EACvB,IAAI,IAAI,WAAW,KAAA,GACjB,MAAM,IAAI,gBACR,sBAAsB,QAAQ,mDAAmD,IAAI,8KAGvF;EAEF,IAAI,MAAM,IAAI,IAAI,MAAM,GACtB,MAAM,IAAI,gBACR,+BAA+B,IAAI,OAAO,iBAAiB,QAAQ,YAAY,IAAI,iEAErF;EAEF,MAAM,IAAI,IAAI,QAAQ,GAAG;CAC3B;CACA,OAAO;AACT;;;;;;;;;;;;;;AAiEA,SAAgB,kBACd,MACA,MACsB;CACtB,MAAM,EAAE,OAAO,kBAAkB,sBAAsB,SAAS,MAAM,IAAI;CAE1E,IAAI,cAAwC;CAC5C,MAAM,eAAyB,CAAC;CAChC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,SAAS,SAAS,KAAA,KAAa,KAAK,UAAU,SAAS,KAAA,GAAW;EAC3E,aAAa,KAAK,KAAK,SAAS,OAAO,IAAI,CAAC;EAC5C,cAAc,KAAK,KAAK,UAAU,OAAO,IAAI,CAAC;CAChD;CACA,IAAI,aAAa,SAAS,GAAG;EAC3B,MAAM,KAAK,QAAQ,cAAc,aAAa;EAC9C,cAAc;GACZ,KAAK,GAAG;GACR,KAAK,GAAG;GACR,SAAS;GACT,gBAAgB,qBAAqB,cAAc,aAAa;EAClE;CACF;CAaA,MAAM,gBAVJ,KAAK,eACL,CACE,GAAG,IAAI,IACL,MAAM,SAAS,MAAM,CACnB,GAAG,OAAO,KAAK,EAAE,SAAS,WAAW,CAAC,CAAC,GACvC,GAAG,OAAO,KAAK,EAAE,UAAU,WAAW,CAAC,CAAC,CAC1C,CAAC,CACH,CACF,CAAC,CAAC,KAAK,EAAA,CAE6C,KAAK,SAAS;EAClE,MAAM,SAAmB,CAAC;EAC1B,MAAM,QAAkB,CAAC;EACzB,IAAI,WAAW;EACf,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,IAAI,YAAY,KAAK,UAAU,IAAI;GACzC,MAAM,IAAI,YAAY,KAAK,WAAW,IAAI;GAC1C,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAAW;IACtC;IACA;GACF;GACA,OAAO,KAAK,CAAC;GACb,MAAM,KAAK,CAAC;EACd;EACA,MAAM,cAAc,OAAO,WAAW,IAAI,OAAO,gBAAgB,QAAQ,OAAO,KAAK,SAAS;EAC9F,OAAO;GACL;GACA,GAAG,OAAO;GACV;GACA,aAAa,aAAa,UAAU;GACpC,WAAW,aAAa,QAAQ;GAChC;GACA,UAAU,OAAO,WAAW,IAAI,OAAO,mBAAmB,QAAQ,KAAK;EACzE;CACF,CAAC;CAED,OAAO;EACL,QAAQ,MAAM;EACd,mBAAmB,iBAAiB;EACpC,oBAAoB,kBAAkB;EACtC;EACA;CACF;AACF;;;;;;;;AA2BA,SAAgB,eACd,cACA,eACsB;CACtB,MAAM,eAAe,iBAAiB,cAAc,UAAU;CAC9D,MAAM,gBAAgB,iBAAiB,eAAe,WAAW;CACjE,yBAAyB,cAAc,UAAU;CACjD,yBAAyB,eAAe,WAAW;CACnD,IAAI,aAAa,WAAW,KAAK,cAAc,WAAW,GACxD,OAAO;EACL,OAAO,CAAC;EACR,kBAAkB,aAAa,KAAK,QAAQ,IAAI,GAAG;EACnD,mBAAmB,cAAc,KAAK,QAAQ,IAAI,GAAG;CACvD;CAGF,MAAM,SAAS,SAAS,CAAC,GAAG,cAAc,GAAG,aAAa,GAAG;EAC3D,aAAa;EACb,cAAc;CAChB,CAAC;CACD,OAAO;EACL,OAAO,OAAO,MAAM,KAAK,SAAS;GAChC,MAAM,WAAW,KAAK;GACtB,MAAM,YAAY,KAAK;GACvB,OAAO;IACL,SAAS,KAAK;IACd,QAAQ,SAAS;IACjB,UAAU,SAAS;IACnB,WAAW,UAAU;GACvB;EACF,CAAC;EACD,kBAAkB,OAAO,iBAAiB,KAAK,QAAS,IAAwB,GAAG;EACnF,mBAAmB,OAAO,kBAAkB,KAAK,QAAS,IAAwB,GAAG;CACvF;AACF;AAEA,SAAS,iBAAiB,MAA4B,KAAgC;CACpF,OAAO,KAAK,KAAK,QAAQ;EACvB,MAAM,aAAa,IAAI,WAAW,KAAK;EACvC,IAAI,CAAC,YACH,MAAM,IAAI,gBACR,wBAAwB,IAAI,MAAM,+EACpC;EAEF,OAAO;GACL,SAAS,KAAK,UAAU,CAAC,IAAI,cAAc,UAAU,CAAC;GACtD,QAAQ,OAAO,IAAI,IAAI;GACvB;GACA;EACF;CACF,CAAC;AACH;AAEA,SAAS,yBAAyB,MAAkC,KAAmB;CACrF,MAAM,4BAAY,IAAI,IAA+B;CACrD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,UAAU,IAAI,IAAI,OAAO,KAAK,CAAC;EAC7C,MAAM,KAAK,GAAG;EACd,UAAU,IAAI,IAAI,SAAS,KAAK;CAClC;CACA,KAAK,MAAM,CAAC,SAAS,UAAU,WAC7B,IAAI,MAAM,SAAS,GAAG,cAAc,OAAO,SAAS,GAAG;AAE3D;AAEA,SAAS,YAAY,KAAmB,MAAkC;CACxE,MAAM,IAAI,IAAI,UAAU;CACxB,IAAI,MAAM,KAAA,GAAW,OAAO,KAAA;CAC5B,IAAI,CAAC,OAAO,SAAS,CAAC,GACpB,MAAM,IAAI,gBACR,mDAAmD,KAAK,gBAAgB,IAAI,QAAQ,UAAU,IAAI,IAAI,MAAM,GAC9G;CAEF,OAAO;AACT"}
|
|
1
|
+
{"version":3,"file":"paired-arms-D4aeIHUy.js","names":[],"sources":["../src/statistics/paired-binary.ts","../src/math/normal.ts","../src/statistics/rank-tests.ts","../src/paired-arms.ts"],"sourcesContent":["import { regularizedIncompleteBeta } from '../math/special-functions'\nimport { binomialSignTwoSided, zQuantile } from './internal'\n\n// ── Binomial proportion + paired-binary + coding-eval estimators ─────\n//\n// The paired family above (pairedBootstrap/pairedTTest/wilcoxonSignedRank)\n// operates on continuous scores. Pass/fail A/B comparisons — \"does treatment\n// X raise the success RATE vs control\" — are binary and paired, so they need\n// their own correct estimators: McNemar for significance (only the discordant\n// pairs carry signal), the paired risk difference for effect size, Wilson for\n// a single-arm proportion CI, and pass@k for the standard k-sample coding-eval\n// metric. The normal approximation is wrong for proportions near 0/1 and for\n// the small discordant counts typical of eval runs, so these are exact /\n// Wilson-based, not Wald.\n\n/** A binomial proportion estimate with a confidence interval. */\nexport interface ProportionInterval {\n /** Point estimate successes / n (0 when n = 0). */\n estimate: number\n /** Lower bound, clamped to [0, 1]. */\n lower: number\n /** Upper bound, clamped to [0, 1]. */\n upper: number\n}\n\n/**\n * Wilson score interval for a binomial proportion. Correct at small n and near\n * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and\n * understates coverage. Use this for any pass-rate / hit-rate / realness-rate\n * CI — the continuous `confidenceInterval` assumes the wrong distribution for a\n * proportion. `n = 0 ⇒ {0, 0, 0}`.\n */\nexport function wilson(successes: number, n: number, confidence = 0.95): ProportionInterval {\n if (n <= 0) return { estimate: 0, lower: 0, upper: 0 }\n if (successes < 0 || successes > n) {\n throw new Error(`wilson: successes (${successes}) must be in [0, ${n}]`)\n }\n const z = zQuantile(1 - (1 - confidence) / 2)\n const p = successes / n\n const z2 = z * z\n const denom = 1 + z2 / n\n const center = (p + z2 / (2 * n)) / denom\n const half = (z * Math.sqrt((p * (1 - p) + z2 / (4 * n)) / n)) / denom\n return {\n estimate: p,\n lower: Math.max(0, center - half),\n upper: Math.min(1, center + half),\n }\n}\n\n/**\n * Are these per-item outcomes binary (every value exactly 0 or 1)?\n *\n * The discriminator a promotion gate needs before choosing a paired statistic.\n * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is\n * normally dominated by zeros (both arms solve, or both arms miss, most items),\n * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in\n * success rate is — and a bootstrap CI on that median collapses to [0, 0].\n * A gate keying on `ci.low > threshold` is then structurally unable to see\n * either a gain or a regression. Detect this shape and switch to the\n * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})\n * instead of silently answering \"no\" forever.\n *\n * Empty input is NOT binary: there is no evidence of the outcome's shape, and\n * defaulting an empty vector into the binary branch would pick a statistic on\n * no data at all.\n *\n * NOT the right discriminator for a gate. It recognises the literal {0, 1}\n * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which\n * judges in this codebase do routinely — reads as non-binary, and a single\n * partial-credit score in an otherwise pass/fail vector flips it to false while\n * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any\n * two-point encoding). This predicate remains for callers that specifically\n * mean \"literally 0/1\".\n */\nexport function isBinaryOutcomeVector(values: ArrayLike<number>): boolean {\n if (values.length === 0) return false\n for (let i = 0; i < values.length; i++) {\n const v = values[i]!\n if (v !== 0 && v !== 1) return false\n }\n return true\n}\n\n/** Result of a McNemar paired-binary significance test. */\nexport interface McNemarResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs (b + c) — the only ones that carry signal. */\n nDiscordant: number\n /** Pairs where treatment succeeded and control failed (\"newly correct\"). */\n b: number\n /** Pairs where control succeeded and treatment failed (\"newly wrong\"). */\n c: number\n /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */\n statistic: number\n /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */\n pValue: number\n}\n\n/**\n * McNemar's test for paired binary outcomes — the correct significance test for\n * \"does treatment change the success rate vs control on the SAME items\". Only\n * discordant pairs (one arm right, the other wrong) carry information; concordant\n * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw\n * rates is wrong here. The p-value is exact: under H0 the b \"treatment-wins\" are\n * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct\n * at the small discordant counts typical of eval runs (no continuity-corrected\n * chi-square approximation needed, though it is returned as `statistic` for\n * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match\n * the module's (before, after) convention. Throws on unequal lengths.\n */\nexport function mcnemar(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n): McNemarResult {\n if (control.length !== treatment.length) {\n throw new Error(`mcnemar: unequal sample sizes (${control.length} vs ${treatment.length})`)\n }\n const n = control.length\n let b = 0 // treatment 1, control 0\n let c = 0 // treatment 0, control 1\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const nDiscordant = b + c\n const statistic = nDiscordant === 0 ? 0 : (Math.abs(b - c) - 1) ** 2 / nDiscordant\n return { n, nDiscordant, b, c, statistic, pValue: binomialSignTwoSided(b, c) }\n}\n\n/** A paired binary effect size (treatment rate − control rate) with a CI. */\nexport interface RiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Lower bound of the CI, clamped to [-1, 1]. */\n lower: number\n /** Upper bound of the CI, clamped to [-1, 1]. */\n upper: number\n /** Confidence level used. */\n confidence: number\n}\n\n/**\n * Paired risk difference (the effect-size companion to {@link mcnemar}): the\n * change in success rate p(treatment) − p(control) on matched items, which for\n * paired binary data equals (b − c) / n. The CI uses the paired variance from\n * the discordant counts, not the independent-samples formula (which overstates\n * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)\n * arrays, control first. Throws on unequal lengths.\n *\n * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald\n * normal approximation, which badly UNDERCOVERS when only a handful of pairs are\n * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,\n * while McNemar's exact test on the same data gives p = 0.50. A gate keying on\n * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose\n * interval is dual to the exact test by construction, for any decision.\n */\nexport function pairedRiskDifference(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): RiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n const n = control.length\n if (n === 0) return { n: 0, b: 0, c: 0, riskDifference: 0, lower: 0, upper: 0, confidence }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const rd = (b - c) / n\n const variance = (b + c - (b - c) ** 2 / n) / (n * n)\n const z = zQuantile(1 - (1 - confidence) / 2)\n const half = z * Math.sqrt(Math.max(0, variance))\n return {\n n,\n b,\n c,\n riskDifference: rd,\n lower: Math.max(-1, rd - half),\n upper: Math.min(1, rd + half),\n confidence,\n }\n}\n\n/** A paired binary effect size with an EXACT interval and the exact test that\n * bounds it — one object so a caller cannot read the estimate without the\n * significance it is entitled to. */\nexport interface ExactRiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Discordant pairs (b + c) — the only ones carrying information. */\n nDiscordant: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */\n lower: number\n /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */\n upper: number\n /** Confidence level used. */\n confidence: number\n /** McNemar's exact two-sided p-value on the same discordant counts. */\n pValue: number\n}\n\n/**\n * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a\n * promotion gate may decide on.\n *\n * Conditional on the number of discordant pairs m = b + c, the treatment-win\n * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the\n * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a\n * Clopper-Pearson exact interval for π maps straight onto RD. This buys the\n * property the Wald interval in {@link pairedRiskDifference} does not have:\n *\n * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**\n *\n * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial\n * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the\n * interval and the test can never disagree, and a gate keyed on `lower` cannot\n * promote what the exact test refuses. The exact p is returned in the same\n * object so the two are impossible to compute apart.\n *\n * The interval is conservative (exact intervals over-cover; conditioning on m\n * discards the concordant pairs' information about m itself). That is the\n * correct direction for a promotion gate: it refuses more often, never less.\n *\n * With m = 0 there are no discordant pairs and π is not identified: the result\n * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —\n * callers must treat a zero-width interval as \"cannot decide\", not as \"no\n * difference\". Inputs are paired 0/1 (or boolean) arrays, control first.\n * Throws on unequal lengths.\n */\nexport function pairedRiskDifferenceExact(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): ExactRiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifferenceExact: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedRiskDifferenceExact: confidence must be in (0,1), got ${confidence}`)\n }\n const n = control.length\n if (n === 0) {\n return {\n n: 0,\n b: 0,\n c: 0,\n nDiscordant: 0,\n riskDifference: 0,\n lower: 0,\n upper: 0,\n confidence,\n pValue: 1,\n }\n }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const m = b + c\n const riskDifference = (b - c) / n\n const pValue = binomialSignTwoSided(b, c)\n if (m === 0) {\n return { n, b, c, nDiscordant: 0, riskDifference: 0, lower: 0, upper: 0, confidence, pValue }\n }\n const alpha = 1 - confidence\n const piLow = b === 0 ? 0 : betaQuantile(alpha / 2, b, m - b + 1)\n const piHigh = b === m ? 1 : betaQuantile(1 - alpha / 2, b + 1, m - b)\n const scale = m / n\n return {\n n,\n b,\n c,\n nDiscordant: m,\n riskDifference,\n lower: Math.max(-1, (2 * piLow - 1) * scale),\n upper: Math.min(1, (2 * piHigh - 1) * scale),\n confidence,\n pValue,\n }\n}\n\n/** Inverse regularized incomplete beta by bisection on\n * {@link regularizedIncompleteBeta}, which is monotone increasing in x. 80\n * halvings of [0,1] resolve to ~8e-25, far past the continued fraction's own\n * 3e-15 tolerance, so the quantile is as exact as the CDF it inverts. */\nfunction betaQuantile(p: number, a: number, b: number): number {\n if (p <= 0) return 0\n if (p >= 1) return 1\n let lo = 0\n let hi = 1\n for (let i = 0; i < 80; i++) {\n const mid = (lo + hi) / 2\n if (regularizedIncompleteBeta(mid, a, b) < p) lo = mid\n else hi = mid\n }\n return (lo + hi) / 2\n}\n\n/** A paired binary effect size with an interval that is valid at a NONZERO\n * margin — the estimator a noninferiority decision may be made on. */\nexport interface ScoreRiskDifferenceResult {\n /** Total paired observations. */\n n: number\n /** Discordant pairs: treatment-win count. */\n b: number\n /** Discordant pairs: control-win count. */\n c: number\n /** Discordant pairs (b + c). */\n nDiscordant: number\n /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */\n riskDifference: number\n /** Score-interval lower bound on the population risk difference. */\n lower: number\n /** Score-interval upper bound on the population risk difference. */\n upper: number\n /** Confidence level used. */\n confidence: number\n}\n\n/**\n * Constrained MLE of q = P(treatment loses) under the hypothesis RD = `delta`.\n *\n * Profiling the two concordant cells out of the multinomial leaves\n * `L(q) = b·log(q+delta) + c·log(q) + e·log(1 − 2q − delta)` with `e = n − b − c`,\n * whose stationary point is the positive root of\n * `2n·q² − [(b + c) − delta·(b + 3c + 2e)]·q − c·delta·(1 − delta) = 0`.\n * At `delta = 0` this returns `(b + c) / 2n`, the familiar null.\n */\nfunction constrainedLossRate(b: number, c: number, n: number, delta: number): number {\n const e = n - b - c\n const quadratic = 2 * n\n const linear = -(b + c - delta * (b + 3 * c + 2 * e))\n const constant = -c * delta * (1 - delta)\n const discriminant = linear * linear - 4 * quadratic * constant\n const root = discriminant > 0 ? Math.sqrt(discriminant) : 0\n const q = (-linear + root) / (2 * quadratic)\n // Clamp into the region where all four cell probabilities stay non-negative.\n return Math.min(Math.max(q, Math.max(0, -delta)), Math.max(0, (1 - delta) / 2))\n}\n\n/** Tango's score statistic for H0: RD = `delta`. `Var(b − c) = n·(2q + delta −\n * delta²)` under that hypothesis, evaluated at the constrained MLE of q. */\nfunction tangoScore(b: number, c: number, n: number, delta: number): number {\n const numerator = b - c - n * delta\n const q = constrainedLossRate(b, c, n, delta)\n const variance = n * (2 * q + delta * (1 - delta))\n if (!(variance > 0)) {\n if (numerator === 0) return 0\n return numerator > 0 ? Number.POSITIVE_INFINITY : Number.NEGATIVE_INFINITY\n }\n return numerator / Math.sqrt(variance)\n}\n\n/**\n * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a\n * promotion gate may decide on **at a nonzero margin**.\n *\n * {@link pairedRiskDifferenceExact} conditions on the observed discordant count\n * `m = b + c`, builds a Clopper-Pearson interval for the win share among those\n * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing\n * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the\n * population risk difference at a nonzero margin, because the sampling\n * variability of `m/n` itself is discarded. The gap is not academic: with the\n * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk\n * difference sits exactly on that margin clears a nominal-95 % `lower > margin`\n * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates\n * each) when the conditional interval decides.\n *\n * Tango's interval inverts the score test of RD = delta, which estimates the\n * nuisance loss rate under each hypothesised delta instead of fixing it at the\n * observed value, so `m` contributes its own uncertainty. It is the method\n * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it\n * is not conditional, so it stays valid as the margin moves away from zero.\n *\n * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is\n * monotone decreasing in delta, so each crossing is unique. Inputs are paired\n * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.\n */\nexport function pairedRiskDifferenceScore(\n control: ArrayLike<number | boolean>,\n treatment: ArrayLike<number | boolean>,\n confidence = 0.95,\n): ScoreRiskDifferenceResult {\n if (control.length !== treatment.length) {\n throw new Error(\n `pairedRiskDifferenceScore: unequal sample sizes (${control.length} vs ${treatment.length})`,\n )\n }\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedRiskDifferenceScore: confidence must be in (0,1), got ${confidence}`)\n }\n const n = control.length\n if (n === 0) {\n return { n: 0, b: 0, c: 0, nDiscordant: 0, riskDifference: 0, lower: -1, upper: 1, confidence }\n }\n let b = 0\n let c = 0\n for (let i = 0; i < n; i++) {\n const ctrl = control[i] ? 1 : 0\n const treat = treatment[i] ? 1 : 0\n if (treat === 1 && ctrl === 0) b++\n else if (treat === 0 && ctrl === 1) c++\n }\n const riskDifference = (b - c) / n\n const z = zQuantile(1 - (1 - confidence) / 2)\n\n // Lower bound: the smallest delta still inside the interval, i.e. the root of\n // score(delta) = +z on [-1, riskDifference]. score(riskDifference) = 0 < z, so\n // the right endpoint is always inside and the bisection is well posed.\n let lo = -1\n let hi = riskDifference\n for (let i = 0; i < 200; i++) {\n const mid = (lo + hi) / 2\n if (tangoScore(b, c, n, mid) > z) lo = mid\n else hi = mid\n }\n const lower = (lo + hi) / 2\n\n // Upper bound: root of score(delta) = -z on [riskDifference, 1].\n let ulo = riskDifference\n let uhi = 1\n for (let i = 0; i < 200; i++) {\n const mid = (ulo + uhi) / 2\n if (tangoScore(b, c, n, mid) > -z) ulo = mid\n else uhi = mid\n }\n const upper = (ulo + uhi) / 2\n\n return {\n n,\n b,\n c,\n nDiscordant: b + c,\n riskDifference,\n lower: Math.max(-1, lower),\n upper: Math.min(1, upper),\n confidence,\n }\n}\n\n/**\n * The common positive level `s` such that EVERY value across both paired arms is\n * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived\n * in. Returns null when the outcomes are not two-point, when the two arms use\n * different levels, or when no positive value was observed at all (all-zero\n * arms: the level is not identified, and there is nothing to decide anyway).\n *\n * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only\n * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as\n * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),\n * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test\n * silently sends it down the median path that cannot see it. Any positive level\n * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired\n * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by\n * s and rescaling the result back into the caller's native units.\n *\n * Non-finite values ⇒ null: an unusable outcome must not be classified as a\n * clean pass/fail shape.\n */\nexport function pairedBinaryScale(\n before: ArrayLike<number>,\n after: ArrayLike<number>,\n): number | null {\n let level: number | null = null\n for (const arm of [before, after]) {\n for (let i = 0; i < arm.length; i++) {\n const v = arm[i]!\n if (!Number.isFinite(v)) return null\n if (v === 0) continue\n if (v < 0) return null\n if (level === null) level = v\n else if (v !== level) return null\n }\n }\n return level\n}\n\n/**\n * Unbiased pass@k for code generation (Chen et al. 2021, \"Evaluating Large\n * Language Models Trained on Code\"). Given `n` independent samples for one\n * problem of which `c` pass, the probability that at least one of a random k of\n * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as \"did any of the\n * first k pass\" is biased high at small n; this is the variance-reduced estimator\n * averaged implicitly over all k-subsets. Average the per-problem values across\n * the suite for the corpus pass@k. Computed in the numerically stable product\n * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.\n */\nexport function passAtK(n: number, c: number, k: number): number {\n if (!Number.isInteger(n) || !Number.isInteger(c) || !Number.isInteger(k)) {\n throw new Error(`passAtK: n, c, k must be integers (got n=${n}, c=${c}, k=${k})`)\n }\n if (k < 1 || k > n || c < 0 || c > n) {\n throw new Error(`passAtK: require 1 ≤ k ≤ n and 0 ≤ c ≤ n (got n=${n}, c=${c}, k=${k})`)\n }\n if (n - c < k) return 1\n let prob = 1\n for (let i = n - c + 1; i <= n; i++) prob *= 1 - k / i\n return 1 - prob\n}\n","/**\n * Standard normal cumulative distribution using Abramowitz and Stegun 7.1.26.\n *\n * The approximation is evaluated as erf(x / sqrt(2)). Computing the negative\n * tail from the complementary term avoids cancellation when x is far below 0.\n * The maximum absolute CDF error is approximately 7.5e-8.\n */\nexport function normalCdf(x: number): number {\n if (x === 0) return 0.5\n\n const a1 = 0.254829592\n const a2 = -0.284496736\n const a3 = 1.421413741\n const a4 = -1.453152027\n const a5 = 1.061405429\n const p = 0.3275911\n\n const scaled = Math.abs(x) / Math.SQRT2\n const t = 1 / (1 + p * scaled)\n const complement = ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-scaled * scaled)\n\n return x < 0 ? complement / 2 : 1 - complement / 2\n}\n","import { ValidationError } from '../errors'\nimport { normalCdf } from '../math/normal'\nimport { lnGamma } from '../math/special-functions'\nimport { assertFiniteSample, makeRng, symmetricTwoSampleSeed } from './internal'\n\n// ── Rank tests: exact by default ─────────────────────────────────────\n//\n// At 3–10 repetitions per arm the binding constraint on a rank test is\n// combinatorial, not numerical. Three versus three admits only 20 splits, so\n// the attainable two-sided p-grid starts at 0.1000 and α = 0.05 is out of\n// reach at that design; a normal approximation reports 0.0495, a p-value that\n// describes no attainable outcome. No better approximation fixes this — adding\n// the tie correction moves `[0,0,0]` versus `[1,1,1]` from 0.0495 to 0.0469,\n// further from its exact 0.1000, not closer.\n//\n// So both null distributions are enumerated exactly inside bounded state and\n// work budgets, by convolution over the observed midranks (identical to\n// enumerating every split / sign pattern, and cheaper), which conditions on the\n// realised tie pattern for free. Above those budgets the default is a seeded\n// Monte Carlo permutation. The asymptotic path is never chosen automatically,\n// and asking for it inside the exact-feasible range throws.\n//\n// Every result carries `method` and `pFloor` so a downstream gate can SEE the\n// discreteness rather than infer it: a gate handed data whose `pFloor` exceeds\n// its alpha is underpowered by construction, which is a true statement about\n// the experiment, not a false one about the effect.\n\n/** How a rank test's p-value was actually computed. */\nexport type RankTestMethod = 'exact' | 'permutation' | 'asymptotic'\n\n/**\n * What the caller asks for. `'auto'` selects `'exact'` inside the enumeration\n * threshold and `'permutation'` above it, and never selects `'asymptotic'`.\n */\nexport type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic'\n\nexport interface RankTestOptions {\n /** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */\n method?: RankTestMethodRequest\n /** Resamples on the Monte Carlo permutation path. Default 100000. */\n permutations?: number\n /** Seed for the permutation path. Omitted ⇒ derived from the data itself, so\n * the result is reproducible either way. */\n seed?: number\n}\n\n/** Maximum dynamic-programming cells used by an exact two-sample rank test. */\nexport const MANN_WHITNEY_EXACT_MAX_STATES = 8_192\n/** Maximum inner-loop transitions used by an exact two-sample rank test. */\nexport const MANN_WHITNEY_EXACT_MAX_WORK = 250_000\n/** Non-zero differences up to which the signed-rank null is enumerated exactly. */\nexport const WILCOXON_EXACT_MAX_N = 20\n/** Resamples used when a rank test falls back to Monte Carlo permutation. */\nexport const DEFAULT_PERMUTATIONS = 100_000\n\nexport interface MannWhitneyResult {\n /** `min(U_a, U_b)` — the conventional reported statistic. */\n u: number\n /** U for sample `a`. Carries the direction of the effect, which `u` discards. */\n uA: number\n /** Two-sided p-value. */\n p: number\n /** How `p` was computed. */\n method: RankTestMethod\n /** Smallest two-sided p this design can produce. `p` can never be below it. */\n pFloor: number\n}\n\n/**\n * Mann-Whitney U — two independent samples, no distributional assumption.\n *\n * Exact conditional (permutation) p by default when the dynamic program fits\n * {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and\n * {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo\n * permutation above those limits. This keeps imbalanced designs such as 1+24\n * exact without admitting expensive balanced designs merely because they have\n * the same total size. Throws on non-finite input and on `method:\n * 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,\n * pFloor = 1` — no design, no attainable evidence.\n */\nexport function mannWhitneyU(\n a: number[],\n b: number[],\n opts: RankTestOptions = {},\n): MannWhitneyResult {\n assertFiniteSample('mannWhitneyU', 'a', a)\n assertFiniteSample('mannWhitneyU', 'b', b)\n\n const n1 = a.length\n const n2 = b.length\n if (n1 === 0 || n2 === 0) return { u: 0, uA: 0, p: 1, method: 'exact', pFloor: 1 }\n\n const total = n1 + n2\n const combined = [\n ...a.map((v) => ({ v, fromA: true })),\n ...b.map((v) => ({ v, fromA: false })),\n ].sort((x, y) => x.v - y.v)\n\n const { midranks, tieTerm } = midranksWithTieTerm(combined.map((entry) => entry.v))\n let rankSumA = 0\n for (let k = 0; k < total; k++) {\n if (combined[k]!.fromA) rankSumA += midranks[k]!\n }\n\n const uA = rankSumA - (n1 * (n1 + 1)) / 2\n const u = Math.min(uA, n1 * n2 - uA)\n // Midranks are integers or halves, so doubling makes the null convolution\n // integral. U is centred on n₁n₂/2, hence a doubled centre of n₁n₂.\n const doubled = midranks.map((rank) => Math.round(rank * 2))\n const doubledDeviation = Math.abs(2 * uA - n1 * n2)\n const selectedN = Math.min(n1, n2)\n const otherN = total - selectedN\n const exactCost = exactTwoSampleCost(doubled, selectedN)\n\n const designFloor = exactTwoSampleFloor(doubled, selectedN)\n const method = selectRankTestMethod(\n 'mannWhitneyU',\n opts.method ?? 'auto',\n `n1=${n1}, n2=${n2}`,\n exactCost.states <= MANN_WHITNEY_EXACT_MAX_STATES &&\n exactCost.work <= MANN_WHITNEY_EXACT_MAX_WORK,\n designFloor,\n `${MANN_WHITNEY_EXACT_MAX_STATES.toLocaleString('en-US')} states and ` +\n `${MANN_WHITNEY_EXACT_MAX_WORK.toLocaleString('en-US')} transitions`,\n )\n\n if (method === 'exact') {\n const { p, pFloor } = exactTwoSampleP(doubled, selectedN, otherN, doubledDeviation)\n return { u, uA, p, method, pFloor }\n }\n\n if (method === 'asymptotic') {\n return {\n u,\n uA,\n p: asymptoticTwoSidedP(doubledDeviation / 2, twoSampleSigma(n1, n2, total, tieTerm)),\n method,\n pFloor: designFloor,\n }\n }\n\n const permutations = resolvePermutations('mannWhitneyU', opts.permutations)\n const rng = opts.seed === undefined ? makeRng(symmetricTwoSampleSeed(a, b)) : makeRng(opts.seed)\n let atLeastAsExtreme = 0\n const pool = [...doubled]\n for (let iteration = 0; iteration < permutations; iteration++) {\n let doubledRankSum = 0\n for (let k = 0; k < selectedN; k++) {\n const pick = k + Math.floor(rng() * (total - k))\n const swapped = pool[pick]!\n pool[pick] = pool[k]!\n pool[k] = swapped\n doubledRankSum += swapped\n }\n if (\n Math.abs(doubledRankSum - selectedN * (selectedN + 1) - selectedN * otherN) >=\n doubledDeviation\n ) {\n atLeastAsExtreme++\n }\n }\n const pFloor = Math.max(1 / (permutations + 1), designFloor)\n return {\n u,\n uA,\n p: Math.max((1 + atLeastAsExtreme) / (permutations + 1), pFloor),\n method,\n pFloor,\n }\n}\n\nexport interface WilcoxonSignedRankResult {\n /** W⁺, the rank sum of the positive differences. (scipy reports\n * `min(W⁺, W⁻)`; compare statistics only after converting.) */\n w: number\n /** Two-sided p-value. */\n p: number\n /** How `p` was computed. */\n method: RankTestMethod\n /** Smallest two-sided p this design can produce. */\n pFloor: number\n /** Non-zero differences — zero differences are dropped and carry no rank. */\n nNonZero: number\n}\n\n/**\n * Wilcoxon signed-rank — paired, no distributional assumption on the deltas.\n *\n * Exact conditional (sign-flip) p by default at `n ≤\n * {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo\n * permutation above it. Throws on non-finite input and on `method:\n * 'asymptotic'` where an exact answer is available.\n *\n * `n` is the count of NON-ZERO differences: exact ties are dropped before\n * ranking, so a run of tied pairs shrinks the design and raises `pFloor`.\n * All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which\n * `pFloor` states rather than leaving `p = 1` to be read as a measured null.\n */\nexport function wilcoxonSignedRank(\n before: number[],\n after: number[],\n opts: RankTestOptions = {},\n): WilcoxonSignedRankResult {\n if (before.length !== after.length) {\n throw new ValidationError(\n `wilcoxonSignedRank: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n assertFiniteSample('wilcoxonSignedRank', 'before', before)\n assertFiniteSample('wilcoxonSignedRank', 'after', after)\n\n const diffs = before.map((b, i) => after[i]! - b).filter((d) => d !== 0)\n const n = diffs.length\n if (n === 0) return { w: 0, p: 1, method: 'exact', pFloor: 1, nNonZero: 0 }\n\n const order = diffs.map((d, i) => ({ abs: Math.abs(d), i })).sort((x, y) => x.abs - y.abs)\n const { midranks, tieTerm } = midranksWithTieTerm(order.map((entry) => entry.abs))\n const ranks: number[] = new Array(n)\n for (let k = 0; k < n; k++) ranks[order[k]!.i] = midranks[k]!\n\n let wPlus = 0\n for (let k = 0; k < n; k++) if (diffs[k]! > 0) wPlus += ranks[k]!\n\n // Midranks are integers or halves; doubling makes the convolution integral.\n // Σ midranks = n(n+1)/2 whatever the tie pattern, so E[W⁺] = n(n+1)/4 and\n // the doubled centre is n(n+1)/2.\n const doubled = midranks.map((rank) => Math.round(rank * 2))\n const doubledDeviation = Math.abs(2 * wPlus - (n * (n + 1)) / 2)\n\n const designFloor = Math.min(1, 2 ** (1 - n))\n const method = selectRankTestMethod(\n 'wilcoxonSignedRank',\n opts.method ?? 'auto',\n `n=${n} non-zero differences`,\n n <= WILCOXON_EXACT_MAX_N,\n designFloor,\n `${WILCOXON_EXACT_MAX_N} non-zero differences`,\n )\n\n if (method === 'exact') {\n const { p, pFloor } = exactSignedRankP(doubled, doubledDeviation)\n return { w: wPlus, p, method, pFloor, nNonZero: n }\n }\n\n if (method === 'asymptotic') {\n const variance = (n * (n + 1) * (2 * n + 1)) / 24 - tieTerm / 48\n return {\n w: wPlus,\n p: asymptoticTwoSidedP(doubledDeviation / 2, Math.sqrt(variance)),\n method,\n pFloor: designFloor,\n nNonZero: n,\n }\n }\n\n const permutations = resolvePermutations('wilcoxonSignedRank', opts.permutations)\n const rng = makeRng(opts.seed, before, after)\n const doubledCentre = (n * (n + 1)) / 2\n let atLeastAsExtreme = 0\n for (let iteration = 0; iteration < permutations; iteration++) {\n let doubledWPlus = 0\n for (let k = 0; k < n; k++) if (rng() < 0.5) doubledWPlus += doubled[k]!\n if (Math.abs(doubledWPlus - doubledCentre) >= doubledDeviation) atLeastAsExtreme++\n }\n return {\n w: wPlus,\n p: (1 + atLeastAsExtreme) / (permutations + 1),\n method,\n pFloor: Math.max(1 / (permutations + 1), designFloor),\n nNonZero: n,\n }\n}\n\n/**\n * Average ranks over an ASCENDING-sorted array, plus `Σ(t³ − t)` over tie\n * groups of size `t` — the correction term both asymptotic rank-test variances\n * need.\n */\nfunction midranksWithTieTerm(sorted: readonly number[]): { midranks: number[]; tieTerm: number } {\n const midranks = new Array<number>(sorted.length)\n let tieTerm = 0\n let i = 0\n while (i < sorted.length) {\n let j = i\n while (j < sorted.length && sorted[j] === sorted[i]) j++\n const average = (i + 1 + j) / 2\n for (let k = i; k < j; k++) midranks[k] = average\n const groupSize = j - i\n if (groupSize > 1) tieTerm += groupSize ** 3 - groupSize\n i = j\n }\n return { midranks, tieTerm }\n}\n\nfunction selectRankTestMethod(\n fn: string,\n request: RankTestMethodRequest,\n design: string,\n exactFeasible: boolean,\n designFloor: number,\n threshold: string,\n): RankTestMethod {\n if (request === 'auto') return exactFeasible ? 'exact' : 'permutation'\n if (request === 'exact') {\n if (exactFeasible) return 'exact'\n throw new ValidationError(\n `${fn}: method 'exact' is out of range at ${design} — enumeration is bounded by ` +\n `${threshold}. Use 'auto' for the seeded Monte Carlo permutation, which converges ` +\n 'to the same answer.',\n )\n }\n if (exactFeasible) {\n throw new ValidationError(\n `${fn}: method 'asymptotic' is refused at ${design} — the exact p-grid at this design ` +\n `starts at ${formatProbability(designFloor)}, so an asymptotic p below it describes no ` +\n `attainable outcome. Use method 'exact' (the default) or add repetitions past ` +\n `${threshold}.`,\n )\n }\n return 'asymptotic'\n}\n\nfunction resolvePermutations(fn: string, permutations: number | undefined): number {\n if (permutations === undefined) return DEFAULT_PERMUTATIONS\n if (!Number.isInteger(permutations) || permutations < 1) {\n throw new ValidationError(`${fn}: permutations must be a positive integer, got ${permutations}`)\n }\n return permutations\n}\n\n/** Two-sided normal-approximation tail with the continuity correction. */\nfunction asymptoticTwoSidedP(deviation: number, sigma: number): number {\n if (!(sigma > 0)) return 1\n return Math.min(1, 2 * (1 - normalCdf(Math.max(0, deviation - 0.5) / sigma)))\n}\n\n/** SD of U under the permutation null, corrected for the realised ties. The\n * tie term reduces (N+1) and reaches it exactly when every value is tied, so\n * the variance floors at 0 rather than going negative. */\nfunction twoSampleSigma(n1: number, n2: number, total: number, tieTerm: number): number {\n if (total < 2) return 0\n const variance = ((n1 * n2) / 12) * (total + 1 - tieTerm / (total * (total - 1)))\n return Math.sqrt(Math.max(0, variance))\n}\n\nfunction logChoose(n: number, k: number): number {\n return lnGamma(n + 1) - lnGamma(k + 1) - lnGamma(n - k + 1)\n}\n\nfunction formatProbability(value: number): string {\n return value >= 1e-4 || value === 0 ? value.toFixed(4) : value.toExponential(3)\n}\n\n/**\n * Exact DP allocation and loop count for this observed rank vector.\n *\n * The smaller arm is sufficient because selecting its complement produces the\n * same two-sided U deviation while using fewer rows in the state table.\n */\nfunction exactTwoSampleCost(\n doubledRanks: readonly number[],\n selectedN: number,\n): { states: number; work: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n let work = 0\n for (let placed = 0; placed < doubledRanks.length; placed++) {\n work += Math.min(selectedN, placed + 1) * (maxSum - doubledRanks[placed]! + 1)\n }\n return {\n states: (selectedN + 1) * (maxSum + 1),\n work,\n }\n}\n\n/**\n * Smallest attainable two-sided p under the observed ties.\n *\n * Only subsets with the minimum or maximum rank sum can attain the largest\n * deviation. Their multiplicity is the number of ways to choose within the\n * tie group at each boundary, so this calculation is exact without allocating\n * the full null distribution.\n */\nfunction exactTwoSampleFloor(doubledRanks: readonly number[], selectedN: number): number {\n const total = doubledRanks.length\n const otherN = total - selectedN\n const minimumSum = doubledRanks.slice(0, selectedN).reduce((sum, rank) => sum + rank, 0)\n const maximumSum = doubledRanks.slice(total - selectedN).reduce((sum, rank) => sum + rank, 0)\n if (minimumSum === maximumSum) return 1\n\n const centre = selectedN * (selectedN + 1) + selectedN * otherN\n const minimumDeviation = Math.abs(minimumSum - centre)\n const maximumDeviation = Math.abs(maximumSum - centre)\n const totalLogWays = logChoose(total, selectedN)\n const minimumMass = Math.exp(\n logExtremeSubsetWays(doubledRanks, selectedN, 'minimum') - totalLogWays,\n )\n const maximumMass = Math.exp(\n logExtremeSubsetWays(doubledRanks, selectedN, 'maximum') - totalLogWays,\n )\n\n if (minimumDeviation > maximumDeviation) return minimumMass\n if (maximumDeviation > minimumDeviation) return maximumMass\n return Math.min(1, minimumMass + maximumMass)\n}\n\nfunction logExtremeSubsetWays(\n sortedRanks: readonly number[],\n selectedN: number,\n side: 'minimum' | 'maximum',\n): number {\n const boundaryIndex = side === 'minimum' ? selectedN - 1 : sortedRanks.length - selectedN\n const boundary = sortedRanks[boundaryIndex]!\n let first = boundaryIndex\n let afterLast = boundaryIndex + 1\n while (first > 0 && sortedRanks[first - 1] === boundary) first--\n while (afterLast < sortedRanks.length && sortedRanks[afterLast] === boundary) afterLast++\n\n const tieSize = afterLast - first\n const fixed = side === 'minimum' ? first : sortedRanks.length - afterLast\n return logChoose(tieSize, selectedN - fixed)\n}\n\n/**\n * Exact conditional two-sided p for the two-sample rank test.\n *\n * Convolves the observed doubled midranks into the null distribution of group\n * a's rank sum over every `C(n₁+n₂, n₁)` split — identical to enumerating the\n * splits, but `O(N·n₁·ΣR)` instead of `O(C(N,n₁)·n₁n₂)`. Conditioning on the\n * realised multiset makes the tie handling exact rather than a correction.\n *\n * The null is symmetric about `n₁n₂/2` (negating every value maps `U → n₁n₂ −\n * U` and permutes the split set onto itself), so the two-sided p is the mass\n * at least as far from the centre as the observation.\n */\nfunction exactTwoSampleP(\n doubledRanks: readonly number[],\n n1: number,\n n2: number,\n doubledDeviation: number,\n): { p: number; pFloor: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n const width = maxSum + 1\n // ways[k][s] = number of size-k subsets whose doubled rank sum is s.\n const ways: Float64Array[] = Array.from({ length: n1 + 1 }, () => new Float64Array(width))\n ways[0]![0] = 1\n let placed = 0\n for (const rank of doubledRanks) {\n for (let k = Math.min(n1, placed + 1); k >= 1; k--) {\n const from = ways[k - 1]!\n const into = ways[k]!\n for (let sum = maxSum - rank; sum >= 0; sum--) {\n const count = from[sum]!\n if (count !== 0) into[sum + rank]! += count\n }\n }\n placed++\n }\n\n // U = rankSum − n₁(n₁+1)/2, so doubled U = s − n₁(n₁+1), and the doubled\n // centre 2·(n₁n₂/2) is n₁n₂.\n const shift = n1 * (n1 + 1) + n1 * n2\n const chosen = ways[n1]!\n let totalWays = 0\n let extremeWays = 0\n let tailWays = 0\n let maxDeviation = -1\n for (let sum = 0; sum < width; sum++) {\n const count = chosen[sum]!\n if (count === 0) continue\n totalWays += count\n const deviation = Math.abs(sum - shift)\n if (deviation >= doubledDeviation) tailWays += count\n if (deviation > maxDeviation) {\n maxDeviation = deviation\n extremeWays = count\n } else if (deviation === maxDeviation) {\n extremeWays += count\n }\n }\n return { p: tailWays / totalWays, pFloor: extremeWays / totalWays }\n}\n\n/**\n * Exact conditional two-sided p for the paired signed-rank test.\n *\n * Convolves the observed doubled absolute midranks over all `2ⁿ` sign\n * assignments in `O(n·ΣR)`. Probabilities rather than counts keep `2ⁿ` off the\n * arithmetic. The null is symmetric about `n(n+1)/4`.\n */\nfunction exactSignedRankP(\n doubledRanks: readonly number[],\n doubledDeviation: number,\n): { p: number; pFloor: number } {\n const maxSum = doubledRanks.reduce((sum, rank) => sum + rank, 0)\n const width = maxSum + 1\n let mass = new Float64Array(width)\n mass[0] = 1\n for (const rank of doubledRanks) {\n const next = new Float64Array(width)\n for (let sum = 0; sum < width; sum++) {\n const probability = mass[sum]!\n if (probability === 0) continue\n next[sum]! += probability * 0.5\n next[sum + rank]! += probability * 0.5\n }\n mass = next\n }\n\n const centre = maxSum / 2\n let tail = 0\n let extreme = 0\n let maxDeviation = -1\n for (let sum = 0; sum < width; sum++) {\n const probability = mass[sum]!\n if (probability === 0) continue\n const deviation = Math.abs(sum - centre)\n if (deviation >= doubledDeviation) tail += probability\n if (deviation > maxDeviation) {\n maxDeviation = deviation\n extreme = probability\n } else if (deviation === maxDeviation) {\n extreme += probability\n }\n }\n return { p: Math.min(1, tail), pFloor: Math.min(1, extreme) }\n}\n","/**\n * Matched-pair arm comparison — \"did the treatment arm beat the baseline arm\n * on the SAME work items?\"\n *\n * An arm A/B over run records is only trustworthy when it is PAIRED: the same\n * task/scenario/seed evaluated under both arms, compared item-by-item, so\n * inter-item difficulty variance cancels instead of masquerading as an arm\n * effect. This module owns the two error-prone steps every consumer otherwise\n * hand-rolls:\n *\n * 1. Pairing — matching rows across arms by `pairKey` (and by `repKey`\n * within multi-rep items), with leftovers REPORTED rather than silently\n * dropped (a silently unbalanced pairing biases every paired statistic\n * downstream). Pairing never keys on outcome content: matching reps by\n * their outcomes deflates discordant-pair counts and makes McNemar\n * anti-conservative, so reps pair only by (`pairKey`, `repKey`) identity.\n * 2. Composition — feeding the matched pairs to the correct paired\n * estimators that already live in `statistics`: `mcnemar` +\n * `pairedRiskDifference` for pass/fail, `pairedBootstrap` +\n * `wilcoxonSignedRank` for continuous metrics. No statistic is\n * re-implemented here.\n *\n * The row shape is deliberately structural — callers project a `RunRecord`\n * (or any record) into `{ pairKey, arm, pass?, metrics? }`. Arm names are\n * caller-supplied parameters; the module ships no domain literal.\n */\n\nimport { ValidationError } from './errors'\nimport type { RunRecord } from './run-record'\nimport type { McNemarResult, PairedBootstrapOptions, PairedBootstrapResult } from './statistics'\nimport {\n mcnemar,\n pairedBootstrap,\n pairedRiskDifference,\n type RiskDifferenceResult,\n wilcoxonSignedRank,\n} from './statistics'\n\n/** One arm observation of one work item. Structural on purpose: callers\n * project their own record type (e.g. a `RunRecord`) into this shape. */\nexport interface PairedArmRow {\n /** Matching key — rows sharing a `pairKey` across both arms form pairs\n * (typically the task/scenario/seed identity). */\n pairKey: string\n /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on\n * every row of a `pairKey` that has more than one rep in either arm; reps\n * then pair only on exact (`pairKey`, `repKey`) match, never on outcome\n * content. Optional when each arm has at most one rep of the item. */\n repKey?: string\n /** Arm label this row was produced under. */\n arm: string\n /** Binary outcome; omit when the comparison has no pass/fail notion. */\n pass?: boolean\n /** Named numeric measurements (score, cost, latency, …). */\n metrics?: Record<string, number>\n}\n\nexport interface PairArmsOptions {\n /** Arm treated as the control side of every pair. */\n baselineArm: string\n /** Arm treated as the treatment side of every pair. */\n treatmentArm: string\n}\n\n/** One matched (baseline, treatment) observation of the same work item. */\nexport interface MatchedPair {\n pairKey: string\n /** 0-based position of this pair within its `pairKey`, ordered by sorted\n * `repKey` (always 0 for a single-rep item). The rep identity itself is on\n * the rows (`baseline.repKey` / `treatment.repKey`). */\n repIndex: number\n baseline: PairedArmRow\n treatment: PairedArmRow\n}\n\nexport interface PairArmsResult {\n /** Matched pairs, ordered by (`pairKey`, `repIndex`). */\n pairs: MatchedPair[]\n /** Baseline rows left without a treatment counterpart — reported, never\n * silently dropped. */\n unpairedBaseline: PairedArmRow[]\n /** Treatment rows left without a baseline counterpart. */\n unpairedTreatment: PairedArmRow[]\n}\n\n/**\n * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.\n *\n * A `pairKey` with at most one row per arm pairs directly, no `repKey`\n * needed. A `pairKey` with multiple reps in either arm requires `repKey` on\n * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)\n * match — pairing is keyed purely on row identity, never on outcome content\n * (outcome-keyed matching deflates discordant counts and biases McNemar), and\n * is therefore independent of input order. Reps whose `repKey` has no\n * counterpart in the other arm, and items present in only one arm, land in\n * the unpaired lists — reported, never truncated.\n *\n * Fail-loud: throws when either named arm has zero rows (an unknown arm\n * name would otherwise read as \"everything unpaired\"), when the two arm\n * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or\n * when a (`pairKey`, arm) group repeats a `repKey` (the match would be\n * ambiguous).\n */\nexport function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult {\n const { baselineArm, treatmentArm } = opts\n if (baselineArm === treatmentArm) {\n throw new ValidationError(\n `pairArms: baselineArm and treatmentArm are both '${baselineArm}' — an arm cannot be compared to itself`,\n )\n }\n\n // arm → pairKey → rows\n const byArm = new Map<string, Map<string, PairedArmRow[]>>()\n const armsSeen = new Set<string>()\n for (const row of rows) {\n armsSeen.add(row.arm)\n if (row.arm !== baselineArm && row.arm !== treatmentArm) continue\n const byKey = byArm.get(row.arm) ?? new Map<string, PairedArmRow[]>()\n const group = byKey.get(row.pairKey) ?? []\n group.push(row)\n byKey.set(row.pairKey, group)\n byArm.set(row.arm, byKey)\n }\n\n for (const arm of [baselineArm, treatmentArm]) {\n if (!byArm.has(arm)) {\n const seen = [...armsSeen].sort().join(', ') || '<none>'\n throw new ValidationError(`pairArms: no rows for arm '${arm}' (arms present: ${seen})`)\n }\n }\n\n const baselineByKey = byArm.get(baselineArm)!\n const treatmentByKey = byArm.get(treatmentArm)!\n\n const allKeys = [...new Set([...baselineByKey.keys(), ...treatmentByKey.keys()])].sort()\n const pairs: MatchedPair[] = []\n const unpairedBaseline: PairedArmRow[] = []\n const unpairedTreatment: PairedArmRow[] = []\n for (const pairKey of allKeys) {\n const b = baselineByKey.get(pairKey) ?? []\n const t = treatmentByKey.get(pairKey) ?? []\n\n if (b.length <= 1 && t.length <= 1) {\n if (b.length === 1 && t.length === 1) {\n const baseline = b[0]!\n const treatment = t[0]!\n if (baseline.repKey !== undefined || treatment.repKey !== undefined) {\n if (\n baseline.repKey === undefined ||\n treatment.repKey === undefined ||\n baseline.repKey !== treatment.repKey\n ) {\n unpairedBaseline.push(baseline)\n unpairedTreatment.push(treatment)\n continue\n }\n }\n pairs.push({ pairKey, repIndex: 0, baseline, treatment })\n } else {\n unpairedBaseline.push(...b)\n unpairedTreatment.push(...t)\n }\n continue\n }\n\n const bByRep = indexByRepKey(b, pairKey, baselineArm)\n const tByRep = indexByRepKey(t, pairKey, treatmentArm)\n const repKeys = [...new Set([...bByRep.keys(), ...tByRep.keys()])].sort()\n let repIndex = 0\n for (const repKey of repKeys) {\n const baseline = bByRep.get(repKey)\n const treatment = tByRep.get(repKey)\n if (baseline !== undefined && treatment !== undefined) {\n pairs.push({ pairKey, repIndex: repIndex++, baseline, treatment })\n } else if (baseline !== undefined) {\n unpairedBaseline.push(baseline)\n } else if (treatment !== undefined) {\n unpairedTreatment.push(treatment)\n }\n }\n }\n\n return { pairs, unpairedBaseline, unpairedTreatment }\n}\n\n/** Index a multi-rep (pairKey, arm) group by `repKey`, enforcing that every\n * row carries one and that no repKey repeats within the group. */\nfunction indexByRepKey(\n group: readonly PairedArmRow[],\n pairKey: string,\n arm: string,\n): Map<string, PairedArmRow> {\n const byRep = new Map<string, PairedArmRow>()\n for (const row of group) {\n if (row.repKey === undefined) {\n throw new ValidationError(\n `pairArms: pairKey '${pairKey}' has multiple reps in an arm, but a row in arm '${arm}' ` +\n `is missing repKey — multi-rep items require an explicit repKey on every row so reps ` +\n `pair by identity (pairing reps by outcome or by index would bias the paired statistics)`,\n )\n }\n if (byRep.has(row.repKey)) {\n throw new ValidationError(\n `pairArms: duplicate repKey '${row.repKey}' for pairKey '${pairKey}' in arm '${arm}' — ` +\n `(pairKey, repKey) must uniquely identify a rep within an arm`,\n )\n }\n byRep.set(row.repKey, row)\n }\n return byRep\n}\n\n/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */\nexport interface PairedCorrectness {\n /** Discordant pairs where the treatment passed and the baseline failed. */\n b10: number\n /** Discordant pairs where the baseline passed and the treatment failed. */\n b01: number\n /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */\n mcnemar: McNemarResult\n /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */\n riskDifference: RiskDifferenceResult\n}\n\n/** Paired delta summary for one named metric (delta = treatment − baseline). */\nexport interface PairedMetricDelta {\n name: string\n /** Pairs where BOTH sides carry a finite value for this metric. */\n n: number\n /** Pairs where at least one side does not carry the metric. */\n nMissing: number\n /** Median paired delta, or null when `n === 0`. */\n medianDelta: number | null\n /** Mean paired delta, or null when `n === 0`. */\n meanDelta: number | null\n /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when\n * `n === 0` — a zero-width [0, 0] interval on no data would read as a\n * measured tight null. */\n bootstrapCi: PairedBootstrapResult | null\n /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */\n wilcoxon: { w: number; p: number } | null\n}\n\nexport interface ComparePairedArmsOptions extends PairArmsOptions {\n /** Metrics to compare. Default: every metric name observed on any matched\n * pair, sorted. A name that appears on no pair is still reported (with\n * `n = 0`) so a misspelled metric is visible instead of vanishing. */\n metricNames?: string[]\n /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */\n bootstrap?: PairedBootstrapOptions\n}\n\nexport interface PairedArmsComparison {\n nPairs: number\n nUnpairedBaseline: number\n nUnpairedTreatment: number\n /** null when no matched pair carries `pass` on both sides — a pass/fail\n * verdict over rows that never measured pass/fail would be fabricated. */\n correctness: PairedCorrectness | null\n metricDeltas: PairedMetricDelta[]\n}\n\n/**\n * Full matched-pair arm comparison: pair via {@link pairArms}, then compose\n * the paired estimators from `statistics` over the matched pairs.\n *\n * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`\n * is that subset's size); each metric uses only the pairs where both sides\n * carry a finite value for it, with the remainder counted in `nMissing`.\n * Deltas are treatment − baseline throughout.\n *\n * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a\n * non-finite metric value — silently treating corrupt telemetry as \"metric\n * absent\" would misreport it as missing coverage.\n */\nexport function comparePairedArms(\n rows: readonly PairedArmRow[],\n opts: ComparePairedArmsOptions,\n): PairedArmsComparison {\n const { pairs, unpairedBaseline, unpairedTreatment } = pairArms(rows, opts)\n\n let correctness: PairedCorrectness | null = null\n const baselinePass: number[] = []\n const treatmentPass: number[] = []\n for (const pair of pairs) {\n if (pair.baseline.pass === undefined || pair.treatment.pass === undefined) continue\n baselinePass.push(pair.baseline.pass ? 1 : 0)\n treatmentPass.push(pair.treatment.pass ? 1 : 0)\n }\n if (baselinePass.length > 0) {\n const mc = mcnemar(baselinePass, treatmentPass)\n correctness = {\n b10: mc.b,\n b01: mc.c,\n mcnemar: mc,\n riskDifference: pairedRiskDifference(baselinePass, treatmentPass),\n }\n }\n\n const metricNames =\n opts.metricNames ??\n [\n ...new Set(\n pairs.flatMap((p) => [\n ...Object.keys(p.baseline.metrics ?? {}),\n ...Object.keys(p.treatment.metrics ?? {}),\n ]),\n ),\n ].sort()\n\n const metricDeltas: PairedMetricDelta[] = metricNames.map((name) => {\n const before: number[] = []\n const after: number[] = []\n let nMissing = 0\n for (const pair of pairs) {\n const b = metricValue(pair.baseline, name)\n const t = metricValue(pair.treatment, name)\n if (b === undefined || t === undefined) {\n nMissing++\n continue\n }\n before.push(b)\n after.push(t)\n }\n const bootstrapCi = before.length === 0 ? null : pairedBootstrap(before, after, opts.bootstrap)\n return {\n name,\n n: before.length,\n nMissing,\n medianDelta: bootstrapCi?.median ?? null,\n meanDelta: bootstrapCi?.mean ?? null,\n bootstrapCi,\n wilcoxon: before.length === 0 ? null : wilcoxonSignedRank(before, after),\n }\n })\n\n return {\n nPairs: pairs.length,\n nUnpairedBaseline: unpairedBaseline.length,\n nUnpairedTreatment: unpairedTreatment.length,\n correctness,\n metricDeltas,\n }\n}\n\nexport interface MatchedRunRecordPair {\n pairKey: string\n repKey: string\n baseline: RunRecord\n treatment: RunRecord\n}\n\nexport interface PairRunRecordsResult {\n pairs: MatchedRunRecordPair[]\n unpairedBaseline: RunRecord[]\n unpairedTreatment: RunRecord[]\n}\n\ninterface RunRecordArmRow extends PairedArmRow {\n run: RunRecord\n repKey: string\n}\n\n/**\n * Pair two RunRecord arms by the identity of the evaluated work:\n * `(experimentId, scenarioId, seed)`.\n *\n * Falling back to array order, candidate id, or experiment id can compare\n * different tasks and fabricate lift. Duplicate identities throw.\n */\nexport function pairRunRecords(\n baselineRuns: readonly RunRecord[],\n treatmentRuns: readonly RunRecord[],\n): PairRunRecordsResult {\n const baselineRows = runRecordArmRows(baselineRuns, 'baseline')\n const treatmentRows = runRecordArmRows(treatmentRuns, 'treatment')\n validateRunRecordArmRows(baselineRows, 'baseline')\n validateRunRecordArmRows(treatmentRows, 'treatment')\n if (baselineRows.length === 0 || treatmentRows.length === 0) {\n return {\n pairs: [],\n unpairedBaseline: baselineRows.map((row) => row.run),\n unpairedTreatment: treatmentRows.map((row) => row.run),\n }\n }\n\n const result = pairArms([...baselineRows, ...treatmentRows], {\n baselineArm: 'baseline',\n treatmentArm: 'treatment',\n })\n return {\n pairs: result.pairs.map((pair) => {\n const baseline = pair.baseline as RunRecordArmRow\n const treatment = pair.treatment as RunRecordArmRow\n return {\n pairKey: pair.pairKey,\n repKey: baseline.repKey,\n baseline: baseline.run,\n treatment: treatment.run,\n }\n }),\n unpairedBaseline: result.unpairedBaseline.map((row) => (row as RunRecordArmRow).run),\n unpairedTreatment: result.unpairedTreatment.map((row) => (row as RunRecordArmRow).run),\n }\n}\n\nfunction runRecordArmRows(runs: readonly RunRecord[], arm: string): RunRecordArmRow[] {\n return runs.map((run) => {\n const scenarioId = run.scenarioId.trim()\n if (!scenarioId) {\n throw new ValidationError(\n `pairRunRecords: run '${run.runId}' is missing scenarioId; paired comparisons require explicit scenario identity`,\n )\n }\n return {\n pairKey: JSON.stringify([run.experimentId, scenarioId]),\n repKey: String(run.seed),\n arm,\n run,\n }\n })\n}\n\nfunction validateRunRecordArmRows(rows: readonly RunRecordArmRow[], arm: string): void {\n const byPairKey = new Map<string, RunRecordArmRow[]>()\n for (const row of rows) {\n const group = byPairKey.get(row.pairKey) ?? []\n group.push(row)\n byPairKey.set(row.pairKey, group)\n }\n for (const [pairKey, group] of byPairKey) {\n if (group.length > 1) indexByRepKey(group, pairKey, arm)\n }\n}\n\nfunction metricValue(row: PairedArmRow, name: string): number | undefined {\n const v = row.metrics?.[name]\n if (v === undefined) return undefined\n if (!Number.isFinite(v)) {\n throw new ValidationError(\n `comparePairedArms: non-finite value for metric '${name}' on pairKey '${row.pairKey}' (arm '${row.arm}'): ${v}`,\n )\n }\n return v\n}\n"],"mappings":";;;;;;;;;;;AAgCA,SAAgB,OAAO,WAAmB,GAAW,aAAa,KAA0B;CAC1F,IAAI,KAAK,GAAG,OAAO;EAAE,UAAU;EAAG,OAAO;EAAG,OAAO;CAAE;CACrD,IAAI,YAAY,KAAK,YAAY,GAC/B,MAAM,IAAI,MAAM,sBAAsB,UAAU,mBAAmB,EAAE,EAAE;CAEzE,MAAM,IAAI,UAAU,KAAK,IAAI,cAAc,CAAC;CAC5C,MAAM,IAAI,YAAY;CACtB,MAAM,KAAK,IAAI;CACf,MAAM,QAAQ,IAAI,KAAK;CACvB,MAAM,UAAU,IAAI,MAAM,IAAI,MAAM;CACpC,MAAM,OAAQ,IAAI,KAAK,MAAM,KAAK,IAAI,KAAK,MAAM,IAAI,MAAM,CAAC,IAAK;CACjE,OAAO;EACL,UAAU;EACV,OAAO,KAAK,IAAI,GAAG,SAAS,IAAI;EAChC,OAAO,KAAK,IAAI,GAAG,SAAS,IAAI;CAClC;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,sBAAsB,QAAoC;CACxE,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;EACtC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,KAAK,MAAM,GAAG,OAAO;CACjC;CACA,OAAO;AACT;;;;;;;;;;;;;AA8BA,SAAgB,QACd,SACA,WACe;CACf,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MAAM,kCAAkC,QAAQ,OAAO,MAAM,UAAU,OAAO,EAAE;CAE5F,MAAM,IAAI,QAAQ;CAClB,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,cAAc,IAAI;CACxB,MAAM,YAAY,gBAAgB,IAAI,KAAK,KAAK,IAAI,IAAI,CAAC,IAAI,MAAM,IAAI;CACvE,OAAO;EAAE;EAAG;EAAa;EAAG;EAAG;EAAW,QAAQ,qBAAqB,GAAG,CAAC;CAAE;AAC/E;;;;;;;;;;;;;;;;AAmCA,SAAgB,qBACd,SACA,WACA,aAAa,KACS;CACtB,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,+CAA+C,QAAQ,OAAO,MAAM,UAAU,OAAO,EACvF;CAEF,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GAAG,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,GAAG;EAAG,gBAAgB;EAAG,OAAO;EAAG,OAAO;EAAG;CAAW;CAC1F,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,MAAM,IAAI,KAAK;CACrB,MAAM,YAAY,IAAI,KAAK,IAAI,MAAM,IAAI,MAAM,IAAI;CAEnD,MAAM,OADI,UAAU,KAAK,IAAI,cAAc,CAC9B,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,CAAC;CAChD,OAAO;EACL;EACA;EACA;EACA,gBAAgB;EAChB,OAAO,KAAK,IAAI,IAAI,KAAK,IAAI;EAC7B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI;EAC5B;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAsDA,SAAgB,0BACd,SACA,WACA,aAAa,KACc;CAC3B,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,oDAAoD,QAAQ,OAAO,MAAM,UAAU,OAAO,EAC5F;CAEF,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,+DAA+D,YAAY;CAE7F,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GACR,OAAO;EACL,GAAG;EACH,GAAG;EACH,GAAG;EACH,aAAa;EACb,gBAAgB;EAChB,OAAO;EACP,OAAO;EACP;EACA,QAAQ;CACV;CAEF,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,IAAI,IAAI;CACd,MAAM,kBAAkB,IAAI,KAAK;CACjC,MAAM,SAAS,qBAAqB,GAAG,CAAC;CACxC,IAAI,MAAM,GACR,OAAO;EAAE;EAAG;EAAG;EAAG,aAAa;EAAG,gBAAgB;EAAG,OAAO;EAAG,OAAO;EAAG;EAAY;CAAO;CAE9F,MAAM,QAAQ,IAAI;CAClB,MAAM,QAAQ,MAAM,IAAI,IAAI,aAAa,QAAQ,GAAG,GAAG,IAAI,IAAI,CAAC;CAChE,MAAM,SAAS,MAAM,IAAI,IAAI,aAAa,IAAI,QAAQ,GAAG,IAAI,GAAG,IAAI,CAAC;CACrE,MAAM,QAAQ,IAAI;CAClB,OAAO;EACL;EACA;EACA;EACA,aAAa;EACb;EACA,OAAO,KAAK,IAAI,KAAK,IAAI,QAAQ,KAAK,KAAK;EAC3C,OAAO,KAAK,IAAI,IAAI,IAAI,SAAS,KAAK,KAAK;EAC3C;EACA;CACF;AACF;;;;;AAMA,SAAS,aAAa,GAAW,GAAW,GAAmB;CAC7D,IAAI,KAAK,GAAG,OAAO;CACnB,IAAI,KAAK,GAAG,OAAO;CACnB,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,KAAK;EAC3B,MAAM,OAAO,KAAK,MAAM;EACxB,IAAI,0BAA0B,KAAK,GAAG,CAAC,IAAI,GAAG,KAAK;OAC9C,KAAK;CACZ;CACA,QAAQ,KAAK,MAAM;AACrB;;;;;;;;;;AAgCA,SAAS,oBAAoB,GAAW,GAAW,GAAW,OAAuB;CACnF,MAAM,IAAI,IAAI,IAAI;CAClB,MAAM,YAAY,IAAI;CACtB,MAAM,SAAS,EAAE,IAAI,IAAI,SAAS,IAAI,IAAI,IAAI,IAAI;CAClD,MAAM,WAAW,CAAC,IAAI,SAAS,IAAI;CACnC,MAAM,eAAe,SAAS,SAAS,IAAI,YAAY;CACvD,MAAM,OAAO,eAAe,IAAI,KAAK,KAAK,YAAY,IAAI;CAC1D,MAAM,KAAK,CAAC,SAAS,SAAS,IAAI;CAElC,OAAO,KAAK,IAAI,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,KAAK,CAAC,GAAG,KAAK,IAAI,IAAI,IAAI,SAAS,CAAC,CAAC;AAChF;;;AAIA,SAAS,WAAW,GAAW,GAAW,GAAW,OAAuB;CAC1E,MAAM,YAAY,IAAI,IAAI,IAAI;CAE9B,MAAM,WAAW,KAAK,IADZ,oBAAoB,GAAG,GAAG,GAAG,KACb,IAAI,SAAS,IAAI;CAC3C,IAAI,EAAE,WAAW,IAAI;EACnB,IAAI,cAAc,GAAG,OAAO;EAC5B,OAAO,YAAY,IAAI,OAAO,oBAAoB,OAAO;CAC3D;CACA,OAAO,YAAY,KAAK,KAAK,QAAQ;AACvC;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,0BACd,SACA,WACA,aAAa,KACc;CAC3B,IAAI,QAAQ,WAAW,UAAU,QAC/B,MAAM,IAAI,MACR,oDAAoD,QAAQ,OAAO,MAAM,UAAU,OAAO,EAC5F;CAEF,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,+DAA+D,YAAY;CAE7F,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GACR,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,GAAG;EAAG,aAAa;EAAG,gBAAgB;EAAG,OAAO;EAAI,OAAO;EAAG;CAAW;CAEhG,IAAI,IAAI;CACR,IAAI,IAAI;CACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,MAAM,OAAO,QAAQ,KAAK,IAAI;EAC9B,MAAM,QAAQ,UAAU,KAAK,IAAI;EACjC,IAAI,UAAU,KAAK,SAAS,GAAG;OAC1B,IAAI,UAAU,KAAK,SAAS,GAAG;CACtC;CACA,MAAM,kBAAkB,IAAI,KAAK;CACjC,MAAM,IAAI,UAAU,KAAK,IAAI,cAAc,CAAC;CAK5C,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,KAAK;EAC5B,MAAM,OAAO,KAAK,MAAM;EACxB,IAAI,WAAW,GAAG,GAAG,GAAG,GAAG,IAAI,GAAG,KAAK;OAClC,KAAK;CACZ;CACA,MAAM,SAAS,KAAK,MAAM;CAG1B,IAAI,MAAM;CACV,IAAI,MAAM;CACV,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,KAAK;EAC5B,MAAM,OAAO,MAAM,OAAO;EAC1B,IAAI,WAAW,GAAG,GAAG,GAAG,GAAG,IAAI,CAAC,GAAG,MAAM;OACpC,MAAM;CACb;CACA,MAAM,SAAS,MAAM,OAAO;CAE5B,OAAO;EACL;EACA;EACA;EACA,aAAa,IAAI;EACjB;EACA,OAAO,KAAK,IAAI,IAAI,KAAK;EACzB,OAAO,KAAK,IAAI,GAAG,KAAK;EACxB;CACF;AACF;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,kBACd,QACA,OACe;CACf,IAAI,QAAuB;CAC3B,KAAK,MAAM,OAAO,CAAC,QAAQ,KAAK,GAC9B,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,MAAM,IAAI,IAAI;EACd,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;EAChC,IAAI,MAAM,GAAG;EACb,IAAI,IAAI,GAAG,OAAO;EAClB,IAAI,UAAU,MAAM,QAAQ;OACvB,IAAI,MAAM,OAAO,OAAO;CAC/B;CAEF,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,QAAQ,GAAW,GAAW,GAAmB;CAC/D,IAAI,CAAC,OAAO,UAAU,CAAC,KAAK,CAAC,OAAO,UAAU,CAAC,KAAK,CAAC,OAAO,UAAU,CAAC,GACrE,MAAM,IAAI,MAAM,4CAA4C,EAAE,MAAM,EAAE,MAAM,EAAE,EAAE;CAElF,IAAI,IAAI,KAAK,IAAI,KAAK,IAAI,KAAK,IAAI,GACjC,MAAM,IAAI,MAAM,mDAAmD,EAAE,MAAM,EAAE,MAAM,EAAE,EAAE;CAEzF,IAAI,IAAI,IAAI,GAAG,OAAO;CACtB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,IAAI,IAAI,GAAG,KAAK,GAAG,KAAK,QAAQ,IAAI,IAAI;CACrD,OAAO,IAAI;AACb;;;;;;;;;;ACzgBA,SAAgB,UAAU,GAAmB;CAC3C,IAAI,MAAM,GAAG,OAAO;CAEpB,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,KAAK;CACX,MAAM,IAAI;CAEV,MAAM,SAAS,KAAK,IAAI,CAAC,IAAI,KAAK;CAClC,MAAM,IAAI,KAAK,IAAI,IAAI;CACvB,MAAM,iBAAiB,KAAK,IAAI,MAAM,IAAI,MAAM,IAAI,MAAM,IAAI,MAAM,IAAI,KAAK,IAAI,CAAC,SAAS,MAAM;CAEjG,OAAO,IAAI,IAAI,aAAa,IAAI,IAAI,aAAa;AACnD;;;;ACyBA,MAAa,gCAAgC;;AAE7C,MAAa,8BAA8B;;AAE3C,MAAa,uBAAuB;;AAEpC,MAAa,uBAAuB;;;;;;;;;;;;;AA2BpC,SAAgB,aACd,GACA,GACA,OAAwB,CAAC,GACN;CACnB,mBAAmB,gBAAgB,KAAK,CAAC;CACzC,mBAAmB,gBAAgB,KAAK,CAAC;CAEzC,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,EAAE;CACb,IAAI,OAAO,KAAK,OAAO,GAAG,OAAO;EAAE,GAAG;EAAG,IAAI;EAAG,GAAG;EAAG,QAAQ;EAAS,QAAQ;CAAE;CAEjF,MAAM,QAAQ,KAAK;CACnB,MAAM,WAAW,CACf,GAAG,EAAE,KAAK,OAAO;EAAE;EAAG,OAAO;CAAK,EAAE,GACpC,GAAG,EAAE,KAAK,OAAO;EAAE;EAAG,OAAO;CAAM,EAAE,CACvC,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,IAAI,EAAE,CAAC;CAE1B,MAAM,EAAE,UAAU,YAAY,oBAAoB,SAAS,KAAK,UAAU,MAAM,CAAC,CAAC;CAClF,IAAI,WAAW;CACf,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,KACzB,IAAI,SAAS,EAAE,CAAE,OAAO,YAAY,SAAS;CAG/C,MAAM,KAAK,WAAY,MAAM,KAAK,KAAM;CACxC,MAAM,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;CAGnC,MAAM,UAAU,SAAS,KAAK,SAAS,KAAK,MAAM,OAAO,CAAC,CAAC;CAC3D,MAAM,mBAAmB,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;CAClD,MAAM,YAAY,KAAK,IAAI,IAAI,EAAE;CACjC,MAAM,SAAS,QAAQ;CACvB,MAAM,YAAY,mBAAmB,SAAS,SAAS;CAEvD,MAAM,cAAc,oBAAoB,SAAS,SAAS;CAC1D,MAAM,SAAS,qBACb,gBACA,KAAK,UAAU,QACf,MAAM,GAAG,OAAO,MAChB,UAAU,UAAA,QACR,UAAU,QAAA,MACZ,aACA,GAAG,8BAA8B,eAAe,OAAO,EAAE,cACpD,4BAA4B,eAAe,OAAO,EAAE,aAC3D;CAEA,IAAI,WAAW,SAAS;EACtB,MAAM,EAAE,GAAG,WAAW,gBAAgB,SAAS,WAAW,QAAQ,gBAAgB;EAClF,OAAO;GAAE;GAAG;GAAI;GAAG;GAAQ;EAAO;CACpC;CAEA,IAAI,WAAW,cACb,OAAO;EACL;EACA;EACA,GAAG,oBAAoB,mBAAmB,GAAG,eAAe,IAAI,IAAI,OAAO,OAAO,CAAC;EACnF;EACA,QAAQ;CACV;CAGF,MAAM,eAAe,oBAAoB,gBAAgB,KAAK,YAAY;CAC1E,MAAM,MAAM,KAAK,SAAS,KAAA,IAAY,QAAQ,uBAAuB,GAAG,CAAC,CAAC,IAAI,QAAQ,KAAK,IAAI;CAC/F,IAAI,mBAAmB;CACvB,MAAM,OAAO,CAAC,GAAG,OAAO;CACxB,KAAK,IAAI,YAAY,GAAG,YAAY,cAAc,aAAa;EAC7D,IAAI,iBAAiB;EACrB,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAAK;GAClC,MAAM,OAAO,IAAI,KAAK,MAAM,IAAI,KAAK,QAAQ,EAAE;GAC/C,MAAM,UAAU,KAAK;GACrB,KAAK,QAAQ,KAAK;GAClB,KAAK,KAAK;GACV,kBAAkB;EACpB;EACA,IACE,KAAK,IAAI,iBAAiB,aAAa,YAAY,KAAK,YAAY,MAAM,KAC1E,kBAEA;CAEJ;CACA,MAAM,SAAS,KAAK,IAAI,KAAK,eAAe,IAAI,WAAW;CAC3D,OAAO;EACL;EACA;EACA,GAAG,KAAK,KAAK,IAAI,qBAAqB,eAAe,IAAI,MAAM;EAC/D;EACA;CACF;AACF;;;;;;;;;;;;;;AA6BA,SAAgB,mBACd,QACA,OACA,OAAwB,CAAC,GACC;CAC1B,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,gBACR,6CAA6C,OAAO,OAAO,MAAM,MAAM,OAAO,EAChF;CAEF,mBAAmB,sBAAsB,UAAU,MAAM;CACzD,mBAAmB,sBAAsB,SAAS,KAAK;CAEvD,MAAM,QAAQ,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC,CAAC,CAAC,QAAQ,MAAM,MAAM,CAAC;CACvE,MAAM,IAAI,MAAM;CAChB,IAAI,MAAM,GAAG,OAAO;EAAE,GAAG;EAAG,GAAG;EAAG,QAAQ;EAAS,QAAQ;EAAG,UAAU;CAAE;CAE1E,MAAM,QAAQ,MAAM,KAAK,GAAG,OAAO;EAAE,KAAK,KAAK,IAAI,CAAC;EAAG;CAAE,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,MAAM,EAAE,GAAG;CACzF,MAAM,EAAE,UAAU,YAAY,oBAAoB,MAAM,KAAK,UAAU,MAAM,GAAG,CAAC;CACjF,MAAM,QAAkB,IAAI,MAAM,CAAC;CACnC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,MAAM,MAAM,EAAE,CAAE,KAAK,SAAS;CAE1D,IAAI,QAAQ;CACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,IAAI,MAAM,KAAM,GAAG,SAAS,MAAM;CAK9D,MAAM,UAAU,SAAS,KAAK,SAAS,KAAK,MAAM,OAAO,CAAC,CAAC;CAC3D,MAAM,mBAAmB,KAAK,IAAI,IAAI,QAAS,KAAK,IAAI,KAAM,CAAC;CAE/D,MAAM,cAAc,KAAK,IAAI,GAAG,MAAM,IAAI,EAAE;CAC5C,MAAM,SAAS,qBACb,sBACA,KAAK,UAAU,QACf,KAAK,EAAE,wBACP,KAAA,IACA,aACA,yBACF;CAEA,IAAI,WAAW,SAAS;EACtB,MAAM,EAAE,GAAG,WAAW,iBAAiB,SAAS,gBAAgB;EAChE,OAAO;GAAE,GAAG;GAAO;GAAG;GAAQ;GAAQ,UAAU;EAAE;CACpD;CAEA,IAAI,WAAW,cAAc;EAC3B,MAAM,WAAY,KAAK,IAAI,MAAM,IAAI,IAAI,KAAM,KAAK,UAAU;EAC9D,OAAO;GACL,GAAG;GACH,GAAG,oBAAoB,mBAAmB,GAAG,KAAK,KAAK,QAAQ,CAAC;GAChE;GACA,QAAQ;GACR,UAAU;EACZ;CACF;CAEA,MAAM,eAAe,oBAAoB,sBAAsB,KAAK,YAAY;CAChF,MAAM,MAAM,QAAQ,KAAK,MAAM,QAAQ,KAAK;CAC5C,MAAM,gBAAiB,KAAK,IAAI,KAAM;CACtC,IAAI,mBAAmB;CACvB,KAAK,IAAI,YAAY,GAAG,YAAY,cAAc,aAAa;EAC7D,IAAI,eAAe;EACnB,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,IAAI,IAAI,IAAI,IAAK,gBAAgB,QAAQ;EACrE,IAAI,KAAK,IAAI,eAAe,aAAa,KAAK,kBAAkB;CAClE;CACA,OAAO;EACL,GAAG;EACH,IAAI,IAAI,qBAAqB,eAAe;EAC5C;EACA,QAAQ,KAAK,IAAI,KAAK,eAAe,IAAI,WAAW;EACpD,UAAU;CACZ;AACF;;;;;;AAOA,SAAS,oBAAoB,QAAoE;CAC/F,MAAM,WAAW,IAAI,MAAc,OAAO,MAAM;CAChD,IAAI,UAAU;CACd,IAAI,IAAI;CACR,OAAO,IAAI,OAAO,QAAQ;EACxB,IAAI,IAAI;EACR,OAAO,IAAI,OAAO,UAAU,OAAO,OAAO,OAAO,IAAI;EACrD,MAAM,WAAW,IAAI,IAAI,KAAK;EAC9B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,SAAS,KAAK;EAC1C,MAAM,YAAY,IAAI;EACtB,IAAI,YAAY,GAAG,WAAW,aAAa,IAAI;EAC/C,IAAI;CACN;CACA,OAAO;EAAE;EAAU;CAAQ;AAC7B;AAEA,SAAS,qBACP,IACA,SACA,QACA,eACA,aACA,WACgB;CAChB,IAAI,YAAY,QAAQ,OAAO,gBAAgB,UAAU;CACzD,IAAI,YAAY,SAAS;EACvB,IAAI,eAAe,OAAO;EAC1B,MAAM,IAAI,gBACR,GAAG,GAAG,sCAAsC,OAAO,+BAC9C,UAAU,yFAEjB;CACF;CACA,IAAI,eACF,MAAM,IAAI,gBACR,GAAG,GAAG,sCAAsC,OAAO,+CACpC,kBAAkB,WAAW,EAAE,0HAEzC,UAAU,EACjB;CAEF,OAAO;AACT;AAEA,SAAS,oBAAoB,IAAY,cAA0C;CACjF,IAAI,iBAAiB,KAAA,GAAW,OAAO;CACvC,IAAI,CAAC,OAAO,UAAU,YAAY,KAAK,eAAe,GACpD,MAAM,IAAI,gBAAgB,GAAG,GAAG,iDAAiD,cAAc;CAEjG,OAAO;AACT;;AAGA,SAAS,oBAAoB,WAAmB,OAAuB;CACrE,IAAI,EAAE,QAAQ,IAAI,OAAO;CACzB,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,UAAU,KAAK,IAAI,GAAG,YAAY,EAAG,IAAI,KAAK,EAAE;AAC9E;;;;AAKA,SAAS,eAAe,IAAY,IAAY,OAAe,SAAyB;CACtF,IAAI,QAAQ,GAAG,OAAO;CACtB,MAAM,WAAa,KAAK,KAAM,MAAO,QAAQ,IAAI,WAAW,SAAS,QAAQ;CAC7E,OAAO,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,CAAC;AACxC;AAEA,SAAS,UAAU,GAAW,GAAmB;CAC/C,OAAO,QAAQ,IAAI,CAAC,IAAI,QAAQ,IAAI,CAAC,IAAI,QAAQ,IAAI,IAAI,CAAC;AAC5D;AAEA,SAAS,kBAAkB,OAAuB;CAChD,OAAO,SAAS,QAAQ,UAAU,IAAI,MAAM,QAAQ,CAAC,IAAI,MAAM,cAAc,CAAC;AAChF;;;;;;;AAQA,SAAS,mBACP,cACA,WACkC;CAClC,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,IAAI,OAAO;CACX,KAAK,IAAI,SAAS,GAAG,SAAS,aAAa,QAAQ,UACjD,QAAQ,KAAK,IAAI,WAAW,SAAS,CAAC,KAAK,SAAS,aAAa,UAAW;CAE9E,OAAO;EACL,SAAS,YAAY,MAAM,SAAS;EACpC;CACF;AACF;;;;;;;;;AAUA,SAAS,oBAAoB,cAAiC,WAA2B;CACvF,MAAM,QAAQ,aAAa;CAC3B,MAAM,SAAS,QAAQ;CACvB,MAAM,aAAa,aAAa,MAAM,GAAG,SAAS,CAAC,CAAC,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CACvF,MAAM,aAAa,aAAa,MAAM,QAAQ,SAAS,CAAC,CAAC,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC5F,IAAI,eAAe,YAAY,OAAO;CAEtC,MAAM,SAAS,aAAa,YAAY,KAAK,YAAY;CACzD,MAAM,mBAAmB,KAAK,IAAI,aAAa,MAAM;CACrD,MAAM,mBAAmB,KAAK,IAAI,aAAa,MAAM;CACrD,MAAM,eAAe,UAAU,OAAO,SAAS;CAC/C,MAAM,cAAc,KAAK,IACvB,qBAAqB,cAAc,WAAW,SAAS,IAAI,YAC7D;CACA,MAAM,cAAc,KAAK,IACvB,qBAAqB,cAAc,WAAW,SAAS,IAAI,YAC7D;CAEA,IAAI,mBAAmB,kBAAkB,OAAO;CAChD,IAAI,mBAAmB,kBAAkB,OAAO;CAChD,OAAO,KAAK,IAAI,GAAG,cAAc,WAAW;AAC9C;AAEA,SAAS,qBACP,aACA,WACA,MACQ;CACR,MAAM,gBAAgB,SAAS,YAAY,YAAY,IAAI,YAAY,SAAS;CAChF,MAAM,WAAW,YAAY;CAC7B,IAAI,QAAQ;CACZ,IAAI,YAAY,gBAAgB;CAChC,OAAO,QAAQ,KAAK,YAAY,QAAQ,OAAO,UAAU;CACzD,OAAO,YAAY,YAAY,UAAU,YAAY,eAAe,UAAU;CAI9E,OAAO,UAFS,YAAY,OAEF,aADZ,SAAS,YAAY,QAAQ,YAAY,SAAS,UACrB;AAC7C;;;;;;;;;;;;;AAcA,SAAS,gBACP,cACA,IACA,IACA,kBAC+B;CAC/B,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,MAAM,QAAQ,SAAS;CAEvB,MAAM,OAAuB,MAAM,KAAK,EAAE,QAAQ,KAAK,EAAE,SAAS,IAAI,aAAa,KAAK,CAAC;CACzF,KAAK,EAAE,CAAE,KAAK;CACd,IAAI,SAAS;CACb,KAAK,MAAM,QAAQ,cAAc;EAC/B,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,SAAS,CAAC,GAAG,KAAK,GAAG,KAAK;GAClD,MAAM,OAAO,KAAK,IAAI;GACtB,MAAM,OAAO,KAAK;GAClB,KAAK,IAAI,MAAM,SAAS,MAAM,OAAO,GAAG,OAAO;IAC7C,MAAM,QAAQ,KAAK;IACnB,IAAI,UAAU,GAAG,KAAK,MAAM,SAAU;GACxC;EACF;EACA;CACF;CAIA,MAAM,QAAQ,MAAM,KAAK,KAAK,KAAK;CACnC,MAAM,SAAS,KAAK;CACpB,IAAI,YAAY;CAChB,IAAI,cAAc;CAClB,IAAI,WAAW;CACf,IAAI,eAAe;CACnB,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;EACpC,MAAM,QAAQ,OAAO;EACrB,IAAI,UAAU,GAAG;EACjB,aAAa;EACb,MAAM,YAAY,KAAK,IAAI,MAAM,KAAK;EACtC,IAAI,aAAa,kBAAkB,YAAY;EAC/C,IAAI,YAAY,cAAc;GAC5B,eAAe;GACf,cAAc;EAChB,OAAO,IAAI,cAAc,cACvB,eAAe;CAEnB;CACA,OAAO;EAAE,GAAG,WAAW;EAAW,QAAQ,cAAc;CAAU;AACpE;;;;;;;;AASA,SAAS,iBACP,cACA,kBAC+B;CAC/B,MAAM,SAAS,aAAa,QAAQ,KAAK,SAAS,MAAM,MAAM,CAAC;CAC/D,MAAM,QAAQ,SAAS;CACvB,IAAI,OAAO,IAAI,aAAa,KAAK;CACjC,KAAK,KAAK;CACV,KAAK,MAAM,QAAQ,cAAc;EAC/B,MAAM,OAAO,IAAI,aAAa,KAAK;EACnC,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;GACpC,MAAM,cAAc,KAAK;GACzB,IAAI,gBAAgB,GAAG;GACvB,KAAK,QAAS,cAAc;GAC5B,KAAK,MAAM,SAAU,cAAc;EACrC;EACA,OAAO;CACT;CAEA,MAAM,SAAS,SAAS;CACxB,IAAI,OAAO;CACX,IAAI,UAAU;CACd,IAAI,eAAe;CACnB,KAAK,IAAI,MAAM,GAAG,MAAM,OAAO,OAAO;EACpC,MAAM,cAAc,KAAK;EACzB,IAAI,gBAAgB,GAAG;EACvB,MAAM,YAAY,KAAK,IAAI,MAAM,MAAM;EACvC,IAAI,aAAa,kBAAkB,QAAQ;EAC3C,IAAI,YAAY,cAAc;GAC5B,eAAe;GACf,UAAU;EACZ,OAAO,IAAI,cAAc,cACvB,WAAW;CAEf;CACA,OAAO;EAAE,GAAG,KAAK,IAAI,GAAG,IAAI;EAAG,QAAQ,KAAK,IAAI,GAAG,OAAO;CAAE;AAC9D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtaA,SAAgB,SAAS,MAA+B,MAAuC;CAC7F,MAAM,EAAE,aAAa,iBAAiB;CACtC,IAAI,gBAAgB,cAClB,MAAM,IAAI,gBACR,oDAAoD,YAAY,wCAClE;CAIF,MAAM,wBAAQ,IAAI,IAAyC;CAC3D,MAAM,2BAAW,IAAI,IAAY;CACjC,KAAK,MAAM,OAAO,MAAM;EACtB,SAAS,IAAI,IAAI,GAAG;EACpB,IAAI,IAAI,QAAQ,eAAe,IAAI,QAAQ,cAAc;EACzD,MAAM,QAAQ,MAAM,IAAI,IAAI,GAAG,qBAAK,IAAI,IAA4B;EACpE,MAAM,QAAQ,MAAM,IAAI,IAAI,OAAO,KAAK,CAAC;EACzC,MAAM,KAAK,GAAG;EACd,MAAM,IAAI,IAAI,SAAS,KAAK;EAC5B,MAAM,IAAI,IAAI,KAAK,KAAK;CAC1B;CAEA,KAAK,MAAM,OAAO,CAAC,aAAa,YAAY,GAC1C,IAAI,CAAC,MAAM,IAAI,GAAG,GAEhB,MAAM,IAAI,gBAAgB,8BAA8B,IAAI,mBAD/C,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,KAAK,SACoC,EAAE;CAI1F,MAAM,gBAAgB,MAAM,IAAI,WAAW;CAC3C,MAAM,iBAAiB,MAAM,IAAI,YAAY;CAE7C,MAAM,UAAU,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,cAAc,KAAK,GAAG,GAAG,eAAe,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;CACvF,MAAM,QAAuB,CAAC;CAC9B,MAAM,mBAAmC,CAAC;CAC1C,MAAM,oBAAoC,CAAC;CAC3C,KAAK,MAAM,WAAW,SAAS;EAC7B,MAAM,IAAI,cAAc,IAAI,OAAO,KAAK,CAAC;EACzC,MAAM,IAAI,eAAe,IAAI,OAAO,KAAK,CAAC;EAE1C,IAAI,EAAE,UAAU,KAAK,EAAE,UAAU,GAAG;GAClC,IAAI,EAAE,WAAW,KAAK,EAAE,WAAW,GAAG;IACpC,MAAM,WAAW,EAAE;IACnB,MAAM,YAAY,EAAE;IACpB,IAAI,SAAS,WAAW,KAAA,KAAa,UAAU,WAAW,KAAA,GAEtD;SAAA,SAAS,WAAW,KAAA,KACpB,UAAU,WAAW,KAAA,KACrB,SAAS,WAAW,UAAU,QAC9B;MACA,iBAAiB,KAAK,QAAQ;MAC9B,kBAAkB,KAAK,SAAS;MAChC;KACF;;IAEF,MAAM,KAAK;KAAE;KAAS,UAAU;KAAG;KAAU;IAAU,CAAC;GAC1D,OAAO;IACL,iBAAiB,KAAK,GAAG,CAAC;IAC1B,kBAAkB,KAAK,GAAG,CAAC;GAC7B;GACA;EACF;EAEA,MAAM,SAAS,cAAc,GAAG,SAAS,WAAW;EACpD,MAAM,SAAS,cAAc,GAAG,SAAS,YAAY;EACrD,MAAM,UAAU,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,GAAG,GAAG,OAAO,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;EACxE,IAAI,WAAW;EACf,KAAK,MAAM,UAAU,SAAS;GAC5B,MAAM,WAAW,OAAO,IAAI,MAAM;GAClC,MAAM,YAAY,OAAO,IAAI,MAAM;GACnC,IAAI,aAAa,KAAA,KAAa,cAAc,KAAA,GAC1C,MAAM,KAAK;IAAE;IAAS,UAAU;IAAY;IAAU;GAAU,CAAC;QAC5D,IAAI,aAAa,KAAA,GACtB,iBAAiB,KAAK,QAAQ;QACzB,IAAI,cAAc,KAAA,GACvB,kBAAkB,KAAK,SAAS;EAEpC;CACF;CAEA,OAAO;EAAE;EAAO;EAAkB;CAAkB;AACtD;;;AAIA,SAAS,cACP,OACA,SACA,KAC2B;CAC3B,MAAM,wBAAQ,IAAI,IAA0B;CAC5C,KAAK,MAAM,OAAO,OAAO;EACvB,IAAI,IAAI,WAAW,KAAA,GACjB,MAAM,IAAI,gBACR,sBAAsB,QAAQ,mDAAmD,IAAI,8KAGvF;EAEF,IAAI,MAAM,IAAI,IAAI,MAAM,GACtB,MAAM,IAAI,gBACR,+BAA+B,IAAI,OAAO,iBAAiB,QAAQ,YAAY,IAAI,iEAErF;EAEF,MAAM,IAAI,IAAI,QAAQ,GAAG;CAC3B;CACA,OAAO;AACT;;;;;;;;;;;;;;AAiEA,SAAgB,kBACd,MACA,MACsB;CACtB,MAAM,EAAE,OAAO,kBAAkB,sBAAsB,SAAS,MAAM,IAAI;CAE1E,IAAI,cAAwC;CAC5C,MAAM,eAAyB,CAAC;CAChC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,SAAS,SAAS,KAAA,KAAa,KAAK,UAAU,SAAS,KAAA,GAAW;EAC3E,aAAa,KAAK,KAAK,SAAS,OAAO,IAAI,CAAC;EAC5C,cAAc,KAAK,KAAK,UAAU,OAAO,IAAI,CAAC;CAChD;CACA,IAAI,aAAa,SAAS,GAAG;EAC3B,MAAM,KAAK,QAAQ,cAAc,aAAa;EAC9C,cAAc;GACZ,KAAK,GAAG;GACR,KAAK,GAAG;GACR,SAAS;GACT,gBAAgB,qBAAqB,cAAc,aAAa;EAClE;CACF;CAaA,MAAM,gBAVJ,KAAK,eACL,CACE,GAAG,IAAI,IACL,MAAM,SAAS,MAAM,CACnB,GAAG,OAAO,KAAK,EAAE,SAAS,WAAW,CAAC,CAAC,GACvC,GAAG,OAAO,KAAK,EAAE,UAAU,WAAW,CAAC,CAAC,CAC1C,CAAC,CACH,CACF,CAAC,CAAC,KAAK,EAAA,CAE6C,KAAK,SAAS;EAClE,MAAM,SAAmB,CAAC;EAC1B,MAAM,QAAkB,CAAC;EACzB,IAAI,WAAW;EACf,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,IAAI,YAAY,KAAK,UAAU,IAAI;GACzC,MAAM,IAAI,YAAY,KAAK,WAAW,IAAI;GAC1C,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAAW;IACtC;IACA;GACF;GACA,OAAO,KAAK,CAAC;GACb,MAAM,KAAK,CAAC;EACd;EACA,MAAM,cAAc,OAAO,WAAW,IAAI,OAAO,gBAAgB,QAAQ,OAAO,KAAK,SAAS;EAC9F,OAAO;GACL;GACA,GAAG,OAAO;GACV;GACA,aAAa,aAAa,UAAU;GACpC,WAAW,aAAa,QAAQ;GAChC;GACA,UAAU,OAAO,WAAW,IAAI,OAAO,mBAAmB,QAAQ,KAAK;EACzE;CACF,CAAC;CAED,OAAO;EACL,QAAQ,MAAM;EACd,mBAAmB,iBAAiB;EACpC,oBAAoB,kBAAkB;EACtC;EACA;CACF;AACF;;;;;;;;AA2BA,SAAgB,eACd,cACA,eACsB;CACtB,MAAM,eAAe,iBAAiB,cAAc,UAAU;CAC9D,MAAM,gBAAgB,iBAAiB,eAAe,WAAW;CACjE,yBAAyB,cAAc,UAAU;CACjD,yBAAyB,eAAe,WAAW;CACnD,IAAI,aAAa,WAAW,KAAK,cAAc,WAAW,GACxD,OAAO;EACL,OAAO,CAAC;EACR,kBAAkB,aAAa,KAAK,QAAQ,IAAI,GAAG;EACnD,mBAAmB,cAAc,KAAK,QAAQ,IAAI,GAAG;CACvD;CAGF,MAAM,SAAS,SAAS,CAAC,GAAG,cAAc,GAAG,aAAa,GAAG;EAC3D,aAAa;EACb,cAAc;CAChB,CAAC;CACD,OAAO;EACL,OAAO,OAAO,MAAM,KAAK,SAAS;GAChC,MAAM,WAAW,KAAK;GACtB,MAAM,YAAY,KAAK;GACvB,OAAO;IACL,SAAS,KAAK;IACd,QAAQ,SAAS;IACjB,UAAU,SAAS;IACnB,WAAW,UAAU;GACvB;EACF,CAAC;EACD,kBAAkB,OAAO,iBAAiB,KAAK,QAAS,IAAwB,GAAG;EACnF,mBAAmB,OAAO,kBAAkB,KAAK,QAAS,IAAwB,GAAG;CACvF;AACF;AAEA,SAAS,iBAAiB,MAA4B,KAAgC;CACpF,OAAO,KAAK,KAAK,QAAQ;EACvB,MAAM,aAAa,IAAI,WAAW,KAAK;EACvC,IAAI,CAAC,YACH,MAAM,IAAI,gBACR,wBAAwB,IAAI,MAAM,+EACpC;EAEF,OAAO;GACL,SAAS,KAAK,UAAU,CAAC,IAAI,cAAc,UAAU,CAAC;GACtD,QAAQ,OAAO,IAAI,IAAI;GACvB;GACA;EACF;CACF,CAAC;AACH;AAEA,SAAS,yBAAyB,MAAkC,KAAmB;CACrF,MAAM,4BAAY,IAAI,IAA+B;CACrD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,UAAU,IAAI,IAAI,OAAO,KAAK,CAAC;EAC7C,MAAM,KAAK,GAAG;EACd,UAAU,IAAI,IAAI,SAAS,KAAK;CAClC;CACA,KAAK,MAAM,CAAC,SAAS,UAAU,WAC7B,IAAI,MAAM,SAAS,GAAG,cAAc,OAAO,SAAS,GAAG;AAE3D;AAEA,SAAS,YAAY,KAAmB,MAAkC;CACxE,MAAM,IAAI,IAAI,UAAU;CACxB,IAAI,MAAM,KAAA,GAAW,OAAO,KAAA;CAC5B,IAAI,CAAC,OAAO,SAAS,CAAC,GACpB,MAAM,IAAI,gBACR,mDAAmD,KAAK,gBAAgB,IAAI,QAAQ,UAAU,IAAI,IAAI,MAAM,GAC9G;CAEF,OAAO;AACT"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
|
-
import { a as medianInPlace, i as makeRng, n as binomialHalfUpperTail, t as assertFiniteSample } from "./internal-
|
|
3
|
-
import { t as studentTCdf } from "./student-t-
|
|
2
|
+
import { a as medianInPlace, i as makeRng, n as binomialHalfUpperTail, t as assertFiniteSample } from "./internal-BMFSR8Ns.js";
|
|
3
|
+
import { t as studentTCdf } from "./student-t-BA-Uy51p.js";
|
|
4
4
|
//#region src/statistics/paired-tests.ts
|
|
5
5
|
/**
|
|
6
6
|
* Paired significance tests on continuous scores: the paired t-test, the
|
|
@@ -210,4 +210,4 @@ const DECISION_PAIRED_DELTA_STATISTIC = "mean";
|
|
|
210
210
|
//#endregion
|
|
211
211
|
export { pairedSignTest as a, pairedDeltaTieFraction as i, DECISION_PAIRED_DELTA_STATISTIC as n, pairedTTest as o, pairedBootstrap as r, BOOTSTRAP_GATE_MIN_N as t };
|
|
212
212
|
|
|
213
|
-
//# sourceMappingURL=paired-tests-
|
|
213
|
+
//# sourceMappingURL=paired-tests-C8iCsioC.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"paired-tests-BHIhYVdu.js","names":[],"sources":["../src/statistics/paired-tests.ts"],"sourcesContent":["/**\n * Paired significance tests on continuous scores: the paired t-test, the\n * promotion-gate paired bootstrap, the exact sign test, and the package-wide\n * paired-delta decision statistic.\n */\n\nimport { ValidationError } from '../errors'\nimport { studentTCdf } from '../math/student-t'\nimport { assertFiniteSample, binomialHalfUpperTail, makeRng, medianInPlace } from './internal'\n\nexport interface PairedTTestResult {\n /** Null when the statistic is undefined — see {@link pairedTTest}. */\n t: number | null\n df: number\n /** Null exactly when `t` is null. */\n p: number | null\n}\n\n/**\n * Paired t-test — before/after measurements on the SAME items.\n * Pairing removes inter-item variance, giving tighter significance than\n * an unpaired test when comparing prompt v1 vs prompt v2 on identical\n * scenarios.\n *\n * Returns `t = p = null` where the statistic is undefined: fewer than two\n * pairs, or a non-zero constant delta whose observed variance is zero. A\n * constant shift carries no information about the variance it would have to\n * be compared against, so the honest answer is \"undefined\", not `p = 0` —\n * three observations cannot buy absolute certainty. This is the same contract\n * {@link pairedCohensDz} states for the same condition. An all-zero delta is\n * different: it is a measured null, and returns `t = 0, p = 1`.\n */\nexport function pairedTTest(before: number[], after: number[]): PairedTTestResult {\n if (before.length !== after.length) {\n throw new ValidationError(\n `pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n assertFiniteSample('pairedTTest', 'before', before)\n assertFiniteSample('pairedTTest', 'after', after)\n const n = before.length\n if (n < 2) return { t: null, df: 0, p: null }\n\n const diffs = before.map((b, i) => after[i]! - b)\n const mean = diffs.reduce((a, b) => a + b, 0) / n\n const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1)\n const se = Math.sqrt(variance / n)\n if (se === 0) {\n return mean === 0 ? { t: 0, df: n - 1, p: 1 } : { t: null, df: n - 1, p: null }\n }\n\n const t = mean / se\n const df = n - 1\n const p = 2 * (1 - studentTCdf(Math.abs(t), df))\n return { t, df, p }\n}\n\n// ── Paired bootstrap (promotion-gate effect size) ────────────────────\n\nexport interface PairedBootstrapResult {\n /** Number of paired observations. */\n n: number\n /** Median of paired deltas (after − before). */\n median: number\n /** Mean of paired deltas. */\n mean: number\n /** Lower bound of the bootstrap CI on the chosen statistic. */\n low: number\n /** Upper bound of the bootstrap CI on the chosen statistic. */\n high: number\n /** Confidence level used (e.g. 0.95). */\n confidence: number\n /** Number of bootstrap resamples used. */\n resamples: number\n /** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */\n gateEligible: boolean\n}\n\n/**\n * Pairs below which a percentile bootstrap interval is descriptive spread only.\n *\n * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000\n * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the\n * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling\n * three points, not an implementation error — scipy's BCa gives 16.0 % on the\n * same n = 3 data — so no change to the estimator moves it. Below this floor\n * the decision belongs to the exact sign test or exact signed-rank test.\n */\nexport const BOOTSTRAP_GATE_MIN_N = 20\n\nexport interface PairedBootstrapOptions {\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resample count. Default 2000. */\n resamples?: number\n /** Statistic to bootstrap. Default 'median'. */\n statistic?: 'median' | 'mean'\n /** Deterministic seed. If omitted, derived from the deltas so the interval\n * is reproducible regardless. */\n seed?: number\n}\n\n/**\n * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen\n * statistic (median by default); pairs are resampled with replacement. Throws\n * on unequal sample sizes.\n *\n * `low > threshold` carries the stated confidence ONLY at `n ≥\n * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the\n * check fires under a true null several times more often than nominal, so the\n * interval is descriptive spread and a promotion must not turn on it.\n */\nexport function pairedBootstrap(\n before: number[],\n after: number[],\n opts: PairedBootstrapOptions = {},\n): PairedBootstrapResult {\n if (before.length !== after.length) {\n throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`)\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const statistic = opts.statistic ?? 'median'\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`)\n }\n\n const n = before.length\n const deltas = before.map((b, i) => after[i]! - b)\n const gateEligible = n >= BOOTSTRAP_GATE_MIN_N\n if (n === 0) {\n return { n: 0, median: 0, mean: 0, low: 0, high: 0, confidence, resamples, gateEligible }\n }\n if (n === 1) {\n const d = deltas[0]!\n return { n: 1, median: d, mean: d, low: d, high: d, confidence, resamples, gateEligible }\n }\n\n const rng = makeRng(opts.seed, deltas)\n const samples = new Array<number>(resamples)\n for (let b = 0; b < resamples; b++) {\n if (statistic === 'mean') {\n let sum = 0\n for (let k = 0; k < n; k++) {\n sum += deltas[Math.floor(rng() * n)]!\n }\n samples[b] = sum / n\n } else {\n const acc = new Array<number>(n)\n for (let k = 0; k < n; k++) {\n acc[k] = deltas[Math.floor(rng() * n)]!\n }\n samples[b] = medianInPlace(acc)\n }\n }\n samples.sort((a, b) => a - b)\n\n const alpha = 1 - confidence\n const lowIdx = Math.floor((alpha / 2) * resamples)\n const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)\n\n return {\n n,\n median: medianInPlace([...deltas]),\n mean: deltas.reduce((s, x) => s + x, 0) / n,\n low: samples[lowIdx]!,\n high: samples[Math.max(highIdx, lowIdx)]!,\n confidence,\n resamples,\n gateEligible,\n }\n}\n\n/** Pre-registered direction for a one-sided paired sign test. */\nexport type SignTestAlternative = 'greater' | 'less'\n\n/** Exact one-sided sign-test result for paired numeric differences. */\nexport interface PairedSignTestResult {\n /** Total supplied differences, including zero ties. */\n n: number\n /** Strictly positive differences. */\n positive: number\n /** Strictly negative differences. */\n negative: number\n /** Zero differences excluded from the binomial test. */\n ties: number\n /** Non-zero differences used by the binomial test. */\n nNonTies: number\n /** Direction of the pre-registered alternative hypothesis. */\n alternative: SignTestAlternative\n /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */\n pValue: number\n}\n\n/**\n * Exact one-sided sign test over paired differences.\n *\n * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`\n * tests whether positive signs are more likely than negative signs and returns\n * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats\n * negative signs as successes instead. With a continuous difference\n * distribution this is the usual directional median test. Exact zero\n * differences are ties and do not enter the binomial denominator. All-tie and\n * empty inputs return p = 1. Every input difference must be finite, and the\n * direction must be chosen explicitly so a caller cannot select it after\n * seeing the signs.\n */\nexport function pairedSignTest(\n differences: readonly number[],\n alternative: SignTestAlternative,\n): PairedSignTestResult {\n if (alternative !== 'greater' && alternative !== 'less') {\n throw new ValidationError(\n `pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`,\n )\n }\n\n let positive = 0\n let negative = 0\n let ties = 0\n for (let i = 0; i < differences.length; i++) {\n const difference = differences[i]!\n if (!Number.isFinite(difference)) {\n throw new ValidationError(\n `pairedSignTest: difference at index ${i} must be finite, got ${difference}`,\n )\n }\n if (difference > 0) positive++\n else if (difference < 0) negative++\n else ties++\n }\n\n const nNonTies = positive + negative\n const successes = alternative === 'greater' ? positive : negative\n return {\n n: differences.length,\n positive,\n negative,\n ties,\n nNonTies,\n alternative,\n pValue: binomialHalfUpperTail(successes, nNonTies),\n }\n}\n\n/** Fraction of paired observations whose delta is an exact tie (|after − before|\n * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */\nexport function pairedDeltaTieFraction(\n before: ArrayLike<number>,\n after: ArrayLike<number>,\n): number {\n if (before.length !== after.length) {\n throw new Error(\n `pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const n = before.length\n if (n === 0) return 0\n let ties = 0\n for (let i = 0; i < n; i++) {\n if (Math.abs(after[i]! - before[i]!) < 1e-9) ties++\n }\n return ties / n\n}\n\n/**\n * The paired-delta statistic a DECISION is computed on, package-wide.\n *\n * The mean paired delta is the estimator that answers the question a promotion\n * gate asks — \"by how much did the candidate move the score\" — in the caller's\n * own units, and it equals the aggregate lift everyone quotes. The MEDIAN\n * answers a different question and loses the answer to this one in every regime\n * eval data actually lands in:\n * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in\n * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI\n * are pinned at exactly 0 however large the shift. (Decide these on\n * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)\n * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by\n * construction, and `ci.low > threshold` then answers \"no\" forever at a\n * non-negative threshold and \"yes\" forever at a negative one.\n * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on\n * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail\n * leaves, put the median on a coarse lattice whose bootstrap percentiles\n * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real\n * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —\n * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.\n * That last case is why there is no tie-fraction threshold here: any cutoff on\n * ties leaves the lattice case open on the other side of it.\n *\n * `heldoutSignificance` has defaulted to the mean since #316 for the same\n * reason. The median remains available per call site for callers who\n * specifically want outlier robustness and accept the blindness.\n */\nexport const DECISION_PAIRED_DELTA_STATISTIC: 'mean' = 'mean'\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,YAAY,QAAkB,OAAoC;CAChF,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,gBACR,sCAAsC,OAAO,OAAO,MAAM,MAAM,OAAO,EACzE;CAEF,mBAAmB,eAAe,UAAU,MAAM;CAClD,mBAAmB,eAAe,SAAS,KAAK;CAChD,MAAM,IAAI,OAAO;CACjB,IAAI,IAAI,GAAG,OAAO;EAAE,GAAG;EAAM,IAAI;EAAG,GAAG;CAAK;CAE5C,MAAM,QAAQ,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CAChD,MAAM,OAAO,MAAM,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAChD,MAAM,WAAW,MAAM,QAAQ,KAAK,MAAM,OAAO,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI;CAC3E,MAAM,KAAK,KAAK,KAAK,WAAW,CAAC;CACjC,IAAI,OAAO,GACT,OAAO,SAAS,IAAI;EAAE,GAAG;EAAG,IAAI,IAAI;EAAG,GAAG;CAAE,IAAI;EAAE,GAAG;EAAM,IAAI,IAAI;EAAG,GAAG;CAAK;CAGhF,MAAM,IAAI,OAAO;CACjB,MAAM,KAAK,IAAI;CAEf,OAAO;EAAE;EAAG;EAAI,GADN,KAAK,IAAI,YAAY,KAAK,IAAI,CAAC,GAAG,EAAE;CAC5B;AACpB;;;;;;;;;;;AAiCA,MAAa,uBAAuB;;;;;;;;;;;AAwBpC,SAAgB,gBACd,QACA,OACA,OAA+B,CAAC,GACT;CACvB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MAAM,0CAA0C,OAAO,OAAO,MAAM,MAAM,OAAO,EAAE;CAE/F,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,qDAAqD,YAAY;CAGnF,MAAM,IAAI,OAAO;CACjB,MAAM,SAAS,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CACjD,MAAM,eAAe,KAAA;CACrB,IAAI,MAAM,GACR,OAAO;EAAE,GAAG;EAAG,QAAQ;EAAG,MAAM;EAAG,KAAK;EAAG,MAAM;EAAG;EAAY;EAAW;CAAa;CAE1F,IAAI,MAAM,GAAG;EACX,MAAM,IAAI,OAAO;EACjB,OAAO;GAAE,GAAG;GAAG,QAAQ;GAAG,MAAM;GAAG,KAAK;GAAG,MAAM;GAAG;GAAY;GAAW;EAAa;CAC1F;CAEA,MAAM,MAAM,QAAQ,KAAK,MAAM,MAAM;CACrC,MAAM,UAAU,IAAI,MAAc,SAAS;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAC7B,IAAI,cAAc,QAAQ;EACxB,IAAI,MAAM;EACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,OAAO,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEpC,QAAQ,KAAK,MAAM;CACrB,OAAO;EACL,MAAM,MAAM,IAAI,MAAc,CAAC;EAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEtC,QAAQ,KAAK,cAAc,GAAG;CAChC;CAEF,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAE5B,MAAM,QAAQ,IAAI;CAClB,MAAM,SAAS,KAAK,MAAO,QAAQ,IAAK,SAAS;CACjD,MAAM,UAAU,KAAK,IAAI,YAAY,GAAG,KAAK,MAAM,IAAI,QAAQ,KAAK,SAAS,IAAI,CAAC;CAElF,OAAO;EACL;EACA,QAAQ,cAAc,CAAC,GAAG,MAAM,CAAC;EACjC,MAAM,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAC1C,KAAK,QAAQ;EACb,MAAM,QAAQ,KAAK,IAAI,SAAS,MAAM;EACtC;EACA;EACA;CACF;AACF;;;;;;;;;;;;;;AAoCA,SAAgB,eACd,aACA,aACsB;CACtB,IAAI,gBAAgB,aAAa,gBAAgB,QAC/C,MAAM,IAAI,gBACR,gEAAgE,aAClE;CAGF,IAAI,WAAW;CACf,IAAI,WAAW;CACf,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,QAAQ,KAAK;EAC3C,MAAM,aAAa,YAAY;EAC/B,IAAI,CAAC,OAAO,SAAS,UAAU,GAC7B,MAAM,IAAI,gBACR,uCAAuC,EAAE,uBAAuB,YAClE;EAEF,IAAI,aAAa,GAAG;OACf,IAAI,aAAa,GAAG;OACpB;CACP;CAEA,MAAM,WAAW,WAAW;CAC5B,MAAM,YAAY,gBAAgB,YAAY,WAAW;CACzD,OAAO;EACL,GAAG,YAAY;EACf;EACA;EACA;EACA;EACA;EACA,QAAQ,sBAAsB,WAAW,QAAQ;CACnD;AACF;;;AAIA,SAAgB,uBACd,QACA,OACQ;CACR,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,iDAAiD,OAAO,OAAO,MAAM,MAAM,OAAO,EACpF;CAEF,MAAM,IAAI,OAAO;CACjB,IAAI,MAAM,GAAG,OAAO;CACpB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,IAAI,MAAM,KAAM,OAAO,EAAG,IAAI,MAAM;CAE/C,OAAO,OAAO;AAChB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8BA,MAAa,kCAA0C"}
|
|
1
|
+
{"version":3,"file":"paired-tests-C8iCsioC.js","names":[],"sources":["../src/statistics/paired-tests.ts"],"sourcesContent":["/**\n * Paired significance tests on continuous scores: the paired t-test, the\n * promotion-gate paired bootstrap, the exact sign test, and the package-wide\n * paired-delta decision statistic.\n */\n\nimport { ValidationError } from '../errors'\nimport { studentTCdf } from '../math/student-t'\nimport { assertFiniteSample, binomialHalfUpperTail, makeRng, medianInPlace } from './internal'\n\nexport interface PairedTTestResult {\n /** Null when the statistic is undefined — see {@link pairedTTest}. */\n t: number | null\n df: number\n /** Null exactly when `t` is null. */\n p: number | null\n}\n\n/**\n * Paired t-test — before/after measurements on the SAME items.\n * Pairing removes inter-item variance, giving tighter significance than\n * an unpaired test when comparing prompt v1 vs prompt v2 on identical\n * scenarios.\n *\n * Returns `t = p = null` where the statistic is undefined: fewer than two\n * pairs, or a non-zero constant delta whose observed variance is zero. A\n * constant shift carries no information about the variance it would have to\n * be compared against, so the honest answer is \"undefined\", not `p = 0` —\n * three observations cannot buy absolute certainty. This is the same contract\n * {@link pairedCohensDz} states for the same condition. An all-zero delta is\n * different: it is a measured null, and returns `t = 0, p = 1`.\n */\nexport function pairedTTest(before: number[], after: number[]): PairedTTestResult {\n if (before.length !== after.length) {\n throw new ValidationError(\n `pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n assertFiniteSample('pairedTTest', 'before', before)\n assertFiniteSample('pairedTTest', 'after', after)\n const n = before.length\n if (n < 2) return { t: null, df: 0, p: null }\n\n const diffs = before.map((b, i) => after[i]! - b)\n const mean = diffs.reduce((a, b) => a + b, 0) / n\n const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1)\n const se = Math.sqrt(variance / n)\n if (se === 0) {\n return mean === 0 ? { t: 0, df: n - 1, p: 1 } : { t: null, df: n - 1, p: null }\n }\n\n const t = mean / se\n const df = n - 1\n const p = 2 * (1 - studentTCdf(Math.abs(t), df))\n return { t, df, p }\n}\n\n// ── Paired bootstrap (promotion-gate effect size) ────────────────────\n\nexport interface PairedBootstrapResult {\n /** Number of paired observations. */\n n: number\n /** Median of paired deltas (after − before). */\n median: number\n /** Mean of paired deltas. */\n mean: number\n /** Lower bound of the bootstrap CI on the chosen statistic. */\n low: number\n /** Upper bound of the bootstrap CI on the chosen statistic. */\n high: number\n /** Confidence level used (e.g. 0.95). */\n confidence: number\n /** Number of bootstrap resamples used. */\n resamples: number\n /** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */\n gateEligible: boolean\n}\n\n/**\n * Pairs below which a percentile bootstrap interval is descriptive spread only.\n *\n * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000\n * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the\n * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling\n * three points, not an implementation error — scipy's BCa gives 16.0 % on the\n * same n = 3 data — so no change to the estimator moves it. Below this floor\n * the decision belongs to the exact sign test or exact signed-rank test.\n */\nexport const BOOTSTRAP_GATE_MIN_N = 20\n\nexport interface PairedBootstrapOptions {\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resample count. Default 2000. */\n resamples?: number\n /** Statistic to bootstrap. Default 'median'. */\n statistic?: 'median' | 'mean'\n /** Deterministic seed. If omitted, derived from the deltas so the interval\n * is reproducible regardless. */\n seed?: number\n}\n\n/**\n * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen\n * statistic (median by default); pairs are resampled with replacement. Throws\n * on unequal sample sizes.\n *\n * `low > threshold` carries the stated confidence ONLY at `n ≥\n * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the\n * check fires under a true null several times more often than nominal, so the\n * interval is descriptive spread and a promotion must not turn on it.\n */\nexport function pairedBootstrap(\n before: number[],\n after: number[],\n opts: PairedBootstrapOptions = {},\n): PairedBootstrapResult {\n if (before.length !== after.length) {\n throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`)\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const statistic = opts.statistic ?? 'median'\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`)\n }\n\n const n = before.length\n const deltas = before.map((b, i) => after[i]! - b)\n const gateEligible = n >= BOOTSTRAP_GATE_MIN_N\n if (n === 0) {\n return { n: 0, median: 0, mean: 0, low: 0, high: 0, confidence, resamples, gateEligible }\n }\n if (n === 1) {\n const d = deltas[0]!\n return { n: 1, median: d, mean: d, low: d, high: d, confidence, resamples, gateEligible }\n }\n\n const rng = makeRng(opts.seed, deltas)\n const samples = new Array<number>(resamples)\n for (let b = 0; b < resamples; b++) {\n if (statistic === 'mean') {\n let sum = 0\n for (let k = 0; k < n; k++) {\n sum += deltas[Math.floor(rng() * n)]!\n }\n samples[b] = sum / n\n } else {\n const acc = new Array<number>(n)\n for (let k = 0; k < n; k++) {\n acc[k] = deltas[Math.floor(rng() * n)]!\n }\n samples[b] = medianInPlace(acc)\n }\n }\n samples.sort((a, b) => a - b)\n\n const alpha = 1 - confidence\n const lowIdx = Math.floor((alpha / 2) * resamples)\n const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)\n\n return {\n n,\n median: medianInPlace([...deltas]),\n mean: deltas.reduce((s, x) => s + x, 0) / n,\n low: samples[lowIdx]!,\n high: samples[Math.max(highIdx, lowIdx)]!,\n confidence,\n resamples,\n gateEligible,\n }\n}\n\n/** Pre-registered direction for a one-sided paired sign test. */\nexport type SignTestAlternative = 'greater' | 'less'\n\n/** Exact one-sided sign-test result for paired numeric differences. */\nexport interface PairedSignTestResult {\n /** Total supplied differences, including zero ties. */\n n: number\n /** Strictly positive differences. */\n positive: number\n /** Strictly negative differences. */\n negative: number\n /** Zero differences excluded from the binomial test. */\n ties: number\n /** Non-zero differences used by the binomial test. */\n nNonTies: number\n /** Direction of the pre-registered alternative hypothesis. */\n alternative: SignTestAlternative\n /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */\n pValue: number\n}\n\n/**\n * Exact one-sided sign test over paired differences.\n *\n * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`\n * tests whether positive signs are more likely than negative signs and returns\n * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats\n * negative signs as successes instead. With a continuous difference\n * distribution this is the usual directional median test. Exact zero\n * differences are ties and do not enter the binomial denominator. All-tie and\n * empty inputs return p = 1. Every input difference must be finite, and the\n * direction must be chosen explicitly so a caller cannot select it after\n * seeing the signs.\n */\nexport function pairedSignTest(\n differences: readonly number[],\n alternative: SignTestAlternative,\n): PairedSignTestResult {\n if (alternative !== 'greater' && alternative !== 'less') {\n throw new ValidationError(\n `pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`,\n )\n }\n\n let positive = 0\n let negative = 0\n let ties = 0\n for (let i = 0; i < differences.length; i++) {\n const difference = differences[i]!\n if (!Number.isFinite(difference)) {\n throw new ValidationError(\n `pairedSignTest: difference at index ${i} must be finite, got ${difference}`,\n )\n }\n if (difference > 0) positive++\n else if (difference < 0) negative++\n else ties++\n }\n\n const nNonTies = positive + negative\n const successes = alternative === 'greater' ? positive : negative\n return {\n n: differences.length,\n positive,\n negative,\n ties,\n nNonTies,\n alternative,\n pValue: binomialHalfUpperTail(successes, nNonTies),\n }\n}\n\n/** Fraction of paired observations whose delta is an exact tie (|after − before|\n * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */\nexport function pairedDeltaTieFraction(\n before: ArrayLike<number>,\n after: ArrayLike<number>,\n): number {\n if (before.length !== after.length) {\n throw new Error(\n `pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const n = before.length\n if (n === 0) return 0\n let ties = 0\n for (let i = 0; i < n; i++) {\n if (Math.abs(after[i]! - before[i]!) < 1e-9) ties++\n }\n return ties / n\n}\n\n/**\n * The paired-delta statistic a DECISION is computed on, package-wide.\n *\n * The mean paired delta is the estimator that answers the question a promotion\n * gate asks — \"by how much did the candidate move the score\" — in the caller's\n * own units, and it equals the aggregate lift everyone quotes. The MEDIAN\n * answers a different question and loses the answer to this one in every regime\n * eval data actually lands in:\n * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in\n * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI\n * are pinned at exactly 0 however large the shift. (Decide these on\n * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)\n * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by\n * construction, and `ci.low > threshold` then answers \"no\" forever at a\n * non-negative threshold and \"yes\" forever at a negative one.\n * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on\n * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail\n * leaves, put the median on a coarse lattice whose bootstrap percentiles\n * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real\n * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —\n * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.\n * That last case is why there is no tie-fraction threshold here: any cutoff on\n * ties leaves the lattice case open on the other side of it.\n *\n * `heldoutSignificance` has defaulted to the mean since #316 for the same\n * reason. The median remains available per call site for callers who\n * specifically want outlier robustness and accept the blindness.\n */\nexport const DECISION_PAIRED_DELTA_STATISTIC: 'mean' = 'mean'\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,YAAY,QAAkB,OAAoC;CAChF,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,gBACR,sCAAsC,OAAO,OAAO,MAAM,MAAM,OAAO,EACzE;CAEF,mBAAmB,eAAe,UAAU,MAAM;CAClD,mBAAmB,eAAe,SAAS,KAAK;CAChD,MAAM,IAAI,OAAO;CACjB,IAAI,IAAI,GAAG,OAAO;EAAE,GAAG;EAAM,IAAI;EAAG,GAAG;CAAK;CAE5C,MAAM,QAAQ,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CAChD,MAAM,OAAO,MAAM,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAChD,MAAM,WAAW,MAAM,QAAQ,KAAK,MAAM,OAAO,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI;CAC3E,MAAM,KAAK,KAAK,KAAK,WAAW,CAAC;CACjC,IAAI,OAAO,GACT,OAAO,SAAS,IAAI;EAAE,GAAG;EAAG,IAAI,IAAI;EAAG,GAAG;CAAE,IAAI;EAAE,GAAG;EAAM,IAAI,IAAI;EAAG,GAAG;CAAK;CAGhF,MAAM,IAAI,OAAO;CACjB,MAAM,KAAK,IAAI;CAEf,OAAO;EAAE;EAAG;EAAI,GADN,KAAK,IAAI,YAAY,KAAK,IAAI,CAAC,GAAG,EAAE;CAC5B;AACpB;;;;;;;;;;;AAiCA,MAAa,uBAAuB;;;;;;;;;;;AAwBpC,SAAgB,gBACd,QACA,OACA,OAA+B,CAAC,GACT;CACvB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MAAM,0CAA0C,OAAO,OAAO,MAAM,MAAM,OAAO,EAAE;CAE/F,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,qDAAqD,YAAY;CAGnF,MAAM,IAAI,OAAO;CACjB,MAAM,SAAS,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CACjD,MAAM,eAAe,KAAA;CACrB,IAAI,MAAM,GACR,OAAO;EAAE,GAAG;EAAG,QAAQ;EAAG,MAAM;EAAG,KAAK;EAAG,MAAM;EAAG;EAAY;EAAW;CAAa;CAE1F,IAAI,MAAM,GAAG;EACX,MAAM,IAAI,OAAO;EACjB,OAAO;GAAE,GAAG;GAAG,QAAQ;GAAG,MAAM;GAAG,KAAK;GAAG,MAAM;GAAG;GAAY;GAAW;EAAa;CAC1F;CAEA,MAAM,MAAM,QAAQ,KAAK,MAAM,MAAM;CACrC,MAAM,UAAU,IAAI,MAAc,SAAS;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAC7B,IAAI,cAAc,QAAQ;EACxB,IAAI,MAAM;EACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,OAAO,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEpC,QAAQ,KAAK,MAAM;CACrB,OAAO;EACL,MAAM,MAAM,IAAI,MAAc,CAAC;EAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEtC,QAAQ,KAAK,cAAc,GAAG;CAChC;CAEF,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAE5B,MAAM,QAAQ,IAAI;CAClB,MAAM,SAAS,KAAK,MAAO,QAAQ,IAAK,SAAS;CACjD,MAAM,UAAU,KAAK,IAAI,YAAY,GAAG,KAAK,MAAM,IAAI,QAAQ,KAAK,SAAS,IAAI,CAAC;CAElF,OAAO;EACL;EACA,QAAQ,cAAc,CAAC,GAAG,MAAM,CAAC;EACjC,MAAM,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAC1C,KAAK,QAAQ;EACb,MAAM,QAAQ,KAAK,IAAI,SAAS,MAAM;EACtC;EACA;EACA;CACF;AACF;;;;;;;;;;;;;;AAoCA,SAAgB,eACd,aACA,aACsB;CACtB,IAAI,gBAAgB,aAAa,gBAAgB,QAC/C,MAAM,IAAI,gBACR,gEAAgE,aAClE;CAGF,IAAI,WAAW;CACf,IAAI,WAAW;CACf,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,QAAQ,KAAK;EAC3C,MAAM,aAAa,YAAY;EAC/B,IAAI,CAAC,OAAO,SAAS,UAAU,GAC7B,MAAM,IAAI,gBACR,uCAAuC,EAAE,uBAAuB,YAClE;EAEF,IAAI,aAAa,GAAG;OACf,IAAI,aAAa,GAAG;OACpB;CACP;CAEA,MAAM,WAAW,WAAW;CAC5B,MAAM,YAAY,gBAAgB,YAAY,WAAW;CACzD,OAAO;EACL,GAAG,YAAY;EACf;EACA;EACA;EACA;EACA;EACA,QAAQ,sBAAsB,WAAW,QAAQ;CACnD;AACF;;;AAIA,SAAgB,uBACd,QACA,OACQ;CACR,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,iDAAiD,OAAO,OAAO,MAAM,MAAM,OAAO,EACpF;CAEF,MAAM,IAAI,OAAO;CACjB,IAAI,MAAM,GAAG,OAAO;CACpB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,IAAI,MAAM,KAAM,OAAO,EAAG,IAAI,MAAM;CAE/C,OAAO,OAAO;AAChB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8BA,MAAa,kCAA0C"}
|
|
@@ -2,7 +2,7 @@ import { f as Run } from "../schema-Bjgdsn73.js";
|
|
|
2
2
|
import { a as RunFilter, s as TraceStore } from "../store-B06JdC56.js";
|
|
3
3
|
import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-CXL8NbEw.js";
|
|
4
4
|
import { n as TrajectoryStep } from "../trajectory-Bi157Gun.js";
|
|
5
|
-
import { c as computeToolUseMetrics, d as judgeAgreementView, f as BudgetBreachFinding, g as BaselineReport, h as BaselineOptions, i as toolWasteView, l as JudgeAgreementReport, m as budgetBreachView, n as ToolWasteOptions, p as BudgetBreachReport, r as ToolWasteReport, t as ToolWasteFinding, u as JudgePair } from "../tool-waste-
|
|
5
|
+
import { c as computeToolUseMetrics, d as judgeAgreementView, f as BudgetBreachFinding, g as BaselineReport, h as BaselineOptions, i as toolWasteView, l as JudgeAgreementReport, m as budgetBreachView, n as ToolWasteOptions, p as BudgetBreachReport, r as ToolWasteReport, t as ToolWasteFinding, u as JudgePair } from "../tool-waste-CKc7bYIg.js";
|
|
6
6
|
//#region src/pipelines/first-divergence.d.ts
|
|
7
7
|
interface DivergenceReport {
|
|
8
8
|
runA: string;
|
|
@@ -24,7 +24,8 @@ declare function firstDivergenceView(store: TraceStore, runA: string, runB: stri
|
|
|
24
24
|
interface RegressionSpec {
|
|
25
25
|
metric: string;
|
|
26
26
|
higherIsBetter: boolean;
|
|
27
|
-
/** Extract a scalar from a run.
|
|
27
|
+
/** Extract a scalar from a run. Omit it and `metric` must name one of
|
|
28
|
+
* `RUN_METRICS`; any other name is refused. */
|
|
28
29
|
extract?: (run: Run, store: TraceStore) => Promise<number | null>;
|
|
29
30
|
}
|
|
30
31
|
interface RegressionOptions extends BaselineOptions {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/first-divergence.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts"],"mappings":";;;;;;UAYiB;EACf;EACA;EACA;EACA,QAAQ;EACR,QAAQ;EACR;;EAEA;;UAGe;;EAEf,cAAc,GAAG,gBAAgB,GAAG;;iBAGhB,oBACpB,OAAO,YACP,cACA,cACA,UAAS,oBACR,QAAQ;;;UCnBM;EACf;EACA
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/first-divergence.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts"],"mappings":";;;;;;UAYiB;EACf;EACA;EACA;EACA,QAAQ;EACR,QAAQ;EACR;;EAEA;;UAGe;;EAEf,cAAc,GAAG,gBAAgB,GAAG;;iBAGhB,oBACpB,OAAO,YACP,cACA,cACA,UAAS,oBACR,QAAQ;;;UCnBM;EACf;EACA;;;EAGA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B,0BAA0B;EACzC,UAAU;EACV,WAAW;;iBAGS,eACpB,OAAO,YACP,SAAS,kBACT,SAAS,oBACR,QAAQ;;;UCFM;EACf;EACA;EACA;;EAEA;EACA;;EAEA;;EAEA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;;EAEf;;EAEA;;;;;EAKA;;EAEA;;iBAGoB,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ"}
|
package/dist/pipelines/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { a as budgetBreachView, i as failureClusterView, n as computeToolUseMetrics, r as judgeAgreementView, t as toolWasteView } from "../tool-waste-
|
|
2
|
-
import { t as compareToBaseline } from "../baseline-
|
|
1
|
+
import { a as budgetBreachView, i as failureClusterView, n as computeToolUseMetrics, r as judgeAgreementView, t as toolWasteView } from "../tool-waste-8BQiUc8K.js";
|
|
2
|
+
import { t as compareToBaseline } from "../baseline-BC-eBZ7U.js";
|
|
3
3
|
import { a as isToolSpan } from "../schema-k6ZBftVv.js";
|
|
4
|
-
import {
|
|
4
|
+
import { a as hasCapturedToolArgs, r as argHash, u as runMetricExtractor } from "../query-_5g6re3_.js";
|
|
5
5
|
import { t as buildTrajectory } from "../trajectory-D_7rLrvE.js";
|
|
6
6
|
import { t as executionTrackByLane } from "../execution-tracks-CpgFPpS5.js";
|
|
7
7
|
//#region src/pipelines/first-divergence.ts
|
|
@@ -67,7 +67,7 @@ async function regressionView(store, metrics, options) {
|
|
|
67
67
|
const baselineRuns = await store.listRuns(options.baseline);
|
|
68
68
|
const candidateRuns = await store.listRuns(options.candidate);
|
|
69
69
|
return compareToBaseline(await Promise.all(metrics.map(async (m) => {
|
|
70
|
-
const extract = m.extract ??
|
|
70
|
+
const extract = m.extract ?? runMetricExtractor(m.metric);
|
|
71
71
|
const baseline = await extractAll(baselineRuns, extract, store);
|
|
72
72
|
const candidate = await extractAll(candidateRuns, extract, store);
|
|
73
73
|
return {
|
|
@@ -86,21 +86,6 @@ async function extractAll(runs, extract, store) {
|
|
|
86
86
|
}
|
|
87
87
|
return out;
|
|
88
88
|
}
|
|
89
|
-
function defaultExtract(metric) {
|
|
90
|
-
return async (run, store) => {
|
|
91
|
-
switch (metric) {
|
|
92
|
-
case "score":
|
|
93
|
-
case "overallScore": return run.outcome?.score ?? null;
|
|
94
|
-
case "pass": return run.outcome?.pass === true ? 1 : 0;
|
|
95
|
-
case "durationMs": return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null;
|
|
96
|
-
case "costUsd": return aggregateLlm(await llmSpans(store, run.runId)).costUsd;
|
|
97
|
-
case "inputTokens": return aggregateLlm(await llmSpans(store, run.runId)).inputTokens;
|
|
98
|
-
case "outputTokens": return aggregateLlm(await llmSpans(store, run.runId)).outputTokens;
|
|
99
|
-
case "failureClass": return runFailureClass(run) === "success" ? 1 : 0;
|
|
100
|
-
default: return null;
|
|
101
|
-
}
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
89
|
//#endregion
|
|
105
90
|
//#region src/pipelines/stuck-loop.ts
|
|
106
91
|
/**
|