@tangle-network/agent-eval 0.144.6 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,682 @@
|
|
|
1
|
+
import { E as pairedBootstrap, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, T as pairedBinaryScale, j as pairedRiskDifferenceExact } from "./statistics-ByxzSiOM.js";
|
|
2
|
+
//#region src/paired-delta-test.ts
|
|
3
|
+
/** Smallest all-positive sample that can clear a one-sided exact sign test. */
|
|
4
|
+
function minimumPairsForPairedDeltaTest(confidence = .95) {
|
|
5
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
|
|
6
|
+
const oneSidedAlpha = (1 - confidence) / 2;
|
|
7
|
+
return Math.ceil(Math.log2(1 / oneSidedAlpha));
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Tests whether a paired candidate-minus-baseline delta clears a threshold.
|
|
11
|
+
*
|
|
12
|
+
* At 20 or more pairs, the percentile bootstrap lower bound carries the
|
|
13
|
+
* decision. Below that point the interval is descriptive only, so the function
|
|
14
|
+
* switches to a pre-registered one-sided exact sign test. The exact path is
|
|
15
|
+
* deliberately conservative: it requires both a point estimate above the
|
|
16
|
+
* threshold and enough consistently positive paired differences.
|
|
17
|
+
*
|
|
18
|
+
* ## A zero-width interval is never significant
|
|
19
|
+
*
|
|
20
|
+
* When every paired delta is identical the resample distribution is a point
|
|
21
|
+
* mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
|
|
22
|
+
* identical deltas of g. Neither says the effect is certain — both say the
|
|
23
|
+
* sample carries no information about how far the estimate could be wrong, and
|
|
24
|
+
* `low > threshold` then answers on the point estimate alone. It fails in both
|
|
25
|
+
* directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
|
|
26
|
+
* tie-dominated pass/fail comparison laundered a regression into a
|
|
27
|
+
* noninferiority pass, and `[g, g]` clears every threshold below g with no
|
|
28
|
+
* spread behind it. Under a bounded asymmetric null whose true mean paired
|
|
29
|
+
* delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
|
|
30
|
+
* every sample that misses the drop is exactly that shape, and deciding on
|
|
31
|
+
* `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
|
|
32
|
+
*
|
|
33
|
+
* So `indeterminate` is reported and `significant` is false whenever the
|
|
34
|
+
* interval has zero width, on BOTH paths: at small n the exact sign test is a
|
|
35
|
+
* test of the MEDIAN and a zero-spread sample is precisely where it stops
|
|
36
|
+
* saying anything about the mean the caller is thresholding.
|
|
37
|
+
*
|
|
38
|
+
* `threshold` may be negative — that is a noninferiority margin, and it is the
|
|
39
|
+
* regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
|
|
40
|
+
* the percentile bootstrap is not a valid interval at a nonzero margin at all;
|
|
41
|
+
* use {@link decidePairedPromotion}, which routes those to Tango's score
|
|
42
|
+
* interval, rather than thresholding this function's bootstrap directly.
|
|
43
|
+
*/
|
|
44
|
+
function pairedDeltaTest(before, after, options = {}) {
|
|
45
|
+
const threshold = options.threshold ?? 0;
|
|
46
|
+
if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
|
|
47
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
|
|
48
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
49
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
50
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
51
|
+
const bootstrap = pairedBootstrap(before, after, options);
|
|
52
|
+
const sufficient = bootstrap.n >= minimumPairs;
|
|
53
|
+
const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
|
|
54
|
+
if (bootstrap.gateEligible) return {
|
|
55
|
+
bootstrap,
|
|
56
|
+
method: "bootstrap-ci",
|
|
57
|
+
pValue: null,
|
|
58
|
+
minimumPairs,
|
|
59
|
+
sufficient,
|
|
60
|
+
indeterminate,
|
|
61
|
+
significant: sufficient && !indeterminate && bootstrap.low > threshold
|
|
62
|
+
};
|
|
63
|
+
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
64
|
+
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
65
|
+
return {
|
|
66
|
+
bootstrap,
|
|
67
|
+
method: "exact-sign",
|
|
68
|
+
pValue: exact.pValue,
|
|
69
|
+
minimumPairs,
|
|
70
|
+
sufficient,
|
|
71
|
+
indeterminate,
|
|
72
|
+
significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
//#endregion
|
|
76
|
+
//#region src/paired-promotion-decision.ts
|
|
77
|
+
/**
|
|
78
|
+
* @module
|
|
79
|
+
* ONE rule for "does this paired interval clear a promotion threshold".
|
|
80
|
+
*
|
|
81
|
+
* The rule below was derived on `HeldOutGate` (#479) after the same estimator
|
|
82
|
+
* bug shipped twice. It then turned out that a SECOND gate — the composable
|
|
83
|
+
* `heldOutGate`, plus everything else routed through `heldoutSignificance` —
|
|
84
|
+
* still carried the original defect, because the rule had been written into one
|
|
85
|
+
* gate's method body rather than into a shared function. Two copies of a
|
|
86
|
+
* statistical rule is how a defect survives in one of them, so there is now
|
|
87
|
+
* exactly one copy and both gates call it.
|
|
88
|
+
*
|
|
89
|
+
* Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
|
|
90
|
+
* does not:
|
|
91
|
+
*
|
|
92
|
+
* 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
|
|
93
|
+
* pass/fail eval the paired delta vector is dominated by ties, so the
|
|
94
|
+
* bootstrap of the mean is a resample of a lattice with three atoms and its
|
|
95
|
+
* percentile interval is not valid at a nonzero margin. The score interval
|
|
96
|
+
* (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
|
|
97
|
+
* each hypothesised margin instead of fixing it at the observed value, which
|
|
98
|
+
* is the only construction that stays a confidence interval as the margin
|
|
99
|
+
* moves off zero — the regime every noninferiority threshold lives in.
|
|
100
|
+
* Measured on the composable gate before this change, at a true risk
|
|
101
|
+
* difference sitting exactly on the production caller's -0.05 margin and a
|
|
102
|
+
* nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
|
|
103
|
+
* 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
|
|
104
|
+
* Redundant with the interval by construction and kept anyway, so that
|
|
105
|
+
* swapping the estimator for one without that duality cannot silently
|
|
106
|
+
* reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
|
|
107
|
+
* c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
|
|
108
|
+
* (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
|
|
109
|
+
* threshold is a noninferiority question, which McNemar's test of "no
|
|
110
|
+
* difference" is not the right test for, so the veto does not apply there.
|
|
111
|
+
* 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
|
|
112
|
+
* cannot tell a gain from a regression and clears every negative threshold.
|
|
113
|
+
* Away from zero it fails the opposite way: n identical positive deltas give
|
|
114
|
+
* [g, g], which clears threshold 0 on no spread at all. Both are an absence
|
|
115
|
+
* of evidence. Measured on the composable gate before this change, under a
|
|
116
|
+
* bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
|
|
117
|
+
* false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
|
|
118
|
+
*
|
|
119
|
+
* Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
|
|
120
|
+
* picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
|
|
121
|
+
* exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
|
|
122
|
+
* Both are needed — an exact sign test applied to a tie-pinned median is still
|
|
123
|
+
* blind, and a mean bootstrap CI at n = 6 is still not a valid test.
|
|
124
|
+
*/
|
|
125
|
+
/**
|
|
126
|
+
* Which estimator {@link decidePairedPromotion} would use on this data, and the
|
|
127
|
+
* shape facts behind it — for callers that must report the shape on a path
|
|
128
|
+
* where no interval is computed at all (an early rejection, or zero pairs).
|
|
129
|
+
* Cheap: no bootstrap, no interval.
|
|
130
|
+
*/
|
|
131
|
+
function pairedDecisionShape(before, after, statistic = "mean") {
|
|
132
|
+
const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
|
|
133
|
+
if (statistic === "median") return {
|
|
134
|
+
statistic: "median_bootstrap",
|
|
135
|
+
binaryScale: null,
|
|
136
|
+
tieFraction
|
|
137
|
+
};
|
|
138
|
+
const binaryScale = pairedBinaryScale(before, after);
|
|
139
|
+
if (binaryScale !== null) return {
|
|
140
|
+
statistic: "paired_risk_difference",
|
|
141
|
+
binaryScale,
|
|
142
|
+
tieFraction
|
|
143
|
+
};
|
|
144
|
+
return {
|
|
145
|
+
statistic: "mean_bootstrap",
|
|
146
|
+
binaryScale: null,
|
|
147
|
+
tieFraction
|
|
148
|
+
};
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* Decide whether a paired candidate-minus-baseline delta clears a promotion
|
|
152
|
+
* threshold. `before` is the baseline arm, `after` the candidate arm, paired by
|
|
153
|
+
* position. Throws on unequal lengths.
|
|
154
|
+
*/
|
|
155
|
+
function decidePairedPromotion(before, after, options = {}) {
|
|
156
|
+
if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
157
|
+
const threshold = options.threshold ?? 0;
|
|
158
|
+
if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
|
|
159
|
+
const confidence = options.confidence ?? .95;
|
|
160
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
|
|
161
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
162
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
163
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
164
|
+
const n = before.length;
|
|
165
|
+
const sufficient = n >= minimumPairs;
|
|
166
|
+
const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
|
|
167
|
+
let core;
|
|
168
|
+
if (binaryScale !== null) {
|
|
169
|
+
const unitControl = before.map((v) => v / binaryScale);
|
|
170
|
+
const unitTreatment = after.map((v) => v / binaryScale);
|
|
171
|
+
const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
|
|
172
|
+
const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
|
|
173
|
+
const low = score.lower * binaryScale;
|
|
174
|
+
core = {
|
|
175
|
+
statistic: "paired_risk_difference",
|
|
176
|
+
method: "score-interval",
|
|
177
|
+
delta: score.riskDifference * binaryScale,
|
|
178
|
+
low,
|
|
179
|
+
high: score.upper * binaryScale,
|
|
180
|
+
bootstrap: null,
|
|
181
|
+
mcnemar: {
|
|
182
|
+
b: exact.b,
|
|
183
|
+
c: exact.c,
|
|
184
|
+
nDiscordant: exact.nDiscordant,
|
|
185
|
+
pValue: exact.pValue
|
|
186
|
+
},
|
|
187
|
+
pValue: null,
|
|
188
|
+
clearsThreshold: low > threshold,
|
|
189
|
+
label: "success-rate",
|
|
190
|
+
methodDetail: ""
|
|
191
|
+
};
|
|
192
|
+
} else {
|
|
193
|
+
const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
|
|
194
|
+
const test = pairedDeltaTest(before, after, {
|
|
195
|
+
confidence,
|
|
196
|
+
resamples: options.resamples,
|
|
197
|
+
statistic: bootstrapStatistic,
|
|
198
|
+
seed: options.seed,
|
|
199
|
+
threshold,
|
|
200
|
+
minPairs: options.minPairs
|
|
201
|
+
});
|
|
202
|
+
const ci = test.bootstrap;
|
|
203
|
+
core = {
|
|
204
|
+
statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
|
|
205
|
+
method: test.method,
|
|
206
|
+
delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
|
|
207
|
+
low: ci.low,
|
|
208
|
+
high: ci.high,
|
|
209
|
+
bootstrap: ci,
|
|
210
|
+
mcnemar: null,
|
|
211
|
+
pValue: test.pValue,
|
|
212
|
+
clearsThreshold: test.significant,
|
|
213
|
+
label: bootstrapStatistic,
|
|
214
|
+
methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
|
|
218
|
+
const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
|
|
219
|
+
const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
|
|
220
|
+
return {
|
|
221
|
+
n,
|
|
222
|
+
threshold,
|
|
223
|
+
confidence,
|
|
224
|
+
binaryScale,
|
|
225
|
+
tieFraction,
|
|
226
|
+
minimumPairs,
|
|
227
|
+
sufficient,
|
|
228
|
+
indeterminate,
|
|
229
|
+
indeterminateCause,
|
|
230
|
+
exactTestVetoes,
|
|
231
|
+
promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
|
|
232
|
+
...core
|
|
233
|
+
};
|
|
234
|
+
}
|
|
235
|
+
function fmt(x) {
|
|
236
|
+
return x.toFixed(4);
|
|
237
|
+
}
|
|
238
|
+
//#endregion
|
|
239
|
+
//#region src/campaign/gates/statistical-heldout.ts
|
|
240
|
+
/**
|
|
241
|
+
* Statistical held-out promotion machinery — the trustworthy core the
|
|
242
|
+
* point-estimate `heldout-delta` gate lacked.
|
|
243
|
+
*
|
|
244
|
+
* The shipped false positive it prevents: a winner re-scored against the
|
|
245
|
+
* baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
|
|
246
|
+
* "+4 lift" and shipped, because the gate compared point estimates with no
|
|
247
|
+
* confidence interval. Here we pair candidate vs baseline holdout observations
|
|
248
|
+
* and bootstrap a CI on the paired delta — a candidate ships only when the CI
|
|
249
|
+
* lower bound clears the effect-size threshold (the gain is real at the
|
|
250
|
+
* confidence level, not noise), and is blocked when a critical dimension
|
|
251
|
+
* (e.g. `hallucination_free` for a legal agent) significantly regresses even if
|
|
252
|
+
* the net composite rose (anti-Goodhart).
|
|
253
|
+
*
|
|
254
|
+
* Two traps this module is built around (both produce a NEW false positive if
|
|
255
|
+
* gotten wrong):
|
|
256
|
+
* 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
|
|
257
|
+
* `scenarioId` (which averages reps away and destroys the within-pair
|
|
258
|
+
* variance reduction that makes a paired bootstrap tighter than unpaired).
|
|
259
|
+
* One paired observation per cell ⇒ reps multiply n.
|
|
260
|
+
* 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
|
|
261
|
+
* threshold + tolerance are interpreted in the judge's NATIVE scale; the
|
|
262
|
+
* per-dimension tolerance auto-scales off the observed baseline magnitudes
|
|
263
|
+
* so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
|
|
264
|
+
*/
|
|
265
|
+
/** Tie fraction at/above which a gate annotates its verdict with the tie share.
|
|
266
|
+
* Tie-domination of the median bites structurally at >= 0.5 (the median is then
|
|
267
|
+
* 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
|
|
268
|
+
* that regime, so an operator sees it before the median goes fully blind. */
|
|
269
|
+
const TIE_WARN_FRACTION = .4;
|
|
270
|
+
/**
|
|
271
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
272
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
273
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
274
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
275
|
+
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
276
|
+
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
277
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
278
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
279
|
+
* means a silent pairing bug, not a soft fallback.
|
|
280
|
+
*/
|
|
281
|
+
function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
282
|
+
const cellValue = (byCell, cellId) => {
|
|
283
|
+
const scores = byCell.get(cellId);
|
|
284
|
+
if (!scores) return void 0;
|
|
285
|
+
const vals = [];
|
|
286
|
+
for (const s of Object.values(scores)) {
|
|
287
|
+
if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
|
|
288
|
+
const v = select(s);
|
|
289
|
+
if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
|
|
290
|
+
if (typeof v === "number") vals.push(v);
|
|
291
|
+
}
|
|
292
|
+
if (vals.length === 0) return void 0;
|
|
293
|
+
return vals.reduce((a, b) => a + b, 0) / vals.length;
|
|
294
|
+
};
|
|
295
|
+
const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
|
|
296
|
+
const candCells = [...candidate.keys()].filter(inScope).sort();
|
|
297
|
+
const baseCells = [...baseline.keys()].filter(inScope).sort();
|
|
298
|
+
if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
|
|
299
|
+
const before = [];
|
|
300
|
+
const after = [];
|
|
301
|
+
const cellIds = [];
|
|
302
|
+
for (const cellId of candCells) {
|
|
303
|
+
const b = cellValue(baseline, cellId);
|
|
304
|
+
const a = cellValue(candidate, cellId);
|
|
305
|
+
if (b === void 0 && a === void 0) continue;
|
|
306
|
+
if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
|
|
307
|
+
before.push(b);
|
|
308
|
+
after.push(a);
|
|
309
|
+
cellIds.push(cellId);
|
|
310
|
+
}
|
|
311
|
+
return {
|
|
312
|
+
before,
|
|
313
|
+
after,
|
|
314
|
+
cellIds
|
|
315
|
+
};
|
|
316
|
+
}
|
|
317
|
+
/**
|
|
318
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
319
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
320
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
321
|
+
* scale.
|
|
322
|
+
*
|
|
323
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
324
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
325
|
+
* also calls. That module's header carries the measurements; the short version
|
|
326
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
327
|
+
*
|
|
328
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
329
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
330
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
331
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
332
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
333
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
334
|
+
*
|
|
335
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
336
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
337
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
338
|
+
* delta is exactly 0.
|
|
339
|
+
*
|
|
340
|
+
* At small n, where the percentile bootstrap is descriptive only, a
|
|
341
|
+
* pre-registered exact sign test still carries the bootstrap path.
|
|
342
|
+
*/
|
|
343
|
+
function heldoutSignificance(paired, opts = {}) {
|
|
344
|
+
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
345
|
+
const confidence = opts.confidence ?? .95;
|
|
346
|
+
const resamples = opts.resamples ?? 2e3;
|
|
347
|
+
const seed = opts.seed ?? 1337;
|
|
348
|
+
const statistic = opts.statistic ?? "mean";
|
|
349
|
+
const decision = decidePairedPromotion(paired.before, paired.after, {
|
|
350
|
+
confidence,
|
|
351
|
+
resamples,
|
|
352
|
+
statistic,
|
|
353
|
+
seed,
|
|
354
|
+
threshold: deltaThreshold,
|
|
355
|
+
minPairs: opts.minProductiveRuns
|
|
356
|
+
});
|
|
357
|
+
const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
|
|
358
|
+
confidence,
|
|
359
|
+
resamples,
|
|
360
|
+
statistic,
|
|
361
|
+
seed
|
|
362
|
+
});
|
|
363
|
+
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
364
|
+
confidence,
|
|
365
|
+
resamples,
|
|
366
|
+
statistic: "median",
|
|
367
|
+
seed
|
|
368
|
+
});
|
|
369
|
+
const n = paired.before.length;
|
|
370
|
+
let ties = 0;
|
|
371
|
+
for (let i = 0; i < n; i += 1) {
|
|
372
|
+
const after = paired.after[i] ?? 0;
|
|
373
|
+
const before = paired.before[i] ?? 0;
|
|
374
|
+
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
375
|
+
}
|
|
376
|
+
const tieFraction = n === 0 ? 0 : ties / n;
|
|
377
|
+
return {
|
|
378
|
+
paired,
|
|
379
|
+
bootstrap,
|
|
380
|
+
medianBootstrap,
|
|
381
|
+
decision,
|
|
382
|
+
decisionStatistic: decision.statistic,
|
|
383
|
+
mcnemar: decision.mcnemar,
|
|
384
|
+
tieFraction,
|
|
385
|
+
n,
|
|
386
|
+
minimumRequired: decision.minimumPairs,
|
|
387
|
+
decisionMethod: decision.method,
|
|
388
|
+
pValue: decision.pValue,
|
|
389
|
+
significant: decision.promote,
|
|
390
|
+
fewRuns: !decision.sufficient
|
|
391
|
+
};
|
|
392
|
+
}
|
|
393
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
394
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
395
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
396
|
+
function detectScale(values) {
|
|
397
|
+
return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
|
|
398
|
+
}
|
|
399
|
+
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
400
|
+
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
401
|
+
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
402
|
+
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
403
|
+
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
404
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
405
|
+
*
|
|
406
|
+
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
407
|
+
* dimension is judged on Tango's score interval rather than a percentile
|
|
408
|
+
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
409
|
+
* is not a valid interval at one. That matters most here because this guard
|
|
410
|
+
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
411
|
+
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
412
|
+
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
413
|
+
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
414
|
+
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
415
|
+
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
416
|
+
* restore the pre-0.134 behaviour. */
|
|
417
|
+
function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
|
|
418
|
+
const out = [];
|
|
419
|
+
for (const dim of criticalDimensions) {
|
|
420
|
+
const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
|
|
421
|
+
if (paired.before.length === 0) continue;
|
|
422
|
+
const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
423
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
424
|
+
const shared = {
|
|
425
|
+
confidence: opts.confidence ?? .95,
|
|
426
|
+
resamples: opts.resamples ?? 2e3,
|
|
427
|
+
statistic: bootstrapStatistic,
|
|
428
|
+
seed: opts.seed ?? 1337
|
|
429
|
+
};
|
|
430
|
+
const guard = decidePairedPromotion(paired.before, paired.after, shared);
|
|
431
|
+
const regression = decidePairedPromotion(paired.after, paired.before, {
|
|
432
|
+
...shared,
|
|
433
|
+
threshold: tolerance
|
|
434
|
+
});
|
|
435
|
+
const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
|
|
436
|
+
out.push({
|
|
437
|
+
dimension: dim,
|
|
438
|
+
bootstrap,
|
|
439
|
+
bootstrapStatistic,
|
|
440
|
+
ci: {
|
|
441
|
+
low: guard.low,
|
|
442
|
+
high: guard.high
|
|
443
|
+
},
|
|
444
|
+
decisionStatistic: guard.statistic,
|
|
445
|
+
mcnemar: guard.mcnemar,
|
|
446
|
+
indeterminate: guard.indeterminate,
|
|
447
|
+
regressed: bootstrap.low < -tolerance || regression.promote,
|
|
448
|
+
tolerance,
|
|
449
|
+
n: paired.before.length
|
|
450
|
+
});
|
|
451
|
+
}
|
|
452
|
+
return out;
|
|
453
|
+
}
|
|
454
|
+
//#endregion
|
|
455
|
+
//#region src/campaign/gates/power-preflight.ts
|
|
456
|
+
/** Two-sided z for the common confidence levels; interpolation is overkill here. */
|
|
457
|
+
function zFor(confidence) {
|
|
458
|
+
if (confidence >= .99) return 2.576;
|
|
459
|
+
if (confidence >= .95) return 1.96;
|
|
460
|
+
if (confidence >= .9) return 1.645;
|
|
461
|
+
return 1.282;
|
|
462
|
+
}
|
|
463
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
464
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
465
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
466
|
+
* observable at this holdout size and worker variance. */
|
|
467
|
+
function powerPreflight(opts) {
|
|
468
|
+
const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
|
|
469
|
+
if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
|
|
470
|
+
const deltaThreshold = opts.deltaThreshold ?? .05;
|
|
471
|
+
const confidence = opts.confidence ?? .95;
|
|
472
|
+
const n = opts.pairedN ?? composites.length;
|
|
473
|
+
if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
|
|
474
|
+
const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
|
|
475
|
+
const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
|
|
476
|
+
const sd = Math.sqrt(variance);
|
|
477
|
+
const z = zFor(confidence);
|
|
478
|
+
const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
|
|
479
|
+
const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
|
|
480
|
+
const headroom = Math.max(0, 1 - mean);
|
|
481
|
+
const underpowered = scaleAssumed && mde > headroom;
|
|
482
|
+
const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
|
|
483
|
+
const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
|
|
484
|
+
return {
|
|
485
|
+
n,
|
|
486
|
+
sd,
|
|
487
|
+
mde,
|
|
488
|
+
baselineMean: mean,
|
|
489
|
+
headroom,
|
|
490
|
+
underpowered,
|
|
491
|
+
scaleAssumed,
|
|
492
|
+
deltaThreshold,
|
|
493
|
+
confidence,
|
|
494
|
+
...sharedChannelCaveat ? { sharedChannelCaveat } : {},
|
|
495
|
+
recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
|
|
496
|
+
};
|
|
497
|
+
}
|
|
498
|
+
//#endregion
|
|
499
|
+
//#region src/campaign/gates/promotion-policy.ts
|
|
500
|
+
/**
|
|
501
|
+
* Promotion policy over the evidence VECTOR — the substrate's answer to "never
|
|
502
|
+
* collapse the multi-objective promotion decision into one scalar." A
|
|
503
|
+
* `defaultProductionGate` is one opinionated composition; this module factors
|
|
504
|
+
* the decision into two reusable pieces so MANY policies can compete over the
|
|
505
|
+
* SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
|
|
506
|
+
*
|
|
507
|
+
* buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
|
|
508
|
+
* PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
|
|
509
|
+
* paretoPolicy(ev) // the default strategy
|
|
510
|
+
* paretoSignificanceGate(options): Gate // bus + policy as a Gate
|
|
511
|
+
*
|
|
512
|
+
* The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
|
|
513
|
+
* potential gain source AND a safety floor (unlike `defaultProductionGate`,
|
|
514
|
+
* where only `composite` can win and `criticalDimensions` are pure floors). A
|
|
515
|
+
* candidate ships iff it weakly DOMINATES the baseline at the confidence level —
|
|
516
|
+
* no objective credibly worse (CI floor breach) AND at least one objective
|
|
517
|
+
* credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
|
|
518
|
+
* (NOT folded into hold: "gather more reps" and "reject" are different actions).
|
|
519
|
+
*
|
|
520
|
+
* Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
|
|
521
|
+
* per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
|
|
522
|
+
* constraints (compose with a budget gate via `composeGate`), not faked CIs.
|
|
523
|
+
*/
|
|
524
|
+
/**
|
|
525
|
+
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
526
|
+
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
527
|
+
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
528
|
+
* a single source of truth governs pairing granularity + scale handling.
|
|
529
|
+
*/
|
|
530
|
+
function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
531
|
+
if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
|
|
532
|
+
const confidence = opts.confidence ?? .95;
|
|
533
|
+
const resamples = opts.resamples ?? 2e3;
|
|
534
|
+
const seed = opts.seed ?? 1337;
|
|
535
|
+
const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores;
|
|
536
|
+
const scenarioIds = new Set(ctx.scenarios.map((s) => s.id));
|
|
537
|
+
const axes = [];
|
|
538
|
+
for (const obj of objectives) {
|
|
539
|
+
let select;
|
|
540
|
+
if (obj.source.kind === "composite") select = (s) => s.composite;
|
|
541
|
+
else {
|
|
542
|
+
const dim = obj.source.dimension;
|
|
543
|
+
select = (s) => s.dimensions[dim];
|
|
544
|
+
}
|
|
545
|
+
const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
|
|
546
|
+
const before = obj.direction === "maximize" ? paired.before : paired.after;
|
|
547
|
+
const after = obj.direction === "maximize" ? paired.after : paired.before;
|
|
548
|
+
const n = paired.before.length;
|
|
549
|
+
const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
550
|
+
const gainThreshold = obj.gainThreshold ?? 0;
|
|
551
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
552
|
+
const improvement = decidePairedPromotion(before, after, {
|
|
553
|
+
confidence,
|
|
554
|
+
resamples,
|
|
555
|
+
statistic: bootstrapStatistic,
|
|
556
|
+
seed,
|
|
557
|
+
threshold: gainThreshold,
|
|
558
|
+
minPairs: opts.minProductiveRuns
|
|
559
|
+
});
|
|
560
|
+
const regression = decidePairedPromotion(after, before, {
|
|
561
|
+
confidence,
|
|
562
|
+
resamples,
|
|
563
|
+
statistic: bootstrapStatistic,
|
|
564
|
+
seed,
|
|
565
|
+
threshold: floorTolerance,
|
|
566
|
+
minPairs: opts.minProductiveRuns
|
|
567
|
+
});
|
|
568
|
+
const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
|
|
569
|
+
confidence,
|
|
570
|
+
resamples,
|
|
571
|
+
statistic: bootstrapStatistic,
|
|
572
|
+
seed
|
|
573
|
+
});
|
|
574
|
+
const floorBreached = bootstrap.low < -floorTolerance || regression.promote;
|
|
575
|
+
const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
|
|
576
|
+
axes.push({
|
|
577
|
+
name: obj.name,
|
|
578
|
+
source: obj.source,
|
|
579
|
+
direction: obj.direction,
|
|
580
|
+
bootstrap,
|
|
581
|
+
bootstrapStatistic,
|
|
582
|
+
ci: {
|
|
583
|
+
low: improvement.low,
|
|
584
|
+
high: improvement.high
|
|
585
|
+
},
|
|
586
|
+
decisionStatistic: improvement.statistic,
|
|
587
|
+
mcnemar: improvement.mcnemar,
|
|
588
|
+
indeterminate: improvement.indeterminate,
|
|
589
|
+
n,
|
|
590
|
+
minimumRequired: improvement.minimumPairs,
|
|
591
|
+
decisionMethod: improvement.method,
|
|
592
|
+
gainThreshold,
|
|
593
|
+
floorTolerance,
|
|
594
|
+
verdict
|
|
595
|
+
});
|
|
596
|
+
}
|
|
597
|
+
const ns = axes.map((a) => a.n).filter((n) => n > 0);
|
|
598
|
+
return {
|
|
599
|
+
axes,
|
|
600
|
+
minN: ns.length > 0 ? Math.min(...ns) : 0,
|
|
601
|
+
cost: {
|
|
602
|
+
candidate: ctx.cost.candidate,
|
|
603
|
+
baseline: ctx.cost.baseline
|
|
604
|
+
}
|
|
605
|
+
};
|
|
606
|
+
}
|
|
607
|
+
/**
|
|
608
|
+
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
609
|
+
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
610
|
+
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
611
|
+
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
612
|
+
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
613
|
+
*/
|
|
614
|
+
const paretoPolicy = (ev) => {
|
|
615
|
+
const contributingGates = ev.axes.map((ax) => ({
|
|
616
|
+
name: `objective:${ax.name}`,
|
|
617
|
+
status: ax.verdict === "regressed" ? "fail" : ax.verdict === "few_runs" ? "not_evaluated" : "pass",
|
|
618
|
+
detail: {
|
|
619
|
+
direction: ax.direction,
|
|
620
|
+
source: ax.source,
|
|
621
|
+
verdict: ax.verdict,
|
|
622
|
+
n: ax.n,
|
|
623
|
+
deltaMedian: ax.bootstrap.median,
|
|
624
|
+
ciLow: ax.ci.low,
|
|
625
|
+
ciHigh: ax.ci.high,
|
|
626
|
+
decisionStatistic: ax.decisionStatistic,
|
|
627
|
+
decisionMethod: ax.decisionMethod,
|
|
628
|
+
mcnemar: ax.mcnemar,
|
|
629
|
+
indeterminate: ax.indeterminate,
|
|
630
|
+
bootstrapCiLow: ax.bootstrap.low,
|
|
631
|
+
bootstrapCiHigh: ax.bootstrap.high,
|
|
632
|
+
confidence: ax.bootstrap.confidence,
|
|
633
|
+
gainThreshold: ax.gainThreshold,
|
|
634
|
+
floorTolerance: ax.floorTolerance
|
|
635
|
+
}
|
|
636
|
+
}));
|
|
637
|
+
const regressed = ev.axes.filter((a) => a.verdict === "regressed");
|
|
638
|
+
const fewRuns = ev.axes.filter((a) => a.verdict === "few_runs");
|
|
639
|
+
const improved = ev.axes.filter((a) => a.verdict === "improved");
|
|
640
|
+
let decision;
|
|
641
|
+
const reasons = [];
|
|
642
|
+
if (regressed.length > 0) {
|
|
643
|
+
decision = "hold";
|
|
644
|
+
for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
|
|
645
|
+
} else if (fewRuns.length > 0) {
|
|
646
|
+
decision = "need_more_work";
|
|
647
|
+
for (const a of fewRuns) reasons.push(`objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`);
|
|
648
|
+
} else if (improved.length > 0) {
|
|
649
|
+
decision = "ship";
|
|
650
|
+
reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; no objective regressed`);
|
|
651
|
+
} else {
|
|
652
|
+
decision = "hold";
|
|
653
|
+
reasons.push("no Pareto improvement: candidate statistically equivalent to baseline on every objective");
|
|
654
|
+
}
|
|
655
|
+
const composite = ev.axes.find((a) => a.source.kind === "composite") ?? ev.axes[0];
|
|
656
|
+
return {
|
|
657
|
+
decision,
|
|
658
|
+
reasons,
|
|
659
|
+
contributingGates,
|
|
660
|
+
delta: composite?.bootstrap.median
|
|
661
|
+
};
|
|
662
|
+
};
|
|
663
|
+
/**
|
|
664
|
+
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
665
|
+
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
666
|
+
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
667
|
+
*/
|
|
668
|
+
function paretoSignificanceGate(options) {
|
|
669
|
+
if (options.objectives.length === 0) throw new Error("paretoSignificanceGate: at least 1 objective required");
|
|
670
|
+
const policy = options.policy ?? paretoPolicy;
|
|
671
|
+
return {
|
|
672
|
+
name: options.name ?? "paretoSignificanceGate",
|
|
673
|
+
async decide(ctx) {
|
|
674
|
+
const ev = buildEvidenceVector(ctx, options.objectives, options);
|
|
675
|
+
return policy(ev);
|
|
676
|
+
}
|
|
677
|
+
};
|
|
678
|
+
}
|
|
679
|
+
//#endregion
|
|
680
|
+
export { TIE_WARN_FRACTION as a, heldoutSignificance as c, pairedDecisionShape as d, minimumPairsForPairedDeltaTest as f, powerPreflight as i, pairHoldout as l, paretoPolicy as n, detectScale as o, pairedDeltaTest as p, paretoSignificanceGate as r, dimensionRegressions as s, buildEvidenceVector as t, decidePairedPromotion as u };
|
|
681
|
+
|
|
682
|
+
//# sourceMappingURL=promotion-policy-CrLrmys8.js.map
|