@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,1108 @@
|
|
|
1
|
+
import { c as ValidationError, n as CaptureIntegrityError } from "../errors-D-LKuDhb.js";
|
|
2
|
+
import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "../pre-registration-DakwTRXk.js";
|
|
3
|
+
import { A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, E as pairedBootstrap, G as wilson, M as pairedRiskDifferenceScore, S as mcnemarRequiredN, b as mcnemar, c as bonferroni, h as holm, j as pairedRiskDifferenceExact, k as pairedMde, m as eProcess, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, x as mcnemarPower, z as requiredPairedSampleSize } from "../statistics-ByxzSiOM.js";
|
|
4
|
+
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "../paired-arms-iZ08VFMN.js";
|
|
5
|
+
import { o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "../experiment-tracker-CnRICnMl.js";
|
|
6
|
+
import { c as heldoutSignificance, i as powerPreflight, l as pairHoldout, n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-CrLrmys8.js";
|
|
7
|
+
import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-D-BLJBKU.js";
|
|
8
|
+
import { n as pairedEvalueSequence } from "../sequential-Br0mAPHA.js";
|
|
9
|
+
//#region src/experiment/ast.ts
|
|
10
|
+
/**
|
|
11
|
+
* The registered-rule AST: every rule an experiment registers is DATA.
|
|
12
|
+
*
|
|
13
|
+
* A closure cannot be canonicalized or hashed; a node tree can. `sealExperiment`
|
|
14
|
+
* hashes the whole tree, and every interpreter in this file takes only a node
|
|
15
|
+
* plus evidence records — no parameter for alpha, threshold, metric, or
|
|
16
|
+
* stopping rule exists on any executable surface. The registered object and
|
|
17
|
+
* the executed object are therefore the same object, and registered-vs-ran
|
|
18
|
+
* drift is unrepresentable rather than checked.
|
|
19
|
+
*
|
|
20
|
+
* Node families:
|
|
21
|
+
* Predicate closed-key comparisons — the only leaf
|
|
22
|
+
* AdmissionRule monotone funnel stages with registered waivers
|
|
23
|
+
* SelectionRule deterministic subsets over a closed field set
|
|
24
|
+
* Estimand what the experiment measures
|
|
25
|
+
* IntervalSpec how uncertainty is computed, seed included
|
|
26
|
+
* Condition decision guards over named derived quantities
|
|
27
|
+
* DecisionRule ordered verdict table, or a registered absence of one
|
|
28
|
+
* Obligation a control that must exist before a verdict class is read
|
|
29
|
+
* ValidityGate pre-spend design checks
|
|
30
|
+
* HaltRule gates as prerequisites — failure refuses the spend
|
|
31
|
+
* BudgetRule spend schedules with a named ledger
|
|
32
|
+
* MatchedBudgetRule arm budget matching as a refusal
|
|
33
|
+
* ReissuePolicy carrier faults are reissued; model outcomes stand
|
|
34
|
+
*/
|
|
35
|
+
/** A decision rule's branches did not cover the evidence. */
|
|
36
|
+
var DecisionTableNotTotalError = class extends ValidationError {};
|
|
37
|
+
/** Read a dot-separated field path. Missing segments yield `undefined`. */
|
|
38
|
+
function readField(record, path) {
|
|
39
|
+
let current = record;
|
|
40
|
+
for (const key of path.split(".")) {
|
|
41
|
+
if (current === null || typeof current !== "object") return void 0;
|
|
42
|
+
current = current[key];
|
|
43
|
+
}
|
|
44
|
+
return current;
|
|
45
|
+
}
|
|
46
|
+
/** Evaluate a predicate against one evidence record. */
|
|
47
|
+
function evaluatePredicate(predicate, record) {
|
|
48
|
+
switch (predicate.kind) {
|
|
49
|
+
case "compare": return compareValues(readField(record, predicate.field), predicate.op, predicate.value);
|
|
50
|
+
case "in": return predicate.values.includes(readField(record, predicate.field));
|
|
51
|
+
case "all": return predicate.of.every((p) => evaluatePredicate(p, record));
|
|
52
|
+
case "any": return predicate.of.some((p) => evaluatePredicate(p, record));
|
|
53
|
+
case "not": return !evaluatePredicate(predicate.of, record);
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
/** Numbers compare numerically; everything else compares as strings. */
|
|
57
|
+
function compareValues(value, op, target) {
|
|
58
|
+
if (op === "eq") return value === target;
|
|
59
|
+
if (op === "ne") return value !== target;
|
|
60
|
+
const numeric = typeof value === "number" && typeof target === "number";
|
|
61
|
+
const left = numeric ? value : String(value);
|
|
62
|
+
const right = numeric ? target : String(target);
|
|
63
|
+
switch (op) {
|
|
64
|
+
case "lt": return left < right;
|
|
65
|
+
case "lte": return left <= right;
|
|
66
|
+
case "gt": return left > right;
|
|
67
|
+
case "gte": return left >= right;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* Execute a selection rule.
|
|
72
|
+
*
|
|
73
|
+
* Round-robin walks groups in lexicographic order and takes ids in
|
|
74
|
+
* within-group order until `take` ids are chosen. Filter-of keeps the ids of
|
|
75
|
+
* `bases[rule.base]` whose record satisfies the predicate, in the registered
|
|
76
|
+
* order. Ids absent from `records` are evaluated on their id alone (fields
|
|
77
|
+
* derived from the id via `idFields`), so a sealed base outlives its source
|
|
78
|
+
* records.
|
|
79
|
+
*/
|
|
80
|
+
function runSelectionRule(rule, records, options) {
|
|
81
|
+
if (rule.kind === "round-robin") {
|
|
82
|
+
const allowed = new Set(rule.reads);
|
|
83
|
+
for (const field of [rule.groupBy, rule.withinOrder.field]) if (!allowed.has(field)) throw new ValidationError(`runSelectionRule: round-robin reads '${field}' but its closed read set is [${rule.reads.join(", ")}]`);
|
|
84
|
+
const byGroup = /* @__PURE__ */ new Map();
|
|
85
|
+
for (const record of records) {
|
|
86
|
+
const group = String(readField(record, rule.groupBy));
|
|
87
|
+
const id = String(readField(record, rule.withinOrder.field));
|
|
88
|
+
const bucket = byGroup.get(group);
|
|
89
|
+
if (bucket) bucket.push(id);
|
|
90
|
+
else byGroup.set(group, [id]);
|
|
91
|
+
}
|
|
92
|
+
for (const ids of byGroup.values()) {
|
|
93
|
+
ids.sort();
|
|
94
|
+
if (rule.withinOrder.dir === "desc") ids.reverse();
|
|
95
|
+
}
|
|
96
|
+
const groups = [...byGroup.keys()].sort();
|
|
97
|
+
const chosen = [];
|
|
98
|
+
let cursor = 0;
|
|
99
|
+
while (chosen.length < rule.take && groups.some((g) => byGroup.get(g).length > 0)) {
|
|
100
|
+
const group = groups[cursor % groups.length];
|
|
101
|
+
const ids = byGroup.get(group);
|
|
102
|
+
if (ids.length > 0) chosen.push(ids.shift());
|
|
103
|
+
cursor += 1;
|
|
104
|
+
}
|
|
105
|
+
return chosen;
|
|
106
|
+
}
|
|
107
|
+
const base = options.bases?.[rule.base];
|
|
108
|
+
if (!base) throw new ValidationError(`runSelectionRule: filter-of base '${rule.base}' was not provided`);
|
|
109
|
+
const index = new Map(records.map((r) => [String(readField(r, options.idField)), r]));
|
|
110
|
+
const sorted = [...base.filter((id) => {
|
|
111
|
+
const record = index.get(id) ?? options.idFields?.(id);
|
|
112
|
+
if (!record) throw new ValidationError(`runSelectionRule: base id '${id}' has no record and no idFields derivation`);
|
|
113
|
+
return evaluatePredicate(rule.keep, record);
|
|
114
|
+
})].sort();
|
|
115
|
+
if (rule.order.dir === "desc") sorted.reverse();
|
|
116
|
+
return sorted;
|
|
117
|
+
}
|
|
118
|
+
function evaluateSetExpr(expr, rows, armField, idField) {
|
|
119
|
+
if (expr.kind === "rows-where") {
|
|
120
|
+
const ids = /* @__PURE__ */ new Set();
|
|
121
|
+
for (const row of rows) {
|
|
122
|
+
if (String(readField(row, armField)) !== expr.arm) continue;
|
|
123
|
+
if (evaluatePredicate(expr.event, row)) ids.add(String(readField(row, idField)));
|
|
124
|
+
}
|
|
125
|
+
return ids;
|
|
126
|
+
}
|
|
127
|
+
if (expr.of.length === 0) throw new ValidationError("computeEstimand: empty intersect");
|
|
128
|
+
const [first, ...rest] = expr.of.map((e) => evaluateSetExpr(e, rows, armField, idField));
|
|
129
|
+
const out = /* @__PURE__ */ new Set();
|
|
130
|
+
for (const id of first) if (rest.every((s) => s.has(id))) out.add(id);
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
/** Compute an estimand over evidence rows. Pure; reads only registered fields. */
|
|
134
|
+
function computeEstimand(estimand, rows) {
|
|
135
|
+
switch (estimand.kind) {
|
|
136
|
+
case "rate": {
|
|
137
|
+
const numerator = rows.filter((r) => evaluatePredicate(estimand.event, r)).length;
|
|
138
|
+
if (rows.length === 0) throw new ValidationError("computeEstimand: rate over zero rows");
|
|
139
|
+
return {
|
|
140
|
+
value: numerator / rows.length,
|
|
141
|
+
numerator,
|
|
142
|
+
denominator: rows.length
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
case "rate-at-least-once": {
|
|
146
|
+
const byGroup = /* @__PURE__ */ new Map();
|
|
147
|
+
for (const row of rows) {
|
|
148
|
+
const group = String(readField(row, estimand.groupBy));
|
|
149
|
+
const hit = evaluatePredicate(estimand.event, row);
|
|
150
|
+
byGroup.set(group, (byGroup.get(group) ?? false) || hit);
|
|
151
|
+
}
|
|
152
|
+
if (byGroup.size === 0) throw new ValidationError("computeEstimand: rate-at-least-once over zero groups");
|
|
153
|
+
const numerator = [...byGroup.values()].filter(Boolean).length;
|
|
154
|
+
return {
|
|
155
|
+
value: numerator / byGroup.size,
|
|
156
|
+
numerator,
|
|
157
|
+
denominator: byGroup.size
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
case "paired-mean-diff": {
|
|
161
|
+
const byPair = /* @__PURE__ */ new Map();
|
|
162
|
+
for (const row of rows) {
|
|
163
|
+
const arm = String(readField(row, estimand.armField));
|
|
164
|
+
if (arm !== estimand.treatment && arm !== estimand.control) continue;
|
|
165
|
+
const pair = String(readField(row, estimand.pairBy));
|
|
166
|
+
const value = readField(row, estimand.value);
|
|
167
|
+
if (typeof value !== "number") throw new ValidationError(`computeEstimand: paired-mean-diff value field '${estimand.value}' is not a number on pair '${pair}'`);
|
|
168
|
+
const slot = byPair.get(pair) ?? {};
|
|
169
|
+
if (arm === estimand.treatment) slot.treatment = value;
|
|
170
|
+
else slot.control = value;
|
|
171
|
+
byPair.set(pair, slot);
|
|
172
|
+
}
|
|
173
|
+
if (byPair.size === 0) throw new ValidationError("computeEstimand: paired-mean-diff over zero pairs");
|
|
174
|
+
let sum = 0;
|
|
175
|
+
for (const slot of byPair.values()) sum += (slot.treatment ?? 0) - (slot.control ?? 0);
|
|
176
|
+
return {
|
|
177
|
+
value: sum / byPair.size,
|
|
178
|
+
numerator: sum,
|
|
179
|
+
denominator: byPair.size
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
case "set-ratio": {
|
|
183
|
+
const numeratorSet = evaluateSetExpr(estimand.numerator, rows, estimand.armField, estimand.idField);
|
|
184
|
+
const denominatorSet = evaluateSetExpr(estimand.denominator, rows, estimand.armField, estimand.idField);
|
|
185
|
+
if (denominatorSet.size === 0) throw new ValidationError("computeEstimand: set-ratio denominator set is empty");
|
|
186
|
+
return {
|
|
187
|
+
value: numeratorSet.size / denominatorSet.size,
|
|
188
|
+
numerator: numeratorSet.size,
|
|
189
|
+
denominator: denominatorSet.size
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Execute an interval spec.
|
|
196
|
+
*
|
|
197
|
+
* Cluster-bootstrap resamples whole clusters of the per-row `value` field and
|
|
198
|
+
* takes percentile bounds of the pooled mean. Clopper-Pearson computes the
|
|
199
|
+
* exact binomial interval and requires `successes`/`trials` evidence instead
|
|
200
|
+
* of rows.
|
|
201
|
+
*/
|
|
202
|
+
function computeInterval(spec, evidence) {
|
|
203
|
+
if (spec.kind === "cluster-bootstrap") {
|
|
204
|
+
if (evidence.kind !== "rows") throw new ValidationError("computeInterval: cluster-bootstrap requires row evidence");
|
|
205
|
+
const clusters = /* @__PURE__ */ new Map();
|
|
206
|
+
for (const row of evidence.rows) {
|
|
207
|
+
const cluster = String(readField(row, spec.clusterBy));
|
|
208
|
+
const value = readField(row, evidence.value);
|
|
209
|
+
if (typeof value !== "number") throw new ValidationError(`computeInterval: value field '${evidence.value}' is not a number in cluster '${cluster}'`);
|
|
210
|
+
const bucket = clusters.get(cluster);
|
|
211
|
+
if (bucket) bucket.push(value);
|
|
212
|
+
else clusters.set(cluster, [value]);
|
|
213
|
+
}
|
|
214
|
+
const clusterValues = [...clusters.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([, values]) => values);
|
|
215
|
+
if (clusterValues.length < 2) throw new ValidationError(`computeInterval: cluster-bootstrap needs >= 2 clusters, got ${clusterValues.length}`);
|
|
216
|
+
const rng = mulberry32(spec.seed);
|
|
217
|
+
const means = new Array(spec.resamples);
|
|
218
|
+
for (let draw = 0; draw < spec.resamples; draw++) {
|
|
219
|
+
let sum = 0;
|
|
220
|
+
let count = 0;
|
|
221
|
+
for (let pick = 0; pick < clusterValues.length; pick++) {
|
|
222
|
+
const cluster = clusterValues[Math.floor(rng() * clusterValues.length)];
|
|
223
|
+
for (const value of cluster) sum += value;
|
|
224
|
+
count += cluster.length;
|
|
225
|
+
}
|
|
226
|
+
means[draw] = sum / count;
|
|
227
|
+
}
|
|
228
|
+
means.sort((a, b) => a - b);
|
|
229
|
+
const alpha = 1 - spec.level;
|
|
230
|
+
const lowerIndex = Math.floor(alpha / 2 * spec.resamples);
|
|
231
|
+
const upperIndex = Math.min(spec.resamples - 1, Math.ceil((1 - alpha / 2) * spec.resamples) - 1);
|
|
232
|
+
return {
|
|
233
|
+
lower: means[lowerIndex],
|
|
234
|
+
upper: means[Math.max(lowerIndex, upperIndex)],
|
|
235
|
+
level: spec.level
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
if (evidence.kind !== "binomial") throw new ValidationError("computeInterval: clopper-pearson requires binomial evidence");
|
|
239
|
+
const { successes, trials } = evidence;
|
|
240
|
+
if (!Number.isInteger(successes) || !Number.isInteger(trials) || trials <= 0 || successes < 0) throw new ValidationError(`computeInterval: clopper-pearson needs 0 <= successes <= trials, got ${successes}/${trials}`);
|
|
241
|
+
if (successes > trials) throw new ValidationError(`computeInterval: clopper-pearson successes ${successes} exceed trials ${trials}`);
|
|
242
|
+
const alpha = 1 - spec.level;
|
|
243
|
+
return {
|
|
244
|
+
lower: successes === 0 ? 0 : binomialQuantile(successes, trials, alpha / 2, "lower"),
|
|
245
|
+
upper: successes === trials ? 1 : binomialQuantile(successes, trials, alpha / 2, "upper"),
|
|
246
|
+
level: spec.level
|
|
247
|
+
};
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Clopper-Pearson bound by bisection on the binomial tail. The lower bound is
|
|
251
|
+
* the p with P(X >= successes | p) = alpha; the upper is the p with
|
|
252
|
+
* P(X <= successes | p) = alpha. Deterministic, no special functions.
|
|
253
|
+
*/
|
|
254
|
+
function binomialQuantile(successes, trials, alpha, side) {
|
|
255
|
+
const tail = (p) => {
|
|
256
|
+
let sum = 0;
|
|
257
|
+
for (let k = 0; k <= trials; k++) {
|
|
258
|
+
if (!(side === "lower" ? k >= successes : k <= successes)) continue;
|
|
259
|
+
sum += Math.exp(logBinomialPmf(k, trials, p));
|
|
260
|
+
}
|
|
261
|
+
return sum;
|
|
262
|
+
};
|
|
263
|
+
let lo = 0;
|
|
264
|
+
let hi = 1;
|
|
265
|
+
for (let iter = 0; iter < 100; iter++) {
|
|
266
|
+
const mid = (lo + hi) / 2;
|
|
267
|
+
if (tail(mid) < alpha) if (side === "lower") lo = mid;
|
|
268
|
+
else hi = mid;
|
|
269
|
+
else if (side === "lower") hi = mid;
|
|
270
|
+
else lo = mid;
|
|
271
|
+
}
|
|
272
|
+
return (lo + hi) / 2;
|
|
273
|
+
}
|
|
274
|
+
function logBinomialPmf(k, n, p) {
|
|
275
|
+
if (p <= 0) return k === 0 ? 0 : Number.NEGATIVE_INFINITY;
|
|
276
|
+
if (p >= 1) return k === n ? 0 : Number.NEGATIVE_INFINITY;
|
|
277
|
+
return logChoose(n, k) + k * Math.log(p) + (n - k) * Math.log(1 - p);
|
|
278
|
+
}
|
|
279
|
+
function logChoose(n, k) {
|
|
280
|
+
return logFactorial(n) - logFactorial(k) - logFactorial(n - k);
|
|
281
|
+
}
|
|
282
|
+
const LOG_FACTORIAL_CACHE = [0];
|
|
283
|
+
function logFactorial(n) {
|
|
284
|
+
for (let i = LOG_FACTORIAL_CACHE.length; i <= n; i++) LOG_FACTORIAL_CACHE[i] = LOG_FACTORIAL_CACHE[i - 1] + Math.log(i);
|
|
285
|
+
return LOG_FACTORIAL_CACHE[n];
|
|
286
|
+
}
|
|
287
|
+
function evaluateCondition(condition, evidence) {
|
|
288
|
+
switch (condition.kind) {
|
|
289
|
+
case "interval-excludes-zero": {
|
|
290
|
+
const interval = evidence.intervals[condition.interval];
|
|
291
|
+
if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
|
|
292
|
+
return (interval.lower > 0 || interval.upper < 0) && (condition.sign === "positive" ? interval.lower > 0 : interval.upper < 0);
|
|
293
|
+
}
|
|
294
|
+
case "interval-includes-zero": {
|
|
295
|
+
const interval = evidence.intervals[condition.interval];
|
|
296
|
+
if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
|
|
297
|
+
return interval.lower <= 0 && interval.upper >= 0;
|
|
298
|
+
}
|
|
299
|
+
case "quantity-threshold": {
|
|
300
|
+
const value = evidence.quantities[condition.quantity];
|
|
301
|
+
if (value === void 0) throw new ValidationError(`evaluateCondition: quantity '${condition.quantity}' is not in the evidence`);
|
|
302
|
+
return compareValues(value, condition.op, condition.value);
|
|
303
|
+
}
|
|
304
|
+
case "obligation-met": return evidence.obligationsMet[condition.obligation] === true;
|
|
305
|
+
case "all": return condition.of.every((c) => evaluateCondition(c, evidence));
|
|
306
|
+
case "any": return condition.of.some((c) => evaluateCondition(c, evidence));
|
|
307
|
+
case "not": return !evaluateCondition(condition.of, evidence);
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
function executeDecisionRule(rule, evidence) {
|
|
311
|
+
if (rule.kind === "report-only") return {
|
|
312
|
+
verdict: "report-only",
|
|
313
|
+
report: [...rule.estimands, ...rule.intervals]
|
|
314
|
+
};
|
|
315
|
+
for (const branch of rule.branches) if (evaluateCondition(branch.when, evidence)) return {
|
|
316
|
+
verdict: branch.verdict,
|
|
317
|
+
report: branch.report
|
|
318
|
+
};
|
|
319
|
+
throw new DecisionTableNotTotalError("executeDecisionRule: decision table is not total — no branch matched the evidence");
|
|
320
|
+
}
|
|
321
|
+
/**
|
|
322
|
+
* Replicate-flip counting over graded states. A state whose replicates split
|
|
323
|
+
* between pass and fail is flipping; its flip rate is the minority share.
|
|
324
|
+
*/
|
|
325
|
+
function evaluateOracleDeterminismGate(id, gate, repsByState) {
|
|
326
|
+
const evidence = {};
|
|
327
|
+
let passed = true;
|
|
328
|
+
for (const [state, reps] of Object.entries(repsByState)) {
|
|
329
|
+
const passes = reps.filter(Boolean).length;
|
|
330
|
+
const flipRate = reps.length === 0 ? 0 : Math.min(passes, reps.length - passes) / reps.length;
|
|
331
|
+
evidence[state] = {
|
|
332
|
+
passes,
|
|
333
|
+
replicates: reps.length,
|
|
334
|
+
flipRate
|
|
335
|
+
};
|
|
336
|
+
if (flipRate > gate.maxFlipRate) passed = false;
|
|
337
|
+
}
|
|
338
|
+
return {
|
|
339
|
+
id,
|
|
340
|
+
passed,
|
|
341
|
+
evidence
|
|
342
|
+
};
|
|
343
|
+
}
|
|
344
|
+
/**
|
|
345
|
+
* Join two population snapshots on `joinOn` and compare the registered fields.
|
|
346
|
+
* Only rows present in both snapshots are compared; a presence change is a
|
|
347
|
+
* different failure and needs its own gate.
|
|
348
|
+
*/
|
|
349
|
+
function evaluatePopulationReproducibilityGate(id, gate, populations) {
|
|
350
|
+
const rightByKey = new Map(populations.right.map((r) => [String(readField(r, gate.joinOn)), r]));
|
|
351
|
+
const changed = [];
|
|
352
|
+
for (const left of populations.left) {
|
|
353
|
+
const key = String(readField(left, gate.joinOn));
|
|
354
|
+
const right = rightByKey.get(key);
|
|
355
|
+
if (!right) continue;
|
|
356
|
+
const moved = gate.compare.filter((f) => readField(left, f) !== readField(right, f));
|
|
357
|
+
if (moved.length > 0) changed.push(`${key} ${moved.map((f) => `${f}:${String(readField(left, f))}->${String(readField(right, f))}`).join(" ")}`);
|
|
358
|
+
}
|
|
359
|
+
return {
|
|
360
|
+
id,
|
|
361
|
+
passed: changed.length <= gate.maxChangedRows,
|
|
362
|
+
evidence: changed
|
|
363
|
+
};
|
|
364
|
+
}
|
|
365
|
+
/** The registered claim about provenance must hold on the provenance record. */
|
|
366
|
+
function evaluateProvenanceGate(id, gate, provenance) {
|
|
367
|
+
const passed = evaluatePredicate(gate.claim, provenance);
|
|
368
|
+
return {
|
|
369
|
+
id,
|
|
370
|
+
passed,
|
|
371
|
+
evidence: { claimHolds: passed }
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
/** Final path segment equality between the pinned and the served identity. */
|
|
375
|
+
function evaluateIdentityGate(id, _gate, identities) {
|
|
376
|
+
const basename = (s) => s.split("/").pop() ?? s;
|
|
377
|
+
const passed = basename(identities.pinned) === basename(identities.served);
|
|
378
|
+
return {
|
|
379
|
+
id,
|
|
380
|
+
passed,
|
|
381
|
+
evidence: {
|
|
382
|
+
pinned: identities.pinned,
|
|
383
|
+
served: identities.served,
|
|
384
|
+
matched: passed
|
|
385
|
+
}
|
|
386
|
+
};
|
|
387
|
+
}
|
|
388
|
+
/**
|
|
389
|
+
* The design's power curve must reach the registered target at some grid
|
|
390
|
+
* effect. The curve must cover the registered effect grid exactly — a curve
|
|
391
|
+
* computed on a different grid is different evidence and is refused.
|
|
392
|
+
*/
|
|
393
|
+
function evaluatePowerFloorGate(id, gate, curve) {
|
|
394
|
+
const byEffect = new Map(curve.map((point) => [point.effect, point.power]));
|
|
395
|
+
const missing = gate.effectGrid.filter((effect) => !byEffect.has(effect));
|
|
396
|
+
if (missing.length > 0) throw new ValidationError(`evaluatePowerFloorGate: curve does not cover registered effects [${missing.join(", ")}]`);
|
|
397
|
+
const powers = gate.effectGrid.map((effect) => byEffect.get(effect));
|
|
398
|
+
const maxPower = Math.max(...powers);
|
|
399
|
+
return {
|
|
400
|
+
id,
|
|
401
|
+
passed: maxPower >= gate.target,
|
|
402
|
+
evidence: {
|
|
403
|
+
target: gate.target,
|
|
404
|
+
maxPower,
|
|
405
|
+
curve: gate.effectGrid.map((effect) => ({
|
|
406
|
+
effect,
|
|
407
|
+
power: byEffect.get(effect)
|
|
408
|
+
}))
|
|
409
|
+
}
|
|
410
|
+
};
|
|
411
|
+
}
|
|
412
|
+
function evaluateHaltRule(halt, gates) {
|
|
413
|
+
const seen = new Map(gates.map((g) => [g.id, g]));
|
|
414
|
+
const missing = halt.when.gates.filter((id) => !seen.has(id));
|
|
415
|
+
if (missing.length > 0) throw new ValidationError(`evaluateHaltRule: halt references gates that were not evaluated: [${missing.join(", ")}]`);
|
|
416
|
+
const failed = halt.when.gates.filter((id) => !seen.get(id).passed);
|
|
417
|
+
return failed.length > 0 ? {
|
|
418
|
+
fired: true,
|
|
419
|
+
action: halt.action,
|
|
420
|
+
failedGates: failed
|
|
421
|
+
} : {
|
|
422
|
+
fired: false,
|
|
423
|
+
action: null,
|
|
424
|
+
failedGates: []
|
|
425
|
+
};
|
|
426
|
+
}
|
|
427
|
+
/**
|
|
428
|
+
* Execute the uniform-pass schedule against measured pass costs. Pass 1 always
|
|
429
|
+
* runs; each later pass runs only when the cumulative spend plus the last
|
|
430
|
+
* measured pass cost stays at or under the registered ceiling. The registered
|
|
431
|
+
* ledger is the pre-spend the ceiling counts.
|
|
432
|
+
*/
|
|
433
|
+
function runUniformPassBudget(rule, measuredPassCosts) {
|
|
434
|
+
let cumulative = rule.ledger.reduce((sum, entry) => sum + entry.usd, 0);
|
|
435
|
+
const decisions = [];
|
|
436
|
+
let uniformN = 0;
|
|
437
|
+
for (let pass = 1; pass <= rule.maxPasses; pass++) {
|
|
438
|
+
if (pass === 1) {
|
|
439
|
+
if (measuredPassCosts[0] === void 0) break;
|
|
440
|
+
cumulative += measuredPassCosts[0];
|
|
441
|
+
uniformN = 1;
|
|
442
|
+
continue;
|
|
443
|
+
}
|
|
444
|
+
const projected = measuredPassCosts[pass - 2];
|
|
445
|
+
if (projected === void 0) break;
|
|
446
|
+
const go = cumulative + projected <= rule.ceilingUsd;
|
|
447
|
+
decisions.push({
|
|
448
|
+
pass,
|
|
449
|
+
cumulativeBefore: cumulative,
|
|
450
|
+
projected,
|
|
451
|
+
go
|
|
452
|
+
});
|
|
453
|
+
if (!go || measuredPassCosts[pass - 1] === void 0) break;
|
|
454
|
+
cumulative += measuredPassCosts[pass - 1];
|
|
455
|
+
uniformN = pass;
|
|
456
|
+
}
|
|
457
|
+
return {
|
|
458
|
+
decisions,
|
|
459
|
+
uniformN
|
|
460
|
+
};
|
|
461
|
+
}
|
|
462
|
+
/**
|
|
463
|
+
* Walk the registered n-ladder and pick the first affordable step. When no
|
|
464
|
+
* step fits the ceiling, the rule refuses and reports the projection instead
|
|
465
|
+
* of shrinking the row set — "never subset rows" is the registered invariant.
|
|
466
|
+
*/
|
|
467
|
+
function projectNLadderBudget(rule, measured) {
|
|
468
|
+
const projections = rule.steps.map((n) => {
|
|
469
|
+
const projectedUsd = measured.unitCostUsd * measured.rows * n;
|
|
470
|
+
return {
|
|
471
|
+
n,
|
|
472
|
+
projectedUsd,
|
|
473
|
+
affordable: projectedUsd <= rule.ceilingUsd
|
|
474
|
+
};
|
|
475
|
+
});
|
|
476
|
+
const first = projections.find((p) => p.affordable);
|
|
477
|
+
if (first) return {
|
|
478
|
+
chosenN: first.n,
|
|
479
|
+
projections,
|
|
480
|
+
refusal: null
|
|
481
|
+
};
|
|
482
|
+
return {
|
|
483
|
+
chosenN: null,
|
|
484
|
+
projections,
|
|
485
|
+
refusal: {
|
|
486
|
+
onExhaust: rule.onExhaust,
|
|
487
|
+
reason: `no ladder step fits the ${rule.ceilingUsd} USD ceiling at ${measured.rows} rows x ${measured.unitCostUsd} USD per unit`
|
|
488
|
+
}
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
/**
|
|
492
|
+
* Classify one rollout event under the registered reissue policy. A carrier
|
|
493
|
+
* event within the issue budget is reissued; a model outcome always stands;
|
|
494
|
+
* a carrier event past `maxIssues` is exhausted and reported, never retried.
|
|
495
|
+
*/
|
|
496
|
+
function classifyReissue(policy, event, issuesSoFar) {
|
|
497
|
+
if (!policy.carrierEvents.includes(event)) return "stands";
|
|
498
|
+
return issuesSoFar < policy.maxIssues ? "reissue" : "exhausted";
|
|
499
|
+
}
|
|
500
|
+
//#endregion
|
|
501
|
+
//#region src/experiment/budget.ts
|
|
502
|
+
/**
|
|
503
|
+
* Matched-budget verification between arms, as a refusal object.
|
|
504
|
+
*
|
|
505
|
+
* A paired contrast is only meaningful when both arms spent comparable
|
|
506
|
+
* resources; "realized prompt and completion tokens must agree within 5%" is
|
|
507
|
+
* a registered rule, so its verification returns a verdict artifact — the
|
|
508
|
+
* refusal lives inside the result, never in prose beside it.
|
|
509
|
+
*/
|
|
510
|
+
/** Arms whose realized budgets diverge past the registered tolerance. */
|
|
511
|
+
var MatchedBudgetError = class extends CaptureIntegrityError {};
|
|
512
|
+
/**
|
|
513
|
+
* Compare realized per-arm spend under the registered tolerance. Requires at
|
|
514
|
+
* least two arms — a single arm has nothing to match against. Negative or
|
|
515
|
+
* non-finite token counts are refused as evidence corruption, not compared.
|
|
516
|
+
*/
|
|
517
|
+
function verifyMatchedBudgets(rule, arms) {
|
|
518
|
+
if (arms.length < 2) throw new MatchedBudgetError(`verifyMatchedBudgets: need >= 2 arms to match budgets, got ${arms.length}`);
|
|
519
|
+
for (const arm of arms) if (!Number.isFinite(arm.realizedTokens) || arm.realizedTokens < 0) throw new MatchedBudgetError(`verifyMatchedBudgets: arm '${arm.armId}' has invalid realized tokens ${arm.realizedTokens}`);
|
|
520
|
+
const sorted = [...arms].sort((a, b) => a.realizedTokens - b.realizedTokens);
|
|
521
|
+
const min = sorted[0];
|
|
522
|
+
const max = sorted[sorted.length - 1];
|
|
523
|
+
const maxRelativeGap = max.realizedTokens === 0 ? 0 : (max.realizedTokens - min.realizedTokens) / max.realizedTokens;
|
|
524
|
+
const matched = maxRelativeGap <= rule.tolerance;
|
|
525
|
+
return {
|
|
526
|
+
rule,
|
|
527
|
+
arms: [...arms],
|
|
528
|
+
maxRelativeGap,
|
|
529
|
+
widestPair: [min.armId, max.armId],
|
|
530
|
+
matched,
|
|
531
|
+
refusal: matched ? null : {
|
|
532
|
+
onFail: rule.onFail,
|
|
533
|
+
reason: `realized ${rule.measure} diverge ${(maxRelativeGap * 100).toFixed(1)}% between '${min.armId}' (${min.realizedTokens}) and '${max.armId}' (${max.realizedTokens}) — registered tolerance is ${(rule.tolerance * 100).toFixed(1)}%; the contrast is refused`
|
|
534
|
+
}
|
|
535
|
+
};
|
|
536
|
+
}
|
|
537
|
+
/** Throw the refusal for callers that gate the contrast on it. */
|
|
538
|
+
function assertMatchedBudgets(rule, arms) {
|
|
539
|
+
const verdict = verifyMatchedBudgets(rule, arms);
|
|
540
|
+
if (verdict.refusal) throw new MatchedBudgetError(verdict.refusal.reason);
|
|
541
|
+
return verdict;
|
|
542
|
+
}
|
|
543
|
+
//#endregion
|
|
544
|
+
//#region src/experiment/funnel.ts
|
|
545
|
+
/**
|
|
546
|
+
* The denominator chain as a first-class object.
|
|
547
|
+
*
|
|
548
|
+
* A benchmark whose denominator is not auditable is not a benchmark. Every
|
|
549
|
+
* stage names what entered, what it removed, and what survived, so
|
|
550
|
+
* `input = surviving + sum(excluded)` reads off the table instead of being
|
|
551
|
+
* trusted. A stage that gains rows is refused at construction — a funnel is
|
|
552
|
+
* monotone by definition, and a non-monotone one is a broken denominator,
|
|
553
|
+
* not a formatting choice.
|
|
554
|
+
*
|
|
555
|
+
* The object is its own JSON render; `renderFunnelTable` is the text render.
|
|
556
|
+
* `executeAdmissionRule` produces one by running a sealed {@link AdmissionRule}
|
|
557
|
+
* over evidence records.
|
|
558
|
+
*/
|
|
559
|
+
/** A funnel stage gained rows, double-counted them, or failed to reconcile. */
|
|
560
|
+
var FunnelIntegrityError = class extends CaptureIntegrityError {};
|
|
561
|
+
/**
|
|
562
|
+
* Build a funnel from counts and refuse anything non-monotone.
|
|
563
|
+
*
|
|
564
|
+
* Refusals: a negative count, a stage that gains rows (excluded < 0 is the
|
|
565
|
+
* only way to gain — `remaining = entering - excluded` by construction, so a
|
|
566
|
+
* gain cannot be smuggled in through `remaining`), named exclusions that do
|
|
567
|
+
* not sum to the stage total, and a partition drawing from an unknown stage
|
|
568
|
+
* or exceeding what that stage excluded.
|
|
569
|
+
*/
|
|
570
|
+
function buildFunnel(input) {
|
|
571
|
+
if (!Number.isInteger(input.input) || input.input < 0) throw new FunnelIntegrityError(`funnel input must be a non-negative integer, got ${input.input}`);
|
|
572
|
+
let entering = input.input;
|
|
573
|
+
const stages = [];
|
|
574
|
+
for (const stage of input.stages) {
|
|
575
|
+
if (!Number.isInteger(stage.excluded)) throw new FunnelIntegrityError(`funnel stage '${stage.id}' excluded count must be an integer, got ${stage.excluded}`);
|
|
576
|
+
if (stage.excluded < 0) throw new FunnelIntegrityError(`funnel stage '${stage.id}' gains ${-stage.excluded} rows — a funnel stage can only remove rows`);
|
|
577
|
+
if (stage.excluded > entering) throw new FunnelIntegrityError(`funnel stage '${stage.id}' excludes ${stage.excluded} rows but only ${entering} entered`);
|
|
578
|
+
if (stage.exclusions) {
|
|
579
|
+
const sum = Object.values(stage.exclusions).reduce((a, b) => a + b, 0);
|
|
580
|
+
if (sum !== stage.excluded) throw new FunnelIntegrityError(`funnel stage '${stage.id}' names exclusions summing to ${sum} but excludes ${stage.excluded}`);
|
|
581
|
+
}
|
|
582
|
+
const remaining = entering - stage.excluded;
|
|
583
|
+
stages.push({
|
|
584
|
+
id: stage.id,
|
|
585
|
+
entering,
|
|
586
|
+
excluded: stage.excluded,
|
|
587
|
+
remaining,
|
|
588
|
+
...stage.exclusions ? { exclusions: { ...stage.exclusions } } : {},
|
|
589
|
+
...stage.waives ? { waives: [...stage.waives] } : {}
|
|
590
|
+
});
|
|
591
|
+
entering = remaining;
|
|
592
|
+
}
|
|
593
|
+
const byStage = new Map(stages.map((s) => [s.id, s]));
|
|
594
|
+
const partitions = (input.partitions ?? []).map((partition) => {
|
|
595
|
+
const source = byStage.get(partition.from);
|
|
596
|
+
if (!source) throw new FunnelIntegrityError(`funnel partition '${partition.id}' draws from unknown stage '${partition.from}'`);
|
|
597
|
+
if (partition.count < 0 || partition.count > source.excluded) throw new FunnelIntegrityError(`funnel partition '${partition.id}' counts ${partition.count} rows but stage '${partition.from}' excluded ${source.excluded}`);
|
|
598
|
+
return {
|
|
599
|
+
id: partition.id,
|
|
600
|
+
from: partition.from,
|
|
601
|
+
count: partition.count,
|
|
602
|
+
pooling: "never"
|
|
603
|
+
};
|
|
604
|
+
});
|
|
605
|
+
const funnel = {
|
|
606
|
+
population: input.population,
|
|
607
|
+
input: input.input,
|
|
608
|
+
stages,
|
|
609
|
+
surviving: entering,
|
|
610
|
+
partitions
|
|
611
|
+
};
|
|
612
|
+
assertFunnelReconciles(funnel);
|
|
613
|
+
return funnel;
|
|
614
|
+
}
|
|
615
|
+
/** A chain that does not add up is a broken denominator, so this throws. */
|
|
616
|
+
function assertFunnelReconciles(funnel) {
|
|
617
|
+
const excluded = funnel.stages.reduce((sum, stage) => sum + stage.excluded, 0);
|
|
618
|
+
if (funnel.input !== funnel.surviving + excluded) throw new FunnelIntegrityError(`funnel '${funnel.population}' does not reconcile: input ${funnel.input} != surviving ${funnel.surviving} + excluded ${excluded}`);
|
|
619
|
+
let entering = funnel.input;
|
|
620
|
+
for (const stage of funnel.stages) {
|
|
621
|
+
if (stage.entering !== entering || stage.remaining !== stage.entering - stage.excluded) throw new FunnelIntegrityError(`funnel '${funnel.population}' stage '${stage.id}' does not chain: entering ${stage.entering} (expected ${entering}), remaining ${stage.remaining}`);
|
|
622
|
+
if (stage.remaining > stage.entering) throw new FunnelIntegrityError(`funnel '${funnel.population}' stage '${stage.id}' gains rows: ${stage.entering} -> ${stage.remaining}`);
|
|
623
|
+
entering = stage.remaining;
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
/**
|
|
627
|
+
* Run a sealed admission rule over evidence records. Each stage keeps the rows
|
|
628
|
+
* its predicate accepts; partitions draw from the rows their source stage
|
|
629
|
+
* dropped. The result embeds the funnel, so the denominator chain and the
|
|
630
|
+
* surviving rows can never disagree.
|
|
631
|
+
*/
|
|
632
|
+
function executeAdmissionRule(rule, records) {
|
|
633
|
+
let current = [...records];
|
|
634
|
+
const droppedAt = /* @__PURE__ */ new Map();
|
|
635
|
+
const stageInputs = [];
|
|
636
|
+
for (const stage of rule.stages) {
|
|
637
|
+
const kept = [];
|
|
638
|
+
const dropped = [];
|
|
639
|
+
for (const record of current) if (evaluatePredicate(stage.keep, record)) kept.push(record);
|
|
640
|
+
else dropped.push(record);
|
|
641
|
+
droppedAt.set(stage.id, dropped);
|
|
642
|
+
stageInputs.push({
|
|
643
|
+
id: stage.id,
|
|
644
|
+
excluded: dropped.length,
|
|
645
|
+
...stage.waives ? { waives: stage.waives } : {}
|
|
646
|
+
});
|
|
647
|
+
current = kept;
|
|
648
|
+
}
|
|
649
|
+
const partitionRows = {};
|
|
650
|
+
const partitionCounts = [];
|
|
651
|
+
for (const partition of rule.partitions ?? []) {
|
|
652
|
+
const source = droppedAt.get(partition.from);
|
|
653
|
+
if (!source) throw new FunnelIntegrityError(`admission partition '${partition.id}' draws from unknown stage '${partition.from}'`);
|
|
654
|
+
const rows = source.filter((record) => evaluatePredicate(partition.keep, record));
|
|
655
|
+
partitionRows[partition.id] = rows;
|
|
656
|
+
partitionCounts.push({
|
|
657
|
+
id: partition.id,
|
|
658
|
+
from: partition.from,
|
|
659
|
+
count: rows.length
|
|
660
|
+
});
|
|
661
|
+
}
|
|
662
|
+
return {
|
|
663
|
+
funnel: buildFunnel({
|
|
664
|
+
population: rule.population,
|
|
665
|
+
input: records.length,
|
|
666
|
+
stages: stageInputs,
|
|
667
|
+
partitions: partitionCounts
|
|
668
|
+
}),
|
|
669
|
+
survivors: current,
|
|
670
|
+
partitionRows
|
|
671
|
+
};
|
|
672
|
+
}
|
|
673
|
+
/**
|
|
674
|
+
* Chain two funnels whose boundary agrees: the second funnel's input must be
|
|
675
|
+
* exactly the first funnel's survivors. Anything else is a gap or an
|
|
676
|
+
* injection, and both are refused.
|
|
677
|
+
*/
|
|
678
|
+
function composeFunnels(first, second) {
|
|
679
|
+
if (second.input !== first.surviving) throw new FunnelIntegrityError(`cannot compose funnels: '${first.population}' survives ${first.surviving} rows but '${second.population}' starts from ${second.input}`);
|
|
680
|
+
return buildFunnel({
|
|
681
|
+
population: first.population,
|
|
682
|
+
input: first.input,
|
|
683
|
+
stages: [...first.stages, ...second.stages].map((stage) => ({
|
|
684
|
+
id: stage.id,
|
|
685
|
+
excluded: stage.excluded,
|
|
686
|
+
...stage.exclusions ? { exclusions: stage.exclusions } : {},
|
|
687
|
+
...stage.waives ? { waives: stage.waives } : {}
|
|
688
|
+
})),
|
|
689
|
+
partitions: [...first.partitions, ...second.partitions].map((partition) => ({
|
|
690
|
+
id: partition.id,
|
|
691
|
+
from: partition.from,
|
|
692
|
+
count: partition.count
|
|
693
|
+
}))
|
|
694
|
+
});
|
|
695
|
+
}
|
|
696
|
+
/**
|
|
697
|
+
* Text render of the chain. One row per stage; the reconciliation line at the
|
|
698
|
+
* bottom restates `input = surviving + excluded` so a reader can check the
|
|
699
|
+
* arithmetic without a tool.
|
|
700
|
+
*/
|
|
701
|
+
function renderFunnelTable(funnel) {
|
|
702
|
+
const rows = funnel.stages.map((stage) => [
|
|
703
|
+
stage.id + (stage.waives?.length ? ` (waives: ${stage.waives.join(", ")})` : ""),
|
|
704
|
+
String(stage.entering),
|
|
705
|
+
String(stage.excluded),
|
|
706
|
+
String(stage.remaining)
|
|
707
|
+
]);
|
|
708
|
+
const header = [
|
|
709
|
+
"stage",
|
|
710
|
+
"entering",
|
|
711
|
+
"excluded",
|
|
712
|
+
"remaining"
|
|
713
|
+
];
|
|
714
|
+
const widths = header.map((h, col) => Math.max(h.length, ...rows.map((r) => r[col].length)));
|
|
715
|
+
const line = (cells) => cells.map((cell, col) => cell.padEnd(widths[col])).join(" ");
|
|
716
|
+
const out = [
|
|
717
|
+
`population: ${funnel.population}`,
|
|
718
|
+
`input: ${funnel.input}`,
|
|
719
|
+
line(header),
|
|
720
|
+
line(widths.map((w) => "-".repeat(w))),
|
|
721
|
+
...rows.map((r) => line(r))
|
|
722
|
+
];
|
|
723
|
+
const excluded = funnel.stages.reduce((sum, stage) => sum + stage.excluded, 0);
|
|
724
|
+
out.push(`surviving: ${funnel.surviving} (input ${funnel.input} = surviving ${funnel.surviving} + excluded ${excluded})`);
|
|
725
|
+
for (const partition of funnel.partitions) out.push(`partition ${partition.id}: ${partition.count} rows from '${partition.from}' — reported separately, never pooled`);
|
|
726
|
+
return out.join("\n");
|
|
727
|
+
}
|
|
728
|
+
//#endregion
|
|
729
|
+
//#region src/experiment/define.ts
|
|
730
|
+
/**
|
|
731
|
+
* Define, seal, and execute experiments whose every rule is registered data.
|
|
732
|
+
*
|
|
733
|
+
* `defineExperiment` validates the cross-references inside a spec and freezes
|
|
734
|
+
* it. `sealExperiment` canonicalizes and hashes the whole tree — arms, row
|
|
735
|
+
* admission, selection, estimands, intervals, decision rule, gates, halt,
|
|
736
|
+
* budget — into one digest that lands on every downstream artifact. Changing
|
|
737
|
+
* what is decided requires a new digest: an amendment is a re-seal with a
|
|
738
|
+
* reason and blindness attestations, and the digest history is the audit
|
|
739
|
+
* trail.
|
|
740
|
+
*
|
|
741
|
+
* `openSealedExperiment` is the only execution surface. It verifies the seal
|
|
742
|
+
* and returns executors bound to the sealed spec; none of them takes a
|
|
743
|
+
* parameter for alpha, threshold, metric, or stopping rule, so
|
|
744
|
+
* registered-vs-ran drift is unrepresentable rather than checked.
|
|
745
|
+
*/
|
|
746
|
+
/** A sealed experiment whose digest no longer matches its spec. */
|
|
747
|
+
var SealIntegrityError = class extends ValidationError {};
|
|
748
|
+
function conditionRefs(condition) {
|
|
749
|
+
switch (condition.kind) {
|
|
750
|
+
case "interval-excludes-zero":
|
|
751
|
+
case "interval-includes-zero": return {
|
|
752
|
+
intervals: [condition.interval],
|
|
753
|
+
quantities: [],
|
|
754
|
+
obligations: []
|
|
755
|
+
};
|
|
756
|
+
case "quantity-threshold": return {
|
|
757
|
+
intervals: [],
|
|
758
|
+
quantities: [condition.quantity],
|
|
759
|
+
obligations: []
|
|
760
|
+
};
|
|
761
|
+
case "obligation-met": return {
|
|
762
|
+
intervals: [],
|
|
763
|
+
quantities: [],
|
|
764
|
+
obligations: [condition.obligation]
|
|
765
|
+
};
|
|
766
|
+
case "all":
|
|
767
|
+
case "any": {
|
|
768
|
+
const nested = condition.of.map(conditionRefs);
|
|
769
|
+
return {
|
|
770
|
+
intervals: nested.flatMap((r) => r.intervals),
|
|
771
|
+
quantities: nested.flatMap((r) => r.quantities),
|
|
772
|
+
obligations: nested.flatMap((r) => r.obligations)
|
|
773
|
+
};
|
|
774
|
+
}
|
|
775
|
+
case "not": return conditionRefs(condition.of);
|
|
776
|
+
}
|
|
777
|
+
}
|
|
778
|
+
/**
|
|
779
|
+
* Validate every cross-reference inside a spec and freeze it.
|
|
780
|
+
*
|
|
781
|
+
* A decision condition may only read a registered interval, a registered
|
|
782
|
+
* estimand, or a registered obligation; a halt rule may only reference
|
|
783
|
+
* registered gates; a filter-of base must resolve to a registered selection
|
|
784
|
+
* or sealed subset. Anything else is refused here, before sealing.
|
|
785
|
+
*/
|
|
786
|
+
function defineExperiment(spec) {
|
|
787
|
+
const problems = [];
|
|
788
|
+
if (!spec.id || spec.id.trim().length === 0) problems.push("id is empty");
|
|
789
|
+
if (spec.arms.length === 0) problems.push("at least one arm is required");
|
|
790
|
+
const armIds = /* @__PURE__ */ new Set();
|
|
791
|
+
for (const arm of spec.arms) {
|
|
792
|
+
if (armIds.has(arm.id)) problems.push(`duplicate arm id '${arm.id}'`);
|
|
793
|
+
armIds.add(arm.id);
|
|
794
|
+
}
|
|
795
|
+
if (!spec.arms.some((arm) => arm.role === "treatment")) problems.push("at least one arm must have role treatment");
|
|
796
|
+
const intervalNames = new Set(Object.keys(spec.intervals ?? {}));
|
|
797
|
+
const estimandNames = new Set(Object.keys(spec.estimands ?? {}));
|
|
798
|
+
const obligationIds = new Set((spec.obligations ?? []).map((o) => o.id));
|
|
799
|
+
const gateNames = new Set(Object.keys(spec.gates ?? {}));
|
|
800
|
+
const selectionNames = new Set(Object.keys(spec.selections ?? {}));
|
|
801
|
+
const sealedSubsetNames = new Set(Object.keys(spec.sealedSubsets ?? {}));
|
|
802
|
+
const checkCondition = (condition, where) => {
|
|
803
|
+
const refs = conditionRefs(condition);
|
|
804
|
+
for (const name of refs.intervals) if (!intervalNames.has(name)) problems.push(`${where} reads unregistered interval '${name}'`);
|
|
805
|
+
for (const name of refs.quantities) if (!estimandNames.has(name)) problems.push(`${where} reads unregistered quantity '${name}'`);
|
|
806
|
+
for (const name of refs.obligations) if (!obligationIds.has(name)) problems.push(`${where} reads unregistered obligation '${name}'`);
|
|
807
|
+
};
|
|
808
|
+
if (spec.decision.kind === "table") {
|
|
809
|
+
if (spec.decision.branches.length === 0) problems.push("decision table has no branches");
|
|
810
|
+
spec.decision.branches.forEach((branch, index) => {
|
|
811
|
+
checkCondition(branch.when, `decision branch ${index} ('${branch.verdict}')`);
|
|
812
|
+
});
|
|
813
|
+
} else {
|
|
814
|
+
for (const name of spec.decision.estimands) if (!estimandNames.has(name)) problems.push(`report-only decision names unregistered estimand '${name}'`);
|
|
815
|
+
for (const name of spec.decision.intervals) if (!intervalNames.has(name)) problems.push(`report-only decision names unregistered interval '${name}'`);
|
|
816
|
+
}
|
|
817
|
+
if (spec.decision.kind === "table" && spec.obligations) {
|
|
818
|
+
const verdicts = new Set(spec.decision.branches.map((b) => b.verdict));
|
|
819
|
+
for (const obligation of spec.obligations) for (const verdict of obligation.appliesToVerdicts) if (!verdicts.has(verdict)) problems.push(`obligation '${obligation.id}' applies to verdict '${verdict}' which no branch produces`);
|
|
820
|
+
}
|
|
821
|
+
if (spec.halt) {
|
|
822
|
+
for (const gate of spec.halt.when.gates) if (!gateNames.has(gate)) problems.push(`halt rule references unregistered gate '${gate}'`);
|
|
823
|
+
}
|
|
824
|
+
for (const [name, selection] of Object.entries(spec.selections ?? {})) if (selection.kind === "filter-of") {
|
|
825
|
+
if (!selectionNames.has(selection.base) && !sealedSubsetNames.has(selection.base)) problems.push(`selection '${name}' filters unregistered base '${selection.base}' (not a selection or sealed subset)`);
|
|
826
|
+
}
|
|
827
|
+
if (spec.admission) {
|
|
828
|
+
const stageIds = /* @__PURE__ */ new Set();
|
|
829
|
+
for (const stage of spec.admission.stages) {
|
|
830
|
+
if (stageIds.has(stage.id)) problems.push(`duplicate admission stage id '${stage.id}'`);
|
|
831
|
+
stageIds.add(stage.id);
|
|
832
|
+
}
|
|
833
|
+
for (const partition of spec.admission.partitions ?? []) if (!stageIds.has(partition.from)) problems.push(`admission partition '${partition.id}' draws from unknown stage '${partition.from}'`);
|
|
834
|
+
}
|
|
835
|
+
if (problems.length > 0) throw new ValidationError(`defineExperiment('${spec.id}'): ${problems.join("; ")}`);
|
|
836
|
+
return deepFreeze(structuredClone(spec));
|
|
837
|
+
}
|
|
838
|
+
function deepFreeze(value) {
|
|
839
|
+
if (value !== null && typeof value === "object") {
|
|
840
|
+
for (const key of Object.keys(value)) deepFreeze(value[key]);
|
|
841
|
+
Object.freeze(value);
|
|
842
|
+
}
|
|
843
|
+
return value;
|
|
844
|
+
}
|
|
845
|
+
/** Validate, canonicalize, and hash a spec into its registration. */
|
|
846
|
+
async function sealExperiment(spec, options = {}) {
|
|
847
|
+
const validated = defineExperiment(spec);
|
|
848
|
+
const digest = await hashJson(validated);
|
|
849
|
+
return {
|
|
850
|
+
spec: validated,
|
|
851
|
+
digest,
|
|
852
|
+
algo: "sha256-content",
|
|
853
|
+
sealedAt: options.sealedAt ?? (/* @__PURE__ */ new Date()).toISOString(),
|
|
854
|
+
initialDigest: digest,
|
|
855
|
+
amendments: []
|
|
856
|
+
};
|
|
857
|
+
}
|
|
858
|
+
/**
|
|
859
|
+
* Amend a sealed experiment. The current seal is verified first, the new spec
|
|
860
|
+
* is validated and re-hashed, and the amendment appends to the digest chain.
|
|
861
|
+
* There is no way to change what is decided without producing a new digest.
|
|
862
|
+
*/
|
|
863
|
+
async function amendExperiment(sealed, amendment) {
|
|
864
|
+
await assertSealIntact(sealed);
|
|
865
|
+
const validated = defineExperiment(amendment.spec);
|
|
866
|
+
const digest = await hashJson(validated);
|
|
867
|
+
return {
|
|
868
|
+
spec: validated,
|
|
869
|
+
digest,
|
|
870
|
+
algo: "sha256-content",
|
|
871
|
+
sealedAt: sealed.sealedAt,
|
|
872
|
+
initialDigest: sealed.initialDigest,
|
|
873
|
+
amendments: [...sealed.amendments, {
|
|
874
|
+
at: amendment.at ?? (/* @__PURE__ */ new Date()).toISOString(),
|
|
875
|
+
reason: amendment.reason,
|
|
876
|
+
blind: [...amendment.blind],
|
|
877
|
+
digest
|
|
878
|
+
}]
|
|
879
|
+
};
|
|
880
|
+
}
|
|
881
|
+
/** True when the sealed digest still matches the spec it carries. */
|
|
882
|
+
async function verifySealedExperiment(sealed) {
|
|
883
|
+
return await hashJson(sealed.spec) === sealed.digest;
|
|
884
|
+
}
|
|
885
|
+
async function assertSealIntact(sealed) {
|
|
886
|
+
if (!await verifySealedExperiment(sealed)) throw new SealIntegrityError(`sealed experiment '${sealed.spec.id}' digest ${sealed.digest} does not match its spec — the registration was tampered with`);
|
|
887
|
+
}
|
|
888
|
+
/**
|
|
889
|
+
* Verify the seal and return executors bound to it. This is the module's only
|
|
890
|
+
* execution surface: a rule that is not in the sealed spec cannot run, and a
|
|
891
|
+
* rule that is cannot run differently.
|
|
892
|
+
*/
|
|
893
|
+
async function openSealedExperiment(sealed) {
|
|
894
|
+
await assertSealIntact(sealed);
|
|
895
|
+
const spec = sealed.spec;
|
|
896
|
+
const need = (value, what) => {
|
|
897
|
+
if (value === void 0) throw new ValidationError(`experiment '${spec.id}' registered no ${what}`);
|
|
898
|
+
return value;
|
|
899
|
+
};
|
|
900
|
+
return {
|
|
901
|
+
sealed,
|
|
902
|
+
decide: (evidence) => executeDecisionRule(spec.decision, evidence),
|
|
903
|
+
admit: (records) => executeAdmissionRule(need(spec.admission, "admission rule"), records),
|
|
904
|
+
select: (name, records, options) => {
|
|
905
|
+
return runSelectionRule(need(spec.selections?.[name], `selection '${name}'`), records, {
|
|
906
|
+
idField: options.idField,
|
|
907
|
+
bases: spec.sealedSubsets
|
|
908
|
+
});
|
|
909
|
+
},
|
|
910
|
+
gate: (name, evidence) => {
|
|
911
|
+
const gate = need(spec.gates?.[name], `gate '${name}'`);
|
|
912
|
+
if (gate.kind !== evidence.kind) throw new ValidationError(`gate '${name}' is registered as ${gate.kind} but received ${evidence.kind} evidence`);
|
|
913
|
+
switch (gate.kind) {
|
|
914
|
+
case "oracle-determinism": return evaluateOracleDeterminismGate(name, gate, evidence.repsByState);
|
|
915
|
+
case "population-reproducibility": return evaluatePopulationReproducibilityGate(name, gate, evidence);
|
|
916
|
+
case "provenance-assertion": return evaluateProvenanceGate(name, gate, evidence.provenance);
|
|
917
|
+
case "identity": return evaluateIdentityGate(name, gate, evidence);
|
|
918
|
+
case "power-floor": return evaluatePowerFloorGate(name, gate, evidence.curve);
|
|
919
|
+
}
|
|
920
|
+
},
|
|
921
|
+
halt: (gates) => evaluateHaltRule(need(spec.halt, "halt rule"), gates),
|
|
922
|
+
runUniformPassBudget: (measuredPassCosts) => {
|
|
923
|
+
const budget = need(spec.budget, "budget rule");
|
|
924
|
+
if (budget.kind !== "uniform-pass") throw new ValidationError(`experiment '${spec.id}' registered a ${budget.kind} budget, not uniform-pass`);
|
|
925
|
+
return runUniformPassBudget(budget, measuredPassCosts);
|
|
926
|
+
},
|
|
927
|
+
projectNLadderBudget: (measured) => {
|
|
928
|
+
const budget = need(spec.budget, "budget rule");
|
|
929
|
+
if (budget.kind !== "n-ladder") throw new ValidationError(`experiment '${spec.id}' registered a ${budget.kind} budget, not n-ladder`);
|
|
930
|
+
return projectNLadderBudget(budget, measured);
|
|
931
|
+
},
|
|
932
|
+
matchedBudgets: (arms) => verifyMatchedBudgets(need(spec.matchedBudget, "matched-budget rule"), arms),
|
|
933
|
+
estimate: (name, rows) => computeEstimand(need(spec.estimands?.[name], `estimand '${name}'`), rows),
|
|
934
|
+
interval: (name, evidence) => computeInterval(need(spec.intervals?.[name], `interval '${name}'`), evidence)
|
|
935
|
+
};
|
|
936
|
+
}
|
|
937
|
+
//#endregion
|
|
938
|
+
//#region src/experiment/power.ts
|
|
939
|
+
/**
|
|
940
|
+
* Design-time power for task-clustered paired designs, with a refusal verdict.
|
|
941
|
+
*
|
|
942
|
+
* The failure this prevents (measured): a pre-registered kill test fixed a
|
|
943
|
+
* task-clustered bootstrap over 14 rows in 4 task clusters. Simulated at the
|
|
944
|
+
* registered seed, the design's power topped out at 0.69 — at a per-row
|
|
945
|
+
* effect of 1.0. "Four clusters cannot certify any effect size, including
|
|
946
|
+
* 1.0" was learned by running the experiment; this module computes it before
|
|
947
|
+
* a dollar is spent.
|
|
948
|
+
*
|
|
949
|
+
* Two floors, one simulation:
|
|
950
|
+
* - Closed form, zero spend: with C independent clusters the exact
|
|
951
|
+
* whole-cluster sign-flip test can never produce a two-sided p below
|
|
952
|
+
* 2^(1-C). C=4 gives 0.125; C=3 gives 0.25 — both above a 0.05 alpha, so
|
|
953
|
+
* those designs are refused at ANY effect size, before simulation.
|
|
954
|
+
* - Seeded simulation: per-row paired contrasts drawn under a registered
|
|
955
|
+
* effect model, a whole-cluster percentile bootstrap on each trial, power =
|
|
956
|
+
* the fraction of trials whose interval excludes zero.
|
|
957
|
+
*
|
|
958
|
+
* The refusal is a verdict INSIDE the returned artifact (the powerPreflight
|
|
959
|
+
* shape, made cluster-aware); `assertDesignAdequate` turns it into a throw for
|
|
960
|
+
* callers that want configuration-time failure.
|
|
961
|
+
*/
|
|
962
|
+
/** A design refused at configuration time, before any spend. */
|
|
963
|
+
var DesignRefusalError = class extends ValidationError {};
|
|
964
|
+
/**
|
|
965
|
+
* Simulate the power of a whole-cluster percentile-bootstrap design and refuse
|
|
966
|
+
* a structure that cannot reach the target at any registered effect.
|
|
967
|
+
*/
|
|
968
|
+
function clusteredPower(options) {
|
|
969
|
+
const clusterSizes = options.clusterSizes;
|
|
970
|
+
if (clusterSizes.length === 0 || clusterSizes.some((n) => !Number.isInteger(n) || n <= 0)) throw new ValidationError(`clusteredPower: clusterSizes must be positive integers, got [${clusterSizes.join(", ")}]`);
|
|
971
|
+
if (options.effects.length === 0) throw new ValidationError("clusteredPower: effects grid is empty");
|
|
972
|
+
if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPower: seed must be an integer, got ${options.seed}`);
|
|
973
|
+
const trials = options.trials ?? 2e3;
|
|
974
|
+
const resamples = options.resamples ?? 4e3;
|
|
975
|
+
if (!Number.isInteger(trials) || trials <= 0 || !Number.isInteger(resamples) || resamples <= 0) throw new ValidationError(`clusteredPower: trials and resamples must be positive integers, got ${trials}/${resamples}`);
|
|
976
|
+
const confidence = options.confidence ?? .95;
|
|
977
|
+
if (confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPower: confidence must be in (0,1), got ${confidence}`);
|
|
978
|
+
const alpha = options.alpha ?? .05;
|
|
979
|
+
const targetPower = options.targetPower ?? .8;
|
|
980
|
+
const baseWinRate = options.baseWinRate ?? .1;
|
|
981
|
+
const baseLossRate = options.baseLossRate ?? .1;
|
|
982
|
+
const noisy = /* @__PURE__ */ new Map();
|
|
983
|
+
for (const cluster of options.noisyClusters ?? []) {
|
|
984
|
+
if (cluster.index < 0 || cluster.index >= clusterSizes.length) throw new ValidationError(`clusteredPower: noisy cluster index ${cluster.index} outside [0,${clusterSizes.length - 1}]`);
|
|
985
|
+
noisy.set(cluster.index, cluster.flipRate);
|
|
986
|
+
}
|
|
987
|
+
const clusterCount = clusterSizes.length;
|
|
988
|
+
const totalRows = clusterSizes.reduce((a, b) => a + b, 0);
|
|
989
|
+
const signFlipFloor = computeSignFlipFloor(clusterCount, alpha);
|
|
990
|
+
const curve = [];
|
|
991
|
+
for (const effect of options.effects) curve.push(simulateEffect(effect, {
|
|
992
|
+
clusterSizes,
|
|
993
|
+
seed: options.seed,
|
|
994
|
+
trials,
|
|
995
|
+
resamples,
|
|
996
|
+
confidence,
|
|
997
|
+
baseWinRate,
|
|
998
|
+
baseLossRate,
|
|
999
|
+
noisy
|
|
1000
|
+
}));
|
|
1001
|
+
const maxPower = Math.max(...curve.map((point) => point.power));
|
|
1002
|
+
const reasons = [];
|
|
1003
|
+
if (!signFlipFloor.certifiableAtAlpha) reasons.push(`${clusterCount} clusters cannot certify any effect size, including 1.0: the exact whole-cluster sign-flip test's smallest two-sided p is 2^(1-${clusterCount}) = ${signFlipFloor.twoSidedP} > alpha ${alpha}; at least ${signFlipFloor.minClustersForAlpha} clusters are needed`);
|
|
1004
|
+
if (maxPower < targetPower) {
|
|
1005
|
+
const best = curve.reduce((a, b) => b.power > a.power ? b : a);
|
|
1006
|
+
reasons.push(`simulated power tops out at ${maxPower.toFixed(3)} (effect ${best.effect}) across the registered grid — below the ${targetPower} target at every effect`);
|
|
1007
|
+
}
|
|
1008
|
+
const adequate = reasons.length === 0;
|
|
1009
|
+
return {
|
|
1010
|
+
clusterCount,
|
|
1011
|
+
totalRows,
|
|
1012
|
+
trials,
|
|
1013
|
+
resamples,
|
|
1014
|
+
seed: options.seed,
|
|
1015
|
+
confidence,
|
|
1016
|
+
targetPower,
|
|
1017
|
+
curve,
|
|
1018
|
+
maxPower,
|
|
1019
|
+
signFlipFloor,
|
|
1020
|
+
adequate,
|
|
1021
|
+
refusal: adequate ? null : {
|
|
1022
|
+
verdict: "underpowered",
|
|
1023
|
+
reasons,
|
|
1024
|
+
recommendation: `Do not spend on this structure. Add independent clusters (>= ${Math.max(signFlipFloor.minClustersForAlpha, clusterCount)}) or register a design whose simulated power reaches ${targetPower}, then re-run clusteredPower.`
|
|
1025
|
+
}
|
|
1026
|
+
};
|
|
1027
|
+
}
|
|
1028
|
+
/** Throw the refusal for callers that want configuration-time failure. */
|
|
1029
|
+
function assertDesignAdequate(result) {
|
|
1030
|
+
if (result.refusal) throw new DesignRefusalError(`design refused (underpowered): ${result.refusal.reasons.join("; ")}`);
|
|
1031
|
+
}
|
|
1032
|
+
function computeSignFlipFloor(clusterCount, alpha) {
|
|
1033
|
+
const twoSidedP = 2 ** (1 - clusterCount);
|
|
1034
|
+
const oneSidedP = 2 ** -clusterCount;
|
|
1035
|
+
let minClusters = 1;
|
|
1036
|
+
while (2 ** (1 - minClusters) > alpha) minClusters += 1;
|
|
1037
|
+
return {
|
|
1038
|
+
twoSidedP,
|
|
1039
|
+
oneSidedP,
|
|
1040
|
+
alpha,
|
|
1041
|
+
certifiableAtAlpha: twoSidedP <= alpha,
|
|
1042
|
+
minClustersForAlpha: minClusters
|
|
1043
|
+
};
|
|
1044
|
+
}
|
|
1045
|
+
/**
|
|
1046
|
+
* One effect point. Per row the paired contrast is +1 with probability
|
|
1047
|
+
* min(baseWin + effect, 1), -1 with the base loss rate (capped by what
|
|
1048
|
+
* remains), else 0. Noisy clusters draw win and loss independently at their
|
|
1049
|
+
* flip rate. Each trial computes a whole-cluster percentile bootstrap of the
|
|
1050
|
+
* pooled row mean; the trial counts toward power when the interval excludes
|
|
1051
|
+
* zero.
|
|
1052
|
+
*/
|
|
1053
|
+
function simulateEffect(effect, config) {
|
|
1054
|
+
const rng = mulberry32(mixSeed(config.seed, effect));
|
|
1055
|
+
const clusterCount = config.clusterSizes.length;
|
|
1056
|
+
let excludes = 0;
|
|
1057
|
+
const widths = new Array(config.trials);
|
|
1058
|
+
const sums = new Array(clusterCount);
|
|
1059
|
+
const means = new Array(config.resamples);
|
|
1060
|
+
for (let trial = 0; trial < config.trials; trial++) {
|
|
1061
|
+
for (let cluster = 0; cluster < clusterCount; cluster++) {
|
|
1062
|
+
const size = config.clusterSizes[cluster];
|
|
1063
|
+
const flipRate = config.noisy.get(cluster);
|
|
1064
|
+
let sum = 0;
|
|
1065
|
+
for (let row = 0; row < size; row++) if (flipRate !== void 0) {
|
|
1066
|
+
const win = rng() < flipRate ? 1 : 0;
|
|
1067
|
+
const loss = rng() < flipRate ? 1 : 0;
|
|
1068
|
+
sum += win - loss;
|
|
1069
|
+
} else {
|
|
1070
|
+
const winRate = Math.min(1, config.baseWinRate + effect);
|
|
1071
|
+
const lossRate = Math.min(1 - winRate, config.baseLossRate);
|
|
1072
|
+
const u = rng();
|
|
1073
|
+
sum += u < winRate ? 1 : u < winRate + lossRate ? -1 : 0;
|
|
1074
|
+
}
|
|
1075
|
+
sums[cluster] = sum;
|
|
1076
|
+
}
|
|
1077
|
+
for (let draw = 0; draw < config.resamples; draw++) {
|
|
1078
|
+
let pooledSum = 0;
|
|
1079
|
+
let pooledRows = 0;
|
|
1080
|
+
for (let pick = 0; pick < clusterCount; pick++) {
|
|
1081
|
+
const index = Math.floor(rng() * clusterCount);
|
|
1082
|
+
pooledSum += sums[index];
|
|
1083
|
+
pooledRows += config.clusterSizes[index];
|
|
1084
|
+
}
|
|
1085
|
+
means[draw] = pooledSum / pooledRows;
|
|
1086
|
+
}
|
|
1087
|
+
means.sort((a, b) => a - b);
|
|
1088
|
+
const tail = 1 - config.confidence;
|
|
1089
|
+
const lower = means[Math.floor(tail / 2 * config.resamples)];
|
|
1090
|
+
const upper = means[Math.max(Math.floor(tail / 2 * config.resamples), Math.min(config.resamples - 1, Math.ceil((1 - tail / 2) * config.resamples) - 1))];
|
|
1091
|
+
widths[trial] = upper - lower;
|
|
1092
|
+
if (lower > 0 || upper < 0) excludes += 1;
|
|
1093
|
+
}
|
|
1094
|
+
widths.sort((a, b) => a - b);
|
|
1095
|
+
return {
|
|
1096
|
+
effect,
|
|
1097
|
+
power: excludes / config.trials,
|
|
1098
|
+
medianCiWidth: widths[Math.floor(config.trials / 2)]
|
|
1099
|
+
};
|
|
1100
|
+
}
|
|
1101
|
+
/** Fold the effect into the seed so every grid point draws an independent stream. */
|
|
1102
|
+
function mixSeed(seed, effect) {
|
|
1103
|
+
return (seed ^ Math.round(effect * 1000003) * 2654435769) >>> 0 | 0;
|
|
1104
|
+
}
|
|
1105
|
+
//#endregion
|
|
1106
|
+
export { BOOTSTRAP_GATE_MIN_N, DecisionTableNotTotalError, DesignRefusalError, ExperimentTracker, FunnelIntegrityError, MatchedBudgetError, SealIntegrityError, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, canonicalize, classifyReissue, clusteredPairedBinary, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, powerPreflight, projectNLadderBudget, readField, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialDecide, sequentialPairedGate, signManifest, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
1107
|
+
|
|
1108
|
+
//# sourceMappingURL=index.js.map
|