@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1,114 +0,0 @@
|
|
|
1
|
-
import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
2
|
-
//#region src/paired-promotion-decision.d.ts
|
|
3
|
-
/** Which paired estimator produced the deciding interval. */
|
|
4
|
-
type PairedDecisionStatistic = 'paired_risk_difference' | 'mean_bootstrap' | 'median_bootstrap';
|
|
5
|
-
/** Which test carried the decision, given the estimator and the sample size. */
|
|
6
|
-
type PairedDecisionMethod = 'score-interval' | 'bootstrap-ci' | 'exact-sign';
|
|
7
|
-
/** McNemar's exact paired-binary evidence, on the two-point path only. */
|
|
8
|
-
interface PairedMcNemarEvidence {
|
|
9
|
-
/** Discordant pairs the treatment won. */
|
|
10
|
-
b: number;
|
|
11
|
-
/** Discordant pairs the control won. */
|
|
12
|
-
c: number;
|
|
13
|
-
/** b + c — the only pairs carrying information. */
|
|
14
|
-
nDiscordant: number;
|
|
15
|
-
/** Two-sided exact p-value. */
|
|
16
|
-
pValue: number;
|
|
17
|
-
}
|
|
18
|
-
interface PairedPromotionDecisionOptions {
|
|
19
|
-
/** Smallest candidate-minus-baseline delta that counts as improvement, in the
|
|
20
|
-
* caller's native units. May be negative (a noninferiority margin). Default 0. */
|
|
21
|
-
threshold?: number;
|
|
22
|
-
/** Confidence level. Default 0.95. */
|
|
23
|
-
confidence?: number;
|
|
24
|
-
/** Bootstrap resamples, on the paths where a bootstrap decides. Default 2000. */
|
|
25
|
-
resamples?: number;
|
|
26
|
-
/** Deterministic bootstrap seed. Omitted ⇒ derived from the deltas. */
|
|
27
|
-
seed?: number;
|
|
28
|
-
/** Caller-required paired observations. The exact test may impose a higher
|
|
29
|
-
* minimum; the effective one is reported as `minimumPairs`. */
|
|
30
|
-
minPairs?: number;
|
|
31
|
-
/**
|
|
32
|
-
* `'mean'` (default) routes by SHAPE: a two-point (pass/fail) outcome on any
|
|
33
|
-
* encoding decides on the score interval, everything else on the mean
|
|
34
|
-
* bootstrap. `'median'` forces the median bootstrap on every input, including
|
|
35
|
-
* shapes where it is structurally blind — kept for callers who want outlier
|
|
36
|
-
* robustness on genuinely continuous outcomes and accept that cost.
|
|
37
|
-
*/
|
|
38
|
-
statistic?: 'mean' | 'median';
|
|
39
|
-
}
|
|
40
|
-
interface PairedPromotionDecision {
|
|
41
|
-
/** Paired observations supplied. */
|
|
42
|
-
n: number;
|
|
43
|
-
/** Threshold the interval was judged against, native units. */
|
|
44
|
-
threshold: number;
|
|
45
|
-
confidence: number;
|
|
46
|
-
statistic: PairedDecisionStatistic;
|
|
47
|
-
method: PairedDecisionMethod;
|
|
48
|
-
/** Common positive level of a two-point outcome ({0,1} ⇒ 1, {0,100} ⇒ 100),
|
|
49
|
-
* or null when the outcome is not two-point. Non-null is exactly the
|
|
50
|
-
* condition for the `paired_risk_difference` path, and it is the factor
|
|
51
|
-
* `delta` / `low` / `high` were rescaled by. */
|
|
52
|
-
binaryScale: number | null;
|
|
53
|
-
/** Exact-tie fraction over the paired deltas; null when there are no pairs. */
|
|
54
|
-
tieFraction: number | null;
|
|
55
|
-
/** Point estimate of the DECIDING statistic, in the caller's native units. */
|
|
56
|
-
delta: number;
|
|
57
|
-
/** Lower bound of the DECIDING interval, native units. */
|
|
58
|
-
low: number;
|
|
59
|
-
/** Upper bound of the DECIDING interval, native units. */
|
|
60
|
-
high: number;
|
|
61
|
-
/** The bootstrap that decided, or null when the score interval did. Callers
|
|
62
|
-
* that need a bootstrap as a diagnostic on the two-point path compute their
|
|
63
|
-
* own — it is not computed here, so the binary path costs no resamples. */
|
|
64
|
-
bootstrap: PairedBootstrapResult | null;
|
|
65
|
-
/** McNemar's exact evidence, or null off the two-point path. */
|
|
66
|
-
mcnemar: PairedMcNemarEvidence | null;
|
|
67
|
-
/** Exact one-sided sign-test p-value on the small-sample bootstrap path;
|
|
68
|
-
* null otherwise. */
|
|
69
|
-
pValue: number | null;
|
|
70
|
-
/** Effective observation minimum after accounting for confidence. */
|
|
71
|
-
minimumPairs: number;
|
|
72
|
-
/** n >= minimumPairs. */
|
|
73
|
-
sufficient: boolean;
|
|
74
|
-
/** The deciding interval is zero-width or non-finite — no evidence in either
|
|
75
|
-
* direction, so it cannot clear any threshold on evidence. */
|
|
76
|
-
indeterminate: boolean;
|
|
77
|
-
/** McNemar's exact test refuses at a non-negative threshold. */
|
|
78
|
-
exactTestVetoes: boolean;
|
|
79
|
-
/** The deciding interval clears the threshold, ignoring the other two guards. */
|
|
80
|
-
clearsThreshold: boolean;
|
|
81
|
-
/** `sufficient && !indeterminate && clearsThreshold && !exactTestVetoes` —
|
|
82
|
-
* the whole rule. */
|
|
83
|
-
promote: boolean;
|
|
84
|
-
/** What `delta` measures, for a reason string. */
|
|
85
|
-
label: 'success-rate' | 'mean' | 'median';
|
|
86
|
-
/** Why a zero-width interval is zero-width; empty when it is not. */
|
|
87
|
-
indeterminateCause: string;
|
|
88
|
-
/** Sentence naming the test when the exact sign test decided; else empty. */
|
|
89
|
-
methodDetail: string;
|
|
90
|
-
}
|
|
91
|
-
/** The shape facts that pick the estimator, without computing an interval. */
|
|
92
|
-
interface PairedDecisionShape {
|
|
93
|
-
statistic: PairedDecisionStatistic;
|
|
94
|
-
/** Common positive level of a two-point outcome; null when not two-point. */
|
|
95
|
-
binaryScale: number | null;
|
|
96
|
-
/** Exact-tie fraction over the paired deltas; null when there are no pairs. */
|
|
97
|
-
tieFraction: number | null;
|
|
98
|
-
}
|
|
99
|
-
/**
|
|
100
|
-
* Which estimator {@link decidePairedPromotion} would use on this data, and the
|
|
101
|
-
* shape facts behind it — for callers that must report the shape on a path
|
|
102
|
-
* where no interval is computed at all (an early rejection, or zero pairs).
|
|
103
|
-
* Cheap: no bootstrap, no interval.
|
|
104
|
-
*/
|
|
105
|
-
declare function pairedDecisionShape(before: number[], after: number[], statistic?: 'mean' | 'median'): PairedDecisionShape;
|
|
106
|
-
/**
|
|
107
|
-
* Decide whether a paired candidate-minus-baseline delta clears a promotion
|
|
108
|
-
* threshold. `before` is the baseline arm, `after` the candidate arm, paired by
|
|
109
|
-
* position. Throws on unequal lengths.
|
|
110
|
-
*/
|
|
111
|
-
declare function decidePairedPromotion(before: number[], after: number[], options?: PairedPromotionDecisionOptions): PairedPromotionDecision;
|
|
112
|
-
//#endregion
|
|
113
|
-
export { PairedPromotionDecision as a, pairedDecisionShape as c, PairedMcNemarEvidence as i, PairedDecisionShape as n, PairedPromotionDecisionOptions as o, PairedDecisionStatistic as r, decidePairedPromotion as s, PairedDecisionMethod as t };
|
|
114
|
-
//# sourceMappingURL=paired-promotion-decision-B6zJ3gYM.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"paired-promotion-decision-B6zJ3gYM.d.ts","names":[],"sources":["../src/paired-promotion-decision.ts"],"mappings":";;;KA2DY;;KAMA;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;;EAGf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;EAQA;;UAGe;;EAEf;;EAEA;EACA;EACA,WAAW;EACX,QAAQ;;;;;EAKR;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,WAAW;;EAEX,SAAS;;;EAGT;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;UAIe;EACf,WAAW;;EAEX;;EAEA;;;;;;;;iBASc,oBACd,kBACA,iBACA,gCACC;;;;;;iBAiBa,sBACd,kBACA,iBACA,UAAS,iCACR"}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"prime-protocol-BfSalTfR.js","names":[],"sources":["../src/analyst/equal-terms.ts","../src/analyst/prime-protocol.ts"],"sourcesContent":["import { ValidationError } from '../errors'\n\n/**\n * The terms every declarative-arm comparison must hold equal before any\n * per-kind check runs. Arm kinds (analyst definitions, repair arms) extend\n * the comparison with their own fields; the refusals here are the floor.\n */\nexport interface DeclarativeArmTerms {\n readonly id: string\n /** Bounded repair turns the arm earns on a malformed reply. */\n readonly repairTurns: number\n}\n\nexport interface EqualDeclarativeTerms {\n readonly ids: readonly string[]\n readonly repairTurns: number\n}\n\n/**\n * Refuses a comparison whose arms are not on equal terms: an empty set, a\n * duplicated id, or unequal repair turns — a retry is a second sample the\n * other arms never got. `noun` names the arm kind in every message so a\n * refusal reads in the caller's vocabulary.\n */\nexport function assertEqualDeclarativeTerms(\n noun: string,\n terms: ReadonlyArray<DeclarativeArmTerms>,\n): EqualDeclarativeTerms {\n if (terms.length === 0) {\n throw new ValidationError(`a ${noun} comparison needs at least one ${noun}`)\n }\n const ids = terms.map((term) => term.id)\n const duplicate = ids.find((id, index) => ids.indexOf(id) !== index)\n if (duplicate !== undefined) {\n throw new ValidationError(`${noun} '${duplicate}' is declared twice`)\n }\n const repairTurns = terms[0]!.repairTurns\n const unequal = terms.find((term) => term.repairTurns !== repairTurns)\n if (unequal) {\n throw new ValidationError(\n `${noun} '${unequal.id}' gets ${unequal.repairTurns} bounded repair turns and ` +\n `${noun} '${terms[0]!.id}' gets ${repairTurns}; a retry is a second sample ` +\n 'the other arms never got',\n )\n }\n return { ids, repairTurns }\n}\n","import { createHash } from 'node:crypto'\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport type { PrimeBridgeTransport, PrimeBridgeTransportResult } from './prime-bridge-transport'\nimport type { ReplyContract, ReplyRowDecoded } from './reply-contract'\nimport type { AnalystUsageReceipt } from './types'\n\n/**\n * The prime analyst protocol: a one-shot RLM reached through an\n * OpenAI-compatible bridge, given the whole trajectory inline because it has no\n * REPL and no trace tools, answering in one fenced JSON block of short strings.\n *\n * This module owns every part of that protocol that does not depend on what the\n * rows MEAN: prompt and contract composition, the bounded repair turn, reply\n * extraction, row decoding order, the render-measure-fall-back-fail-loud\n * projection ladder, usage normalization, and the protocol identity digest.\n *\n * The cut: the core speaks RAW ROWS and never names a finding type. It knows a\n * prime reply carries an `answer` string plus an array of rows under a named\n * field, and that each row is decoded by a caller-supplied function. Consumers\n * — the CodeTraceBench benchmark runner here, a span-grounded trace analyzer\n * elsewhere — map decoded rows into their own finding shape on the way out. So\n * nothing in this file may import a block type, a finding type, or a trace\n * store, and no consumer needs another copy of the protocol.\n */\n\n// ── Reply contract ───────────────────────────────────────────────────\n\n/**\n * The prime reply grammar is the general analyst reply contract: these aliases\n * keep every prime-protocol consumer compiling against the shared type in\n * ./reply-contract. The exchange below reads the base fields (`rowsField`,\n * contract lines, `decodeRow`, `maxRows`); the strict-envelope knobs serve\n * one-shot JSON arms and are inert here.\n */\nexport type PrimeRowDecoded<TRow> = ReplyRowDecoded<TRow>\n\nexport type PrimeReplyContract<TRow> = ReplyContract<TRow>\n\n// ── Prompt composition ───────────────────────────────────────────────\n\nexport interface PrimePromptSpec {\n question: string\n /** Task definition spliced under a `TASK DEFINITION:` heading. Omit when the question is the whole task. */\n taskDefinition?: string\n contractLines: readonly string[]\n /** Line introducing the trajectory, e.g. `TRAJECTORY (trace_id X; 12 assistant step spans; ...):`. */\n trajectoryHeader: string\n renderedTrajectory: string\n /** Material appended after the trajectory, e.g. the final-verification spans. */\n trailer?: string\n}\n\nexport function buildPrimePrompt(spec: PrimePromptSpec): string {\n return [\n `QUESTION: ${spec.question}`,\n '',\n ...(spec.taskDefinition === undefined ? [] : ['TASK DEFINITION:', spec.taskDefinition, '']),\n ...spec.contractLines,\n '',\n spec.trajectoryHeader,\n spec.renderedTrajectory,\n ...(spec.trailer === undefined ? [] : ['', spec.trailer]),\n ].join('\\n')\n}\n\nexport interface PrimeRepairPromptSpec {\n defect: string\n previousReply: string\n repairContractLines: readonly string[]\n}\n\n/** Carries the malformed reply and the contract — never the trajectory. */\nexport function buildPrimeRepairPrompt(spec: PrimeRepairPromptSpec): string {\n return [\n 'Your previous reply to a trace-analysis task was structurally malformed and could not be parsed',\n `(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,\n 'fenced ```json block, no other text, no tools. The JSON object has exactly two fields:',\n ...spec.repairContractLines,\n '',\n 'PREVIOUS REPLY:',\n spec.previousReply,\n ].join('\\n')\n}\n\n// ── Reply extraction ─────────────────────────────────────────────────\n\n/**\n * Recover the reply's JSON object.\n *\n * Distinct from `extractJsonPayload` in ../llm-client, which serves a response\n * that DECLARES a JSON root and therefore must not scan onward. A prime reply\n * is prose plus a fenced block, and when the model emits several fences the\n * last one is its answer — so fences are scanned in reverse, and only then is a\n * brace-to-brace slice tried.\n */\nexport function extractPrimeJsonObject(text: string): Record<string, unknown> | null {\n const direct = parsePrimeJsonObject(text)\n if (direct) return direct\n const fenced = [...text.matchAll(/```(?:json)?\\s*\\n?([\\s\\S]*?)```/g)]\n for (let index = fenced.length - 1; index >= 0; index -= 1) {\n const candidate = parsePrimeJsonObject(fenced[index]![1]!)\n if (candidate) return candidate\n }\n const start = text.indexOf('{')\n const end = text.lastIndexOf('}')\n if (start >= 0 && end > start) {\n const candidate = parsePrimeJsonObject(text.slice(start, end + 1))\n if (candidate) return candidate\n }\n return null\n}\n\n/** Why the reply cannot be read as a prime answer, or null when it can. */\nexport function primeReplyDefect(\n parsed: Record<string, unknown> | null,\n rowsField: string,\n): string | null {\n if (parsed === null) return 'no parseable JSON object'\n if (!Array.isArray(parsed[rowsField])) return `JSON has no \"${rowsField}\" array`\n return null\n}\n\nfunction parsePrimeJsonObject(text: string): Record<string, unknown> | null {\n try {\n const value: unknown = JSON.parse(text.trim())\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n ? (value as Record<string, unknown>)\n : null\n } catch {\n return null\n }\n}\n\n// ── Usage ────────────────────────────────────────────────────────────\n\n/**\n * Token accounting exactly as the bridge reported it. Lossless on purpose: a\n * side the bridge did not report stays null rather than becoming a zero, and a\n * side it DID report survives even when its partner is missing.\n */\nexport interface PrimeRawUsage {\n /**\n * Model-usage records the bridge observed. NOT a count of bridge round trips:\n * one prime turn routes several internal model calls, so a single turn\n * typically reports 3-5.\n */\n calls: number | null\n /** Prompt tokens the bridge reported; null when it reported none. */\n inputTokens: number | null\n /** Completion tokens the bridge reported; null when it reported none. */\n outputTokens: number | null\n /**\n * True when the bridge DERIVED the counts (from character lengths) because\n * the backend reported no usage of its own. Pricing derived counts produces a\n * guess about a guess, which a reader must be able to tell apart from a\n * rate-estimated cost over exact tokens.\n */\n bridgeEstimated: boolean\n}\n\nexport function emptyPrimeRawUsage(): PrimeRawUsage {\n return { calls: null, inputTokens: null, outputTokens: null, bridgeEstimated: false }\n}\n\n/** Read the bridge's OpenAI-shaped `usage` object. */\nexport function normalizePrimeUsage(raw: unknown): PrimeRawUsage {\n if (typeof raw !== 'object' || raw === null) return emptyPrimeRawUsage()\n const record = raw as Record<string, unknown>\n return {\n calls: tokenCountOrNull(record.model_requests),\n inputTokens: tokenCountOrNull(record.prompt_tokens),\n outputTokens: tokenCountOrNull(record.completion_tokens),\n bridgeEstimated: record.estimated === true,\n }\n}\n\n/**\n * Sum two turns. Each side poisons independently: two turns that both report\n * input and neither report output yield a real input total beside a null\n * output, because discarding a measured count is as wrong as inventing one.\n */\nexport function mergePrimeRawUsage(a: PrimeRawUsage, b: PrimeRawUsage): PrimeRawUsage {\n return {\n calls: sumOrNull(a.calls, b.calls),\n inputTokens: sumOrNull(a.inputTokens, b.inputTokens),\n outputTokens: sumOrNull(a.outputTokens, b.outputTokens),\n bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated,\n }\n}\n\nfunction sumOrNull(a: number | null, b: number | null): number | null {\n return a !== null && b !== null ? a + b : null\n}\n\nfunction tokenCountOrNull(value: unknown): number | null {\n return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0 ? value : null\n}\n\n/**\n * Bind raw prime usage to agent-eval's typed receipt.\n *\n * `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count\n * cannot round-trip through `tokens` without writing a zero nobody measured.\n * The complete-accounting field therefore stays null, the reported side is\n * carried verbatim in `partialTokens`, and its price becomes the receipt's\n * `knownCostUsd` lower bound.\n *\n * Only agent-eval calls this; consumers with no pricing table read\n * `PrimeRawUsage` directly.\n */\nexport function analystUsageReceiptFromPrimeUsage(\n usage: PrimeRawUsage,\n pricing: CustomTokenPricing,\n): AnalystUsageReceipt {\n const { calls, inputTokens, outputTokens, bridgeEstimated } = usage\n const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {}\n if (inputTokens !== null && outputTokens !== null) {\n return {\n calls,\n tokens: { input: inputTokens, output: outputTokens },\n cost: { kind: 'estimated', usd: priceTokens(inputTokens, outputTokens, pricing) },\n ...estimatedTokens,\n }\n }\n if (inputTokens === null && outputTokens === null) {\n return { calls, tokens: null, cost: { kind: 'uncaptured', usd: null }, ...estimatedTokens }\n }\n return {\n calls,\n tokens: null,\n partialTokens: { input: inputTokens, output: outputTokens },\n cost: { kind: 'uncaptured', usd: null },\n knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),\n ...estimatedTokens,\n }\n}\n\nfunction priceTokens(input: number, output: number, pricing: CustomTokenPricing): number {\n return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1_000_000\n}\n\n// ── Two-turn exchange ────────────────────────────────────────────────\n\n/**\n * Why an exchange ended without a readable reply. `deadline` and `aborted` are\n * separate members from `transport` on purpose: a model that ran past its\n * deadline, a caller who cancelled, and a bridge that refused the connection\n * demand three different responses, and a record that conflates them cannot be\n * acted on.\n */\nexport type PrimeFailure =\n | {\n kind: 'transport' | 'unparseable-json' | 'no-content' | 'deadline' | 'malformed-reply'\n message: string\n }\n | { kind: 'http-status'; message: string; status: number; bodySnippet: string }\n | { kind: 'aborted'; message: string; cause: unknown }\n\nexport interface PrimeRepairState {\n attempted: boolean\n /** Null when no repair turn ran, or when the repair turn never returned a reply. */\n succeeded: boolean | null\n}\n\n/** One completed bridge turn. A turn that failed produces no record. */\nexport interface PrimeTurnRecord {\n turn: 'first' | 'repair'\n usage: PrimeRawUsage\n /** The reply's `usage` object verbatim; null when the bridge sent none. */\n rawUsage: unknown\n}\n\nexport interface PrimeRejectedRow {\n index: number\n reason: string\n}\n\nexport type PrimeExchangeOutcome<TRow> =\n | {\n ok: true\n /** The reply's `answer` field; null when absent or not a string. */\n answer: string | null\n rows: TRow[]\n rejected: PrimeRejectedRow[]\n /** Rows the model reported, before decoding or the count cap. */\n reportedRows: number\n /** Valid rows dropped by `contract.maxRows`. */\n overflow: number\n usage: PrimeRawUsage\n turns: PrimeTurnRecord[]\n repair: PrimeRepairState\n reply: string\n }\n | {\n ok: false\n failure: PrimeFailure\n usage: PrimeRawUsage\n turns: PrimeTurnRecord[]\n repair: PrimeRepairState\n /** The last reply received, when one arrived — the diagnostic artifact for a malformed case. */\n reply?: string\n }\n\nexport interface PrimeExchangeOptions<TRow> {\n contract: PrimeReplyContract<TRow>\n prompt: string\n transport: PrimeBridgeTransport\n /** Full endpoint, e.g. `http://localhost:4181/v1/chat/completions`. */\n url: string\n model: string\n /** Deadline for ONE turn. Prime analyses routinely exceed five minutes. */\n timeoutMs: number\n /** Whether a structurally malformed reply gets one bounded repair turn. */\n repair: boolean\n signal?: AbortSignal\n}\n\n/**\n * Run the protocol: one call, one bounded repair turn on a structurally\n * malformed reply, then decode. Zero valid rows from a well-formed reply is an\n * honest null, not a failure.\n */\nexport async function runPrimeExchange<TRow>(\n options: PrimeExchangeOptions<TRow>,\n): Promise<PrimeExchangeOutcome<TRow>> {\n const { contract } = options\n const turns: PrimeTurnRecord[] = []\n const repair: PrimeRepairState = { attempted: false, succeeded: null }\n\n const first = await callPrimeTurn(options, options.prompt)\n if (!first.ok) {\n return { ok: false, failure: first.failure, usage: mergeTurns(turns), turns, repair }\n }\n turns.push({ turn: 'first', usage: first.usage, rawUsage: first.rawUsage })\n\n let reply = first.content\n let parsed = extractPrimeJsonObject(reply)\n let defect = primeReplyDefect(parsed, contract.rowsField)\n\n if (defect !== null && options.repair) {\n repair.attempted = true\n const second = await callPrimeTurn(\n options,\n buildPrimeRepairPrompt({\n defect,\n previousReply: reply,\n repairContractLines: contract.repairContractLines,\n }),\n )\n if (!second.ok) {\n return { ok: false, failure: second.failure, usage: mergeTurns(turns), turns, repair, reply }\n }\n turns.push({ turn: 'repair', usage: second.usage, rawUsage: second.rawUsage })\n reply = second.content\n parsed = extractPrimeJsonObject(reply)\n defect = primeReplyDefect(parsed, contract.rowsField)\n repair.succeeded = defect === null\n }\n\n const usage = mergeTurns(turns)\n if (defect !== null) {\n return {\n ok: false,\n failure: {\n kind: 'malformed-reply',\n message: `${defect} in prime reply${repair.attempted ? ' even after the bounded repair turn' : ''}`,\n },\n usage,\n turns,\n repair,\n reply,\n }\n }\n\n const rawRows = parsed![contract.rowsField] as unknown[]\n const rows: TRow[] = []\n const rejected: PrimeRejectedRow[] = []\n let overflow = 0\n rawRows.forEach((row, index) => {\n const decoded = contract.decodeRow(row, index)\n if (!decoded.ok) {\n rejected.push({ index, reason: decoded.reason })\n return\n }\n // Shape before count: a malformed row must never consume an accepted slot.\n if (contract.maxRows !== undefined && rows.length >= contract.maxRows) {\n overflow += 1\n return\n }\n rows.push(decoded.row)\n })\n const answer = parsed!.answer\n return {\n ok: true,\n answer: typeof answer === 'string' ? answer : null,\n rows,\n rejected,\n reportedRows: rawRows.length,\n overflow,\n usage,\n turns,\n repair,\n reply,\n }\n}\n\ntype PrimeTurnOutcome =\n | { ok: true; content: string; usage: PrimeRawUsage; rawUsage: unknown }\n | { ok: false; failure: PrimeFailure }\n\nasync function callPrimeTurn<TRow>(\n options: PrimeExchangeOptions<TRow>,\n content: string,\n): Promise<PrimeTurnOutcome> {\n const { transport, url, model, timeoutMs, signal } = options\n const controller = new AbortController()\n const forwardAbort = () => controller.abort(signal?.reason)\n if (signal?.aborted) controller.abort(signal.reason)\n else signal?.addEventListener('abort', forwardAbort, { once: true })\n const deadline = setTimeout(() => controller.abort(), timeoutMs)\n let result: PrimeBridgeTransportResult\n try {\n result = await transport({\n url,\n body: { model, messages: [{ role: 'user', content }] },\n signal: controller.signal,\n })\n } catch (error) {\n if (signal?.aborted) {\n return {\n ok: false,\n failure: {\n kind: 'aborted',\n message: 'prime exchange cancelled by the caller',\n cause: error,\n },\n }\n }\n if (controller.signal.aborted) {\n return {\n ok: false,\n failure: { kind: 'deadline', message: `bridge call exceeded ${timeoutMs}ms` },\n }\n }\n return {\n ok: false,\n failure: {\n kind: 'transport',\n message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`,\n },\n }\n } finally {\n clearTimeout(deadline)\n signal?.removeEventListener('abort', forwardAbort)\n }\n\n if (result.status !== 200) {\n const bodySnippet = result.text.slice(0, 500)\n return {\n ok: false,\n failure: {\n kind: 'http-status',\n message: `bridge HTTP ${result.status}: ${bodySnippet}`,\n status: result.status,\n bodySnippet,\n },\n }\n }\n let response: unknown\n try {\n response = JSON.parse(result.text)\n } catch {\n return {\n ok: false,\n failure: {\n kind: 'unparseable-json',\n message: `bridge returned unparseable JSON (${result.text.length} bytes)`,\n },\n }\n }\n const replyContent = primeReplyContent(response)\n if (replyContent === null) {\n return {\n ok: false,\n failure: { kind: 'no-content', message: 'bridge reply carries no message content' },\n }\n }\n const rawUsage = primeReplyUsage(response)\n return { ok: true, content: replyContent, usage: normalizePrimeUsage(rawUsage), rawUsage }\n}\n\n/**\n * Fold from the FIRST turn, never from an empty receipt: an all-null identity\n * would poison every side it merged with and erase counts the bridge reported.\n */\nfunction mergeTurns(turns: readonly PrimeTurnRecord[]): PrimeRawUsage {\n if (turns.length === 0) return emptyPrimeRawUsage()\n return turns\n .slice(1)\n .reduce<PrimeRawUsage>((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0]!.usage)\n}\n\nfunction primeReplyContent(response: unknown): string | null {\n if (typeof response !== 'object' || response === null) return null\n const choices = (response as { choices?: unknown }).choices\n if (!Array.isArray(choices) || choices.length === 0) return null\n const message = (choices[0] as { message?: unknown })?.message\n if (typeof message !== 'object' || message === null) return null\n const content = (message as { content?: unknown }).content\n return typeof content === 'string' && content.length > 0 ? content : null\n}\n\nfunction primeReplyUsage(response: unknown): unknown {\n if (typeof response !== 'object' || response === null) return null\n return (response as { usage?: unknown }).usage ?? null\n}\n\n// ── Trajectory projection ────────────────────────────────────────────\n\n/**\n * Where the inlined trajectory comes from. The protocol needs only two moves —\n * fetch it all, or fetch a reduced version — so no store, artifact path, or\n * span type enters the shared signature.\n */\nexport interface PrimeProjectionSource<TItem> {\n /** The full projection, or null when the source declares itself oversized. */\n full(): Promise<readonly TItem[] | null>\n /** A reduced projection of the same material. */\n capped(): Promise<readonly TItem[]>\n /** How `capped` reduces, named for the failure message, e.g. `per-attribute cap 1200`. */\n cappedDescription: string\n}\n\nexport interface PrimeProjectionDelivery {\n mode: 'inline-json'\n fetch: 'full' | 'capped'\n renderedChars: number\n}\n\nexport type PrimeProjectionOutcome<TItem> =\n | {\n ok: true\n items: readonly TItem[]\n rendered: string\n delivery: PrimeProjectionDelivery\n }\n | { ok: false; reason: string; renderedChars: number }\n\n/**\n * Render, measure, fall back to the capped projection, re-measure, fail loud.\n *\n * Inline is the only delivery prime has, so an oversized trajectory is a\n * refusal rather than a silent truncation: dropping spans would understate the\n * trajectory and the analyst would answer a question about a different run.\n */\nexport async function projectPrimeTrajectory<TItem>(\n source: PrimeProjectionSource<TItem>,\n limits: { maxInlineChars: number },\n): Promise<PrimeProjectionOutcome<TItem>> {\n let fetch: PrimeProjectionDelivery['fetch'] = 'full'\n let items = await source.full()\n if (items === null) {\n fetch = 'capped'\n items = await source.capped()\n }\n let rendered = JSON.stringify(items)\n if (rendered.length > limits.maxInlineChars && fetch === 'full') {\n fetch = 'capped'\n items = await source.capped()\n rendered = JSON.stringify(items)\n }\n if (rendered.length > limits.maxInlineChars) {\n return {\n ok: false,\n reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,\n renderedChars: rendered.length,\n }\n }\n return {\n ok: true,\n items,\n rendered,\n delivery: { mode: 'inline-json', fetch, renderedChars: rendered.length },\n }\n}\n\n// ── Protocol identity ────────────────────────────────────────────────\n\nexport interface PrimeProtocolIdentity {\n question: string\n taskDefinition?: string\n contractLines: readonly string[]\n repairContractLines: readonly string[]\n /** Numeric limits the contract states, e.g. row and width caps. */\n limits: Readonly<Record<string, number>>\n}\n\n/**\n * Digest of everything a consumer can send to the bridge under the prime\n * protocol, recorded per observation so a prime result names the exact contract\n * that produced it.\n *\n * Computed over the ACTUALLY composed contract, so two consumers that both\n * stamp `analyst_id: 'prime'` while asking materially different questions get\n * different digests by construction. That is what makes 'prime' a reproducible\n * claim rather than a label.\n */\nexport function primeProtocolSha256(identity: PrimeProtocolIdentity): string {\n return createHash('sha256')\n .update(\n JSON.stringify({\n kind: 'prime-analyst-protocol',\n question: identity.question,\n taskPrompt: identity.taskDefinition ?? null,\n outputContract: identity.contractLines,\n repairContract: buildPrimeRepairPrompt({\n defect: '<defect>',\n previousReply: '<previous-reply>',\n repairContractLines: identity.repairContractLines,\n }),\n limits: identity.limits,\n }),\n )\n .digest('hex')\n}\n"],"mappings":";;;;;;;;;AAwBA,SAAgB,4BACd,MACA,OACuB;CACvB,IAAI,MAAM,WAAW,GACnB,MAAM,IAAI,gBAAgB,KAAK,KAAK,iCAAiC,MAAM;CAE7E,MAAM,MAAM,MAAM,KAAK,SAAS,KAAK,EAAE;CACvC,MAAM,YAAY,IAAI,MAAM,IAAI,UAAU,IAAI,QAAQ,EAAE,MAAM,KAAK;CACnE,IAAI,cAAc,KAAA,GAChB,MAAM,IAAI,gBAAgB,GAAG,KAAK,IAAI,UAAU,oBAAoB;CAEtE,MAAM,cAAc,MAAM,EAAE,CAAE;CAC9B,MAAM,UAAU,MAAM,MAAM,SAAS,KAAK,gBAAgB,WAAW;CACrE,IAAI,SACF,MAAM,IAAI,gBACR,GAAG,KAAK,IAAI,QAAQ,GAAG,SAAS,QAAQ,YAAY,4BAC/C,KAAK,IAAI,MAAM,EAAE,CAAE,GAAG,SAAS,YAAY,sDAElD;CAEF,OAAO;EAAE;EAAK;CAAY;AAC5B;;;ACMA,SAAgB,iBAAiB,MAA+B;CAC9D,OAAO;EACL,aAAa,KAAK;EAClB;EACA,GAAI,KAAK,mBAAmB,KAAA,IAAY,CAAC,IAAI;GAAC;GAAoB,KAAK;GAAgB;EAAE;EACzF,GAAG,KAAK;EACR;EACA,KAAK;EACL,KAAK;EACL,GAAI,KAAK,YAAY,KAAA,IAAY,CAAC,IAAI,CAAC,IAAI,KAAK,OAAO;CACzD,CAAC,CAAC,KAAK,IAAI;AACb;;AASA,SAAgB,uBAAuB,MAAqC;CAC1E,OAAO;EACL;EACA,IAAI,KAAK,OAAO;EAChB;EACA,GAAG,KAAK;EACR;EACA;EACA,KAAK;CACP,CAAC,CAAC,KAAK,IAAI;AACb;;;;;;;;;;AAaA,SAAgB,uBAAuB,MAA8C;CACnF,MAAM,SAAS,qBAAqB,IAAI;CACxC,IAAI,QAAQ,OAAO;CACnB,MAAM,SAAS,CAAC,GAAG,KAAK,SAAS,kCAAkC,CAAC;CACpE,KAAK,IAAI,QAAQ,OAAO,SAAS,GAAG,SAAS,GAAG,SAAS,GAAG;EAC1D,MAAM,YAAY,qBAAqB,OAAO,MAAM,CAAE,EAAG;EACzD,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,QAAQ,KAAK,QAAQ,GAAG;CAC9B,MAAM,MAAM,KAAK,YAAY,GAAG;CAChC,IAAI,SAAS,KAAK,MAAM,OAAO;EAC7B,MAAM,YAAY,qBAAqB,KAAK,MAAM,OAAO,MAAM,CAAC,CAAC;EACjE,IAAI,WAAW,OAAO;CACxB;CACA,OAAO;AACT;;AAGA,SAAgB,iBACd,QACA,WACe;CACf,IAAI,WAAW,MAAM,OAAO;CAC5B,IAAI,CAAC,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,gBAAgB,UAAU;CACxE,OAAO;AACT;AAEA,SAAS,qBAAqB,MAA8C;CAC1E,IAAI;EACF,MAAM,QAAiB,KAAK,MAAM,KAAK,KAAK,CAAC;EAC7C,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK,IACrE,QACD;CACN,QAAQ;EACN,OAAO;CACT;AACF;AA6BA,SAAgB,qBAAoC;CAClD,OAAO;EAAE,OAAO;EAAM,aAAa;EAAM,cAAc;EAAM,iBAAiB;CAAM;AACtF;;AAGA,SAAgB,oBAAoB,KAA6B;CAC/D,IAAI,OAAO,QAAQ,YAAY,QAAQ,MAAM,OAAO,mBAAmB;CACvE,MAAM,SAAS;CACf,OAAO;EACL,OAAO,iBAAiB,OAAO,cAAc;EAC7C,aAAa,iBAAiB,OAAO,aAAa;EAClD,cAAc,iBAAiB,OAAO,iBAAiB;EACvD,iBAAiB,OAAO,cAAc;CACxC;AACF;;;;;;AAOA,SAAgB,mBAAmB,GAAkB,GAAiC;CACpF,OAAO;EACL,OAAO,UAAU,EAAE,OAAO,EAAE,KAAK;EACjC,aAAa,UAAU,EAAE,aAAa,EAAE,WAAW;EACnD,cAAc,UAAU,EAAE,cAAc,EAAE,YAAY;EACtD,iBAAiB,EAAE,mBAAmB,EAAE;CAC1C;AACF;AAEA,SAAS,UAAU,GAAkB,GAAiC;CACpE,OAAO,MAAM,QAAQ,MAAM,OAAO,IAAI,IAAI;AAC5C;AAEA,SAAS,iBAAiB,OAA+B;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,cAAc,KAAK,KAAK,SAAS,IAAI,QAAQ;AAC1F;;;;;;;;;;;;;AAcA,SAAgB,kCACd,OACA,SACqB;CACrB,MAAM,EAAE,OAAO,aAAa,cAAc,oBAAoB;CAC9D,MAAM,kBAAkB,kBAAkB,EAAE,iBAAiB,KAAK,IAAI,CAAC;CACvE,IAAI,gBAAgB,QAAQ,iBAAiB,MAC3C,OAAO;EACL;EACA,QAAQ;GAAE,OAAO;GAAa,QAAQ;EAAa;EACnD,MAAM;GAAE,MAAM;GAAa,KAAK,YAAY,aAAa,cAAc,OAAO;EAAE;EAChF,GAAG;CACL;CAEF,IAAI,gBAAgB,QAAQ,iBAAiB,MAC3C,OAAO;EAAE;EAAO,QAAQ;EAAM,MAAM;GAAE,MAAM;GAAc,KAAK;EAAK;EAAG,GAAG;CAAgB;CAE5F,OAAO;EACL;EACA,QAAQ;EACR,eAAe;GAAE,OAAO;GAAa,QAAQ;EAAa;EAC1D,MAAM;GAAE,MAAM;GAAc,KAAK;EAAK;EACtC,cAAc,YAAY,eAAe,GAAG,gBAAgB,GAAG,OAAO;EACtE,GAAG;CACL;AACF;AAEA,SAAS,YAAY,OAAe,QAAgB,SAAqC;CACvF,QAAQ,QAAQ,QAAQ,qBAAqB,SAAS,QAAQ,uBAAuB;AACvF;;;;;;AAmFA,eAAsB,iBACpB,SACqC;CACrC,MAAM,EAAE,aAAa;CACrB,MAAM,QAA2B,CAAC;CAClC,MAAM,SAA2B;EAAE,WAAW;EAAO,WAAW;CAAK;CAErE,MAAM,QAAQ,MAAM,cAAc,SAAS,QAAQ,MAAM;CACzD,IAAI,CAAC,MAAM,IACT,OAAO;EAAE,IAAI;EAAO,SAAS,MAAM;EAAS,OAAO,WAAW,KAAK;EAAG;EAAO;CAAO;CAEtF,MAAM,KAAK;EAAE,MAAM;EAAS,OAAO,MAAM;EAAO,UAAU,MAAM;CAAS,CAAC;CAE1E,IAAI,QAAQ,MAAM;CAClB,IAAI,SAAS,uBAAuB,KAAK;CACzC,IAAI,SAAS,iBAAiB,QAAQ,SAAS,SAAS;CAExD,IAAI,WAAW,QAAQ,QAAQ,QAAQ;EACrC,OAAO,YAAY;EACnB,MAAM,SAAS,MAAM,cACnB,SACA,uBAAuB;GACrB;GACA,eAAe;GACf,qBAAqB,SAAS;EAChC,CAAC,CACH;EACA,IAAI,CAAC,OAAO,IACV,OAAO;GAAE,IAAI;GAAO,SAAS,OAAO;GAAS,OAAO,WAAW,KAAK;GAAG;GAAO;GAAQ;EAAM;EAE9F,MAAM,KAAK;GAAE,MAAM;GAAU,OAAO,OAAO;GAAO,UAAU,OAAO;EAAS,CAAC;EAC7E,QAAQ,OAAO;EACf,SAAS,uBAAuB,KAAK;EACrC,SAAS,iBAAiB,QAAQ,SAAS,SAAS;EACpD,OAAO,YAAY,WAAW;CAChC;CAEA,MAAM,QAAQ,WAAW,KAAK;CAC9B,IAAI,WAAW,MACb,OAAO;EACL,IAAI;EACJ,SAAS;GACP,MAAM;GACN,SAAS,GAAG,OAAO,iBAAiB,OAAO,YAAY,wCAAwC;EACjG;EACA;EACA;EACA;EACA;CACF;CAGF,MAAM,UAAU,OAAQ,SAAS;CACjC,MAAM,OAAe,CAAC;CACtB,MAAM,WAA+B,CAAC;CACtC,IAAI,WAAW;CACf,QAAQ,SAAS,KAAK,UAAU;EAC9B,MAAM,UAAU,SAAS,UAAU,KAAK,KAAK;EAC7C,IAAI,CAAC,QAAQ,IAAI;GACf,SAAS,KAAK;IAAE;IAAO,QAAQ,QAAQ;GAAO,CAAC;GAC/C;EACF;EAEA,IAAI,SAAS,YAAY,KAAA,KAAa,KAAK,UAAU,SAAS,SAAS;GACrE,YAAY;GACZ;EACF;EACA,KAAK,KAAK,QAAQ,GAAG;CACvB,CAAC;CACD,MAAM,SAAS,OAAQ;CACvB,OAAO;EACL,IAAI;EACJ,QAAQ,OAAO,WAAW,WAAW,SAAS;EAC9C;EACA;EACA,cAAc,QAAQ;EACtB;EACA;EACA;EACA;EACA;CACF;AACF;AAMA,eAAe,cACb,SACA,SAC2B;CAC3B,MAAM,EAAE,WAAW,KAAK,OAAO,WAAW,WAAW;CACrD,MAAM,aAAa,IAAI,gBAAgB;CACvC,MAAM,qBAAqB,WAAW,MAAM,QAAQ,MAAM;CAC1D,IAAI,QAAQ,SAAS,WAAW,MAAM,OAAO,MAAM;MAC9C,QAAQ,iBAAiB,SAAS,cAAc,EAAE,MAAM,KAAK,CAAC;CACnE,MAAM,WAAW,iBAAiB,WAAW,MAAM,GAAG,SAAS;CAC/D,IAAI;CACJ,IAAI;EACF,SAAS,MAAM,UAAU;GACvB;GACA,MAAM;IAAE;IAAO,UAAU,CAAC;KAAE,MAAM;KAAQ;IAAQ,CAAC;GAAE;GACrD,QAAQ,WAAW;EACrB,CAAC;CACH,SAAS,OAAO;EACd,IAAI,QAAQ,SACV,OAAO;GACL,IAAI;GACJ,SAAS;IACP,MAAM;IACN,SAAS;IACT,OAAO;GACT;EACF;EAEF,IAAI,WAAW,OAAO,SACpB,OAAO;GACL,IAAI;GACJ,SAAS;IAAE,MAAM;IAAY,SAAS,wBAAwB,UAAU;GAAI;EAC9E;EAEF,OAAO;GACL,IAAI;GACJ,SAAS;IACP,MAAM;IACN,SAAS,6BAA6B,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;GAC7F;EACF;CACF,UAAU;EACR,aAAa,QAAQ;EACrB,QAAQ,oBAAoB,SAAS,YAAY;CACnD;CAEA,IAAI,OAAO,WAAW,KAAK;EACzB,MAAM,cAAc,OAAO,KAAK,MAAM,GAAG,GAAG;EAC5C,OAAO;GACL,IAAI;GACJ,SAAS;IACP,MAAM;IACN,SAAS,eAAe,OAAO,OAAO,IAAI;IAC1C,QAAQ,OAAO;IACf;GACF;EACF;CACF;CACA,IAAI;CACJ,IAAI;EACF,WAAW,KAAK,MAAM,OAAO,IAAI;CACnC,QAAQ;EACN,OAAO;GACL,IAAI;GACJ,SAAS;IACP,MAAM;IACN,SAAS,qCAAqC,OAAO,KAAK,OAAO;GACnE;EACF;CACF;CACA,MAAM,eAAe,kBAAkB,QAAQ;CAC/C,IAAI,iBAAiB,MACnB,OAAO;EACL,IAAI;EACJ,SAAS;GAAE,MAAM;GAAc,SAAS;EAA0C;CACpF;CAEF,MAAM,WAAW,gBAAgB,QAAQ;CACzC,OAAO;EAAE,IAAI;EAAM,SAAS;EAAc,OAAO,oBAAoB,QAAQ;EAAG;CAAS;AAC3F;;;;;AAMA,SAAS,WAAW,OAAkD;CACpE,IAAI,MAAM,WAAW,GAAG,OAAO,mBAAmB;CAClD,OAAO,MACJ,MAAM,CAAC,CAAC,CACR,QAAuB,OAAO,SAAS,mBAAmB,OAAO,KAAK,KAAK,GAAG,MAAM,EAAE,CAAE,KAAK;AAClG;AAEA,SAAS,kBAAkB,UAAkC;CAC3D,IAAI,OAAO,aAAa,YAAY,aAAa,MAAM,OAAO;CAC9D,MAAM,UAAW,SAAmC;CACpD,IAAI,CAAC,MAAM,QAAQ,OAAO,KAAK,QAAQ,WAAW,GAAG,OAAO;CAC5D,MAAM,UAAW,QAAQ,EAAE,EAA4B;CACvD,IAAI,OAAO,YAAY,YAAY,YAAY,MAAM,OAAO;CAC5D,MAAM,UAAW,QAAkC;CACnD,OAAO,OAAO,YAAY,YAAY,QAAQ,SAAS,IAAI,UAAU;AACvE;AAEA,SAAS,gBAAgB,UAA4B;CACnD,IAAI,OAAO,aAAa,YAAY,aAAa,MAAM,OAAO;CAC9D,OAAQ,SAAiC,SAAS;AACpD;;;;;;;;AAwCA,eAAsB,uBACpB,QACA,QACwC;CACxC,IAAI,QAA0C;CAC9C,IAAI,QAAQ,MAAM,OAAO,KAAK;CAC9B,IAAI,UAAU,MAAM;EAClB,QAAQ;EACR,QAAQ,MAAM,OAAO,OAAO;CAC9B;CACA,IAAI,WAAW,KAAK,UAAU,KAAK;CACnC,IAAI,SAAS,SAAS,OAAO,kBAAkB,UAAU,QAAQ;EAC/D,QAAQ;EACR,QAAQ,MAAM,OAAO,OAAO;EAC5B,WAAW,KAAK,UAAU,KAAK;CACjC;CACA,IAAI,SAAS,SAAS,OAAO,gBAC3B,OAAO;EACL,IAAI;EACJ,QAAQ,yBAAyB,SAAS,OAAO,iBAAiB,OAAO,kBAAkB;EAC3F,eAAe,SAAS;CAC1B;CAEF,OAAO;EACL,IAAI;EACJ;EACA;EACA,UAAU;GAAE,MAAM;GAAe;GAAO,eAAe,SAAS;EAAO;CACzE;AACF;;;;;;;;;;;AAuBA,SAAgB,oBAAoB,UAAyC;CAC3E,OAAO,WAAW,QAAQ,CAAC,CACxB,OACC,KAAK,UAAU;EACb,MAAM;EACN,UAAU,SAAS;EACnB,YAAY,SAAS,kBAAkB;EACvC,gBAAgB,SAAS;EACzB,gBAAgB,uBAAuB;GACrC,QAAQ;GACR,eAAe;GACf,qBAAqB,SAAS;EAChC,CAAC;EACD,QAAQ,SAAS;CACnB,CAAC,CACH,CAAC,CACA,OAAO,KAAK;AACjB"}
|
|
@@ -1,289 +0,0 @@
|
|
|
1
|
-
import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-BnjdJ70P.js";
|
|
2
|
-
import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
3
|
-
import { i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-B6zJ3gYM.js";
|
|
4
|
-
//#region src/campaign/gates/power-preflight.d.ts
|
|
5
|
-
/**
|
|
6
|
-
* Power preflight — "can this budget detect the effect you are hunting?"
|
|
7
|
-
*
|
|
8
|
-
* The failure it prevents (measured, twice): a live prompt-improvement campaign ran
|
|
9
|
-
* 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
|
|
10
|
-
* (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
|
|
11
|
-
* that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
|
|
12
|
-
* any effect a prompt change plausibly produces. The budget was spent learning what
|
|
13
|
-
* a 30-second calculation on the baseline cells already knew. No eval framework we
|
|
14
|
-
* know of surfaces this; every underpowered improvement run everywhere ends in an
|
|
15
|
-
* uninformative "hold".
|
|
16
|
-
*
|
|
17
|
-
* Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
|
|
18
|
-
* bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
|
|
19
|
-
* true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
|
|
20
|
-
* before the candidate exists; we bound it by the zero-correlation case
|
|
21
|
-
* `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
|
|
22
|
-
* direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
|
|
23
|
-
*
|
|
24
|
-
* Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
|
|
25
|
-
* live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
|
|
26
|
-
* it to every result and warns when the run was structurally unable to ship.
|
|
27
|
-
*/
|
|
28
|
-
interface PowerPreflightOptions {
|
|
29
|
-
/** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
|
|
30
|
-
baselineComposites: number[];
|
|
31
|
-
/** Paired observations the budgeted comparison will produce
|
|
32
|
-
* (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
|
|
33
|
-
pairedN?: number;
|
|
34
|
-
/** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
|
|
35
|
-
deltaThreshold?: number;
|
|
36
|
-
/** CI confidence the gate uses. Default 0.95. */
|
|
37
|
-
confidence?: number;
|
|
38
|
-
/** True when the holdout is scored by the SAME judge/scorer family as the gate
|
|
39
|
-
* (selfImprove's default composition — one judge scores everything). Under a
|
|
40
|
-
* shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
|
|
41
|
-
* systematic judge bias is untouched, so the MDE here is a lower bound and the
|
|
42
|
-
* only full debiaser is an independent second scoring channel
|
|
43
|
-
* (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
|
|
44
|
-
sharedScorerChannel?: boolean;
|
|
45
|
-
}
|
|
46
|
-
interface PowerPreflight {
|
|
47
|
-
/** Paired observations the comparison will have. */
|
|
48
|
-
n: number;
|
|
49
|
-
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
50
|
-
sd: number;
|
|
51
|
-
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
52
|
-
mde: number;
|
|
53
|
-
/** Baseline holdout composite mean. */
|
|
54
|
-
baselineMean: number;
|
|
55
|
-
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
56
|
-
headroom: number;
|
|
57
|
-
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
58
|
-
* the run is structurally unable to ship regardless of proposal quality.
|
|
59
|
-
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
60
|
-
underpowered: boolean;
|
|
61
|
-
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
62
|
-
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
63
|
-
scaleAssumed: boolean;
|
|
64
|
-
deltaThreshold: number;
|
|
65
|
-
confidence: number;
|
|
66
|
-
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
67
|
-
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
68
|
-
sharedChannelCaveat?: string;
|
|
69
|
-
/** One actionable sentence for humans and logs. */
|
|
70
|
-
recommendation: string;
|
|
71
|
-
}
|
|
72
|
-
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
73
|
-
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
74
|
-
* spending a search to learn whether the effect you are hunting is even
|
|
75
|
-
* observable at this holdout size and worker variance. */
|
|
76
|
-
declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
|
|
77
|
-
//#endregion
|
|
78
|
-
//#region src/pareto.d.ts
|
|
79
|
-
/**
|
|
80
|
-
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
81
|
-
*
|
|
82
|
-
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
83
|
-
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
84
|
-
* ttfb), you rarely have a single "winner" — you have a set of
|
|
85
|
-
* non-dominated candidates. This module exposes:
|
|
86
|
-
*
|
|
87
|
-
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
88
|
-
* - `dominates`: does A dominate B across all objectives?
|
|
89
|
-
*
|
|
90
|
-
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
91
|
-
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
92
|
-
* `objective(candidate)` accessor.
|
|
93
|
-
*/
|
|
94
|
-
type Direction = 'maximize' | 'minimize';
|
|
95
|
-
interface Objective<T> {
|
|
96
|
-
/** Stable label used in reports. */
|
|
97
|
-
name: string;
|
|
98
|
-
direction: Direction;
|
|
99
|
-
value: (candidate: T) => number;
|
|
100
|
-
}
|
|
101
|
-
interface ParetoResult<T> {
|
|
102
|
-
frontier: T[];
|
|
103
|
-
dominated: T[];
|
|
104
|
-
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
105
|
-
dominanceMap: Array<{
|
|
106
|
-
dominator: T;
|
|
107
|
-
dominated: T[];
|
|
108
|
-
}>;
|
|
109
|
-
}
|
|
110
|
-
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
111
|
-
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
112
|
-
/**
|
|
113
|
-
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
114
|
-
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
115
|
-
* iff no other candidate dominates it.
|
|
116
|
-
*/
|
|
117
|
-
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
118
|
-
/**
|
|
119
|
-
* Weighted-sum scalarisation. Use as a tie-break / single-winner selector
|
|
120
|
-
* when callers don't want to consume a frontier. Each objective contributes
|
|
121
|
-
* its normalised value (0..1 via min-max across the candidate pool) times
|
|
122
|
-
* its weight; missing weights default to 1/N.
|
|
123
|
-
*
|
|
124
|
-
* Direction is honoured automatically — `minimize` axes have their values
|
|
125
|
-
* inverted before scaling so "higher scalar = better" always holds.
|
|
126
|
-
*/
|
|
127
|
-
declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
|
|
128
|
-
weights?: Partial<Record<string, number>>;
|
|
129
|
-
}): Array<{
|
|
130
|
-
candidate: T;
|
|
131
|
-
score: number;
|
|
132
|
-
}>;
|
|
133
|
-
/**
|
|
134
|
-
* NSGA-II crowding distance — secondary sort for ties on the frontier.
|
|
135
|
-
*
|
|
136
|
-
* When the Pareto front collapses to a single point (or many candidates tie
|
|
137
|
-
* on dominance), naive selection picks arbitrarily and the population
|
|
138
|
-
* degenerates over generations. NSGA-II preserves diversity by preferring
|
|
139
|
-
* candidates with more empty space around them on the frontier.
|
|
140
|
-
*
|
|
141
|
-
* Returns an array of `{ candidate, distance }` in the SAME order as the
|
|
142
|
-
* input. Higher distance = more isolated = should be preferred when
|
|
143
|
-
* preserving diversity.
|
|
144
|
-
*/
|
|
145
|
-
declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
146
|
-
candidate: T;
|
|
147
|
-
distance: number;
|
|
148
|
-
}>;
|
|
149
|
-
/**
|
|
150
|
-
* Pareto frontier with tie-break by crowding distance — the canonical
|
|
151
|
-
* NSGA-II selection step. Returns the frontier sorted by descending crowding
|
|
152
|
-
* distance so callers can `.slice(0, k)` to pick K diverse winners.
|
|
153
|
-
*/
|
|
154
|
-
declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
155
|
-
candidate: T;
|
|
156
|
-
distance: number;
|
|
157
|
-
}>;
|
|
158
|
-
//#endregion
|
|
159
|
-
//#region src/campaign/gates/promotion-policy.d.ts
|
|
160
|
-
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
161
|
-
* judge's composite; `dimension` reads a named per-dimension score. */
|
|
162
|
-
type ObjectiveSource = {
|
|
163
|
-
kind: 'composite';
|
|
164
|
-
} | {
|
|
165
|
-
kind: 'dimension';
|
|
166
|
-
dimension: string;
|
|
167
|
-
};
|
|
168
|
-
interface PromotionObjective {
|
|
169
|
-
/** Stable label used in reports + `contributingGates`. */
|
|
170
|
-
name: string;
|
|
171
|
-
source: ObjectiveSource;
|
|
172
|
-
/** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
|
|
173
|
-
* the paired delta so a positive bootstrap always means "candidate better". */
|
|
174
|
-
direction: Direction;
|
|
175
|
-
/** The good-direction paired-delta CI lower bound must EXCEED this to count
|
|
176
|
-
* as a significant gain on this axis. Interpreted in the judge's native
|
|
177
|
-
* scale. Default 0 (⇒ "confidently better"). */
|
|
178
|
-
gainThreshold?: number;
|
|
179
|
-
/** A floor breach (regression) is declared when the good-direction CI lower
|
|
180
|
-
* bound is below −floorTolerance, or when the exact small-sample test proves
|
|
181
|
-
* a drop past it. When omitted it auto-scales off observed magnitudes
|
|
182
|
-
* (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
|
|
183
|
-
floorTolerance?: number;
|
|
184
|
-
}
|
|
185
|
-
/** Per-axis verdict from the good-direction paired bootstrap. */
|
|
186
|
-
type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
|
|
187
|
-
interface AxisEvidence {
|
|
188
|
-
name: string;
|
|
189
|
-
source: ObjectiveSource;
|
|
190
|
-
direction: Direction;
|
|
191
|
-
/** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
|
|
192
|
-
* a positive value means the candidate is better on this axis.
|
|
193
|
-
*
|
|
194
|
-
* DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's
|
|
195
|
-
* score interval instead, because a percentile bootstrap over a three-atom
|
|
196
|
-
* delta lattice is not a valid interval at the nonzero margin `floorTolerance`
|
|
197
|
-
* and `gainThreshold` create. `ci` carries the interval that decided. */
|
|
198
|
-
bootstrap: PairedBootstrapResult;
|
|
199
|
-
/** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the
|
|
200
|
-
* caller asked for the median — on a pass/fail axis the median and its whole
|
|
201
|
-
* CI are pinned at 0 by tie domination and can see neither a gain nor a
|
|
202
|
-
* regression. `bootstrap.median` still carries the median point estimate. */
|
|
203
|
-
bootstrapStatistic: 'median' | 'mean';
|
|
204
|
-
/** The interval the axis verdict was actually decided on, good-direction and
|
|
205
|
-
* in the axis's native units. */
|
|
206
|
-
ci: {
|
|
207
|
-
low: number;
|
|
208
|
-
high: number;
|
|
209
|
-
};
|
|
210
|
-
/** Which estimator produced `ci`. */
|
|
211
|
-
decisionStatistic: PairedDecisionStatistic;
|
|
212
|
-
/** McNemar's exact evidence on a pass/fail axis; null otherwise. */
|
|
213
|
-
mcnemar: PairedMcNemarEvidence | null;
|
|
214
|
-
/** `ci` has zero width — no evidence in either direction, so the axis is
|
|
215
|
-
* neither improved nor regressed however the point estimate sits. */
|
|
216
|
-
indeterminate: boolean;
|
|
217
|
-
/** Paired observations contributing to this axis. */
|
|
218
|
-
n: number;
|
|
219
|
-
minimumRequired: number;
|
|
220
|
-
decisionMethod: PairedDecisionMethod;
|
|
221
|
-
gainThreshold: number;
|
|
222
|
-
floorTolerance: number;
|
|
223
|
-
verdict: AxisVerdict;
|
|
224
|
-
}
|
|
225
|
-
interface EvidenceVector {
|
|
226
|
-
/** One entry per objective — NOTHING averaged across axes. */
|
|
227
|
-
axes: AxisEvidence[];
|
|
228
|
-
/** Smallest paired n across axes that produced observations — the binding
|
|
229
|
-
* evidence-sufficiency constraint. 0 when no axis produced observations. */
|
|
230
|
-
minN: number;
|
|
231
|
-
/** Aggregate per-side cost from the gate context (a constraint input, not a
|
|
232
|
-
* CI axis — see the module header). */
|
|
233
|
-
cost: {
|
|
234
|
-
candidate: number;
|
|
235
|
-
baseline: number;
|
|
236
|
-
};
|
|
237
|
-
}
|
|
238
|
-
/** A promotion strategy: a pure function from the evidence vector to a verdict.
|
|
239
|
-
* Many policies can run over the same `EvidenceVector` and disagree — that's
|
|
240
|
-
* the point (competing strategies, shared evidence). */
|
|
241
|
-
type PromotionPolicy = (ev: EvidenceVector) => GateResult;
|
|
242
|
-
interface BuildEvidenceVectorOptions {
|
|
243
|
-
/** Minimum paired observations before an axis can claim significance; below
|
|
244
|
-
* it the axis is `few_runs`. The exact small-sample test may require more
|
|
245
|
-
* observations at the selected confidence. */
|
|
246
|
-
minProductiveRuns?: number;
|
|
247
|
-
/** Confidence level for every axis bootstrap. Default 0.95. */
|
|
248
|
-
confidence?: number;
|
|
249
|
-
/** Bootstrap resamples. Default 2000. */
|
|
250
|
-
resamples?: number;
|
|
251
|
-
/** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
|
|
252
|
-
seed?: number;
|
|
253
|
-
/** Paired statistic every axis CI is computed on. Default `'mean'` — see
|
|
254
|
-
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
255
|
-
statistic?: 'mean' | 'median';
|
|
256
|
-
}
|
|
257
|
-
/**
|
|
258
|
-
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
259
|
-
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
260
|
-
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
261
|
-
* a single source of truth governs pairing granularity + scale handling.
|
|
262
|
-
*/
|
|
263
|
-
declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
|
|
264
|
-
/**
|
|
265
|
-
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
266
|
-
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
267
|
-
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
268
|
-
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
269
|
-
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
270
|
-
*/
|
|
271
|
-
declare const paretoPolicy: PromotionPolicy;
|
|
272
|
-
interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
273
|
-
/** The objective vector. Every axis is both a gain source and a safety floor. */
|
|
274
|
-
objectives: PromotionObjective[];
|
|
275
|
-
/** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
|
|
276
|
-
* to run a stricter/looser strategy over the SAME bus (competing policies). */
|
|
277
|
-
policy?: PromotionPolicy;
|
|
278
|
-
/** Override the gate name in reports. */
|
|
279
|
-
name?: string;
|
|
280
|
-
}
|
|
281
|
-
/**
|
|
282
|
-
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
283
|
-
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
284
|
-
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
285
|
-
*/
|
|
286
|
-
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
287
|
-
//#endregion
|
|
288
|
-
export { powerPreflight as S, paretoFrontier as _, ObjectiveSource as a, PowerPreflight as b, PromotionPolicy as c, paretoSignificanceGate as d, Direction as f, dominates as g, crowdingDistance as h, EvidenceVector as i, buildEvidenceVector as l, ParetoResult as m, AxisVerdict as n, ParetoSignificanceGateOptions as o, Objective as p, BuildEvidenceVectorOptions as r, PromotionObjective as s, AxisEvidence as t, paretoPolicy as u, paretoFrontierWithCrowding as v, PowerPreflightOptions as x, scalarScore as y };
|
|
289
|
-
//# sourceMappingURL=promotion-policy-Ckjhzg_4.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"promotion-policy-Ckjhzg_4.d.ts","names":[],"sources":["../src/campaign/gates/power-preflight.ts","../src/pareto.ts","../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;UAwBiB;;EAEf;;;EAGA;;EAEA;;EAEA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;;EAGA;EACA;EACA;;;EAGA;;EAEA;;;;;;iBAec,eAAe,MAAM,wBAAwB;;;;;;;;;;;;;;;;;;KClEjD;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa;;;;;;;;;;iBA4B7E,YAAY,GAC1B,YAAY,KACZ,YAAY,UAAU,MACtB;EAAW,UAAU,QAAQ;IAC5B;EAAQ,WAAW;EAAG;;;;;;;;;;;;;;iBAyCT,iBAAiB,GAC/B,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;;iBA6BT,2BAA2B,GACzC,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;KCtHb;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
|