@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
|
+
import { a as medianInPlace, i as makeRng, n as binomialHalfUpperTail, t as assertFiniteSample } from "./internal-BDHPCnjk.js";
|
|
3
|
+
import { t as studentTCdf } from "./student-t-CvBq2mve.js";
|
|
4
|
+
//#region src/statistics/paired-tests.ts
|
|
5
|
+
/**
|
|
6
|
+
* Paired significance tests on continuous scores: the paired t-test, the
|
|
7
|
+
* promotion-gate paired bootstrap, the exact sign test, and the package-wide
|
|
8
|
+
* paired-delta decision statistic.
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* Paired t-test — before/after measurements on the SAME items.
|
|
12
|
+
* Pairing removes inter-item variance, giving tighter significance than
|
|
13
|
+
* an unpaired test when comparing prompt v1 vs prompt v2 on identical
|
|
14
|
+
* scenarios.
|
|
15
|
+
*
|
|
16
|
+
* Returns `t = p = null` where the statistic is undefined: fewer than two
|
|
17
|
+
* pairs, or a non-zero constant delta whose observed variance is zero. A
|
|
18
|
+
* constant shift carries no information about the variance it would have to
|
|
19
|
+
* be compared against, so the honest answer is "undefined", not `p = 0` —
|
|
20
|
+
* three observations cannot buy absolute certainty. This is the same contract
|
|
21
|
+
* {@link pairedCohensDz} states for the same condition. An all-zero delta is
|
|
22
|
+
* different: it is a measured null, and returns `t = 0, p = 1`.
|
|
23
|
+
*/
|
|
24
|
+
function pairedTTest(before, after) {
|
|
25
|
+
if (before.length !== after.length) throw new ValidationError(`pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
26
|
+
assertFiniteSample("pairedTTest", "before", before);
|
|
27
|
+
assertFiniteSample("pairedTTest", "after", after);
|
|
28
|
+
const n = before.length;
|
|
29
|
+
if (n < 2) return {
|
|
30
|
+
t: null,
|
|
31
|
+
df: 0,
|
|
32
|
+
p: null
|
|
33
|
+
};
|
|
34
|
+
const diffs = before.map((b, i) => after[i] - b);
|
|
35
|
+
const mean = diffs.reduce((a, b) => a + b, 0) / n;
|
|
36
|
+
const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1);
|
|
37
|
+
const se = Math.sqrt(variance / n);
|
|
38
|
+
if (se === 0) return mean === 0 ? {
|
|
39
|
+
t: 0,
|
|
40
|
+
df: n - 1,
|
|
41
|
+
p: 1
|
|
42
|
+
} : {
|
|
43
|
+
t: null,
|
|
44
|
+
df: n - 1,
|
|
45
|
+
p: null
|
|
46
|
+
};
|
|
47
|
+
const t = mean / se;
|
|
48
|
+
const df = n - 1;
|
|
49
|
+
return {
|
|
50
|
+
t,
|
|
51
|
+
df,
|
|
52
|
+
p: 2 * (1 - studentTCdf(Math.abs(t), df))
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Pairs below which a percentile bootstrap interval is descriptive spread only.
|
|
57
|
+
*
|
|
58
|
+
* `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000
|
|
59
|
+
* seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the
|
|
60
|
+
* median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling
|
|
61
|
+
* three points, not an implementation error — scipy's BCa gives 16.0 % on the
|
|
62
|
+
* same n = 3 data — so no change to the estimator moves it. Below this floor
|
|
63
|
+
* the decision belongs to the exact sign test or exact signed-rank test.
|
|
64
|
+
*/
|
|
65
|
+
const BOOTSTRAP_GATE_MIN_N = 20;
|
|
66
|
+
/**
|
|
67
|
+
* Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
|
|
68
|
+
* statistic (median by default); pairs are resampled with replacement. Throws
|
|
69
|
+
* on unequal sample sizes.
|
|
70
|
+
*
|
|
71
|
+
* `low > threshold` carries the stated confidence ONLY at `n ≥
|
|
72
|
+
* {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the
|
|
73
|
+
* check fires under a true null several times more often than nominal, so the
|
|
74
|
+
* interval is descriptive spread and a promotion must not turn on it.
|
|
75
|
+
*/
|
|
76
|
+
function pairedBootstrap(before, after, opts = {}) {
|
|
77
|
+
if (before.length !== after.length) throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
78
|
+
const confidence = opts.confidence ?? .95;
|
|
79
|
+
const resamples = opts.resamples ?? 2e3;
|
|
80
|
+
const statistic = opts.statistic ?? "median";
|
|
81
|
+
if (confidence <= 0 || confidence >= 1) throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`);
|
|
82
|
+
const n = before.length;
|
|
83
|
+
const deltas = before.map((b, i) => after[i] - b);
|
|
84
|
+
const gateEligible = n >= 20;
|
|
85
|
+
if (n === 0) return {
|
|
86
|
+
n: 0,
|
|
87
|
+
median: 0,
|
|
88
|
+
mean: 0,
|
|
89
|
+
low: 0,
|
|
90
|
+
high: 0,
|
|
91
|
+
confidence,
|
|
92
|
+
resamples,
|
|
93
|
+
gateEligible
|
|
94
|
+
};
|
|
95
|
+
if (n === 1) {
|
|
96
|
+
const d = deltas[0];
|
|
97
|
+
return {
|
|
98
|
+
n: 1,
|
|
99
|
+
median: d,
|
|
100
|
+
mean: d,
|
|
101
|
+
low: d,
|
|
102
|
+
high: d,
|
|
103
|
+
confidence,
|
|
104
|
+
resamples,
|
|
105
|
+
gateEligible
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
const rng = makeRng(opts.seed, deltas);
|
|
109
|
+
const samples = new Array(resamples);
|
|
110
|
+
for (let b = 0; b < resamples; b++) if (statistic === "mean") {
|
|
111
|
+
let sum = 0;
|
|
112
|
+
for (let k = 0; k < n; k++) sum += deltas[Math.floor(rng() * n)];
|
|
113
|
+
samples[b] = sum / n;
|
|
114
|
+
} else {
|
|
115
|
+
const acc = new Array(n);
|
|
116
|
+
for (let k = 0; k < n; k++) acc[k] = deltas[Math.floor(rng() * n)];
|
|
117
|
+
samples[b] = medianInPlace(acc);
|
|
118
|
+
}
|
|
119
|
+
samples.sort((a, b) => a - b);
|
|
120
|
+
const alpha = 1 - confidence;
|
|
121
|
+
const lowIdx = Math.floor(alpha / 2 * resamples);
|
|
122
|
+
const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1);
|
|
123
|
+
return {
|
|
124
|
+
n,
|
|
125
|
+
median: medianInPlace([...deltas]),
|
|
126
|
+
mean: deltas.reduce((s, x) => s + x, 0) / n,
|
|
127
|
+
low: samples[lowIdx],
|
|
128
|
+
high: samples[Math.max(highIdx, lowIdx)],
|
|
129
|
+
confidence,
|
|
130
|
+
resamples,
|
|
131
|
+
gateEligible
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* Exact one-sided sign test over paired differences.
|
|
136
|
+
*
|
|
137
|
+
* Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
|
|
138
|
+
* tests whether positive signs are more likely than negative signs and returns
|
|
139
|
+
* `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
|
|
140
|
+
* negative signs as successes instead. With a continuous difference
|
|
141
|
+
* distribution this is the usual directional median test. Exact zero
|
|
142
|
+
* differences are ties and do not enter the binomial denominator. All-tie and
|
|
143
|
+
* empty inputs return p = 1. Every input difference must be finite, and the
|
|
144
|
+
* direction must be chosen explicitly so a caller cannot select it after
|
|
145
|
+
* seeing the signs.
|
|
146
|
+
*/
|
|
147
|
+
function pairedSignTest(differences, alternative) {
|
|
148
|
+
if (alternative !== "greater" && alternative !== "less") throw new ValidationError(`pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`);
|
|
149
|
+
let positive = 0;
|
|
150
|
+
let negative = 0;
|
|
151
|
+
let ties = 0;
|
|
152
|
+
for (let i = 0; i < differences.length; i++) {
|
|
153
|
+
const difference = differences[i];
|
|
154
|
+
if (!Number.isFinite(difference)) throw new ValidationError(`pairedSignTest: difference at index ${i} must be finite, got ${difference}`);
|
|
155
|
+
if (difference > 0) positive++;
|
|
156
|
+
else if (difference < 0) negative++;
|
|
157
|
+
else ties++;
|
|
158
|
+
}
|
|
159
|
+
const nNonTies = positive + negative;
|
|
160
|
+
const successes = alternative === "greater" ? positive : negative;
|
|
161
|
+
return {
|
|
162
|
+
n: differences.length,
|
|
163
|
+
positive,
|
|
164
|
+
negative,
|
|
165
|
+
ties,
|
|
166
|
+
nNonTies,
|
|
167
|
+
alternative,
|
|
168
|
+
pValue: binomialHalfUpperTail(successes, nNonTies)
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
/** Fraction of paired observations whose delta is an exact tie (|after − before|
|
|
172
|
+
* < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
|
|
173
|
+
function pairedDeltaTieFraction(before, after) {
|
|
174
|
+
if (before.length !== after.length) throw new Error(`pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
175
|
+
const n = before.length;
|
|
176
|
+
if (n === 0) return 0;
|
|
177
|
+
let ties = 0;
|
|
178
|
+
for (let i = 0; i < n; i++) if (Math.abs(after[i] - before[i]) < 1e-9) ties++;
|
|
179
|
+
return ties / n;
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* The paired-delta statistic a DECISION is computed on, package-wide.
|
|
183
|
+
*
|
|
184
|
+
* The mean paired delta is the estimator that answers the question a promotion
|
|
185
|
+
* gate asks — "by how much did the candidate move the score" — in the caller's
|
|
186
|
+
* own units, and it equals the aggregate lift everyone quotes. The MEDIAN
|
|
187
|
+
* answers a different question and loses the answer to this one in every regime
|
|
188
|
+
* eval data actually lands in:
|
|
189
|
+
* - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
|
|
190
|
+
* {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
|
|
191
|
+
* are pinned at exactly 0 however large the shift. (Decide these on
|
|
192
|
+
* {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
|
|
193
|
+
* - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
|
|
194
|
+
* construction, and `ci.low > threshold` then answers "no" forever at a
|
|
195
|
+
* non-negative threshold and "yes" forever at a negative one.
|
|
196
|
+
* - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
|
|
197
|
+
* integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
|
|
198
|
+
* leaves, put the median on a coarse lattice whose bootstrap percentiles
|
|
199
|
+
* land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
|
|
200
|
+
* +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
|
|
201
|
+
* lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
|
|
202
|
+
* That last case is why there is no tie-fraction threshold here: any cutoff on
|
|
203
|
+
* ties leaves the lattice case open on the other side of it.
|
|
204
|
+
*
|
|
205
|
+
* `heldoutSignificance` has defaulted to the mean since #316 for the same
|
|
206
|
+
* reason. The median remains available per call site for callers who
|
|
207
|
+
* specifically want outlier robustness and accept the blindness.
|
|
208
|
+
*/
|
|
209
|
+
const DECISION_PAIRED_DELTA_STATISTIC = "mean";
|
|
210
|
+
//#endregion
|
|
211
|
+
export { pairedSignTest as a, pairedDeltaTieFraction as i, DECISION_PAIRED_DELTA_STATISTIC as n, pairedTTest as o, pairedBootstrap as r, BOOTSTRAP_GATE_MIN_N as t };
|
|
212
|
+
|
|
213
|
+
//# sourceMappingURL=paired-tests-BHIhYVdu.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"paired-tests-BHIhYVdu.js","names":[],"sources":["../src/statistics/paired-tests.ts"],"sourcesContent":["/**\n * Paired significance tests on continuous scores: the paired t-test, the\n * promotion-gate paired bootstrap, the exact sign test, and the package-wide\n * paired-delta decision statistic.\n */\n\nimport { ValidationError } from '../errors'\nimport { studentTCdf } from '../math/student-t'\nimport { assertFiniteSample, binomialHalfUpperTail, makeRng, medianInPlace } from './internal'\n\nexport interface PairedTTestResult {\n /** Null when the statistic is undefined — see {@link pairedTTest}. */\n t: number | null\n df: number\n /** Null exactly when `t` is null. */\n p: number | null\n}\n\n/**\n * Paired t-test — before/after measurements on the SAME items.\n * Pairing removes inter-item variance, giving tighter significance than\n * an unpaired test when comparing prompt v1 vs prompt v2 on identical\n * scenarios.\n *\n * Returns `t = p = null` where the statistic is undefined: fewer than two\n * pairs, or a non-zero constant delta whose observed variance is zero. A\n * constant shift carries no information about the variance it would have to\n * be compared against, so the honest answer is \"undefined\", not `p = 0` —\n * three observations cannot buy absolute certainty. This is the same contract\n * {@link pairedCohensDz} states for the same condition. An all-zero delta is\n * different: it is a measured null, and returns `t = 0, p = 1`.\n */\nexport function pairedTTest(before: number[], after: number[]): PairedTTestResult {\n if (before.length !== after.length) {\n throw new ValidationError(\n `pairedTTest: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n assertFiniteSample('pairedTTest', 'before', before)\n assertFiniteSample('pairedTTest', 'after', after)\n const n = before.length\n if (n < 2) return { t: null, df: 0, p: null }\n\n const diffs = before.map((b, i) => after[i]! - b)\n const mean = diffs.reduce((a, b) => a + b, 0) / n\n const variance = diffs.reduce((acc, d) => acc + (d - mean) ** 2, 0) / (n - 1)\n const se = Math.sqrt(variance / n)\n if (se === 0) {\n return mean === 0 ? { t: 0, df: n - 1, p: 1 } : { t: null, df: n - 1, p: null }\n }\n\n const t = mean / se\n const df = n - 1\n const p = 2 * (1 - studentTCdf(Math.abs(t), df))\n return { t, df, p }\n}\n\n// ── Paired bootstrap (promotion-gate effect size) ────────────────────\n\nexport interface PairedBootstrapResult {\n /** Number of paired observations. */\n n: number\n /** Median of paired deltas (after − before). */\n median: number\n /** Mean of paired deltas. */\n mean: number\n /** Lower bound of the bootstrap CI on the chosen statistic. */\n low: number\n /** Upper bound of the bootstrap CI on the chosen statistic. */\n high: number\n /** Confidence level used (e.g. 0.95). */\n confidence: number\n /** Number of bootstrap resamples used. */\n resamples: number\n /** False below {@link BOOTSTRAP_GATE_MIN_N}. See {@link pairedBootstrap}. */\n gateEligible: boolean\n}\n\n/**\n * Pairs below which a percentile bootstrap interval is descriptive spread only.\n *\n * `P(low > 0)` under a true null, against a nominal 2.5 %, measured over 4000\n * seeded trials: 13.53 % at n = 3, 3.52 % at n = 10, 3.10 % at n = 20 on the\n * median; 13.85 %, 4.90 %, 3.80 % on the mean. This is intrinsic to resampling\n * three points, not an implementation error — scipy's BCa gives 16.0 % on the\n * same n = 3 data — so no change to the estimator moves it. Below this floor\n * the decision belongs to the exact sign test or exact signed-rank test.\n */\nexport const BOOTSTRAP_GATE_MIN_N = 20\n\nexport interface PairedBootstrapOptions {\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resample count. Default 2000. */\n resamples?: number\n /** Statistic to bootstrap. Default 'median'. */\n statistic?: 'median' | 'mean'\n /** Deterministic seed. If omitted, derived from the deltas so the interval\n * is reproducible regardless. */\n seed?: number\n}\n\n/**\n * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen\n * statistic (median by default); pairs are resampled with replacement. Throws\n * on unequal sample sizes.\n *\n * `low > threshold` carries the stated confidence ONLY at `n ≥\n * {@link BOOTSTRAP_GATE_MIN_N}`, which `gateEligible` reports. Below it the\n * check fires under a true null several times more often than nominal, so the\n * interval is descriptive spread and a promotion must not turn on it.\n */\nexport function pairedBootstrap(\n before: number[],\n after: number[],\n opts: PairedBootstrapOptions = {},\n): PairedBootstrapResult {\n if (before.length !== after.length) {\n throw new Error(`pairedBootstrap: unequal sample sizes (${before.length} vs ${after.length})`)\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const statistic = opts.statistic ?? 'median'\n if (confidence <= 0 || confidence >= 1) {\n throw new Error(`pairedBootstrap: confidence must be in (0,1), got ${confidence}`)\n }\n\n const n = before.length\n const deltas = before.map((b, i) => after[i]! - b)\n const gateEligible = n >= BOOTSTRAP_GATE_MIN_N\n if (n === 0) {\n return { n: 0, median: 0, mean: 0, low: 0, high: 0, confidence, resamples, gateEligible }\n }\n if (n === 1) {\n const d = deltas[0]!\n return { n: 1, median: d, mean: d, low: d, high: d, confidence, resamples, gateEligible }\n }\n\n const rng = makeRng(opts.seed, deltas)\n const samples = new Array<number>(resamples)\n for (let b = 0; b < resamples; b++) {\n if (statistic === 'mean') {\n let sum = 0\n for (let k = 0; k < n; k++) {\n sum += deltas[Math.floor(rng() * n)]!\n }\n samples[b] = sum / n\n } else {\n const acc = new Array<number>(n)\n for (let k = 0; k < n; k++) {\n acc[k] = deltas[Math.floor(rng() * n)]!\n }\n samples[b] = medianInPlace(acc)\n }\n }\n samples.sort((a, b) => a - b)\n\n const alpha = 1 - confidence\n const lowIdx = Math.floor((alpha / 2) * resamples)\n const highIdx = Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)\n\n return {\n n,\n median: medianInPlace([...deltas]),\n mean: deltas.reduce((s, x) => s + x, 0) / n,\n low: samples[lowIdx]!,\n high: samples[Math.max(highIdx, lowIdx)]!,\n confidence,\n resamples,\n gateEligible,\n }\n}\n\n/** Pre-registered direction for a one-sided paired sign test. */\nexport type SignTestAlternative = 'greater' | 'less'\n\n/** Exact one-sided sign-test result for paired numeric differences. */\nexport interface PairedSignTestResult {\n /** Total supplied differences, including zero ties. */\n n: number\n /** Strictly positive differences. */\n positive: number\n /** Strictly negative differences. */\n negative: number\n /** Zero differences excluded from the binomial test. */\n ties: number\n /** Non-zero differences used by the binomial test. */\n nNonTies: number\n /** Direction of the pre-registered alternative hypothesis. */\n alternative: SignTestAlternative\n /** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */\n pValue: number\n}\n\n/**\n * Exact one-sided sign test over paired differences.\n *\n * Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`\n * tests whether positive signs are more likely than negative signs and returns\n * `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats\n * negative signs as successes instead. With a continuous difference\n * distribution this is the usual directional median test. Exact zero\n * differences are ties and do not enter the binomial denominator. All-tie and\n * empty inputs return p = 1. Every input difference must be finite, and the\n * direction must be chosen explicitly so a caller cannot select it after\n * seeing the signs.\n */\nexport function pairedSignTest(\n differences: readonly number[],\n alternative: SignTestAlternative,\n): PairedSignTestResult {\n if (alternative !== 'greater' && alternative !== 'less') {\n throw new ValidationError(\n `pairedSignTest: alternative must be 'greater' or 'less', got ${alternative}`,\n )\n }\n\n let positive = 0\n let negative = 0\n let ties = 0\n for (let i = 0; i < differences.length; i++) {\n const difference = differences[i]!\n if (!Number.isFinite(difference)) {\n throw new ValidationError(\n `pairedSignTest: difference at index ${i} must be finite, got ${difference}`,\n )\n }\n if (difference > 0) positive++\n else if (difference < 0) negative++\n else ties++\n }\n\n const nNonTies = positive + negative\n const successes = alternative === 'greater' ? positive : negative\n return {\n n: differences.length,\n positive,\n negative,\n ties,\n nNonTies,\n alternative,\n pValue: binomialHalfUpperTail(successes, nNonTies),\n }\n}\n\n/** Fraction of paired observations whose delta is an exact tie (|after − before|\n * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */\nexport function pairedDeltaTieFraction(\n before: ArrayLike<number>,\n after: ArrayLike<number>,\n): number {\n if (before.length !== after.length) {\n throw new Error(\n `pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const n = before.length\n if (n === 0) return 0\n let ties = 0\n for (let i = 0; i < n; i++) {\n if (Math.abs(after[i]! - before[i]!) < 1e-9) ties++\n }\n return ties / n\n}\n\n/**\n * The paired-delta statistic a DECISION is computed on, package-wide.\n *\n * The mean paired delta is the estimator that answers the question a promotion\n * gate asks — \"by how much did the candidate move the score\" — in the caller's\n * own units, and it equals the aggregate lift everyone quotes. The MEDIAN\n * answers a different question and loses the answer to this one in every regime\n * eval data actually lands in:\n * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in\n * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI\n * are pinned at exactly 0 however large the shift. (Decide these on\n * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)\n * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by\n * construction, and `ci.low > threshold` then answers \"no\" forever at a\n * non-negative threshold and \"yes\" forever at a negative one.\n * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on\n * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail\n * leaves, put the median on a coarse lattice whose bootstrap percentiles\n * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real\n * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —\n * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.\n * That last case is why there is no tie-fraction threshold here: any cutoff on\n * ties leaves the lattice case open on the other side of it.\n *\n * `heldoutSignificance` has defaulted to the mean since #316 for the same\n * reason. The median remains available per call site for callers who\n * specifically want outlier robustness and accept the blindness.\n */\nexport const DECISION_PAIRED_DELTA_STATISTIC: 'mean' = 'mean'\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,YAAY,QAAkB,OAAoC;CAChF,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,gBACR,sCAAsC,OAAO,OAAO,MAAM,MAAM,OAAO,EACzE;CAEF,mBAAmB,eAAe,UAAU,MAAM;CAClD,mBAAmB,eAAe,SAAS,KAAK;CAChD,MAAM,IAAI,OAAO;CACjB,IAAI,IAAI,GAAG,OAAO;EAAE,GAAG;EAAM,IAAI;EAAG,GAAG;CAAK;CAE5C,MAAM,QAAQ,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CAChD,MAAM,OAAO,MAAM,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAChD,MAAM,WAAW,MAAM,QAAQ,KAAK,MAAM,OAAO,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI;CAC3E,MAAM,KAAK,KAAK,KAAK,WAAW,CAAC;CACjC,IAAI,OAAO,GACT,OAAO,SAAS,IAAI;EAAE,GAAG;EAAG,IAAI,IAAI;EAAG,GAAG;CAAE,IAAI;EAAE,GAAG;EAAM,IAAI,IAAI;EAAG,GAAG;CAAK;CAGhF,MAAM,IAAI,OAAO;CACjB,MAAM,KAAK,IAAI;CAEf,OAAO;EAAE;EAAG;EAAI,GADN,KAAK,IAAI,YAAY,KAAK,IAAI,CAAC,GAAG,EAAE;CAC5B;AACpB;;;;;;;;;;;AAiCA,MAAa,uBAAuB;;;;;;;;;;;AAwBpC,SAAgB,gBACd,QACA,OACA,OAA+B,CAAC,GACT;CACvB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MAAM,0CAA0C,OAAO,OAAO,MAAM,MAAM,OAAO,EAAE;CAE/F,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,cAAc,KAAK,cAAc,GACnC,MAAM,IAAI,MAAM,qDAAqD,YAAY;CAGnF,MAAM,IAAI,OAAO;CACjB,MAAM,SAAS,OAAO,KAAK,GAAG,MAAM,MAAM,KAAM,CAAC;CACjD,MAAM,eAAe,KAAA;CACrB,IAAI,MAAM,GACR,OAAO;EAAE,GAAG;EAAG,QAAQ;EAAG,MAAM;EAAG,KAAK;EAAG,MAAM;EAAG;EAAY;EAAW;CAAa;CAE1F,IAAI,MAAM,GAAG;EACX,MAAM,IAAI,OAAO;EACjB,OAAO;GAAE,GAAG;GAAG,QAAQ;GAAG,MAAM;GAAG,KAAK;GAAG,MAAM;GAAG;GAAY;GAAW;EAAa;CAC1F;CAEA,MAAM,MAAM,QAAQ,KAAK,MAAM,MAAM;CACrC,MAAM,UAAU,IAAI,MAAc,SAAS;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAC7B,IAAI,cAAc,QAAQ;EACxB,IAAI,MAAM;EACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,OAAO,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEpC,QAAQ,KAAK,MAAM;CACrB,OAAO;EACL,MAAM,MAAM,IAAI,MAAc,CAAC;EAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,OAAO,KAAK,MAAM,IAAI,IAAI,CAAC;EAEtC,QAAQ,KAAK,cAAc,GAAG;CAChC;CAEF,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAE5B,MAAM,QAAQ,IAAI;CAClB,MAAM,SAAS,KAAK,MAAO,QAAQ,IAAK,SAAS;CACjD,MAAM,UAAU,KAAK,IAAI,YAAY,GAAG,KAAK,MAAM,IAAI,QAAQ,KAAK,SAAS,IAAI,CAAC;CAElF,OAAO;EACL;EACA,QAAQ,cAAc,CAAC,GAAG,MAAM,CAAC;EACjC,MAAM,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAC1C,KAAK,QAAQ;EACb,MAAM,QAAQ,KAAK,IAAI,SAAS,MAAM;EACtC;EACA;EACA;CACF;AACF;;;;;;;;;;;;;;AAoCA,SAAgB,eACd,aACA,aACsB;CACtB,IAAI,gBAAgB,aAAa,gBAAgB,QAC/C,MAAM,IAAI,gBACR,gEAAgE,aAClE;CAGF,IAAI,WAAW;CACf,IAAI,WAAW;CACf,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,QAAQ,KAAK;EAC3C,MAAM,aAAa,YAAY;EAC/B,IAAI,CAAC,OAAO,SAAS,UAAU,GAC7B,MAAM,IAAI,gBACR,uCAAuC,EAAE,uBAAuB,YAClE;EAEF,IAAI,aAAa,GAAG;OACf,IAAI,aAAa,GAAG;OACpB;CACP;CAEA,MAAM,WAAW,WAAW;CAC5B,MAAM,YAAY,gBAAgB,YAAY,WAAW;CACzD,OAAO;EACL,GAAG,YAAY;EACf;EACA;EACA;EACA;EACA;EACA,QAAQ,sBAAsB,WAAW,QAAQ;CACnD;AACF;;;AAIA,SAAgB,uBACd,QACA,OACQ;CACR,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,iDAAiD,OAAO,OAAO,MAAM,MAAM,OAAO,EACpF;CAEF,MAAM,IAAI,OAAO;CACjB,IAAI,MAAM,GAAG,OAAO;CACpB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,IAAI,KAAK,IAAI,MAAM,KAAM,OAAO,EAAG,IAAI,MAAM;CAE/C,OAAO,OAAO;AAChB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8BA,MAAa,kCAA0C"}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
//#region src/campaign/gates/power-preflight.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* Power preflight — "can this budget detect the effect you are hunting?"
|
|
4
|
+
*
|
|
5
|
+
* The failure it prevents (measured, twice): a live prompt-improvement campaign ran
|
|
6
|
+
* 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
|
|
7
|
+
* (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
|
|
8
|
+
* that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
|
|
9
|
+
* any effect a prompt change plausibly produces. The budget was spent learning what
|
|
10
|
+
* a 30-second calculation on the baseline cells already knew. No eval framework we
|
|
11
|
+
* know of surfaces this; every underpowered improvement run everywhere ends in an
|
|
12
|
+
* uninformative "hold".
|
|
13
|
+
*
|
|
14
|
+
* Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
|
|
15
|
+
* bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
|
|
16
|
+
* true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
|
|
17
|
+
* before the candidate exists; we bound it by the zero-correlation case
|
|
18
|
+
* `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
|
|
19
|
+
* direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
|
|
20
|
+
*
|
|
21
|
+
* Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
|
|
22
|
+
* live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
|
|
23
|
+
* it to every result and warns when the run was structurally unable to ship.
|
|
24
|
+
*/
|
|
25
|
+
interface PowerPreflightOptions {
|
|
26
|
+
/** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
|
|
27
|
+
baselineComposites: number[];
|
|
28
|
+
/** Paired observations the budgeted comparison will produce
|
|
29
|
+
* (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
|
|
30
|
+
pairedN?: number;
|
|
31
|
+
/** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
|
|
32
|
+
deltaThreshold?: number;
|
|
33
|
+
/** CI confidence the gate uses. Default 0.95. */
|
|
34
|
+
confidence?: number;
|
|
35
|
+
/** True when the holdout is scored by the SAME judge/scorer family as the gate
|
|
36
|
+
* (selfImprove's default composition — one judge scores everything). Under a
|
|
37
|
+
* shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
|
|
38
|
+
* systematic judge bias is untouched, so the MDE here is a lower bound and the
|
|
39
|
+
* only full debiaser is an independent second scoring channel
|
|
40
|
+
* (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
|
|
41
|
+
sharedScorerChannel?: boolean;
|
|
42
|
+
}
|
|
43
|
+
interface PowerPreflight {
|
|
44
|
+
/** Paired observations the comparison will have. */
|
|
45
|
+
n: number;
|
|
46
|
+
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
47
|
+
sd: number;
|
|
48
|
+
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
49
|
+
mde: number;
|
|
50
|
+
/** Baseline holdout composite mean. */
|
|
51
|
+
baselineMean: number;
|
|
52
|
+
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
53
|
+
headroom: number;
|
|
54
|
+
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
55
|
+
* the run is structurally unable to ship regardless of proposal quality.
|
|
56
|
+
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
57
|
+
underpowered: boolean;
|
|
58
|
+
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
59
|
+
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
60
|
+
scaleAssumed: boolean;
|
|
61
|
+
deltaThreshold: number;
|
|
62
|
+
confidence: number;
|
|
63
|
+
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
64
|
+
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
65
|
+
sharedChannelCaveat?: string;
|
|
66
|
+
/** One actionable sentence for humans and logs. */
|
|
67
|
+
recommendation: string;
|
|
68
|
+
}
|
|
69
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
70
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
71
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
72
|
+
* observable at this holdout size and worker variance. */
|
|
73
|
+
declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
|
|
74
|
+
//#endregion
|
|
75
|
+
//#region src/pareto.d.ts
|
|
76
|
+
/**
|
|
77
|
+
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
78
|
+
*
|
|
79
|
+
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
80
|
+
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
81
|
+
* ttfb), you rarely have a single "winner" — you have a set of
|
|
82
|
+
* non-dominated candidates. This module exposes:
|
|
83
|
+
*
|
|
84
|
+
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
85
|
+
* - `dominates`: does A dominate B across all objectives?
|
|
86
|
+
*
|
|
87
|
+
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
88
|
+
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
89
|
+
* `objective(candidate)` accessor.
|
|
90
|
+
*/
|
|
91
|
+
type Direction = 'maximize' | 'minimize';
|
|
92
|
+
interface Objective<T> {
|
|
93
|
+
/** Stable label used in reports. */
|
|
94
|
+
name: string;
|
|
95
|
+
direction: Direction;
|
|
96
|
+
value: (candidate: T) => number;
|
|
97
|
+
}
|
|
98
|
+
interface ParetoResult<T> {
|
|
99
|
+
frontier: T[];
|
|
100
|
+
dominated: T[];
|
|
101
|
+
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
102
|
+
dominanceMap: Array<{
|
|
103
|
+
dominator: T;
|
|
104
|
+
dominated: T[];
|
|
105
|
+
}>;
|
|
106
|
+
}
|
|
107
|
+
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
108
|
+
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
109
|
+
/**
|
|
110
|
+
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
111
|
+
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
112
|
+
* iff no other candidate dominates it.
|
|
113
|
+
*/
|
|
114
|
+
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
115
|
+
//#endregion
|
|
116
|
+
export { paretoFrontier as a, powerPreflight as c, dominates as i, Objective as n, PowerPreflight as o, ParetoResult as r, PowerPreflightOptions as s, Direction as t };
|
|
117
|
+
//# sourceMappingURL=pareto-BqNW3LJR.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pareto-BqNW3LJR.d.ts","names":[],"sources":["../src/campaign/gates/power-preflight.ts","../src/pareto.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;UAwBiB;;EAEf;;;EAGA;;EAEA;;EAEA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;;EAGA;EACA;EACA;;;EAGA;;EAEA;;;;;;iBAec,eAAe,MAAM,wBAAwB;;;;;;;;;;;;;;;;;;KClEjD;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa"}
|
|
@@ -1,33 +1,8 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { f as Run } from "../schema-BtVldJ3T.js";
|
|
2
2
|
import { a as RunFilter, s as TraceStore } from "../store-CT9YIIve.js";
|
|
3
|
-
import {
|
|
4
|
-
import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-CqcvCcdR.js";
|
|
5
|
-
import { n as BaselineReport, p as computeToolUseMetrics, t as BaselineOptions } from "../baseline-CavEbRyH.js";
|
|
3
|
+
import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-BLURuWG4.js";
|
|
6
4
|
import { n as TrajectoryStep } from "../trajectory-YC15QDYQ.js";
|
|
7
|
-
|
|
8
|
-
interface BudgetBreachFinding {
|
|
9
|
-
runId: string;
|
|
10
|
-
scenarioId: string;
|
|
11
|
-
variantId?: string;
|
|
12
|
-
dimension: keyof BudgetSpec;
|
|
13
|
-
limit: number;
|
|
14
|
-
consumed: number;
|
|
15
|
-
excessRatio: number;
|
|
16
|
-
timestamp: number;
|
|
17
|
-
}
|
|
18
|
-
interface BudgetBreachReport {
|
|
19
|
-
findings: BudgetBreachFinding[];
|
|
20
|
-
byDimension: Record<string, number>;
|
|
21
|
-
byScenario: Record<string, number>;
|
|
22
|
-
byVariant: Record<string, number>;
|
|
23
|
-
totalRuns: number;
|
|
24
|
-
breachedRunRatio: number;
|
|
25
|
-
}
|
|
26
|
-
declare function budgetBreachView(store: TraceStore, options?: {
|
|
27
|
-
scenarioId?: string;
|
|
28
|
-
variantId?: string;
|
|
29
|
-
}): Promise<BudgetBreachReport>;
|
|
30
|
-
//#endregion
|
|
5
|
+
import { c as computeToolUseMetrics, d as judgeAgreementView, f as BudgetBreachFinding, g as BaselineReport, h as BaselineOptions, i as toolWasteView, l as JudgeAgreementReport, m as budgetBreachView, n as ToolWasteOptions, p as BudgetBreachReport, r as ToolWasteReport, t as ToolWasteFinding, u as JudgePair } from "../tool-waste-DjRDEsuI.js";
|
|
31
6
|
//#region src/pipelines/first-divergence.d.ts
|
|
32
7
|
interface DivergenceReport {
|
|
33
8
|
runA: string;
|
|
@@ -45,23 +20,6 @@ interface DivergenceOptions {
|
|
|
45
20
|
}
|
|
46
21
|
declare function firstDivergenceView(store: TraceStore, runA: string, runB: string, options?: DivergenceOptions): Promise<DivergenceReport>;
|
|
47
22
|
//#endregion
|
|
48
|
-
//#region src/pipelines/judge-agreement.d.ts
|
|
49
|
-
interface JudgePair {
|
|
50
|
-
judgeA: string;
|
|
51
|
-
judgeB: string;
|
|
52
|
-
dimension: string;
|
|
53
|
-
/** Number of (targetSpanId, dimension) tuples both judges scored. */
|
|
54
|
-
commonItems: number;
|
|
55
|
-
pearson: number;
|
|
56
|
-
krippendorff: number;
|
|
57
|
-
}
|
|
58
|
-
interface JudgeAgreementReport {
|
|
59
|
-
pairs: JudgePair[];
|
|
60
|
-
dimensions: string[];
|
|
61
|
-
judgeIds: string[];
|
|
62
|
-
}
|
|
63
|
-
declare function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport>;
|
|
64
|
-
//#endregion
|
|
65
23
|
//#region src/pipelines/regression.d.ts
|
|
66
24
|
interface RegressionSpec {
|
|
67
25
|
metric: string;
|
|
@@ -108,24 +66,5 @@ interface StuckLoopOptions {
|
|
|
108
66
|
}
|
|
109
67
|
declare function stuckLoopView(store: TraceStore, options?: StuckLoopOptions): Promise<StuckLoopReport>;
|
|
110
68
|
//#endregion
|
|
111
|
-
//#region src/pipelines/tool-waste.d.ts
|
|
112
|
-
interface ToolWasteFinding {
|
|
113
|
-
runId: string;
|
|
114
|
-
wastedCalls: number;
|
|
115
|
-
totalCalls: number;
|
|
116
|
-
wasteRate: number;
|
|
117
|
-
}
|
|
118
|
-
interface ToolWasteReport {
|
|
119
|
-
byRun: ToolWasteFinding[];
|
|
120
|
-
overallWasteRate: number;
|
|
121
|
-
}
|
|
122
|
-
interface ToolWasteOptions {
|
|
123
|
-
runId?: string;
|
|
124
|
-
usageOracle?: (tool: ToolSpan, later: {
|
|
125
|
-
llm: Awaited<ReturnType<typeof llmSpans>>;
|
|
126
|
-
}) => boolean;
|
|
127
|
-
}
|
|
128
|
-
declare function toolWasteView(store: TraceStore, options?: ToolWasteOptions): Promise<ToolWasteReport>;
|
|
129
|
-
//#endregion
|
|
130
69
|
export { BudgetBreachFinding, BudgetBreachReport, DivergenceOptions, DivergenceReport, FailureCluster, FailureClusterReport, JudgeAgreementReport, JudgePair, RegressionOptions, RegressionSpec, StuckLoopFinding, StuckLoopOptions, StuckLoopReport, ToolWasteFinding, ToolWasteOptions, ToolWasteReport, budgetBreachView, computeToolUseMetrics, failureClusterView, firstDivergenceView, judgeAgreementView, regressionView, stuckLoopView, toolWasteView };
|
|
131
70
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/first-divergence.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts"],"mappings":";;;;;;UAYiB;EACf;EACA;EACA;EACA,QAAQ;EACR,QAAQ;EACR;;EAEA;;UAGe;;EAEf,cAAc,GAAG,gBAAgB,GAAG;;iBAGhB,oBACpB,OAAO,YACP,cACA,cACA,UAAS,oBACR,QAAQ;;;UCnBM;EACf;EACA;;EAEA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B,0BAA0B;EACzC,UAAU;EACV,WAAW;;iBAGS,eACpB,OAAO,YACP,SAAS,kBACT,SAAS,oBACR,QAAQ;;;UCDM;EACf;EACA;EACA;;EAEA;EACA;;EAEA;;EAEA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;;EAEf;;EAEA;;;;;EAKA;;EAEA;;iBAGoB,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ"}
|