@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
+
import { p as ChatClient } from "./types-Cx3YUh2r.js";
|
|
3
|
+
import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-DLQx4mKU.js";
|
|
4
|
+
import { i as ExactAnalystRunEvent, o as ExactAnalystRunResult, u as ExactExecutionComponentIdentity } from "./exact-types-BH1twmAJ.js";
|
|
5
|
+
//#region src/analyst/registry.d.ts
|
|
6
|
+
interface AnalystHooks {
|
|
7
|
+
/** Legacy runs may mutate ctx; exact runs provide a frozen observational context. */
|
|
8
|
+
onBeforeAnalyze?(args: {
|
|
9
|
+
analyst: Analyst;
|
|
10
|
+
ctx: AnalystContext;
|
|
11
|
+
runId: string;
|
|
12
|
+
}): void | Promise<void>;
|
|
13
|
+
/** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
|
|
14
|
+
onAfterAnalyze?(args: {
|
|
15
|
+
analyst: Analyst;
|
|
16
|
+
summary: AnalystRunSummary;
|
|
17
|
+
findings: AnalystFinding[];
|
|
18
|
+
runId: string;
|
|
19
|
+
}): void | Promise<void>;
|
|
20
|
+
/**
|
|
21
|
+
* On analyst exception. Hook MAY return findings to convert the
|
|
22
|
+
* error into structured findings; the summary still reports 'failed'.
|
|
23
|
+
* Return void to keep the default empty-findings behavior.
|
|
24
|
+
*/
|
|
25
|
+
onError?(args: {
|
|
26
|
+
analyst: Analyst;
|
|
27
|
+
error: Error;
|
|
28
|
+
runId: string;
|
|
29
|
+
}): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
|
|
30
|
+
/** Once after registry.run() completes. Use for final aggregation, persistence. */
|
|
31
|
+
onComplete?(args: {
|
|
32
|
+
result: AnalystRunResult;
|
|
33
|
+
}): void | Promise<void>;
|
|
34
|
+
}
|
|
35
|
+
interface BudgetPolicy {
|
|
36
|
+
/** Overall USD cap across the registry.run(). */
|
|
37
|
+
totalUsd?: number;
|
|
38
|
+
/** Per-analyst weight for the default allocator. Missing ids get weight 1. */
|
|
39
|
+
weights?: Record<string, number>;
|
|
40
|
+
/**
|
|
41
|
+
* Custom allocator — receives the analyst, remaining/total budget, and
|
|
42
|
+
* the count of analysts that will run. Returns the per-analyst budget
|
|
43
|
+
* (or undefined only when the run has no overall cap). Overrides weights
|
|
44
|
+
* when set.
|
|
45
|
+
*/
|
|
46
|
+
allocate?: (args: {
|
|
47
|
+
analyst: Analyst;
|
|
48
|
+
totalUsd: number | undefined;
|
|
49
|
+
remainingUsd: number | undefined;
|
|
50
|
+
runningCount: number;
|
|
51
|
+
}) => number | undefined;
|
|
52
|
+
}
|
|
53
|
+
interface AnalystRegistryOptions {
|
|
54
|
+
/** Shared chat client passed to every LLM analyst via AnalystContext. */
|
|
55
|
+
chat?: ChatClient;
|
|
56
|
+
/** Logger callback. Defaults to a no-op. */
|
|
57
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
58
|
+
/** Hooks invoked around analyze() — observability + customization seam. */
|
|
59
|
+
hooks?: AnalystHooks;
|
|
60
|
+
/** Required identity/config when `runExact` applies `hooks`. */
|
|
61
|
+
hooksIdentity?: ExactExecutionComponentIdentity;
|
|
62
|
+
/** Default budget when run() doesn't override. */
|
|
63
|
+
defaultBudget?: BudgetPolicy;
|
|
64
|
+
/** Required identity/config when `runExact` provides `chat`. */
|
|
65
|
+
chatIdentity?: ExactExecutionComponentIdentity;
|
|
66
|
+
}
|
|
67
|
+
interface RegistryRunOpts {
|
|
68
|
+
/** Restrict to a subset of registered analysts by id. */
|
|
69
|
+
only?: string[];
|
|
70
|
+
/** Skip these analysts even if registered. Useful for cheap iteration. */
|
|
71
|
+
skip?: string[];
|
|
72
|
+
/** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
|
|
73
|
+
budget?: BudgetPolicy;
|
|
74
|
+
/** Active-work cap for the complete registry run. Model receipt settlement may follow. */
|
|
75
|
+
timeoutMs?: number;
|
|
76
|
+
/** Abort signal — forwarded into every analyst's context. */
|
|
77
|
+
signal?: AbortSignal;
|
|
78
|
+
/** Shared paid-call account forwarded to every analyst. */
|
|
79
|
+
costLedger?: CostLedgerHandle;
|
|
80
|
+
/** Attribution phase for calls written to `costLedger`. */
|
|
81
|
+
costPhase?: string;
|
|
82
|
+
/** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
|
|
83
|
+
tags?: Record<string, string>;
|
|
84
|
+
/**
|
|
85
|
+
* Prior-run findings made available as retrieval context to every
|
|
86
|
+
* analyst via `ctx.priorFindings`. The registry forwards the slice
|
|
87
|
+
* whose `analyst_id` matches each registered analyst so a kind sees
|
|
88
|
+
* only its own history. Pass `{ '*': findings }` to broadcast to
|
|
89
|
+
* every analyst (useful when several kinds share the same historical
|
|
90
|
+
* context). For findings from this run, use `chainFindings` instead.
|
|
91
|
+
*/
|
|
92
|
+
priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
|
|
93
|
+
/**
|
|
94
|
+
* Pass findings produced earlier in this registry run to each later analyst
|
|
95
|
+
* via `ctx.upstreamFindings`. Registration order is dependency order.
|
|
96
|
+
* Disabled by default because independent analyst suites must opt in.
|
|
97
|
+
*/
|
|
98
|
+
chainFindings?: boolean;
|
|
99
|
+
}
|
|
100
|
+
/** A caller-selected allocation rule for an exact analyst run. */
|
|
101
|
+
type ExactAnalystBudgetPolicy = {
|
|
102
|
+
readonly kind: 'equal';
|
|
103
|
+
readonly totalUsd: number;
|
|
104
|
+
} | {
|
|
105
|
+
readonly kind: 'weighted';
|
|
106
|
+
readonly totalUsd: number;
|
|
107
|
+
/** One explicit weight for every selected analyst id. */
|
|
108
|
+
readonly weights: Readonly<Record<string, number>>;
|
|
109
|
+
};
|
|
110
|
+
/**
|
|
111
|
+
* Complete per-run analyst policy.
|
|
112
|
+
*
|
|
113
|
+
* Every field is required. `null` explicitly disables an optional resource or context channel.
|
|
114
|
+
* `analystIds` is execution order; it is not filtered through registry insertion order.
|
|
115
|
+
* Missing-input behavior is explicit so callers cannot accidentally inherit it by omission.
|
|
116
|
+
* Exact runs are serial; more elaborate scheduling belongs in the caller's runtime.
|
|
117
|
+
*/
|
|
118
|
+
interface ExactRegistryRunOpts {
|
|
119
|
+
readonly analystIds: readonly string[];
|
|
120
|
+
readonly budget: ExactAnalystBudgetPolicy | null;
|
|
121
|
+
readonly totalTimeoutMs: number | null;
|
|
122
|
+
readonly signal: AbortSignal | null;
|
|
123
|
+
readonly costLedger: CostLedgerHandle | null;
|
|
124
|
+
readonly costLedgerIdentity: ExactExecutionComponentIdentity | null;
|
|
125
|
+
readonly costPhase: string | null;
|
|
126
|
+
readonly tags: Readonly<Record<string, string>> | null;
|
|
127
|
+
readonly priorFindings: ReadonlyArray<AnalystFinding> | Readonly<Record<string, ReadonlyArray<AnalystFinding>>> | null;
|
|
128
|
+
readonly chainFindings: boolean;
|
|
129
|
+
readonly missingInputMode: 'skip' | 'abort';
|
|
130
|
+
readonly applyRegistryHooks: boolean;
|
|
131
|
+
readonly useRegistryChat: boolean;
|
|
132
|
+
}
|
|
133
|
+
/** A post-start exact-run failure; completed work remains attached for accounting and review. */
|
|
134
|
+
declare class ExactAnalystRunExecutionError extends Error {
|
|
135
|
+
readonly name: string;
|
|
136
|
+
readonly result: ExactAnalystRunResult;
|
|
137
|
+
constructor(message: string, result: ExactAnalystRunResult, options?: ErrorOptions);
|
|
138
|
+
}
|
|
139
|
+
declare class AnalystRegistry {
|
|
140
|
+
private readonly analysts;
|
|
141
|
+
private readonly options;
|
|
142
|
+
constructor(options?: AnalystRegistryOptions);
|
|
143
|
+
register(analyst: Analyst): void;
|
|
144
|
+
list(): ReadonlyArray<{
|
|
145
|
+
id: string;
|
|
146
|
+
description: string;
|
|
147
|
+
version: string;
|
|
148
|
+
cost: Analyst['cost'];
|
|
149
|
+
}>;
|
|
150
|
+
run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
|
|
151
|
+
/** Run exactly the ordered analysts and complete policy supplied by the caller. */
|
|
152
|
+
runExact(runId: string, inputs: AnalystRunInputs, runOpts: ExactRegistryRunOpts): Promise<ExactAnalystRunResult>;
|
|
153
|
+
/** Streaming counterpart to {@link runExact}. */
|
|
154
|
+
runExactStream(runId: string, inputs: AnalystRunInputs, runOpts: ExactRegistryRunOpts): AsyncGenerator<ExactAnalystRunEvent, void, void>;
|
|
155
|
+
/**
|
|
156
|
+
* Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
|
|
157
|
+
* in real time — `run-started`, then per-analyst `skipped` /
|
|
158
|
+
* `started` / `completed`, then a terminal `run-completed` whose
|
|
159
|
+
* payload is the full `AnalystRunResult`. UIs use this to render
|
|
160
|
+
* progress; persistence consumers use `run()` and read the result.
|
|
161
|
+
*
|
|
162
|
+
* Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
|
|
163
|
+
* `onComplete`) fire as before — streaming is additive, not a hook
|
|
164
|
+
* replacement.
|
|
165
|
+
*/
|
|
166
|
+
runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
|
|
167
|
+
private normalizeLegacyPlan;
|
|
168
|
+
private normalizeExactPlan;
|
|
169
|
+
private executePlanStream;
|
|
170
|
+
private selectAnalysts;
|
|
171
|
+
private selectExactAnalysts;
|
|
172
|
+
private routeInput;
|
|
173
|
+
}
|
|
174
|
+
/** Validate the canonical exact-run policy before any analyst can start. */
|
|
175
|
+
declare function assertExactRegistryRunOpts(value: unknown): asserts value is ExactRegistryRunOpts;
|
|
176
|
+
//#endregion
|
|
177
|
+
export { ExactAnalystBudgetPolicy as a, RegistryRunOpts as c, BudgetPolicy as i, assertExactRegistryRunOpts as l, AnalystRegistry as n, ExactAnalystRunExecutionError as o, AnalystRegistryOptions as r, ExactRegistryRunOpts as s, AnalystHooks as t };
|
|
178
|
+
//# sourceMappingURL=registry-oJeeI4-a.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"registry-oJeeI4-a.d.ts","names":[],"sources":["../src/analyst/registry.ts"],"mappings":";;;;;UAqDiB;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;EAEhB,gBAAgB;;EAEhB,eAAe;;UAGA;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;;KAIU;WAEG;WACA;;WAGA;WACA;;WAEA,SAAS,SAAS;;;;;;;;;;UAWhB;WACN;WACA,QAAQ;WACR;WACA,QAAQ;WACR,YAAY;WACZ,oBAAoB;WACpB;WACA,MAAM,SAAS;WACf,eACL,cAAc,kBACd,SAAS,eAAe,cAAc;WAEjC;WACA;WACA;WACA;;;cAqCE,sCAAsC;WACxC;WACA,QAAQ;EAEjB,YAAY,iBAAiB,QAAQ,uBAAuB,UAAU;;cAU3D;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAsBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;EAUL,SACJ,eACA,QAAQ,kBACR,SAAS,uBACR,QAAQ;;EAQJ,eACL,eACA,QAAQ,kBACR,SAAS,uBACR,eAAe;;;;;;;;;;;;EAmBX,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UAIV;UA8BA;UAwFO;UA2aP;UAaA;UAUA;;;iBAiGM,2BAA2B,yBAAyB,SAAS"}
|
|
@@ -1,7 +1,182 @@
|
|
|
1
1
|
import { o as FailureClass } from "./schema-BtVldJ3T.js";
|
|
2
|
-
import { a as RunRecord, s as RunSplitTag } from "./run-record-
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
2
|
+
import { a as RunRecord, s as RunSplitTag } from "./run-record-CKiihE6f.js";
|
|
3
|
+
import { i as DatasetSplit, n as DatasetManifest, r as DatasetScenario } from "./dataset-CJjKqQfA.js";
|
|
4
|
+
import { b as GateDecision } from "./summary-report-CaL-Hnxt.js";
|
|
5
|
+
//#region src/statistics/rank-tests.d.ts
|
|
6
|
+
/** How a rank test's p-value was actually computed. */
|
|
7
|
+
type RankTestMethod = 'exact' | 'permutation' | 'asymptotic';
|
|
8
|
+
/**
|
|
9
|
+
* What the caller asks for. `'auto'` selects `'exact'` inside the enumeration
|
|
10
|
+
* threshold and `'permutation'` above it, and never selects `'asymptotic'`.
|
|
11
|
+
*/
|
|
12
|
+
type RankTestMethodRequest = 'auto' | 'exact' | 'asymptotic';
|
|
13
|
+
interface RankTestOptions {
|
|
14
|
+
/** Default `'auto'`. `'asymptotic'` inside the exact-feasible range throws. */
|
|
15
|
+
method?: RankTestMethodRequest;
|
|
16
|
+
/** Resamples on the Monte Carlo permutation path. Default 100000. */
|
|
17
|
+
permutations?: number;
|
|
18
|
+
/** Seed for the permutation path. Omitted ⇒ derived from the data itself, so
|
|
19
|
+
* the result is reproducible either way. */
|
|
20
|
+
seed?: number;
|
|
21
|
+
}
|
|
22
|
+
/** Maximum dynamic-programming cells used by an exact two-sample rank test. */
|
|
23
|
+
declare const MANN_WHITNEY_EXACT_MAX_STATES = 8192;
|
|
24
|
+
/** Maximum inner-loop transitions used by an exact two-sample rank test. */
|
|
25
|
+
declare const MANN_WHITNEY_EXACT_MAX_WORK = 250000;
|
|
26
|
+
/** Non-zero differences up to which the signed-rank null is enumerated exactly. */
|
|
27
|
+
declare const WILCOXON_EXACT_MAX_N = 20;
|
|
28
|
+
/** Resamples used when a rank test falls back to Monte Carlo permutation. */
|
|
29
|
+
declare const DEFAULT_PERMUTATIONS = 100000;
|
|
30
|
+
interface MannWhitneyResult {
|
|
31
|
+
/** `min(U_a, U_b)` — the conventional reported statistic. */
|
|
32
|
+
u: number;
|
|
33
|
+
/** U for sample `a`. Carries the direction of the effect, which `u` discards. */
|
|
34
|
+
uA: number;
|
|
35
|
+
/** Two-sided p-value. */
|
|
36
|
+
p: number;
|
|
37
|
+
/** How `p` was computed. */
|
|
38
|
+
method: RankTestMethod;
|
|
39
|
+
/** Smallest two-sided p this design can produce. `p` can never be below it. */
|
|
40
|
+
pFloor: number;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Mann-Whitney U — two independent samples, no distributional assumption.
|
|
44
|
+
*
|
|
45
|
+
* Exact conditional (permutation) p by default when the dynamic program fits
|
|
46
|
+
* {@link MANN_WHITNEY_EXACT_MAX_STATES} cells and
|
|
47
|
+
* {@link MANN_WHITNEY_EXACT_MAX_WORK} transitions, seeded Monte Carlo
|
|
48
|
+
* permutation above those limits. This keeps imbalanced designs such as 1+24
|
|
49
|
+
* exact without admitting expensive balanced designs merely because they have
|
|
50
|
+
* the same total size. Throws on non-finite input and on `method:
|
|
51
|
+
* 'asymptotic'` where an exact answer is available. Empty input yields `p = 1,
|
|
52
|
+
* pFloor = 1` — no design, no attainable evidence.
|
|
53
|
+
*/
|
|
54
|
+
declare function mannWhitneyU(a: number[], b: number[], opts?: RankTestOptions): MannWhitneyResult;
|
|
55
|
+
interface WilcoxonSignedRankResult {
|
|
56
|
+
/** W⁺, the rank sum of the positive differences. (scipy reports
|
|
57
|
+
* `min(W⁺, W⁻)`; compare statistics only after converting.) */
|
|
58
|
+
w: number;
|
|
59
|
+
/** Two-sided p-value. */
|
|
60
|
+
p: number;
|
|
61
|
+
/** How `p` was computed. */
|
|
62
|
+
method: RankTestMethod;
|
|
63
|
+
/** Smallest two-sided p this design can produce. */
|
|
64
|
+
pFloor: number;
|
|
65
|
+
/** Non-zero differences — zero differences are dropped and carry no rank. */
|
|
66
|
+
nNonZero: number;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Wilcoxon signed-rank — paired, no distributional assumption on the deltas.
|
|
70
|
+
*
|
|
71
|
+
* Exact conditional (sign-flip) p by default at `n ≤
|
|
72
|
+
* {@link WILCOXON_EXACT_MAX_N}` non-zero differences, seeded Monte Carlo
|
|
73
|
+
* permutation above it. Throws on non-finite input and on `method:
|
|
74
|
+
* 'asymptotic'` where an exact answer is available.
|
|
75
|
+
*
|
|
76
|
+
* `n` is the count of NON-ZERO differences: exact ties are dropped before
|
|
77
|
+
* ranking, so a run of tied pairs shrinks the design and raises `pFloor`.
|
|
78
|
+
* All-tied input yields `p = 1, pFloor = 1` — no attainable evidence, which
|
|
79
|
+
* `pFloor` states rather than leaving `p = 1` to be read as a measured null.
|
|
80
|
+
*/
|
|
81
|
+
declare function wilcoxonSignedRank(before: number[], after: number[], opts?: RankTestOptions): WilcoxonSignedRankResult;
|
|
82
|
+
//#endregion
|
|
83
|
+
//#region src/promotion-gate.d.ts
|
|
84
|
+
/**
|
|
85
|
+
* Bootstrap-CI promotion gate.
|
|
86
|
+
*
|
|
87
|
+
* In any iterative-improvement loop (GEPA, prompt evolution, dataset
|
|
88
|
+
* curation), the question is "did this generation actually improve, or are
|
|
89
|
+
* we celebrating noise?". With small N and noisy outcomes, point-estimate
|
|
90
|
+
* deltas lie. Bootstrap confidence intervals tell the operator whether the
|
|
91
|
+
* delta is real before code or prompts get promoted.
|
|
92
|
+
*
|
|
93
|
+
* This module is pure functions — no I/O, no model calls. Easy to unit-test
|
|
94
|
+
* and to compose into any verdict gate.
|
|
95
|
+
*
|
|
96
|
+
* Default gate:
|
|
97
|
+
* - Bootstrap mean baseline vs candidate (1k resamples).
|
|
98
|
+
* - Compute the delta distribution; pass if the lower CI bound > 0.
|
|
99
|
+
* - Tunable confidence (default 95%) and resample count.
|
|
100
|
+
*
|
|
101
|
+
* Verdict semantics intentionally match the existing `experiments.jsonl`
|
|
102
|
+
* vocabulary:
|
|
103
|
+
* - ADVANCE: candidate's CI lower bound > baseline mean (real win)
|
|
104
|
+
* - KEEP: overlap, but candidate point estimate >= baseline (neutral)
|
|
105
|
+
* - REVERT: candidate's CI upper bound < baseline mean (real regression)
|
|
106
|
+
* - INCONCLUSIVE: not enough samples or CI straddles zero with no signal
|
|
107
|
+
*/
|
|
108
|
+
type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE';
|
|
109
|
+
interface BootstrapResult {
|
|
110
|
+
baselineMean: number;
|
|
111
|
+
candidateMean: number;
|
|
112
|
+
/** candidateMean - baselineMean, point estimate. */
|
|
113
|
+
delta: number;
|
|
114
|
+
/** Lower bound of the (1 - alpha) CI on the delta. */
|
|
115
|
+
ciLower: number;
|
|
116
|
+
/** Upper bound of the (1 - alpha) CI on the delta. */
|
|
117
|
+
ciUpper: number;
|
|
118
|
+
/** Number of bootstrap resamples used. */
|
|
119
|
+
iterations: number;
|
|
120
|
+
alpha: number;
|
|
121
|
+
verdict: Verdict;
|
|
122
|
+
}
|
|
123
|
+
interface BootstrapOptions {
|
|
124
|
+
/** Confidence level alpha (default 0.05 → 95% CI). */
|
|
125
|
+
alpha?: number;
|
|
126
|
+
/** Number of resamples (default 1000). */
|
|
127
|
+
iterations?: number;
|
|
128
|
+
/**
|
|
129
|
+
* Minimum total samples (baseline + candidate) below which we always
|
|
130
|
+
* return INCONCLUSIVE — bootstrap with too few samples is meaningless.
|
|
131
|
+
* Default 6 (combined).
|
|
132
|
+
*/
|
|
133
|
+
minTotalSamples?: number;
|
|
134
|
+
/** RNG seed for reproducibility. Default: Math.random. */
|
|
135
|
+
seed?: number;
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.
|
|
139
|
+
*
|
|
140
|
+
* Uses simple percentile bootstrap on the difference of resampled means.
|
|
141
|
+
* That's the standard non-parametric primitive — no distributional
|
|
142
|
+
* assumptions, robust to skew, easy to reason about.
|
|
143
|
+
*/
|
|
144
|
+
declare function bootstrapCi(baseline: number[], candidate: number[], options?: BootstrapOptions): BootstrapResult;
|
|
145
|
+
/**
|
|
146
|
+
* Judge-replay promotion gate.
|
|
147
|
+
*
|
|
148
|
+
* The cheap inner-loop judge that drives an evolution run is by definition
|
|
149
|
+
* fast and noisy. When you're about to promote a winning variant to the
|
|
150
|
+
* canonical default, you want a STRONGER judge (a more expensive model, a
|
|
151
|
+
* human grader, a separately-trained reward model) to confirm the win
|
|
152
|
+
* generalises beyond the inner loop.
|
|
153
|
+
*
|
|
154
|
+
* This helper takes raw winner + baseline outputs, scores both through the
|
|
155
|
+
* stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger
|
|
156
|
+
* judge agrees the winner is real with the configured confidence. Doesn't
|
|
157
|
+
* matter what shape your "output" is — pass a string, an object, anything
|
|
158
|
+
* the judge can read.
|
|
159
|
+
*/
|
|
160
|
+
interface JudgeReplayGateArgs<TOutput> {
|
|
161
|
+
baselineOutputs: TOutput[];
|
|
162
|
+
candidateOutputs: TOutput[];
|
|
163
|
+
/** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */
|
|
164
|
+
judge: (output: TOutput) => Promise<number> | number;
|
|
165
|
+
alpha?: number;
|
|
166
|
+
iterations?: number;
|
|
167
|
+
/** RNG seed for reproducibility. */
|
|
168
|
+
seed?: number;
|
|
169
|
+
/** Maximum concurrent judge calls. Default 4. */
|
|
170
|
+
judgeConcurrency?: number;
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
|
|
174
|
+
*/
|
|
175
|
+
declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
|
|
176
|
+
baselineSamples: number;
|
|
177
|
+
candidateSamples: number;
|
|
178
|
+
}>;
|
|
179
|
+
//#endregion
|
|
5
180
|
//#region src/release-confidence.d.ts
|
|
6
181
|
/** Severity of an actionable finding attached to a run/trace. */
|
|
7
182
|
type AsiSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
@@ -133,112 +308,5 @@ interface ReleaseConfidenceScorecard {
|
|
|
133
308
|
declare function evaluateReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
|
|
134
309
|
declare function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
|
|
135
310
|
//#endregion
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
* Bootstrap-CI promotion gate.
|
|
139
|
-
*
|
|
140
|
-
* In any iterative-improvement loop (GEPA, prompt evolution, dataset
|
|
141
|
-
* curation), the question is "did this generation actually improve, or are
|
|
142
|
-
* we celebrating noise?". With small N and noisy outcomes, point-estimate
|
|
143
|
-
* deltas lie. Bootstrap confidence intervals tell the operator whether the
|
|
144
|
-
* delta is real before code or prompts get promoted.
|
|
145
|
-
*
|
|
146
|
-
* This module is pure functions — no I/O, no model calls. Easy to unit-test
|
|
147
|
-
* and to compose into any verdict gate.
|
|
148
|
-
*
|
|
149
|
-
* Default gate:
|
|
150
|
-
* - Bootstrap mean baseline vs candidate (1k resamples).
|
|
151
|
-
* - Compute the delta distribution; pass if the lower CI bound > 0.
|
|
152
|
-
* - Tunable confidence (default 95%) and resample count.
|
|
153
|
-
*
|
|
154
|
-
* Verdict semantics intentionally match the existing `experiments.jsonl`
|
|
155
|
-
* vocabulary:
|
|
156
|
-
* - ADVANCE: candidate's CI lower bound > baseline mean (real win)
|
|
157
|
-
* - KEEP: overlap, but candidate point estimate >= baseline (neutral)
|
|
158
|
-
* - REVERT: candidate's CI upper bound < baseline mean (real regression)
|
|
159
|
-
* - INCONCLUSIVE: not enough samples or CI straddles zero with no signal
|
|
160
|
-
*/
|
|
161
|
-
type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE';
|
|
162
|
-
interface BootstrapResult {
|
|
163
|
-
baselineMean: number;
|
|
164
|
-
candidateMean: number;
|
|
165
|
-
/** candidateMean - baselineMean, point estimate. */
|
|
166
|
-
delta: number;
|
|
167
|
-
/** Lower bound of the (1 - alpha) CI on the delta. */
|
|
168
|
-
ciLower: number;
|
|
169
|
-
/** Upper bound of the (1 - alpha) CI on the delta. */
|
|
170
|
-
ciUpper: number;
|
|
171
|
-
/** Number of bootstrap resamples used. */
|
|
172
|
-
iterations: number;
|
|
173
|
-
alpha: number;
|
|
174
|
-
verdict: Verdict;
|
|
175
|
-
}
|
|
176
|
-
interface BootstrapOptions {
|
|
177
|
-
/** Confidence level alpha (default 0.05 → 95% CI). */
|
|
178
|
-
alpha?: number;
|
|
179
|
-
/** Number of resamples (default 1000). */
|
|
180
|
-
iterations?: number;
|
|
181
|
-
/**
|
|
182
|
-
* Minimum total samples (baseline + candidate) below which we always
|
|
183
|
-
* return INCONCLUSIVE — bootstrap with too few samples is meaningless.
|
|
184
|
-
* Default 6 (combined).
|
|
185
|
-
*/
|
|
186
|
-
minTotalSamples?: number;
|
|
187
|
-
/** RNG seed for reproducibility. Default: Math.random. */
|
|
188
|
-
seed?: number;
|
|
189
|
-
}
|
|
190
|
-
/**
|
|
191
|
-
* Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.
|
|
192
|
-
*
|
|
193
|
-
* Uses simple percentile bootstrap on the difference of resampled means.
|
|
194
|
-
* That's the standard non-parametric primitive — no distributional
|
|
195
|
-
* assumptions, robust to skew, easy to reason about.
|
|
196
|
-
*/
|
|
197
|
-
declare function bootstrapCi(baseline: number[], candidate: number[], options?: BootstrapOptions): BootstrapResult;
|
|
198
|
-
/**
|
|
199
|
-
* Judge-replay promotion gate.
|
|
200
|
-
*
|
|
201
|
-
* The cheap inner-loop judge that drives an evolution run is by definition
|
|
202
|
-
* fast and noisy. When you're about to promote a winning variant to the
|
|
203
|
-
* canonical default, you want a STRONGER judge (a more expensive model, a
|
|
204
|
-
* human grader, a separately-trained reward model) to confirm the win
|
|
205
|
-
* generalises beyond the inner loop.
|
|
206
|
-
*
|
|
207
|
-
* This helper takes raw winner + baseline outputs, scores both through the
|
|
208
|
-
* stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger
|
|
209
|
-
* judge agrees the winner is real with the configured confidence. Doesn't
|
|
210
|
-
* matter what shape your "output" is — pass a string, an object, anything
|
|
211
|
-
* the judge can read.
|
|
212
|
-
*/
|
|
213
|
-
interface JudgeReplayGateArgs<TOutput> {
|
|
214
|
-
baselineOutputs: TOutput[];
|
|
215
|
-
candidateOutputs: TOutput[];
|
|
216
|
-
/** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */
|
|
217
|
-
judge: (output: TOutput) => Promise<number> | number;
|
|
218
|
-
alpha?: number;
|
|
219
|
-
iterations?: number;
|
|
220
|
-
/** RNG seed for reproducibility. */
|
|
221
|
-
seed?: number;
|
|
222
|
-
/** Maximum concurrent judge calls. Default 4. */
|
|
223
|
-
judgeConcurrency?: number;
|
|
224
|
-
}
|
|
225
|
-
/**
|
|
226
|
-
* Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
|
|
227
|
-
*/
|
|
228
|
-
declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
|
|
229
|
-
baselineSamples: number;
|
|
230
|
-
candidateSamples: number;
|
|
231
|
-
}>;
|
|
232
|
-
//#endregion
|
|
233
|
-
//#region src/release-report.d.ts
|
|
234
|
-
interface RenderReleaseReportOptions {
|
|
235
|
-
title?: string;
|
|
236
|
-
runs?: readonly RunRecord[];
|
|
237
|
-
comparator?: string;
|
|
238
|
-
traceAnalystFindings?: readonly string[];
|
|
239
|
-
nextActions?: readonly string[];
|
|
240
|
-
}
|
|
241
|
-
declare function renderReleaseReport(scorecard: ReleaseConfidenceScorecard, options?: RenderReleaseReportOptions): string;
|
|
242
|
-
//#endregion
|
|
243
|
-
export { ReleaseConfidenceStatus as _, JudgeReplayGateArgs as a, assertReleaseConfidence as b, judgeReplayGate as c, ReleaseConfidenceAxis as d, ReleaseConfidenceAxisName as f, ReleaseConfidenceScorecard as g, ReleaseConfidenceMetrics as h, BootstrapResult as i, ActionableSideInfo as l, ReleaseConfidenceIssue as m, renderReleaseReport as n, Verdict as o, ReleaseConfidenceInput as p, BootstrapOptions as r, bootstrapCi as s, RenderReleaseReportOptions as t, AsiSeverity as u, ReleaseConfidenceThresholds as v, evaluateReleaseConfidence as x, ReleaseTraceEvidence as y };
|
|
244
|
-
//# sourceMappingURL=release-report-DA2BCu5a.d.ts.map
|
|
311
|
+
export { RankTestMethod as C, WilcoxonSignedRankResult as D, WILCOXON_EXACT_MAX_N as E, mannWhitneyU as O, MannWhitneyResult as S, RankTestOptions as T, bootstrapCi as _, ReleaseConfidenceIssue as a, MANN_WHITNEY_EXACT_MAX_STATES as b, ReleaseConfidenceStatus as c, assertReleaseConfidence as d, evaluateReleaseConfidence as f, Verdict as g, JudgeReplayGateArgs as h, ReleaseConfidenceInput as i, wilcoxonSignedRank as k, ReleaseConfidenceThresholds as l, BootstrapResult as m, ReleaseConfidenceAxis as n, ReleaseConfidenceMetrics as o, BootstrapOptions as p, ReleaseConfidenceAxisName as r, ReleaseConfidenceScorecard as s, ActionableSideInfo as t, ReleaseTraceEvidence as u, judgeReplayGate as v, RankTestMethodRequest as w, MANN_WHITNEY_EXACT_MAX_WORK as x, DEFAULT_PERMUTATIONS as y };
|
|
312
|
+
//# sourceMappingURL=release-confidence-BFRE5WSp.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"release-confidence-BFRE5WSp.d.ts","names":[],"sources":["../src/statistics/rank-tests.ts","../src/promotion-gate.ts","../src/release-confidence.ts"],"mappings":";;;;;;KA4BY;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;UAuFc;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;;;;;;;;;;;;;;;KCjLS;UAEK;EACf;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA,SAAS;;UAGM;;EAEf;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;iBAUc,YACd,oBACA,qBACA,UAAS,mBACR;;;;;;;;;;;;;;;;UAsHc,oBAAoB;EACnC,iBAAiB;EACjB,kBAAkB;;EAElB,QAAQ,QAAQ,YAAY;EAC5B;EACA;;EAEA;;EAEA;;;;;iBAMoB,gBAAgB,SACpC,MAAM,oBAAoB,WACzB,QAAQ;EAAoB;EAAyB;;;;;KCtL5C;;UAGK;;EAEf;;EAEA;EACA,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;EACA,WAAW;;KAGD;KACA;UAQK;EACf;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;EAEA,eAAe;EACf,MAAM;EACN,WAAW;;UAGI;;EAEf;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,UAAU;EACV,qBAAqB;EACrB,gBAAgB;EAChB,kBAAkB;EAClB,eAAe;EACf,aAAa;;UAGE;EACf,MAAM;EACN,QAAQ;EACR;EACA;;UAGe;EACf,MAAM;EACN;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,aAAa,OAAO;EACpB,cAAc;EACd,oBAAoB,QAAQ,OAAO;EACnC,0BAA0B;;;;;;;EAO1B;;UAGe;EACf;EACA;EACA;EACA,QAAQ;EACR;EACA,MAAM;EACN,QAAQ;EACR,SAAS;EACT,SAAS;EACT,cAAc;EACd;;iBAkBc,0BACd,OAAO,yBACN;iBA4Ha,wBAAwB,OAAO,yBAAyB"}
|