@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1,755 +0,0 @@
|
|
|
1
|
-
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { i as fsCampaignStorage } from "./single-run-lock-DFWHEB09.js";
|
|
3
|
-
import { U as runCampaign } from "./skillopt-optimization-method-CQdVeM8k.js";
|
|
4
|
-
import "./campaign-C2TTzQII.js";
|
|
5
|
-
import { join } from "node:path";
|
|
6
|
-
//#region src/benchmarks/calibration.ts
|
|
7
|
-
async function calibrateBenchmarkMetric(options) {
|
|
8
|
-
const weak = await options.adapter.evaluate(options.item, options.weakArtifact);
|
|
9
|
-
const strong = await options.adapter.evaluate(options.item, options.strongArtifact);
|
|
10
|
-
const weakScore = clamp01$1(weak.score);
|
|
11
|
-
const strongScore = clamp01$1(strong.score);
|
|
12
|
-
const maxWeakScore = options.maxWeakScore ?? .3;
|
|
13
|
-
const minStrongScore = options.minStrongScore ?? .7;
|
|
14
|
-
const minGap = options.minGap ?? .4;
|
|
15
|
-
const gap = strongScore - weakScore;
|
|
16
|
-
const reasons = [
|
|
17
|
-
weakScore <= maxWeakScore ? void 0 : `weak score ${weakScore.toFixed(3)} exceeds max ${maxWeakScore.toFixed(3)}`,
|
|
18
|
-
strongScore >= minStrongScore ? void 0 : `strong score ${strongScore.toFixed(3)} below min ${minStrongScore.toFixed(3)}`,
|
|
19
|
-
gap >= minGap ? void 0 : `gap ${gap.toFixed(3)} below min ${minGap.toFixed(3)}`
|
|
20
|
-
].filter((reason) => Boolean(reason));
|
|
21
|
-
return {
|
|
22
|
-
passed: reasons.length === 0,
|
|
23
|
-
weak,
|
|
24
|
-
strong,
|
|
25
|
-
weakScore,
|
|
26
|
-
strongScore,
|
|
27
|
-
gap,
|
|
28
|
-
reasons
|
|
29
|
-
};
|
|
30
|
-
}
|
|
31
|
-
function clamp01$1(value) {
|
|
32
|
-
if (!Number.isFinite(value)) return 0;
|
|
33
|
-
if (value < 0) return 0;
|
|
34
|
-
if (value > 1) return 1;
|
|
35
|
-
return value;
|
|
36
|
-
}
|
|
37
|
-
//#endregion
|
|
38
|
-
//#region src/benchmarks/types.ts
|
|
39
|
-
/**
|
|
40
|
-
* 32-bit FNV-1a hash. Stable, allocation-free, deterministic across
|
|
41
|
-
* runtimes. We use it to assign items to splits rather than depending
|
|
42
|
-
* on a polyfilled crypto.subtle path.
|
|
43
|
-
*/
|
|
44
|
-
function fnv1a32(input) {
|
|
45
|
-
let h = 2166136261;
|
|
46
|
-
for (let i = 0; i < input.length; i++) {
|
|
47
|
-
h ^= input.charCodeAt(i) & 255;
|
|
48
|
-
h = h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24)) >>> 0;
|
|
49
|
-
}
|
|
50
|
-
return h >>> 0;
|
|
51
|
-
}
|
|
52
|
-
/** Split-assignment seed shared across all benchmarks. Bumping this
|
|
53
|
-
* value reshuffles every split — do NOT do that lightly. */
|
|
54
|
-
const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
|
|
55
|
-
/**
|
|
56
|
-
* Assign an item id to one of `'search' | 'dev' | 'holdout'` using a
|
|
57
|
-
* stable 32-bit hash of `${seed}::${id}`. Default proportions:
|
|
58
|
-
*
|
|
59
|
-
* search: 60% (optimization-readable)
|
|
60
|
-
* dev: 20% (held-out for tuning, leak-on-purpose during dev)
|
|
61
|
-
* holdout:20% (paper-grade held-out, gated reads)
|
|
62
|
-
*/
|
|
63
|
-
function deterministicSplit(itemId, seed = BENCHMARK_SPLIT_SEED) {
|
|
64
|
-
const pos = fnv1a32(`${seed}::${itemId}`) / 4294967296;
|
|
65
|
-
if (pos < .6) return "search";
|
|
66
|
-
if (pos < .8) return "dev";
|
|
67
|
-
return "holdout";
|
|
68
|
-
}
|
|
69
|
-
//#endregion
|
|
70
|
-
//#region src/benchmarks/routing/dataset.ts
|
|
71
|
-
const ROUTING_DATASET = [
|
|
72
|
-
{
|
|
73
|
-
id: "file_001",
|
|
74
|
-
category: "file",
|
|
75
|
-
prompt: "Save the meeting notes to /tmp/notes-2025-04.md as markdown.",
|
|
76
|
-
route: "fs.write",
|
|
77
|
-
synonyms: ["filesystem.write", "write_file"],
|
|
78
|
-
hardNegatives: ["fs.read", "chat.reply"]
|
|
79
|
-
},
|
|
80
|
-
{
|
|
81
|
-
id: "file_002",
|
|
82
|
-
category: "file",
|
|
83
|
-
prompt: "Read the contents of /etc/hosts and summarize the entries.",
|
|
84
|
-
route: "fs.read",
|
|
85
|
-
synonyms: ["filesystem.read", "read_file"],
|
|
86
|
-
hardNegatives: ["fs.write", "search.web"]
|
|
87
|
-
},
|
|
88
|
-
{
|
|
89
|
-
id: "file_003",
|
|
90
|
-
category: "file",
|
|
91
|
-
prompt: "List every Python file under src/ recursively.",
|
|
92
|
-
route: "fs.list",
|
|
93
|
-
synonyms: ["filesystem.list", "list_files"],
|
|
94
|
-
hardNegatives: ["fs.read", "search.code"]
|
|
95
|
-
},
|
|
96
|
-
{
|
|
97
|
-
id: "file_004",
|
|
98
|
-
category: "file",
|
|
99
|
-
prompt: "Delete the cached build at .turbo/cache.",
|
|
100
|
-
route: "fs.delete",
|
|
101
|
-
synonyms: ["filesystem.delete", "remove_file"],
|
|
102
|
-
hardNegatives: ["fs.write", "fs.list"]
|
|
103
|
-
},
|
|
104
|
-
{
|
|
105
|
-
id: "math_001",
|
|
106
|
-
category: "math",
|
|
107
|
-
prompt: "What is the integral of 3x^2 + 2x from 0 to 5?",
|
|
108
|
-
route: "math.integral",
|
|
109
|
-
synonyms: ["calculator.integral", "math.solve"],
|
|
110
|
-
hardNegatives: ["math.derivative", "chat.reply"]
|
|
111
|
-
},
|
|
112
|
-
{
|
|
113
|
-
id: "math_002",
|
|
114
|
-
category: "math",
|
|
115
|
-
prompt: "Compute the derivative of sin(x) * cos(x).",
|
|
116
|
-
route: "math.derivative",
|
|
117
|
-
synonyms: ["calculator.derivative", "math.solve"],
|
|
118
|
-
hardNegatives: ["math.integral", "math.algebra"]
|
|
119
|
-
},
|
|
120
|
-
{
|
|
121
|
-
id: "math_003",
|
|
122
|
-
category: "math",
|
|
123
|
-
prompt: "Solve 2x + 7 = 19 for x.",
|
|
124
|
-
route: "math.algebra",
|
|
125
|
-
synonyms: ["calculator.algebra", "math.solve"],
|
|
126
|
-
hardNegatives: ["math.derivative", "math.integral"]
|
|
127
|
-
},
|
|
128
|
-
{
|
|
129
|
-
id: "math_004",
|
|
130
|
-
category: "math",
|
|
131
|
-
prompt: "What is the prime factorization of 360?",
|
|
132
|
-
route: "math.numbertheory",
|
|
133
|
-
synonyms: ["calculator.factor", "math.solve"],
|
|
134
|
-
hardNegatives: ["math.algebra", "search.web"]
|
|
135
|
-
},
|
|
136
|
-
{
|
|
137
|
-
id: "search_001",
|
|
138
|
-
category: "search",
|
|
139
|
-
prompt: "Find recent papers on agent prompt optimization with held-out promotion gates.",
|
|
140
|
-
route: "search.web",
|
|
141
|
-
synonyms: ["web.search", "search.papers"],
|
|
142
|
-
hardNegatives: ["search.code", "chat.reply"]
|
|
143
|
-
},
|
|
144
|
-
{
|
|
145
|
-
id: "search_002",
|
|
146
|
-
category: "search",
|
|
147
|
-
prompt: "Search the codebase for every call site of `runProposeReview`.",
|
|
148
|
-
route: "search.code",
|
|
149
|
-
synonyms: ["code.search", "grep"],
|
|
150
|
-
hardNegatives: ["search.web", "fs.read"]
|
|
151
|
-
},
|
|
152
|
-
{
|
|
153
|
-
id: "search_003",
|
|
154
|
-
category: "search",
|
|
155
|
-
prompt: "What is the latest release of the Tangle network on GitHub?",
|
|
156
|
-
route: "search.web",
|
|
157
|
-
synonyms: ["web.search", "github.releases"],
|
|
158
|
-
hardNegatives: ["search.code", "chat.reply"]
|
|
159
|
-
},
|
|
160
|
-
{
|
|
161
|
-
id: "search_004",
|
|
162
|
-
category: "search",
|
|
163
|
-
prompt: "Find all TODO comments in the agent-eval src tree.",
|
|
164
|
-
route: "search.code",
|
|
165
|
-
synonyms: ["code.search", "grep"],
|
|
166
|
-
hardNegatives: ["search.web", "fs.list"]
|
|
167
|
-
},
|
|
168
|
-
{
|
|
169
|
-
id: "chat_001",
|
|
170
|
-
category: "chat",
|
|
171
|
-
prompt: "Hi there, how are you doing today?",
|
|
172
|
-
route: "chat.reply",
|
|
173
|
-
synonyms: ["conversation.reply"],
|
|
174
|
-
hardNegatives: ["search.web", "fs.read"]
|
|
175
|
-
},
|
|
176
|
-
{
|
|
177
|
-
id: "chat_002",
|
|
178
|
-
category: "chat",
|
|
179
|
-
prompt: "Please explain the difference between an LLM and a foundation model.",
|
|
180
|
-
route: "chat.reply",
|
|
181
|
-
synonyms: ["conversation.reply", "qa.answer"],
|
|
182
|
-
hardNegatives: ["search.web", "math.algebra"]
|
|
183
|
-
},
|
|
184
|
-
{
|
|
185
|
-
id: "chat_003",
|
|
186
|
-
category: "chat",
|
|
187
|
-
prompt: "Tell me a short joke about distributed systems.",
|
|
188
|
-
route: "chat.reply",
|
|
189
|
-
synonyms: ["conversation.reply"],
|
|
190
|
-
hardNegatives: ["search.web", "fs.read"]
|
|
191
|
-
},
|
|
192
|
-
{
|
|
193
|
-
id: "chat_004",
|
|
194
|
-
category: "chat",
|
|
195
|
-
prompt: "Acknowledge my last message with a thumbs up.",
|
|
196
|
-
route: "chat.reply",
|
|
197
|
-
synonyms: ["conversation.reply", "react"],
|
|
198
|
-
hardNegatives: ["fs.write", "search.web"]
|
|
199
|
-
}
|
|
200
|
-
];
|
|
201
|
-
//#endregion
|
|
202
|
-
//#region src/benchmarks/routing/index.ts
|
|
203
|
-
var routing_exports = /* @__PURE__ */ __exportAll({
|
|
204
|
-
ROUTING_DATASET: () => ROUTING_DATASET,
|
|
205
|
-
RoutingAdapter: () => RoutingAdapter,
|
|
206
|
-
assignSplit: () => assignSplit,
|
|
207
|
-
evaluate: () => evaluate,
|
|
208
|
-
extractRouteTokens: () => extractRouteTokens,
|
|
209
|
-
loadDataset: () => loadDataset
|
|
210
|
-
});
|
|
211
|
-
var RoutingAdapter = class {
|
|
212
|
-
id = "first-party/routing";
|
|
213
|
-
family = "first-party";
|
|
214
|
-
taskKind = "routing";
|
|
215
|
-
description = "Synthetic fixed-route classification smoke benchmark";
|
|
216
|
-
defaultMetric = "route_exact_match";
|
|
217
|
-
async loadDataset(split) {
|
|
218
|
-
return ROUTING_DATASET.map((item) => ({
|
|
219
|
-
id: item.id,
|
|
220
|
-
payload: item
|
|
221
|
-
})).filter((it) => assignSplitImpl(it.id) === split);
|
|
222
|
-
}
|
|
223
|
-
async evaluate(item, response) {
|
|
224
|
-
const tokens = extractRouteTokens(response);
|
|
225
|
-
const correct = new Set([item.payload.route, ...item.payload.synonyms].map((s) => s.toLowerCase()));
|
|
226
|
-
const hardNeg = new Set(item.payload.hardNegatives.map((s) => s.toLowerCase()));
|
|
227
|
-
const firstMatch = tokens.find((t) => correct.has(t.toLowerCase())) ?? null;
|
|
228
|
-
const firstHardNeg = tokens.find((t) => hardNeg.has(t.toLowerCase())) ?? null;
|
|
229
|
-
return {
|
|
230
|
-
score: firstMatch ? 1 : 0,
|
|
231
|
-
raw: {
|
|
232
|
-
firstToken: tokens[0] ?? null,
|
|
233
|
-
matchedRoute: firstMatch,
|
|
234
|
-
hitHardNegative: Boolean(firstHardNeg),
|
|
235
|
-
hardNegativeRoute: firstHardNeg,
|
|
236
|
-
category: item.payload.category
|
|
237
|
-
}
|
|
238
|
-
};
|
|
239
|
-
}
|
|
240
|
-
assignSplit(itemId) {
|
|
241
|
-
return assignSplitImpl(itemId);
|
|
242
|
-
}
|
|
243
|
-
};
|
|
244
|
-
function assignSplitImpl(itemId) {
|
|
245
|
-
return deterministicSplit(`routing::${itemId}`);
|
|
246
|
-
}
|
|
247
|
-
/**
|
|
248
|
-
* Pull route-shaped tokens out of a model response. Routes look like
|
|
249
|
-
* `category.action` (`fs.write`, `chat.reply`). Bare alphanumerics
|
|
250
|
-
* are not routes, but `category.action` patterns are robust to most
|
|
251
|
-
* model wrappers (JSON output, prose explanations, code fences).
|
|
252
|
-
*/
|
|
253
|
-
function extractRouteTokens(response) {
|
|
254
|
-
return response.match(/[a-z][a-z0-9_]*\.[a-z][a-z0-9_]*/gi) ?? [];
|
|
255
|
-
}
|
|
256
|
-
const adapter = new RoutingAdapter();
|
|
257
|
-
const loadDataset = adapter.loadDataset.bind(adapter);
|
|
258
|
-
const evaluate = adapter.evaluate.bind(adapter);
|
|
259
|
-
const assignSplit = adapter.assignSplit.bind(adapter);
|
|
260
|
-
//#endregion
|
|
261
|
-
//#region src/benchmarks/runner.ts
|
|
262
|
-
async function runBenchmarkAdapter(options) {
|
|
263
|
-
const storage = options.storage ?? fsCampaignStorage();
|
|
264
|
-
const benchmarkId = benchmarkIdFor(options.adapter);
|
|
265
|
-
const scenarios = await loadBenchmarkScenarios(options.adapter, options.splits);
|
|
266
|
-
const judge = benchmarkAdapterJudge(options.adapter);
|
|
267
|
-
const dispatch = async (scenario, context) => {
|
|
268
|
-
return options.respond({
|
|
269
|
-
scenario,
|
|
270
|
-
item: scenario.item,
|
|
271
|
-
context
|
|
272
|
-
});
|
|
273
|
-
};
|
|
274
|
-
const campaign = await runCampaign({
|
|
275
|
-
scenarios,
|
|
276
|
-
dispatch,
|
|
277
|
-
dispatchRef: `benchmark:${benchmarkId}`,
|
|
278
|
-
judges: [judge],
|
|
279
|
-
seed: options.seed,
|
|
280
|
-
reps: options.reps,
|
|
281
|
-
resumable: options.resumable,
|
|
282
|
-
costCeiling: options.costCeiling,
|
|
283
|
-
maxConcurrency: options.maxConcurrency,
|
|
284
|
-
dispatchTimeoutMs: options.dispatchTimeoutMs,
|
|
285
|
-
expectUsage: options.expectUsage ?? "off",
|
|
286
|
-
runDir: options.runDir,
|
|
287
|
-
repo: options.repo,
|
|
288
|
-
storage,
|
|
289
|
-
now: options.now
|
|
290
|
-
});
|
|
291
|
-
const report = summarizeBenchmarkCampaign({
|
|
292
|
-
adapter: options.adapter,
|
|
293
|
-
scenarios,
|
|
294
|
-
campaign
|
|
295
|
-
});
|
|
296
|
-
storage.ensureDir(campaign.runDir);
|
|
297
|
-
const reportJsonPath = join(campaign.runDir, "benchmark-report.json");
|
|
298
|
-
const reportMarkdownPath = join(campaign.runDir, "benchmark-report.md");
|
|
299
|
-
storage.write(reportJsonPath, `${JSON.stringify(report, null, 2)}\n`);
|
|
300
|
-
storage.write(reportMarkdownPath, renderBenchmarkReportMarkdown(report));
|
|
301
|
-
return {
|
|
302
|
-
scenarios,
|
|
303
|
-
campaign,
|
|
304
|
-
report,
|
|
305
|
-
reportJsonPath,
|
|
306
|
-
reportMarkdownPath
|
|
307
|
-
};
|
|
308
|
-
}
|
|
309
|
-
async function loadBenchmarkScenarios(adapter, splits = [
|
|
310
|
-
"search",
|
|
311
|
-
"dev",
|
|
312
|
-
"holdout"
|
|
313
|
-
]) {
|
|
314
|
-
const benchmarkId = benchmarkIdFor(adapter);
|
|
315
|
-
const family = adapter.family ?? "custom";
|
|
316
|
-
const taskKind = adapter.taskKind ?? "custom";
|
|
317
|
-
const out = [];
|
|
318
|
-
const seen = /* @__PURE__ */ new Set();
|
|
319
|
-
for (const split of splits) {
|
|
320
|
-
const items = await adapter.loadDataset(split);
|
|
321
|
-
for (const item of items) {
|
|
322
|
-
const splitTag = item.split ?? adapter.assignSplit(item.id);
|
|
323
|
-
if (splitTag !== split) continue;
|
|
324
|
-
const key = `${benchmarkId}:${item.id}`;
|
|
325
|
-
if (seen.has(key)) continue;
|
|
326
|
-
seen.add(key);
|
|
327
|
-
out.push({
|
|
328
|
-
id: key,
|
|
329
|
-
kind: "benchmark",
|
|
330
|
-
benchmarkId,
|
|
331
|
-
family: item.family ?? family,
|
|
332
|
-
taskKind: item.taskKind ?? taskKind,
|
|
333
|
-
splitTag,
|
|
334
|
-
tags: [.../* @__PURE__ */ new Set([splitTag, ...item.tags ?? []])],
|
|
335
|
-
item
|
|
336
|
-
});
|
|
337
|
-
}
|
|
338
|
-
}
|
|
339
|
-
return out;
|
|
340
|
-
}
|
|
341
|
-
function benchmarkAdapterJudge(adapter) {
|
|
342
|
-
const benchmarkId = benchmarkIdFor(adapter);
|
|
343
|
-
return {
|
|
344
|
-
name: `${benchmarkId}:score`,
|
|
345
|
-
dimensions: [{
|
|
346
|
-
key: "score",
|
|
347
|
-
description: `Primary ${benchmarkId} benchmark score`
|
|
348
|
-
}, {
|
|
349
|
-
key: "passed",
|
|
350
|
-
description: "Binary pass projection for aggregate reporting"
|
|
351
|
-
}],
|
|
352
|
-
appliesTo: (scenario) => {
|
|
353
|
-
return scenario.kind === "benchmark" && scenario.id.startsWith(`${benchmarkId}:`);
|
|
354
|
-
},
|
|
355
|
-
async score({ artifact, scenario }) {
|
|
356
|
-
const evaluation = await adapter.evaluate(scenario.item, artifact);
|
|
357
|
-
return {
|
|
358
|
-
dimensions: normalizeEvaluationDimensions(evaluation),
|
|
359
|
-
composite: clamp01(evaluation.score),
|
|
360
|
-
notes: evaluation.notes ?? ""
|
|
361
|
-
};
|
|
362
|
-
}
|
|
363
|
-
};
|
|
364
|
-
}
|
|
365
|
-
function summarizeBenchmarkCampaign(input) {
|
|
366
|
-
const scenarioById = new Map(input.scenarios.map((scenario) => [scenario.id, scenario]));
|
|
367
|
-
const successful = input.campaign.cells.map((cell) => {
|
|
368
|
-
const scenario = scenarioById.get(cell.scenarioId);
|
|
369
|
-
const judge = firstJudgeScore(cell.judgeScores);
|
|
370
|
-
const score = judge?.composite ?? 0;
|
|
371
|
-
return {
|
|
372
|
-
cell,
|
|
373
|
-
scenario,
|
|
374
|
-
score,
|
|
375
|
-
passed: (judge?.dimensions.passed ?? (score > 0 ? 1 : 0)) >= 1,
|
|
376
|
-
dimensions: judge?.dimensions ?? {}
|
|
377
|
-
};
|
|
378
|
-
}).filter((row) => !row.cell.error);
|
|
379
|
-
const scoreValues = successful.map((row) => row.score);
|
|
380
|
-
return {
|
|
381
|
-
benchmarkId: benchmarkIdFor(input.adapter),
|
|
382
|
-
family: input.adapter.family ?? "custom",
|
|
383
|
-
taskKind: input.adapter.taskKind ?? "custom",
|
|
384
|
-
...input.adapter.source ? { source: input.adapter.source } : {},
|
|
385
|
-
runDir: input.campaign.runDir,
|
|
386
|
-
manifestHash: input.campaign.manifestHash,
|
|
387
|
-
seed: input.campaign.seed,
|
|
388
|
-
startedAt: input.campaign.startedAt,
|
|
389
|
-
endedAt: input.campaign.endedAt,
|
|
390
|
-
durationMs: input.campaign.durationMs,
|
|
391
|
-
totalItems: input.scenarios.length,
|
|
392
|
-
totalCells: input.campaign.cells.length,
|
|
393
|
-
cellsFailed: input.campaign.aggregates.cellsFailed,
|
|
394
|
-
cellsCached: input.campaign.aggregates.cellsCached,
|
|
395
|
-
totalCostUsd: input.campaign.aggregates.cost.totalCostUsd,
|
|
396
|
-
splits: summarizeSlices(successful, (row) => row.scenario?.splitTag ?? "unknown", [
|
|
397
|
-
"search",
|
|
398
|
-
"dev",
|
|
399
|
-
"holdout"
|
|
400
|
-
]),
|
|
401
|
-
tags: summarizeSlices(successful, (row) => row.scenario?.tags ?? []),
|
|
402
|
-
dimensions: summarizeDimensions(successful.map((row) => row.dimensions)),
|
|
403
|
-
score: distribution(scoreValues),
|
|
404
|
-
costUsd: distribution(successful.map((row) => row.cell.costUsd)),
|
|
405
|
-
latencyMs: distribution(successful.map((row) => row.cell.durationMs))
|
|
406
|
-
};
|
|
407
|
-
}
|
|
408
|
-
function renderBenchmarkReportMarkdown(report) {
|
|
409
|
-
const splitRows = Object.entries(report.splits).map(([split, summary]) => {
|
|
410
|
-
return `| ${split} | ${summary.n} | ${fmt(summary.meanScore)} | ${fmt(summary.passRate)} | ${fmt(summary.score.p90)} | ${fmt(summary.costUsd.mean)} | ${fmt(summary.latencyMs.p90)} |`;
|
|
411
|
-
}).join("\n");
|
|
412
|
-
const dimRows = Object.entries(report.dimensions).sort(([a], [b]) => a.localeCompare(b)).map(([key, dist]) => `| ${key} | ${dist.n} | ${fmt(dist.mean)} | ${fmt(dist.p90)} |`).join("\n");
|
|
413
|
-
return [
|
|
414
|
-
`# Benchmark Report: ${report.benchmarkId}`,
|
|
415
|
-
"",
|
|
416
|
-
`- family: ${report.family}`,
|
|
417
|
-
`- task kind: ${report.taskKind}`,
|
|
418
|
-
`- run dir: ${report.runDir}`,
|
|
419
|
-
`- manifest: ${report.manifestHash}`,
|
|
420
|
-
`- cells: ${report.totalCells} total, ${report.cellsFailed} failed, ${report.cellsCached} cached`,
|
|
421
|
-
`- cost: $${fmt(report.totalCostUsd)}`,
|
|
422
|
-
`- score: mean ${fmt(report.score.mean)}, median ${fmt(report.score.median)}, p90 ${fmt(report.score.p90)}, n=${report.score.n}`,
|
|
423
|
-
"",
|
|
424
|
-
"## Splits",
|
|
425
|
-
"",
|
|
426
|
-
"| split | n | mean score | pass rate | score p90 | mean cost | latency p90 ms |",
|
|
427
|
-
"| --- | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
428
|
-
splitRows || "| none | 0 | 0 | 0 | 0 | 0 | 0 |",
|
|
429
|
-
"",
|
|
430
|
-
"## Dimensions",
|
|
431
|
-
"",
|
|
432
|
-
"| dimension | n | mean | p90 |",
|
|
433
|
-
"| --- | ---: | ---: | ---: |",
|
|
434
|
-
dimRows || "| none | 0 | 0 | 0 |",
|
|
435
|
-
""
|
|
436
|
-
].join("\n");
|
|
437
|
-
}
|
|
438
|
-
function benchmarkIdFor(adapter) {
|
|
439
|
-
return adapter.id ?? `${adapter.family ?? "custom"}/${adapter.taskKind ?? "custom"}`;
|
|
440
|
-
}
|
|
441
|
-
function normalizeEvaluationDimensions(evaluation) {
|
|
442
|
-
const dimensions = { score: clamp01(evaluation.score) };
|
|
443
|
-
for (const [key, value] of Object.entries(evaluation.dimensions ?? {})) if (Number.isFinite(value)) dimensions[key] = value;
|
|
444
|
-
dimensions.passed = evaluation.passed ?? evaluation.score > 0 ? 1 : 0;
|
|
445
|
-
return dimensions;
|
|
446
|
-
}
|
|
447
|
-
function summarizeDimensions(rows) {
|
|
448
|
-
const values = /* @__PURE__ */ new Map();
|
|
449
|
-
for (const row of rows) for (const [key, value] of Object.entries(row)) {
|
|
450
|
-
if (!Number.isFinite(value)) continue;
|
|
451
|
-
const list = values.get(key) ?? [];
|
|
452
|
-
list.push(value);
|
|
453
|
-
values.set(key, list);
|
|
454
|
-
}
|
|
455
|
-
return Object.fromEntries([...values.entries()].map(([key, vals]) => [key, distribution(vals)]));
|
|
456
|
-
}
|
|
457
|
-
function summarizeSlices(rows, keyOf, knownKeys = []) {
|
|
458
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
459
|
-
for (const key of knownKeys) grouped.set(key, []);
|
|
460
|
-
for (const row of rows) {
|
|
461
|
-
const keys = keyOf(row);
|
|
462
|
-
for (const key of Array.isArray(keys) ? keys : [keys]) {
|
|
463
|
-
const list = grouped.get(key) ?? [];
|
|
464
|
-
list.push(row);
|
|
465
|
-
grouped.set(key, list);
|
|
466
|
-
}
|
|
467
|
-
}
|
|
468
|
-
const out = {};
|
|
469
|
-
for (const [key, list] of grouped) {
|
|
470
|
-
const withShape = list;
|
|
471
|
-
out[key] = {
|
|
472
|
-
n: list.length,
|
|
473
|
-
meanScore: mean(withShape.map((row) => row.score)),
|
|
474
|
-
passRate: mean(withShape.map((row) => row.passed ? 1 : 0)),
|
|
475
|
-
score: distribution(withShape.map((row) => row.score)),
|
|
476
|
-
costUsd: distribution(withShape.map((row) => row.cell.costUsd)),
|
|
477
|
-
latencyMs: distribution(withShape.map((row) => row.cell.durationMs))
|
|
478
|
-
};
|
|
479
|
-
}
|
|
480
|
-
return out;
|
|
481
|
-
}
|
|
482
|
-
function firstJudgeScore(judgeScores) {
|
|
483
|
-
return Object.values(judgeScores)[0];
|
|
484
|
-
}
|
|
485
|
-
function distribution(values) {
|
|
486
|
-
const finite = [...values].filter(Number.isFinite).sort((a, b) => a - b);
|
|
487
|
-
if (finite.length === 0) return {
|
|
488
|
-
n: 0,
|
|
489
|
-
min: 0,
|
|
490
|
-
mean: 0,
|
|
491
|
-
median: 0,
|
|
492
|
-
p90: 0,
|
|
493
|
-
max: 0
|
|
494
|
-
};
|
|
495
|
-
return {
|
|
496
|
-
n: finite.length,
|
|
497
|
-
min: finite[0],
|
|
498
|
-
mean: mean(finite),
|
|
499
|
-
median: percentile(finite, .5),
|
|
500
|
-
p90: percentile(finite, .9),
|
|
501
|
-
max: finite[finite.length - 1]
|
|
502
|
-
};
|
|
503
|
-
}
|
|
504
|
-
function percentile(sortedValues, p) {
|
|
505
|
-
if (sortedValues.length === 0) return 0;
|
|
506
|
-
return sortedValues[Math.min(sortedValues.length - 1, Math.max(0, Math.ceil(p * sortedValues.length) - 1))];
|
|
507
|
-
}
|
|
508
|
-
function mean(values) {
|
|
509
|
-
const finite = values.filter(Number.isFinite);
|
|
510
|
-
if (finite.length === 0) return 0;
|
|
511
|
-
return finite.reduce((sum, value) => sum + value, 0) / finite.length;
|
|
512
|
-
}
|
|
513
|
-
function clamp01(value) {
|
|
514
|
-
if (!Number.isFinite(value)) return 0;
|
|
515
|
-
if (value < 0) return 0;
|
|
516
|
-
if (value > 1) return 1;
|
|
517
|
-
return value;
|
|
518
|
-
}
|
|
519
|
-
function fmt(value) {
|
|
520
|
-
if (!Number.isFinite(value)) return "0";
|
|
521
|
-
return value.toFixed(value === 0 || Math.abs(value) >= 10 ? 0 : 3);
|
|
522
|
-
}
|
|
523
|
-
//#endregion
|
|
524
|
-
//#region src/benchmarks/standard-formats.ts
|
|
525
|
-
function parseJsonlRows(text) {
|
|
526
|
-
return text.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).map((line, index) => {
|
|
527
|
-
try {
|
|
528
|
-
return JSON.parse(line);
|
|
529
|
-
} catch (error) {
|
|
530
|
-
throw new Error(`invalid JSONL row ${index + 1}: ${error.message}`);
|
|
531
|
-
}
|
|
532
|
-
});
|
|
533
|
-
}
|
|
534
|
-
function parseTsvRows(text) {
|
|
535
|
-
return text.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).filter((line) => !line.startsWith("#")).map((line) => line.split(/\t|\s+/));
|
|
536
|
-
}
|
|
537
|
-
function parseQrels(text) {
|
|
538
|
-
return parseTsvRows(text).flatMap((parts, index) => {
|
|
539
|
-
if (parts.length < 3) return [];
|
|
540
|
-
const [queryId, maybeZeroOrDocId, maybeDocIdOrScore, maybeScore] = parts;
|
|
541
|
-
if (!queryId || !maybeZeroOrDocId || !maybeDocIdOrScore) return [];
|
|
542
|
-
if (queryId.toLowerCase() === "query-id" || queryId.toLowerCase() === "qid") return [];
|
|
543
|
-
const documentId = maybeScore === void 0 ? maybeZeroOrDocId : maybeDocIdOrScore;
|
|
544
|
-
const score = Number(maybeScore === void 0 ? maybeDocIdOrScore : maybeScore);
|
|
545
|
-
if (!documentId || !Number.isFinite(score)) throw new Error(`invalid qrels row ${index + 1}: expected query id, doc id, score`);
|
|
546
|
-
return [{
|
|
547
|
-
queryId,
|
|
548
|
-
documentId,
|
|
549
|
-
score
|
|
550
|
-
}];
|
|
551
|
-
});
|
|
552
|
-
}
|
|
553
|
-
function parseBeirCorpusJsonl(text) {
|
|
554
|
-
return parseJsonlRows(text).map((row, index) => {
|
|
555
|
-
const id = stringField(row, "_id") ?? stringField(row, "id");
|
|
556
|
-
const body = stringField(row, "text") ?? stringField(row, "contents");
|
|
557
|
-
if (!id || body === void 0) throw new Error(`invalid BEIR corpus row ${index + 1}: expected _id/id and text/contents`);
|
|
558
|
-
return {
|
|
559
|
-
id,
|
|
560
|
-
title: stringField(row, "title"),
|
|
561
|
-
text: body,
|
|
562
|
-
metadata: stripKnown(row, [
|
|
563
|
-
"_id",
|
|
564
|
-
"id",
|
|
565
|
-
"title",
|
|
566
|
-
"text",
|
|
567
|
-
"contents"
|
|
568
|
-
])
|
|
569
|
-
};
|
|
570
|
-
});
|
|
571
|
-
}
|
|
572
|
-
function parseBeirQueriesJsonl(text) {
|
|
573
|
-
return parseJsonlRows(text).map((row, index) => {
|
|
574
|
-
const id = stringField(row, "_id") ?? stringField(row, "id") ?? stringField(row, "query_id");
|
|
575
|
-
const query = stringField(row, "text") ?? stringField(row, "query");
|
|
576
|
-
if (!id || !query) throw new Error(`invalid BEIR query row ${index + 1}: expected _id/id and text/query`);
|
|
577
|
-
return {
|
|
578
|
-
id,
|
|
579
|
-
text: query,
|
|
580
|
-
metadata: stripKnown(row, [
|
|
581
|
-
"_id",
|
|
582
|
-
"id",
|
|
583
|
-
"query_id",
|
|
584
|
-
"text",
|
|
585
|
-
"query"
|
|
586
|
-
])
|
|
587
|
-
};
|
|
588
|
-
});
|
|
589
|
-
}
|
|
590
|
-
function buildStandardRetrievalItems(options) {
|
|
591
|
-
const qrelsByQuery = /* @__PURE__ */ new Map();
|
|
592
|
-
for (const qrel of options.qrels) {
|
|
593
|
-
if (qrel.score <= 0) continue;
|
|
594
|
-
const list = qrelsByQuery.get(qrel.queryId) ?? [];
|
|
595
|
-
list.push(qrel);
|
|
596
|
-
qrelsByQuery.set(qrel.queryId, list);
|
|
597
|
-
}
|
|
598
|
-
const corpus = options.includeCorpusInPayload && options.corpus ? Object.fromEntries(options.corpus.map((document) => [document.id, document])) : void 0;
|
|
599
|
-
return options.queries.flatMap((query) => {
|
|
600
|
-
const qrels = qrelsByQuery.get(query.id) ?? [];
|
|
601
|
-
if (qrels.length === 0) return [];
|
|
602
|
-
const split = options.splitOf?.(query.id) ?? deterministicSplit(`${options.benchmarkId}:${query.id}`);
|
|
603
|
-
return [{
|
|
604
|
-
id: query.id,
|
|
605
|
-
split,
|
|
606
|
-
family: options.family,
|
|
607
|
-
taskKind: "retrieval",
|
|
608
|
-
tags: [.../* @__PURE__ */ new Set([...options.tags ?? [], split])],
|
|
609
|
-
...options.source ? { source: options.source } : {},
|
|
610
|
-
...query.metadata ? { metadata: query.metadata } : {},
|
|
611
|
-
payload: {
|
|
612
|
-
queryId: query.id,
|
|
613
|
-
query: query.text,
|
|
614
|
-
expectedDocumentIds: qrels.map((qrel) => qrel.documentId),
|
|
615
|
-
expectedScores: Object.fromEntries(qrels.map((qrel) => [qrel.documentId, qrel.score])),
|
|
616
|
-
...corpus ? { corpus } : {},
|
|
617
|
-
...query.metadata ? { metadata: query.metadata } : {}
|
|
618
|
-
}
|
|
619
|
-
}];
|
|
620
|
-
});
|
|
621
|
-
}
|
|
622
|
-
function createRetrievalIdBenchmarkAdapter(options) {
|
|
623
|
-
const items = buildStandardRetrievalItems(options);
|
|
624
|
-
return {
|
|
625
|
-
id: options.benchmarkId,
|
|
626
|
-
family: options.family,
|
|
627
|
-
taskKind: "retrieval",
|
|
628
|
-
source: options.source,
|
|
629
|
-
defaultMetric: options.primaryMetric ?? "ndcg@10",
|
|
630
|
-
async loadDataset(split) {
|
|
631
|
-
return items.filter((item) => item.split === split);
|
|
632
|
-
},
|
|
633
|
-
async evaluate(item, artifact) {
|
|
634
|
-
return evaluateStandardRetrieval(item.payload, artifact, options);
|
|
635
|
-
},
|
|
636
|
-
assignSplit(itemId) {
|
|
637
|
-
return options.splitOf?.(itemId) ?? deterministicSplit(`${options.benchmarkId}:${itemId}`);
|
|
638
|
-
}
|
|
639
|
-
};
|
|
640
|
-
}
|
|
641
|
-
function evaluateStandardRetrieval(payload, artifact, options = {}) {
|
|
642
|
-
const rankedDocumentIds = normalizeRetrievedDocumentIds(artifact, options.responseIdPattern ?? /[A-Za-z0-9_.:/-]+/g);
|
|
643
|
-
const cutoffs = normalizeCutoffs(options.cutoffs ?? [
|
|
644
|
-
1,
|
|
645
|
-
3,
|
|
646
|
-
5,
|
|
647
|
-
10
|
|
648
|
-
]);
|
|
649
|
-
const dimensions = {
|
|
650
|
-
expected_count: payload.expectedDocumentIds.length,
|
|
651
|
-
returned_count: rankedDocumentIds.length
|
|
652
|
-
};
|
|
653
|
-
for (const cutoff of cutoffs) Object.assign(dimensions, retrievalMetricsAtCutoff({
|
|
654
|
-
rankedDocumentIds,
|
|
655
|
-
expectedScores: payload.expectedScores,
|
|
656
|
-
cutoff
|
|
657
|
-
}));
|
|
658
|
-
const primaryMetric = options.primaryMetric ?? "ndcg@10";
|
|
659
|
-
const passMetric = options.passMetric ?? "hit@10";
|
|
660
|
-
const score = dimensions[primaryMetric] ?? dimensions[`ndcg@${cutoffs[cutoffs.length - 1]}`] ?? 0;
|
|
661
|
-
return {
|
|
662
|
-
score,
|
|
663
|
-
passed: (dimensions[passMetric] ?? dimensions[`hit@${cutoffs[cutoffs.length - 1]}`] ?? score) >= (options.passThreshold ?? 1),
|
|
664
|
-
dimensions,
|
|
665
|
-
raw: {
|
|
666
|
-
rankedDocumentIds,
|
|
667
|
-
expectedDocumentIds: payload.expectedDocumentIds,
|
|
668
|
-
expectedScores: payload.expectedScores
|
|
669
|
-
}
|
|
670
|
-
};
|
|
671
|
-
}
|
|
672
|
-
function normalizeRetrievedDocumentIds(artifact, responseIdPattern = /[A-Za-z0-9_.:/-]+/g) {
|
|
673
|
-
if (typeof artifact === "string") return uniqueOrdered((artifact.match(responseIdPattern) ?? []).map((value) => value.trim()));
|
|
674
|
-
if (Array.isArray(artifact)) return uniqueOrdered(artifact.flatMap((entry) => {
|
|
675
|
-
if (typeof entry === "string") return [entry];
|
|
676
|
-
return [entry.documentId ?? entry.docId ?? entry.id ?? ""];
|
|
677
|
-
}));
|
|
678
|
-
const objectArtifact = artifact;
|
|
679
|
-
return uniqueOrdered([
|
|
680
|
-
...objectArtifact.documentIds ?? [],
|
|
681
|
-
...objectArtifact.ids ?? [],
|
|
682
|
-
...(objectArtifact.results ?? []).map((entry) => entry.documentId ?? entry.docId ?? entry.id ?? "")
|
|
683
|
-
].map((value) => value.trim()));
|
|
684
|
-
}
|
|
685
|
-
function retrievalMetricsAtCutoff(input) {
|
|
686
|
-
const top = input.rankedDocumentIds.slice(0, input.cutoff);
|
|
687
|
-
const relevantIds = Object.keys(input.expectedScores).filter((id) => input.expectedScores[id] > 0);
|
|
688
|
-
const relevant = new Set(relevantIds);
|
|
689
|
-
const hits = top.filter((id) => relevant.has(id)).length;
|
|
690
|
-
const firstRelevantRank = top.findIndex((id) => relevant.has(id));
|
|
691
|
-
const prefix = `@${input.cutoff}`;
|
|
692
|
-
return {
|
|
693
|
-
[`hit${prefix}`]: hits > 0 ? 1 : 0,
|
|
694
|
-
[`recall${prefix}`]: relevant.size === 0 ? 1 : hits / relevant.size,
|
|
695
|
-
[`precision${prefix}`]: hits / input.cutoff,
|
|
696
|
-
[`mrr${prefix}`]: firstRelevantRank === -1 ? 0 : 1 / (firstRelevantRank + 1),
|
|
697
|
-
[`ndcg${prefix}`]: ndcgAt(input.rankedDocumentIds, input.expectedScores, input.cutoff)
|
|
698
|
-
};
|
|
699
|
-
}
|
|
700
|
-
function stringField(row, key) {
|
|
701
|
-
const value = row[key];
|
|
702
|
-
return typeof value === "string" ? value : void 0;
|
|
703
|
-
}
|
|
704
|
-
function stripKnown(row, keys) {
|
|
705
|
-
const known = new Set(keys);
|
|
706
|
-
return Object.fromEntries(Object.entries(row).filter(([key]) => !known.has(key)));
|
|
707
|
-
}
|
|
708
|
-
function normalizeCutoffs(cutoffs) {
|
|
709
|
-
return [...new Set(cutoffs.map((cutoff) => Math.trunc(cutoff)).filter((cutoff) => cutoff > 0))].sort((a, b) => a - b);
|
|
710
|
-
}
|
|
711
|
-
function uniqueOrdered(values) {
|
|
712
|
-
const seen = /* @__PURE__ */ new Set();
|
|
713
|
-
const out = [];
|
|
714
|
-
for (const value of values) {
|
|
715
|
-
const trimmed = value.trim();
|
|
716
|
-
if (!trimmed || seen.has(trimmed)) continue;
|
|
717
|
-
seen.add(trimmed);
|
|
718
|
-
out.push(trimmed);
|
|
719
|
-
}
|
|
720
|
-
return out;
|
|
721
|
-
}
|
|
722
|
-
function ndcgAt(rankedDocumentIds, expectedScores, cutoff) {
|
|
723
|
-
const dcg = rankedDocumentIds.slice(0, cutoff).reduce((sum, documentId, index) => sum + discountedGain(expectedScores[documentId] ?? 0, index), 0);
|
|
724
|
-
const ideal = Object.values(expectedScores).filter((score) => score > 0).sort((a, b) => b - a).slice(0, cutoff).reduce((sum, score, index) => sum + discountedGain(score, index), 0);
|
|
725
|
-
return ideal === 0 ? 1 : dcg / ideal;
|
|
726
|
-
}
|
|
727
|
-
function discountedGain(relevance, zeroBasedRank) {
|
|
728
|
-
if (relevance <= 0) return 0;
|
|
729
|
-
return (2 ** relevance - 1) / Math.log2(zeroBasedRank + 2);
|
|
730
|
-
}
|
|
731
|
-
//#endregion
|
|
732
|
-
//#region src/benchmarks/index.ts
|
|
733
|
-
var benchmarks_exports = /* @__PURE__ */ __exportAll({
|
|
734
|
-
BENCHMARK_SPLIT_SEED: () => BENCHMARK_SPLIT_SEED,
|
|
735
|
-
buildStandardRetrievalItems: () => buildStandardRetrievalItems,
|
|
736
|
-
calibrateBenchmarkMetric: () => calibrateBenchmarkMetric,
|
|
737
|
-
createRetrievalIdBenchmarkAdapter: () => createRetrievalIdBenchmarkAdapter,
|
|
738
|
-
deterministicSplit: () => deterministicSplit,
|
|
739
|
-
evaluateStandardRetrieval: () => evaluateStandardRetrieval,
|
|
740
|
-
normalizeRetrievedDocumentIds: () => normalizeRetrievedDocumentIds,
|
|
741
|
-
parseBeirCorpusJsonl: () => parseBeirCorpusJsonl,
|
|
742
|
-
parseBeirQueriesJsonl: () => parseBeirQueriesJsonl,
|
|
743
|
-
parseJsonlRows: () => parseJsonlRows,
|
|
744
|
-
parseQrels: () => parseQrels,
|
|
745
|
-
parseTsvRows: () => parseTsvRows,
|
|
746
|
-
renderBenchmarkReportMarkdown: () => renderBenchmarkReportMarkdown,
|
|
747
|
-
retrievalMetricsAtCutoff: () => retrievalMetricsAtCutoff,
|
|
748
|
-
routing: () => routing_exports,
|
|
749
|
-
runBenchmarkAdapter: () => runBenchmarkAdapter,
|
|
750
|
-
summarizeBenchmarkCampaign: () => summarizeBenchmarkCampaign
|
|
751
|
-
});
|
|
752
|
-
//#endregion
|
|
753
|
-
export { deterministicSplit as _, normalizeRetrievedDocumentIds as a, parseJsonlRows as c, retrievalMetricsAtCutoff as d, renderBenchmarkReportMarkdown as f, BENCHMARK_SPLIT_SEED as g, routing_exports as h, evaluateStandardRetrieval as i, parseQrels as l, summarizeBenchmarkCampaign as m, buildStandardRetrievalItems as n, parseBeirCorpusJsonl as o, runBenchmarkAdapter as p, createRetrievalIdBenchmarkAdapter as r, parseBeirQueriesJsonl as s, benchmarks_exports as t, parseTsvRows as u, calibrateBenchmarkMetric as v };
|
|
754
|
-
|
|
755
|
-
//# sourceMappingURL=benchmarks-Dzs8CKb1.js.map
|