@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-record-BmSPWXJR.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - `model` MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: bare model aliases are not paper-grade.\n if (!modelHasSnapshot(obj.model as string)) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;;;;;;;;AAkPA,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAG5C,IAAI,CAAC,iBAAiB,IAAI,KAAe,GACvC,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,wEACpB,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAEF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
|
|
1
|
+
{"version":3,"file":"run-record-BvHPVS-i.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - `model` MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: bare model aliases are not paper-grade.\n if (!modelHasSnapshot(obj.model as string)) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;;;;;;;;AAkPA,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAG5C,IAAI,CAAC,iBAAiB,IAAI,KAAe,GACvC,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,wEACpB,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAEF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { r as AgentProfileCell } from "./agent-profile-cell-
|
|
3
|
-
import { p as CostProvenance } from "./cost-ledger-
|
|
1
|
+
import { c as ValidationError } from "./errors-DEE6u6ot.js";
|
|
2
|
+
import { r as AgentProfileCell } from "./agent-profile-cell-BkcRDikH.js";
|
|
3
|
+
import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
4
4
|
import { o as FailureClass } from "./schema-BtVldJ3T.js";
|
|
5
5
|
//#region src/run-record.d.ts
|
|
6
6
|
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
@@ -239,4 +239,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
|
|
|
239
239
|
declare function modelHasSnapshot(model: string): boolean;
|
|
240
240
|
//#endregion
|
|
241
241
|
export { RunRecord as a, RunTaskFailure as c, isRunRecord as d, modelHasSnapshot as f, validateRunRecord as g, runTaskScore as h, RunOutcome as i, RunTerminalOutcome as l, roundTripRunRecord as m, RunCostProvenance as n, RunRecordValidationError as o, parseRunRecordSafe as p, RunJudgeMetadata as r, RunSplitTag as s, JudgeScoresRecord as t, RunTokenUsage as u };
|
|
242
|
-
//# sourceMappingURL=run-record-
|
|
242
|
+
//# sourceMappingURL=run-record-CKiihE6f.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-record-
|
|
1
|
+
{"version":3,"file":"run-record-CKiihE6f.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;UAEK;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;UAmB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAuOnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
|
|
@@ -8,35 +8,6 @@ function combineAbortSignals(...signals) {
|
|
|
8
8
|
return AbortSignal.any(active);
|
|
9
9
|
}
|
|
10
10
|
//#endregion
|
|
11
|
-
//#region src/run-score.ts
|
|
12
|
-
const DEFAULT_RUN_SCORE_WEIGHTS = {
|
|
13
|
-
success: 4,
|
|
14
|
-
goalProgress: 2,
|
|
15
|
-
repoGroundedness: 1.5,
|
|
16
|
-
driftPenalty: -1.5,
|
|
17
|
-
toolUseQuality: 1,
|
|
18
|
-
patchQuality: 1.25,
|
|
19
|
-
testReality: 1.5,
|
|
20
|
-
finalGate: 3,
|
|
21
|
-
reviewerBlockers: -2,
|
|
22
|
-
costUsd: -.2,
|
|
23
|
-
wallSeconds: -.1
|
|
24
|
-
};
|
|
25
|
-
function aggregateRunScore(score, weights = {}) {
|
|
26
|
-
const w = {
|
|
27
|
-
...DEFAULT_RUN_SCORE_WEIGHTS,
|
|
28
|
-
...weights
|
|
29
|
-
};
|
|
30
|
-
return w.success * clamp01(score.success) + w.goalProgress * clamp01(score.goalProgress) + w.repoGroundedness * clamp01(score.repoGroundedness) + w.driftPenalty * clamp01(score.driftPenalty) + w.toolUseQuality * clamp01(score.toolUseQuality) + w.patchQuality * clamp01(score.patchQuality) + w.testReality * clamp01(score.testReality) + w.finalGate * clamp01(score.finalGate) + w.reviewerBlockers * clamp01(score.reviewerBlockers) + w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) + w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60);
|
|
31
|
-
}
|
|
32
|
-
function clamp01(value) {
|
|
33
|
-
if (!Number.isFinite(value)) return 0;
|
|
34
|
-
return Math.max(0, Math.min(1, value));
|
|
35
|
-
}
|
|
36
|
-
function finiteOrZero(value) {
|
|
37
|
-
return Number.isFinite(value) ? value : 0;
|
|
38
|
-
}
|
|
39
|
-
//#endregion
|
|
40
11
|
//#region src/analyst/proposal-findings.ts
|
|
41
12
|
const ProposalFindingSchema = z.object({
|
|
42
13
|
schema_version: z.literal("1.0.0"),
|
|
@@ -93,6 +64,35 @@ function findingLabel(finding, index) {
|
|
|
93
64
|
return typeof id === "string" && id.length > 0 ? id : `index ${index}`;
|
|
94
65
|
}
|
|
95
66
|
//#endregion
|
|
96
|
-
|
|
67
|
+
//#region src/run-score.ts
|
|
68
|
+
const DEFAULT_RUN_SCORE_WEIGHTS = {
|
|
69
|
+
success: 4,
|
|
70
|
+
goalProgress: 2,
|
|
71
|
+
repoGroundedness: 1.5,
|
|
72
|
+
driftPenalty: -1.5,
|
|
73
|
+
toolUseQuality: 1,
|
|
74
|
+
patchQuality: 1.25,
|
|
75
|
+
testReality: 1.5,
|
|
76
|
+
finalGate: 3,
|
|
77
|
+
reviewerBlockers: -2,
|
|
78
|
+
costUsd: -.2,
|
|
79
|
+
wallSeconds: -.1
|
|
80
|
+
};
|
|
81
|
+
function aggregateRunScore(score, weights = {}) {
|
|
82
|
+
const w = {
|
|
83
|
+
...DEFAULT_RUN_SCORE_WEIGHTS,
|
|
84
|
+
...weights
|
|
85
|
+
};
|
|
86
|
+
return w.success * clamp01(score.success) + w.goalProgress * clamp01(score.goalProgress) + w.repoGroundedness * clamp01(score.repoGroundedness) + w.driftPenalty * clamp01(score.driftPenalty) + w.toolUseQuality * clamp01(score.toolUseQuality) + w.patchQuality * clamp01(score.patchQuality) + w.testReality * clamp01(score.testReality) + w.finalGate * clamp01(score.finalGate) + w.reviewerBlockers * clamp01(score.reviewerBlockers) + w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) + w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60);
|
|
87
|
+
}
|
|
88
|
+
function clamp01(value) {
|
|
89
|
+
if (!Number.isFinite(value)) return 0;
|
|
90
|
+
return Math.max(0, Math.min(1, value));
|
|
91
|
+
}
|
|
92
|
+
function finiteOrZero(value) {
|
|
93
|
+
return Number.isFinite(value) ? value : 0;
|
|
94
|
+
}
|
|
95
|
+
//#endregion
|
|
96
|
+
export { combineAbortSignals as a, isProposalFinding as i, clamp01 as n, assertProposalFindings as r, aggregateRunScore as t };
|
|
97
97
|
|
|
98
|
-
//# sourceMappingURL=
|
|
98
|
+
//# sourceMappingURL=run-score-lDzV0X8j.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-score-lDzV0X8j.js","names":[],"sources":["../src/abort-signal.ts","../src/analyst/proposal-findings.ts","../src/run-score.ts"],"sourcesContent":["/** Combine active cancellation sources without wrapping a single source. */\nexport function combineAbortSignals(\n ...signals: Array<AbortSignal | undefined>\n): AbortSignal | undefined {\n const active = [\n ...new Set(signals.filter((signal): signal is AbortSignal => signal !== undefined)),\n ]\n if (active.length === 0) return undefined\n if (active.length === 1) return active[0]\n return AbortSignal.any(active)\n}\n","import { z } from 'zod'\nimport type { ProposalFinding } from './types'\n\nconst ProposalFindingSchema = z\n .object({\n schema_version: z.literal('1.0.0'),\n finding_id: z.string().min(1),\n analyst_id: z.string().min(1),\n produced_at: z.string().min(1),\n severity: z.enum(['critical', 'high', 'medium', 'low', 'info']),\n area: z.string().min(1),\n claim: z.string().min(1),\n rationale: z.string().optional(),\n evidence_refs: z.array(\n z\n .object({\n kind: z.enum(['span', 'event', 'artifact', 'finding', 'metric']),\n uri: z.string().min(1),\n excerpt: z.string().optional(),\n })\n .strict(),\n ),\n recommended_action: z.string().optional(),\n validation_plan: z.string().optional(),\n confidence: z.number().min(0).max(1),\n subject: z.string().optional(),\n derived_from_judge: z.boolean().optional(),\n metadata: z.record(z.string(), z.unknown()).optional(),\n proposal_origin: z.enum(['search', 'production']),\n })\n .strict() satisfies z.ZodType<ProposalFinding>\n\n/** True when a finding names a source candidate generation may learn from. */\nexport function isProposalFinding(finding: unknown): finding is ProposalFinding {\n return ProposalFindingSchema.safeParse(finding).success\n}\n\n/**\n * Reject findings whose source has not been explicitly admitted for candidate\n * generation. Search feedback and observed production behavior are allowed;\n * final evaluation data has no allowed origin.\n */\nexport function assertProposalFindings(\n findings: unknown,\n context = 'proposal findings',\n): ReadonlyArray<ProposalFinding> {\n if (!Array.isArray(findings)) {\n throw new TypeError(`${context}: expected an array`)\n }\n const rejected = findings.flatMap((finding, index) =>\n isProposalFinding(finding) ? [] : [findingLabel(finding, index)],\n )\n if (rejected.length > 0) {\n throw new Error(\n `${context}: every finding must match AnalystFinding and declare ` +\n `proposal_origin as search or production. ` +\n `Rejected findings: [${rejected.join(', ')}].`,\n )\n }\n return findings as ReadonlyArray<ProposalFinding>\n}\n\nfunction findingLabel(finding: unknown, index: number): string {\n if (typeof finding !== 'object' || finding === null) return `index ${index}`\n const id = (finding as { finding_id?: unknown }).finding_id\n return typeof id === 'string' && id.length > 0 ? id : `index ${index}`\n}\n","export interface RunScore {\n success: number\n goalProgress: number\n repoGroundedness: number\n driftPenalty: number\n toolUseQuality: number\n patchQuality: number\n testReality: number\n finalGate: number\n reviewerBlockers: number\n costUsd: number\n wallSeconds: number\n notes?: string[]\n}\n\nexport interface RunScoreWeights {\n success: number\n goalProgress: number\n repoGroundedness: number\n driftPenalty: number\n toolUseQuality: number\n patchQuality: number\n testReality: number\n finalGate: number\n reviewerBlockers: number\n costUsd: number\n wallSeconds: number\n}\n\nexport const DEFAULT_RUN_SCORE_WEIGHTS: RunScoreWeights = {\n success: 4,\n goalProgress: 2,\n repoGroundedness: 1.5,\n driftPenalty: -1.5,\n toolUseQuality: 1,\n patchQuality: 1.25,\n testReality: 1.5,\n finalGate: 3,\n reviewerBlockers: -2,\n costUsd: -0.2,\n wallSeconds: -0.1,\n}\n\nexport function aggregateRunScore(score: RunScore, weights: Partial<RunScoreWeights> = {}): number {\n const w = { ...DEFAULT_RUN_SCORE_WEIGHTS, ...weights }\n return (\n w.success * clamp01(score.success) +\n w.goalProgress * clamp01(score.goalProgress) +\n w.repoGroundedness * clamp01(score.repoGroundedness) +\n w.driftPenalty * clamp01(score.driftPenalty) +\n w.toolUseQuality * clamp01(score.toolUseQuality) +\n w.patchQuality * clamp01(score.patchQuality) +\n w.testReality * clamp01(score.testReality) +\n w.finalGate * clamp01(score.finalGate) +\n w.reviewerBlockers * clamp01(score.reviewerBlockers) +\n w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) +\n w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60)\n )\n}\n\nexport function clamp01(value: number): number {\n if (!Number.isFinite(value)) return 0\n return Math.max(0, Math.min(1, value))\n}\n\nfunction finiteOrZero(value: number): number {\n return Number.isFinite(value) ? value : 0\n}\n"],"mappings":";;;AACA,SAAgB,oBACd,GAAG,SACsB;CACzB,MAAM,SAAS,CACb,GAAG,IAAI,IAAI,QAAQ,QAAQ,WAAkC,WAAW,KAAA,CAAS,CAAC,CACpF;CACA,IAAI,OAAO,WAAW,GAAG,OAAO,KAAA;CAChC,IAAI,OAAO,WAAW,GAAG,OAAO,OAAO;CACvC,OAAO,YAAY,IAAI,MAAM;AAC/B;;;ACPA,MAAM,wBAAwB,EAC3B,OAAO;CACN,gBAAgB,EAAE,QAAQ,OAAO;CACjC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,aAAa,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC7B,UAAU,EAAE,KAAK;EAAC;EAAY;EAAQ;EAAU;EAAO;CAAM,CAAC;CAC9D,MAAM,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACtB,OAAO,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACvB,WAAW,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,eAAe,EAAE,MACf,EACG,OAAO;EACN,MAAM,EAAE,KAAK;GAAC;GAAQ;GAAS;GAAY;GAAW;EAAQ,CAAC;EAC/D,KAAK,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;EACrB,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,CAAC,CAAC,CACD,OAAO,CACZ;CACA,oBAAoB,EAAE,OAAO,CAAC,CAAC,SAAS;CACxC,iBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS;CACrC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;CACnC,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC7B,oBAAoB,EAAE,QAAQ,CAAC,CAAC,SAAS;CACzC,UAAU,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,QAAQ,CAAC,CAAC,CAAC,SAAS;CACrD,iBAAiB,EAAE,KAAK,CAAC,UAAU,YAAY,CAAC;AAClD,CAAC,CAAC,CACD,OAAO;;AAGV,SAAgB,kBAAkB,SAA8C;CAC9E,OAAO,sBAAsB,UAAU,OAAO,CAAC,CAAC;AAClD;;;;;;AAOA,SAAgB,uBACd,UACA,UAAU,qBACsB;CAChC,IAAI,CAAC,MAAM,QAAQ,QAAQ,GACzB,MAAM,IAAI,UAAU,GAAG,QAAQ,oBAAoB;CAErD,MAAM,WAAW,SAAS,SAAS,SAAS,UAC1C,kBAAkB,OAAO,IAAI,CAAC,IAAI,CAAC,aAAa,SAAS,KAAK,CAAC,CACjE;CACA,IAAI,SAAS,SAAS,GACpB,MAAM,IAAI,MACR,GAAG,QAAQ,qHAEc,SAAS,KAAK,IAAI,EAAE,GAC/C;CAEF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,OAAuB;CAC7D,IAAI,OAAO,YAAY,YAAY,YAAY,MAAM,OAAO,SAAS;CACrE,MAAM,KAAM,QAAqC;CACjD,OAAO,OAAO,OAAO,YAAY,GAAG,SAAS,IAAI,KAAK,SAAS;AACjE;;;ACrCA,MAAa,4BAA6C;CACxD,SAAS;CACT,cAAc;CACd,kBAAkB;CAClB,cAAc;CACd,gBAAgB;CAChB,cAAc;CACd,aAAa;CACb,WAAW;CACX,kBAAkB;CAClB,SAAS;CACT,aAAa;AACf;AAEA,SAAgB,kBAAkB,OAAiB,UAAoC,CAAC,GAAW;CACjG,MAAM,IAAI;EAAE,GAAG;EAA2B,GAAG;CAAQ;CACrD,OACE,EAAE,UAAU,QAAQ,MAAM,OAAO,IACjC,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,mBAAmB,QAAQ,MAAM,gBAAgB,IACnD,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,iBAAiB,QAAQ,MAAM,cAAc,IAC/C,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,cAAc,QAAQ,MAAM,WAAW,IACzC,EAAE,YAAY,QAAQ,MAAM,SAAS,IACrC,EAAE,mBAAmB,QAAQ,MAAM,gBAAgB,IACnD,EAAE,UAAU,KAAK,IAAI,GAAG,aAAa,MAAM,OAAO,CAAC,IACnD,EAAE,cAAc,KAAK,IAAI,GAAG,aAAa,MAAM,WAAW,IAAI,EAAE;AAEpE;AAEA,SAAgB,QAAQ,OAAuB;CAC7C,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG,OAAO;CACpC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,KAAK,CAAC;AACvC;AAEA,SAAS,aAAa,OAAuB;CAC3C,OAAO,OAAO,SAAS,KAAK,IAAI,QAAQ;AAC1C"}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
//#region src/sandbox-harness.d.ts
|
|
2
|
+
interface HarnessConfig {
|
|
3
|
+
/** Setup command (e.g. "pnpm install"). Non-zero exit fails the run. */
|
|
4
|
+
setupCommand?: string;
|
|
5
|
+
/** Run command (e.g. "pnpm build"). */
|
|
6
|
+
runCommand?: string;
|
|
7
|
+
/** Test command (e.g. "pnpm test --run"). Drives the test count + pass count. */
|
|
8
|
+
testCommand?: string;
|
|
9
|
+
/** Absolute cwd for the subprocess driver. Ignored by docker driver. */
|
|
10
|
+
cwd?: string;
|
|
11
|
+
/** Max wall-clock per phase in ms. Default 10 minutes. */
|
|
12
|
+
timeoutMs?: number;
|
|
13
|
+
/**
|
|
14
|
+
* Cap on captured stdout+stderr bytes per phase. A runaway process can
|
|
15
|
+
* otherwise grow the in-memory buffer without bound. Once hit, further
|
|
16
|
+
* output is dropped and `outputTruncated` is set. Default 16 MiB.
|
|
17
|
+
*/
|
|
18
|
+
maxOutputBytes?: number;
|
|
19
|
+
/** Image for the docker driver. */
|
|
20
|
+
image?: string;
|
|
21
|
+
/** Extra env vars (validated; shell-escaped). */
|
|
22
|
+
env?: Record<string, string>;
|
|
23
|
+
/** Parser for the test output — maps stdout/stderr/exit code → pass count. */
|
|
24
|
+
testParser?: TestOutputParser;
|
|
25
|
+
}
|
|
26
|
+
interface TestOutputParser {
|
|
27
|
+
id: string;
|
|
28
|
+
parse(stdout: string, stderr: string, exitCode: number): {
|
|
29
|
+
testsTotal: number;
|
|
30
|
+
testsPassed: number;
|
|
31
|
+
} | undefined;
|
|
32
|
+
}
|
|
33
|
+
interface SandboxResult {
|
|
34
|
+
phase: 'setup' | 'run' | 'test';
|
|
35
|
+
exitCode: number;
|
|
36
|
+
stdout: string;
|
|
37
|
+
stderr: string;
|
|
38
|
+
wallMs: number;
|
|
39
|
+
testsTotal?: number;
|
|
40
|
+
testsPassed?: number;
|
|
41
|
+
/**
|
|
42
|
+
* True when the process was killed because it exceeded `timeoutMs`. A
|
|
43
|
+
* SIGKILLed child can still close with exit code 0; callers MUST treat
|
|
44
|
+
* a timed-out phase as a hard failure regardless of `exitCode`, never
|
|
45
|
+
* as a pass. `undefined`/`false` means the process completed on its own.
|
|
46
|
+
*/
|
|
47
|
+
killedByTimeout?: boolean;
|
|
48
|
+
/**
|
|
49
|
+
* True when captured stdout/stderr hit `maxOutputBytes` and further
|
|
50
|
+
* output was dropped. The result is still returned (the process was
|
|
51
|
+
* not killed for this), but downstream parsers see truncated text.
|
|
52
|
+
*/
|
|
53
|
+
outputTruncated?: boolean;
|
|
54
|
+
}
|
|
55
|
+
interface SandboxDriver {
|
|
56
|
+
id: string;
|
|
57
|
+
exec(phase: SandboxResult['phase'], command: string, config: HarnessConfig): Promise<SandboxResult>;
|
|
58
|
+
}
|
|
59
|
+
interface SandboxHarnessResult {
|
|
60
|
+
passed: boolean;
|
|
61
|
+
setup?: SandboxResult;
|
|
62
|
+
run?: SandboxResult;
|
|
63
|
+
test?: SandboxResult;
|
|
64
|
+
totalWallMs: number;
|
|
65
|
+
/** Final score — 0 when no tests; otherwise testsPassed/testsTotal. */
|
|
66
|
+
score: number;
|
|
67
|
+
}
|
|
68
|
+
//#endregion
|
|
69
|
+
export { SandboxDriver as n, SandboxHarnessResult as r, HarnessConfig as t };
|
|
70
|
+
//# sourceMappingURL=sandbox-harness-BlSOu4LX.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sandbox-harness-BlSOu4LX.d.ts","names":[],"sources":["../src/sandbox-harness.ts"],"mappings":";UAkBiB;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;EAMA;;EAEA;;EAEA,MAAM;;EAEN,aAAa;;UAGE;EACf;EACA,MACE,gBACA,gBACA;IACG;IAAoB;;;UAGV;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;;;;;EAOA;;;;;;EAMA;;UAGe;EACf;EACA,KACE,OAAO,wBACP,iBACA,QAAQ,gBACP,QAAQ;;UA4PI;EACf;EACA,QAAQ;EACR,MAAM;EACN,OAAO;EACP;;EAEA"}
|
|
@@ -405,4 +405,4 @@ declare function assertMinted(value: unknown, context?: string): MintedRolloutLi
|
|
|
405
405
|
declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
|
|
406
406
|
//#endregion
|
|
407
407
|
export { isTrainableSplit as A, TRAINABLE_SPLITS as C, assertRolloutLine as D, assertMintedLines as E, gateGamedOutcome as O, RolloutTask as S, assertMinted as T, RolloutPolicy as _, GatedEvidence as a, RolloutSplit as b, ROLLOUT_CAPTURES as c, ROLLOUT_SPLITS as d, RolloutArtifacts as f, RolloutOutcome as g, RolloutLine as h, ChatToolCall as i, validateRolloutLine as j, isRolloutLine as k, ROLLOUT_ROLES as l, RolloutCostBlock as m, ChatMessage as n, MintedRolloutLine as o, RolloutCapture as p, ChatRole as r, MintedRolloutOutcome as s, CHAT_ROLES as t, ROLLOUT_SCHEMA as u, RolloutProvenance as v, ToolDef as w, RolloutStep as x, RolloutRole as y };
|
|
408
|
-
//# sourceMappingURL=schema-
|
|
408
|
+
//# sourceMappingURL=schema-Cef2cFmb.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"schema-Cef2cFmb.d.ts","names":[],"sources":["../src/rollout/schema.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cAmDa;;KAGD;cACC,wBAAwB;;KAUzB;cACC,yBAAyB;;cAEzB,2BAA2B;iBAExB,iBAAiB,OAAO;;KAK5B;cACC,2BAA2B;KAM5B;cACC,qBAAqB;UAEjB;EACf;EACA;EACA;IACE;;IAEA;;;UAIa;EACf,MAAM;EACN;;EAEA;EACA,aAAa;;EAEb;EACA;;;;;;;;EAQA;;UAGe;EACf;EACA;IACE;IACA;IACA,aAAa;;;;;;;;UASA;EACf;EACA;;EAEA;;EAEA;EACA;EACA;;;;;EAUA;;EAEA;;EAEA;;;;;;EAMA;;UAOe;;EAEf;EACA;EACA,OAAO;;EAEP;;EAEA;;UAGe;;EAEf;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;;;;EAKf;;EAEA;;EAEA;;EAEA,SAAS;EACT;EACA;EACA;;;;;;;;;;;;;EAaA;;;;;;;;;;;;;;;;;;;;;EAqBA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;;;EAKA;;UAGe;EACf;EACA;;EAEA;;;;;;;UAQe;;EAEf,UAAU;;EAEV;;;;;;;;;EASA;;UAGe;EACf;EACA,SAAS;;;;;;;EAOT;;;;;;;EAOA,iBAAiB;;UAGF;EACf,eAAe;EACf;;EAEA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,MAAM;EACN,MAAM;EACN,QAAQ;;EAER,UAAU;EACV,WAAW;;EAEX,QAAQ;EACR,SAAS;EACT,MAAM;EACN,WAAW;EACX,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6GE,iBAAiB,MAAM,cAAc;iBAqDrC,oBAAoB;iBAiNpB,kBACd,gBACA,2BACS,SAAS;iBAOJ,cAAc,iBAAiB,SAAS;;;;;;cAa1C;;;;;;UAOG,6BAA6B;EAC5C;;;;;;;;;;;;;;;;;;;;;;KAuBU,oBAAoB,KAAK;YACzB;EACV,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsCK,aAAa,gBAAgB,mBAA2B;;iBAiBxD,kBACd,4BACA,mBACC"}
|