@tangle-network/agent-eval 0.150.1 → 0.161.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +178 -1
- package/README.md +7 -3
- package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
- package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
- package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +19 -36
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
- package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
- package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
- package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
- package/dist/{benchmark-command-CAFwbH0L.js → benchmark-command-BVtaq_ve.js} +26 -31
- package/dist/benchmark-command-BVtaq_ve.js.map +1 -0
- package/dist/benchmarks/index.d.ts +6 -19
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +4 -4
- package/dist/benchmarks/index.js.map +1 -1
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +3 -3
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-CN_7xJdV.js → campaign-BSmOwskD.js} +77 -795
- package/dist/campaign-BSmOwskD.js.map +1 -0
- package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
- package/dist/canonical-IL-Bu-14.js.map +1 -0
- package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
- package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
- package/dist/{chat-client-2bVfrzhN.js → chat-client-DlMlAeYI.js} +5 -55
- package/dist/{chat-client-2bVfrzhN.js.map → chat-client-DlMlAeYI.js.map} +1 -1
- package/dist/chat-json-call-6g5sJobJ.js +53 -0
- package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
- package/dist/cli.js +54 -19
- package/dist/cli.js.map +1 -1
- package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
- package/dist/client-CX7KqIdB.js.map +1 -0
- package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
- package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -27
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +14 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
- package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
- package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
- package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
- package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
- package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
- package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
- package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
- package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
- package/dist/{define-agent-eval-rqNyVhVV.js → define-agent-eval-h-s-sI-v.js} +17 -11
- package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
- package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
- package/dist/descriptive-jDOuI6mz.js.map +1 -0
- package/dist/{dspy-rlm-engine-DHI0WrUU.js → dspy-rlm-engine-DptEII26.js} +95 -15
- package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
- package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
- package/dist/emitter-BpYFQPj4.js.map +1 -0
- package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
- package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
- package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
- package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
- package/dist/{eval-campaign-C4jmuM-b.js → eval-campaign-BsXWL2-2.js} +17 -28
- package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
- package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
- package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
- package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
- package/dist/exec-D9WpA2p-.js.map +1 -0
- package/dist/experiment/index.d.ts +45 -11
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +30 -11
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
- package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
- package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
- package/dist/exporters-Df7TgHFv.js.map +1 -0
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Bhmzngf-.js → external-optimizer-process-WosTBChy.js} +4 -4
- package/dist/{external-optimizer-process-Bhmzngf-.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-Dn90UqN2.js → external-optimizer-subprocess-BIWbHpgD.js} +10 -6
- package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +1 -0
- package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
- package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
- package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -14
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +2 -2
- package/dist/{index-CTKpu9ry.d.ts → index-CGtH1piv.d.ts} +48 -26
- package/dist/index-CGtH1piv.d.ts.map +1 -0
- package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
- package/dist/index-D-IiQIBB.d.ts.map +1 -0
- package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
- package/dist/index-D-V8gCs_.d.ts.map +1 -0
- package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
- package/dist/index-lfaSeKSD.d.ts.map +1 -0
- package/dist/index-vrJugRal.d.ts +1 -0
- package/dist/index.d.ts +68 -125
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +61 -112
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
- package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
- package/dist/{integrity-DysDBWDu.js → integrity-CyWSSoQS.js} +17 -4
- package/dist/integrity-CyWSSoQS.js.map +1 -0
- package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
- package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
- package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
- package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
- package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
- package/dist/kind-factory-DY8FdoXf.js.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +3 -3
- package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
- package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
- package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
- package/dist/llm-client-hgDieDNN.js.map +1 -0
- package/dist/{llm-judge-CVq33oz1.js → llm-judge-BhasIPFT.js} +1136 -78
- package/dist/llm-judge-BhasIPFT.js.map +1 -0
- package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
- package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
- package/dist/meta-eval/index.d.ts +8 -6
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +9 -7
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
- package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +2 -8
- package/dist/multishot/golden/index.d.ts.map +1 -1
- package/dist/multishot/golden/index.js +56 -86
- package/dist/multishot/golden/index.js.map +1 -1
- package/dist/multishot/index.d.ts +11 -46
- package/dist/multishot/index.d.ts.map +1 -1
- package/dist/multishot/index.js +30 -83
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
- package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
- package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
- package/dist/pre-registration-KN9jkh58.js +110 -0
- package/dist/pre-registration-KN9jkh58.js.map +1 -0
- package/dist/{produced-state-jfk8Du3b.js → produced-state-DZ89riy5.js} +8 -8
- package/dist/produced-state-DZ89riy5.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +31 -5
- package/dist/profile-cell.js.map +1 -1
- package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
- package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
- package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
- package/dist/query-CHmMP42p.js.map +1 -0
- package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
- package/dist/query-DxPYqpmT.d.ts.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
- package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
- package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
- package/dist/release-confidence-DKfD2RYU.js.map +1 -0
- package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
- package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +3 -3
- package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
- package/dist/researcher-Cz565b7D.d.ts.map +1 -0
- package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
- package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
- package/dist/{reward-hacking-DKI9T52l.js → reward-hacking-t4lB1yt8.js} +3 -3
- package/dist/{reward-hacking-DKI9T52l.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
- package/dist/rl.d.ts +11 -42
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +41 -24
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +3 -3
- package/dist/rollout/index.js +7 -7
- package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
- package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
- package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CK8SCOg-.js} +6 -16
- package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
- package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
- package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
- package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
- package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
- package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
- package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
- package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
- package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
- package/dist/schema-BzWDXhOR.d.ts.map +1 -0
- package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
- package/dist/schema-C1aaAxTf.js.map +1 -0
- package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
- package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
- package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
- package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
- package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
- package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
- package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
- package/dist/sequential-rYW-Ophm.js.map +1 -0
- package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
- package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
- package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
- package/dist/server-BtFd4uzB.js.map +1 -0
- package/dist/{skillopt-optimization-method-DLeUcK-K.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
- package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
- package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
- package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
- package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
- package/dist/steps-CiNVJry_.d.ts.map +1 -0
- package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
- package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
- package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
- package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
- package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
- package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
- package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-Blysd6Z2.js → summary-report-BI5hUtvK.js} +7 -17
- package/dist/summary-report-BI5hUtvK.js.map +1 -0
- package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
- package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +27 -18
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +104 -26
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
- package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
- package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
- package/dist/{tool-waste-BDdBZG1F.js → tool-waste-BqzmVdJk.js} +4 -4
- package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
- package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
- package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +4 -77
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -15
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +13 -23
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +9 -20
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
- package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +5 -5
- package/dist/trajectory-replay/index.js +5 -5
- package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
- package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
- package/dist/types-B3jzCp0p.js.map +1 -1
- package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
- package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts +443 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -0
- package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
- package/dist/types-D4s7Z6nq.d.ts.map +1 -0
- package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
- package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
- package/dist/{types-yLK8gXE9.d.ts → types-DeIUdzNd.d.ts} +160 -10
- package/dist/types-DeIUdzNd.d.ts.map +1 -0
- package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
- package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
- package/dist/verdict-cache-CdVVTVmn.js +88 -0
- package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
- package/dist/wire/index.d.ts +21 -111
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/adapters-observability.md +9 -23
- package/docs/building-doctrine.md +3 -3
- package/docs/campaign-proposers.md +54 -15
- package/docs/concepts.md +3 -4
- package/docs/design/statistics-decisions.md +89 -1
- package/docs/eval-surface-map.md +14 -0
- package/docs/experiment.md +19 -2
- package/docs/feedback-trajectories.md +1 -1
- package/docs/multishot-golden-records.md +4 -4
- package/docs/public-api.md +1616 -0
- package/docs/research-report-methodology.md +1 -1
- package/docs/search-history-receipts.md +39 -1
- package/docs/trace-analysis.md +1 -1
- package/docs/trace-repair-admission.md +1 -1
- package/docs/trace-repair-continuation.md +1 -1
- package/docs/verdicts.md +24 -0
- package/docs/wire-protocol.md +1 -1
- package/package.json +6 -2
- package/dist/active-curriculum-C4mk67HP.js.map +0 -1
- package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
- package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
- package/dist/benchmark-command-CAFwbH0L.js.map +0 -1
- package/dist/campaign-CN_7xJdV.js.map +0 -1
- package/dist/canonical-D-XsTQ6_.js.map +0 -1
- package/dist/client-LIuo-KPv.js.map +0 -1
- package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
- package/dist/define-agent-eval-rqNyVhVV.js.map +0 -1
- package/dist/descriptive-B5MwKfbf.js.map +0 -1
- package/dist/dspy-rlm-engine-DHI0WrUU.js.map +0 -1
- package/dist/emitter-CPBAhxum.js.map +0 -1
- package/dist/eval-campaign-C4jmuM-b.js.map +0 -1
- package/dist/exec-y-DCLqK7.js.map +0 -1
- package/dist/exporters-q9iL-2Jf.js.map +0 -1
- package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +0 -1
- package/dist/index-B8Ui1mr1.d.ts.map +0 -1
- package/dist/index-BNPtkBPf.d.ts.map +0 -1
- package/dist/index-C1ravkGA.d.ts +0 -1
- package/dist/index-CTKpu9ry.d.ts.map +0 -1
- package/dist/index-IQccV3Ou.d.ts.map +0 -1
- package/dist/integrity-DysDBWDu.js.map +0 -1
- package/dist/kind-factory-DmAa0h3K.js.map +0 -1
- package/dist/llm-client-Bg32RW0j.js.map +0 -1
- package/dist/llm-judge-CVq33oz1.js.map +0 -1
- package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
- package/dist/pre-registration-DakwTRXk.js +0 -96
- package/dist/pre-registration-DakwTRXk.js.map +0 -1
- package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
- package/dist/produced-state-jfk8Du3b.js.map +0 -1
- package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
- package/dist/query-CJ_DX8vl.d.ts.map +0 -1
- package/dist/query-Di7eEQ79.js.map +0 -1
- package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
- package/dist/release-confidence-BknrpBnO.js.map +0 -1
- package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-Cwwyd7ah.js.map +0 -1
- package/dist/schema-C6DW4ZHR.js.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
- package/dist/sequential-C458DXNf.js.map +0 -1
- package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
- package/dist/server-dIWwF3j_.js.map +0 -1
- package/dist/skillopt-optimization-method-DLeUcK-K.js.map +0 -1
- package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
- package/dist/steps-AmkT-GIM.d.ts.map +0 -1
- package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
- package/dist/summary-report-Blysd6Z2.js.map +0 -1
- package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
- package/dist/types-CLAwnY-L.d.ts.map +0 -1
- package/dist/types-DdFNuyxQ.d.ts.map +0 -1
- package/dist/types-jUBXJ7Iz.d.ts +0 -884
- package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
- package/dist/types-yLK8gXE9.d.ts.map +0 -1
- package/dist/verdict-cache-mZf5FEiY.js +0 -107
- package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-record-D2lDdSAz.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\n/** Explicit model value for a row that never produced a served model snapshot. */\nexport const UNKNOWN_MODEL = 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - successful rows MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n * - a failed, cancelled, incomplete, or otherwise unknown row may use\n * `UNKNOWN_MODEL` when no served model was observed. This is an explicit\n * absence marker, not a fabricated snapshot.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: successful rows require a served model snapshot.\n // Non-success rows may carry the explicit absence marker when execution\n // stopped before a model identity was observed.\n if (\n !modelHasSnapshot(obj.model as string) &&\n !(obj.model === UNKNOWN_MODEL && obj.terminalOutcome !== 'succeeded')\n ) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD', or '${UNKNOWN_MODEL}' for a non-success row without a served model)`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n // A record must never read as a total it cannot support, so an uncaptured\n // cost carries no number at all. A matrix `CellResult` keeps its known\n // subtotal instead, because a cost ceiling must charge the part it can see;\n // converting one to the other drops that subtotal.\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;AAiDA,MAAa,gBAAgB;;;;;;;;;AAuM7B,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAK5C,IACE,CAAC,iBAAiB,IAAI,KAAe,KACrC,EAAE,IAAI,UAAA,aAA2B,IAAI,oBAAoB,cAEzD,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,4EAA4E,cAAc,kDAC9G,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAMF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
|
|
1
|
+
{"version":3,"file":"run-record-BC0ebuRP.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\n/** Explicit model value for a row that never produced a served model snapshot. */\nexport const UNKNOWN_MODEL = 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - successful rows MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n * - a failed, cancelled, incomplete, or otherwise unknown row may use\n * `UNKNOWN_MODEL` when no served model was observed. This is an explicit\n * absence marker, not a fabricated snapshot.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: successful rows require a served model snapshot.\n // Non-success rows may carry the explicit absence marker when execution\n // stopped before a model identity was observed.\n if (\n !modelHasSnapshot(obj.model as string) &&\n !(obj.model === UNKNOWN_MODEL && obj.terminalOutcome !== 'succeeded')\n ) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD', or '${UNKNOWN_MODEL}' for a non-success row without a served model)`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n // A record must never read as a total it cannot support, so an uncaptured\n // cost carries no number at all. A matrix `CellResult` keeps its known\n // subtotal instead, because a cost ceiling must charge the part it can see;\n // converting one to the other drops that subtotal.\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;AAiDA,MAAa,gBAAgB;;;;;;;;;AAuM7B,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAK5C,IACE,CAAC,iBAAiB,IAAI,KAAe,KACrC,EAAE,IAAI,UAAA,aAA2B,IAAI,oBAAoB,cAEzD,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,4EAA4E,cAAc,kDAC9G,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAMF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { c as ValidationError } from "./errors-DEE6u6ot.js";
|
|
2
|
-
import { r as AgentProfileCell } from "./agent-profile-cell-
|
|
2
|
+
import { r as AgentProfileCell } from "./agent-profile-cell-CTOZJUuE.js";
|
|
3
3
|
import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
4
|
-
import { o as FailureClass } from "./schema-
|
|
4
|
+
import { o as FailureClass } from "./schema-Bjgdsn73.js";
|
|
5
5
|
//#region src/run-record.d.ts
|
|
6
6
|
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
7
7
|
* combined train+test pool that the optimizer is allowed to read. */
|
|
@@ -244,4 +244,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
|
|
|
244
244
|
declare function modelHasSnapshot(model: string): boolean;
|
|
245
245
|
//#endregion
|
|
246
246
|
export { validateRunRecord as _, RunRecord as a, RunTaskFailure as c, UNKNOWN_MODEL as d, isRunRecord as f, runTaskScore as g, roundTripRunRecord as h, RunOutcome as i, RunTerminalOutcome as l, parseRunRecordSafe as m, RunCostProvenance as n, RunRecordValidationError as o, modelHasSnapshot as p, RunJudgeMetadata as r, RunSplitTag as s, JudgeScoresRecord as t, RunTokenUsage as u };
|
|
247
|
-
//# sourceMappingURL=run-record-
|
|
247
|
+
//# sourceMappingURL=run-record-VVy4T9OW.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-record-
|
|
1
|
+
{"version":3,"file":"run-record-VVy4T9OW.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;;cAGC;UAEI;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;;;;UAsB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAgPnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
|
|
@@ -198,9 +198,7 @@ type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'to
|
|
|
198
198
|
declare const FAILURE_CLASSES: readonly FailureClass[];
|
|
199
199
|
declare function isLlmSpan(s: Span): s is LlmSpan;
|
|
200
200
|
declare function isToolSpan(s: Span): s is ToolSpan;
|
|
201
|
-
declare function isRetrievalSpan(s: Span): s is RetrievalSpan;
|
|
202
201
|
declare function isJudgeSpan(s: Span): s is JudgeSpan;
|
|
203
|
-
declare function isSandboxSpan(s: Span): s is SandboxSpan;
|
|
204
202
|
//#endregion
|
|
205
|
-
export { TraceEvent as C,
|
|
206
|
-
//# sourceMappingURL=schema-
|
|
203
|
+
export { TraceEvent as C, isToolSpan as E, ToolSpan as S, isLlmSpan as T, Span as _, FAILURE_CLASSES as a, SpanStatus as b, JudgeSpan as c, RetrievalSpan as d, Run as f, SandboxSpan as g, RunStatus as h, EventKind as i, LlmSpan as l, RunOutcome as m, BudgetLedgerEntry as n, FailureClass as o, RunLayer as p, BudgetSpec as r, GenericSpan as s, Artifact as t, Message as u, SpanBase as v, isJudgeSpan as w, TRACE_SCHEMA_VERSION as x, SpanKind as y };
|
|
204
|
+
//# sourceMappingURL=schema-Bjgdsn73.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"schema-
|
|
1
|
+
{"version":3,"file":"schema-Bjgdsn73.d.ts","names":[],"sources":["../src/trace/schema.ts"],"mappings":";;;;;;;;;;;;cAYa;KAID;UAEK;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,eAAe;EACf;;;;;;;;;KAUU;UAEK;EACf;;;;;;;;;;;EAWA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;EAEA,iBAAiB;;EAEjB;;;EAGA;;EAEA;;EAEA;;EAEA,QAAQ;EACR;EACA;EACA,QAAQ;EACR,UAAU;EACV,SAAS;;EAET,OAAO;;KAKG;KAEA;UAEK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,SAAS;EACT;;EAEA,aAAa;;UAGE;EACf;EACA;EACA;;EAEA,SAAS;IAAQ;IAAqB;IAAc;;;UAGrC,gBAAgB;EAC/B;EACA;EACA,UAAU;EACV;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;UAGe,iBAAiB;EAChC;EACA;EACA;;EAEA;EACA;EACA;;UAGe,sBAAsB;EACrC;EACA;EACA,MAAM;IAAQ;IAAe;IAAe;;;UAG7B,kBAAkB;EACjC;EACA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe,oBAAoB;EACnC;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;UAGe,oBAAoB;EACnC;;KAGU,OAAO,UAAU,WAAW,gBAAgB,YAAY,cAAc;KAItE;UAUK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA,SAAS;;UAKM;EACf;EACA,iBAAiB;EACjB;EACA;EACA;EACA;EACA;;EAEA;;UAKe;EACf;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;KAKU;cAqCC,0BAA0B;iBAwCvB,UAAU,GAAG,OAAO,KAAK;iBAGzB,WAAW,GAAG,OAAO,KAAK;iBAG1B,YAAY,GAAG,OAAO,KAAK"}
|
|
@@ -54,14 +54,11 @@ declare const ROLLOUT_ROLES: readonly RolloutRole[];
|
|
|
54
54
|
/** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
|
|
55
55
|
type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
|
|
56
56
|
declare const ROLLOUT_SPLITS: readonly RolloutSplit[];
|
|
57
|
-
/** Splits that may ship in training exports. Everything else is fail-closed excluded. */
|
|
58
|
-
declare const TRAINABLE_SPLITS: readonly RolloutSplit[];
|
|
59
57
|
declare function isTrainableSplit(split: RolloutSplit): boolean;
|
|
60
58
|
/** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
|
|
61
59
|
type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
|
|
62
60
|
declare const ROLLOUT_CAPTURES: readonly RolloutCapture[];
|
|
63
61
|
type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
|
|
64
|
-
declare const CHAT_ROLES: readonly ChatRole[];
|
|
65
62
|
interface ChatToolCall {
|
|
66
63
|
id: string;
|
|
67
64
|
type: 'function';
|
|
@@ -404,5 +401,5 @@ declare function assertMinted(value: unknown, context?: string): MintedRolloutLi
|
|
|
404
401
|
/** `assertMinted` over a batch, naming the offending index in the error. */
|
|
405
402
|
declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
|
|
406
403
|
//#endregion
|
|
407
|
-
export {
|
|
408
|
-
//# sourceMappingURL=schema-
|
|
404
|
+
export { assertMinted as C, isRolloutLine as D, gateGamedOutcome as E, isTrainableSplit as O, ToolDef as S, assertRolloutLine as T, RolloutProvenance as _, MintedRolloutLine as a, RolloutStep as b, ROLLOUT_ROLES as c, RolloutArtifacts as d, RolloutCapture as f, RolloutPolicy as g, RolloutOutcome as h, GatedEvidence as i, validateRolloutLine as k, ROLLOUT_SCHEMA as l, RolloutLine as m, ChatRole as n, MintedRolloutOutcome as o, RolloutCostBlock as p, ChatToolCall as r, ROLLOUT_CAPTURES as s, ChatMessage as t, ROLLOUT_SPLITS as u, RolloutRole as v, assertMintedLines as w, RolloutTask as x, RolloutSplit as y };
|
|
405
|
+
//# sourceMappingURL=schema-BzWDXhOR.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"schema-BzWDXhOR.d.ts","names":[],"sources":["../src/rollout/schema.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cAmDa;;KAGD;cACC,wBAAwB;;KAUzB;cACC,yBAAyB;iBAItB,iBAAiB,OAAO;;KAK5B;cACC,2BAA2B;KAM5B;UAGK;EACf;EACA;EACA;IACE;;IAEA;;;UAIa;EACf,MAAM;EACN;;EAEA;EACA,aAAa;;EAEb;EACA;;;;;;;;EAQA;;UAGe;EACf;EACA;IACE;IACA;IACA,aAAa;;;;;;;;UASA;EACf;EACA;;EAEA;;EAEA;EACA;EACA;;;;;EAUA;;EAEA;;EAEA;;;;;;EAMA;;UAOe;;EAEf;EACA;EACA,OAAO;;EAEP;;EAEA;;UAGe;;EAEf;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;;;;EAKf;;EAEA;;EAEA;;EAEA,SAAS;EACT;EACA;EACA;;;;;;;;;;;;;EAaA;;;;;;;;;;;;;;;;;;;;;EAqBA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;;;EAKA;;UAGe;EACf;EACA;;EAEA;;;;;;;UAQe;;EAEf,UAAU;;EAEV;;;;;;;;;EASA;;UAGe;EACf;EACA,SAAS;;;;;;;EAOT;;;;;;;EAOA,iBAAiB;;UAGF;EACf,eAAe;EACf;;EAEA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,MAAM;EACN,MAAM;EACN,QAAQ;;EAER,UAAU;EACV,WAAW;;EAEX,QAAQ;EACR,SAAS;EACT,MAAM;EACN,WAAW;EACX,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6GE,iBAAiB,MAAM,cAAc;iBAqDrC,oBAAoB;iBAiNpB,kBACd,gBACA,2BACS,SAAS;iBAOJ,cAAc,iBAAiB,SAAS;;;;;;cAa1C;;;;;;UAOG,6BAA6B;EAC5C;;;;;;;;;;;;;;;;;;;;;;KAuBU,oBAAoB,KAAK;YACzB;EACV,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsCK,aAAa,gBAAgB,mBAA2B;;iBAiBxD,kBACd,4BACA,mBACC"}
|
|
@@ -816,6 +816,6 @@ function assertMintedLines(values, context = "rollout line") {
|
|
|
816
816
|
return values.map((value, i) => assertMinted(value, `${context} [${i}]`));
|
|
817
817
|
}
|
|
818
818
|
//#endregion
|
|
819
|
-
export {
|
|
819
|
+
export { gatedEvidenceOf as _, assertMinted as a, assertRolloutLine as c, isTrainableSplit as d, validateRolloutLine as f, gateErrors as g, GATE_POLICIES as h, ROLLOUT_SPLITS as i, gateGamedOutcome as l, GATE_CHECK_IDS as m, ROLLOUT_ROLES as n, assertMintedLines as o, GATE_CHECKS as p, ROLLOUT_SCHEMA as r, assertRewardGate as s, ROLLOUT_CAPTURES as t, isRolloutLine as u, undeclaredStepPayload as v };
|
|
820
820
|
|
|
821
|
-
//# sourceMappingURL=schema-
|
|
821
|
+
//# sourceMappingURL=schema-C1aaAxTf.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"schema-C1aaAxTf.js","names":[],"sources":["../src/rollout/gate-checks.ts","../src/rollout/schema.ts"],"sourcesContent":["/**\n * THE canonical list of anti-Goodhart gate checks, plus the TOTAL policy every\n * entry point has to declare over it.\n *\n * Four rounds of adversarial review found the same defect four times, and it\n * was never the check itself: it was the COMPOSITION. `validateRolloutLine`\n * composed one check, `assertMinted` composed two, `assertRewardGate` composed\n * two of the three, `assertGateReport` composed its own pair — each by hand, in\n * its own file. So a check added to the package applied wherever its author\n * happened to remember, and the guard that forgot it looked exactly like the\n * guard that didn't. The last leak was literally that: `assertRewardGate`\n * composed `reward-relationship` + `gated-evidence` and not `unscreened-reward`,\n * so a never-screened positive reward that `assertMinted` correctly REFUSED\n * walked through all four waist exporters at full value.\n *\n * The fix is to make hand-composition impossible rather than to add a third\n * call to the two places that had two:\n *\n * - `GATE_CHECKS` is a TOTAL map over `GateCheckId`. A new id with no\n * implementation does not compile.\n * - `GatePolicy` is a TOTAL map over `GateCheckId`. Every entry point\n * declares one, so a new id makes EVERY entry point's policy a type error\n * until it is wired. Wiring it means writing `enforced`, `repairedBy(...)`\n * or `omittedBecause(...)` — and the last two force a written reason, so a\n * silent gap is not expressible.\n * - every check carries a `tripwire`: the minimal outcome it must refuse.\n * `gate-checks.test.ts` feeds each tripwire to every entry point that\n * declares `enforced` and requires a rejection, so wiring a check to the\n * wrong disposition is a TEST failure even when it type-checks.\n *\n * Adding a check is therefore: append the id, write the check, and the compiler\n * enumerates every place that has to decide about it.\n */\n\nimport type { GatedEvidence, RolloutOutcome, RolloutStep } from './schema'\n\n/**\n * Every gate check in the package, in the order they are applied.\n *\n * Order is load-bearing only for which message a caller sees first: a line that\n * trips two checks reports the earlier one, and `reward-relationship` is first\n * because it is the invariant the other three protect.\n */\nexport const GATE_CHECK_IDS = [\n 'reward-relationship',\n 'gated-evidence',\n 'undeclared-step-payload',\n 'unscreened-reward',\n] as const\n\nexport type GateCheckId = (typeof GATE_CHECK_IDS)[number]\n\n/**\n * An outcome as it reaches a check.\n *\n * Deliberately accepts a raw record as well as the typed shape: the checks are\n * the RUNTIME half of the gate, and the callers they exist for — JSON off a\n * ledger, a plain-JavaScript consumer of the published package — arrive with no\n * types at all. `Partial` because a tripwire states only the fields it trips on.\n */\nexport type GateCheckedOutcome = Partial<RolloutOutcome> | Readonly<Record<string, unknown>>\n\n/**\n * What a gate check reads: the reward-bearing surface of ONE LINE.\n *\n * For three rounds the subject was the OUTCOME alone, and that assumption is\n * what produced the next leak rather than any missing check: `steps[]` sits on\n * the LINE, outside `outcome`, so a per-step reward on a gated line was read by\n * no check at all while `toRewardRows` copied it out verbatim — through the\n * MINTED door, not merely the raw one. Widening the subject is what makes\n * \"somewhere else on the line\" a place the checks can see.\n *\n * `outcome` is REQUIRED, and that is the point: a bare `RolloutOutcome` is then\n * not assignable to a subject, so every call site that used to pass one is a\n * COMPILE error until it passes the line instead. A subject with an optional\n * `outcome` would have let the old call sites keep compiling while silently\n * checking nothing — the exact failure this module exists to make impossible.\n */\nexport interface GateSubject {\n outcome: GateCheckedOutcome\n /** The line's trajectory steps, when it carries any. */\n steps?: unknown\n}\n\n/** One untyped field read off the outcome, so every check narrows from the same place. */\nconst field = (subject: GateSubject, name: string): unknown =>\n (subject.outcome as Readonly<Record<string, unknown>>)?.[name]\n\nexport interface GateCheck {\n id: GateCheckId\n /** One sentence: what this check refuses. */\n refuses: string\n /** One dotted-path message per defect; `[]` when the line is clean. */\n errors: (subject: GateSubject) => string[]\n /**\n * Every minimal subject that MUST trip `errors` — the executable form of\n * `refuses`, and the reason a check cannot be added without being provable.\n * The calibration test feeds each one to every entry point declaring\n * `enforced`.\n *\n * A LIST rather than one case: a check that refuses two distinct populations\n * (a positive reward AND a reward it cannot read as a number) proved able to\n * hold for the first while silently passing the second, so each population\n * states its own tripwire and each is exercised separately.\n */\n tripwires: GateSubject[]\n}\n\n// ---------------------------------------------------------------------------\n// Total readers. Every \"is this positive?\" / \"is this populated?\" question the\n// checks ask goes through one of these.\n// ---------------------------------------------------------------------------\n\n/**\n * How the gate reads one `outcome.reward`.\n *\n * TOTAL over `unknown`, with an explicit `unreadable` case, because the\n * recurring defect across four rounds was a hand-written type test whose\n * NEGATIVE branch silently passed: `typeof reward !== 'number' || !(reward > 0)`\n * returned \"clean\" for `reward: \"0.95\"`, so a plain-JavaScript producer that\n * stringifies its numbers — the exact population the runtime backstop exists\n * for — shipped a gamed run at full value through three waist exporters.\n *\n * A value the gate cannot read as a number is not evidence the reward is\n * cleared; it is the absence of that evidence, and the gate fails closed on it.\n */\nexport type RewardReading =\n /** `null`/`undefined` — a labeled gap, which is never a positive signal. */\n | { kind: 'absent' }\n /** A number at or below zero: the gate's own verdict, applied. */\n | { kind: 'cleared' }\n /** A number above zero, `Infinity` and the 5e-324 denormal included. */\n | { kind: 'positive'; shown: string }\n /** Anything the wire format does not permit here, plus `NaN`. */\n | { kind: 'unreadable'; shown: string }\n\nexport function readReward(value: unknown): RewardReading {\n if (value === null || value === undefined) return { kind: 'absent' }\n if (typeof value === 'number') {\n if (Number.isNaN(value)) return { kind: 'unreadable', shown: 'NaN' }\n return value > 0 ? { kind: 'positive', shown: String(value) } : { kind: 'cleared' }\n }\n return {\n kind: 'unreadable',\n shown: `${JSON.stringify(value) ?? String(value)} (${typeof value})`,\n }\n}\n\nconst isPlainRecord = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null && !Array.isArray(value)\n\n/**\n * Whether a reward-derived field carries anything at all.\n *\n * The previous emptiness test was `Object.keys(value).length > 0`, which reads\n * `[]` for a NUMBER, a BOOLEAN and the empty string — so `metrics: 0.95` on a\n * gated line counted as empty and shipped verbatim in the verifiers format's\n * per-rubric score dict. `Object.keys` being total over primitives is not the\n * same property as being CORRECT over them.\n *\n * The rule that has no such hole: only `undefined`, `null` and a record with no\n * keys are empty. Everything else is payload, whatever its type.\n */\nexport function payloadIsPopulated(value: unknown): boolean {\n if (value === undefined || value === null) return false\n if (isPlainRecord(value)) return Object.keys(value).length > 0\n return true\n}\n\n/**\n * Every key `tangle.rollout.v1` declares on a step, as a TOTAL map over the\n * interface: a field added to `RolloutStep` and not to this list does not\n * compile, so the list cannot silently fall behind the schema it mirrors.\n *\n * This is the producer's OWN classification, which is the only partition\n * `gatedEvidenceOf` accepts — a key-name heuristic (\"strip anything matching\n * /reward|score/\") holds until someone names a field `credit` and the next\n * per-step signal ships at full value.\n */\nconst DECLARED_STEP_KEYS: { readonly [K in keyof Required<RolloutStep>]: true } = {\n kind: true,\n name: true,\n input: true,\n output: true,\n status: true,\n durationMs: true,\n llm_call_count: true,\n prompt_token_ids: true,\n completion_token_ids: true,\n logprobs: true,\n}\n\n/**\n * The part of `steps` the wire format has no name for — a per-step reward, a\n * per-step score, whatever a producer decided to hang there.\n *\n * `undefined` when the steps carry nothing undeclared.\n */\nexport function undeclaredStepPayload(steps: unknown): unknown {\n if (steps === undefined || steps === null) return undefined\n // The format says `steps` is a list. Anything else is wholly unclassified.\n if (!Array.isArray(steps)) return steps\n const extras: Array<Record<string, unknown>> = []\n let found = false\n for (const step of steps) {\n const extra: Record<string, unknown> = {}\n if (isPlainRecord(step)) {\n for (const [key, value] of Object.entries(step)) {\n if (key in DECLARED_STEP_KEYS) continue\n extra[key] = value\n found = true\n }\n } else if (step !== undefined && step !== null) {\n // A step that is not even a record carries no declared key to keep.\n extras.push({ step } as Record<string, unknown>)\n found = true\n continue\n }\n extras.push(extra)\n }\n return found ? extras : undefined\n}\n\n/** `steps` projected down to the keys the wire format declares. */\nexport function declaredSteps(steps: unknown): RolloutStep[] | undefined {\n if (!Array.isArray(steps)) return undefined\n const kept: RolloutStep[] = []\n for (const step of steps) {\n if (!isPlainRecord(step)) continue\n const projected: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(step)) {\n if (key in DECLARED_STEP_KEYS) projected[key] = value\n }\n kept.push(projected as unknown as RolloutStep)\n }\n return kept\n}\n\n/**\n * THE anti-Goodhart invariant, checked as a RELATIONSHIP between two fields\n * rather than as two independent type checks.\n *\n * Everything upstream of a training export is allowed to be wrong; this is the\n * one thing that cannot be. `realness_gated: true` means the run faked its\n * success signal, so its reward is a fabrication, and a fabrication above zero\n * is precisely what a trainer would learn to reproduce. Validating only that\n * `reward` is a number and `realness_gated` is a boolean is what let a line\n * claiming `{reward: 0.95, realness_gated: true}` validate clean and walk\n * through every exporter.\n */\nconst rewardRelationship: GateCheck = {\n id: 'reward-relationship',\n refuses: 'a positive — or unreadable — reward on a line the authenticity screen flagged as gamed',\n tripwires: [\n { outcome: { reward: 0.95, realness_gated: true } },\n // The second population, stated separately because the check held for the\n // first while passing this one for a full round.\n { outcome: { reward: '0.95' as unknown as number, realness_gated: true } },\n ],\n errors: (subject) => {\n if (field(subject, 'realness_gated') !== true) return []\n const reading = readReward(field(subject, 'reward'))\n if (reading.kind === 'absent' || reading.kind === 'cleared') return []\n if (reading.kind === 'unreadable') {\n return [\n `outcome.reward: ${reading.shown} with outcome.realness_gated: true — the wire format ` +\n 'declares this field `number | null`, so the gate cannot read this value as a number ' +\n 'and cannot establish that the reward was cleared. On a run flagged as gamed that is ' +\n 'not evidence of a zero, it is the absence of it, and a consumer that coerces the ' +\n 'value (`\"0.95\"`, `\"2\"`) reads a positive reward off a faked success. Emit a number ' +\n 'or null — `trainingReward` / `trainingScore` from `rollout/reward.ts` produce one.',\n ]\n }\n return [\n `outcome.reward: ${reading.shown} with outcome.realness_gated: true — a run flagged as ` +\n 'gamed may not carry a positive reward. The anti-Goodhart gate forces the reward to 0 (a ' +\n 'real verdict: the gate decided) before the line is written, so a fine-tune cannot learn ' +\n 'from a faked success. Derive the reward with `trainingReward` / `trainingScore` from ' +\n '`rollout/reward.ts`, or drop the line.',\n ]\n },\n}\n\n/**\n * The reward-bearing outcome fields that are NOT the scalar: the numbers the\n * reward was computed from, and the verdict record that claimed it.\n *\n * Returned as one block rather than filtered key-by-key. A key-name heuristic\n * (\"zero anything matching `layer.*` or `/score/`\") is the same defect shape as\n * the line-oriented regex the AST score guard replaced: it holds until someone\n * names a metric `pass_fraction`, and the next reward-shaped key ships at full\n * value. The producer's OWN classification — \"this is the scalar, that is\n * everything else\" — is the only partition that cannot be out-guessed.\n */\nexport function gatedEvidenceOf(subject: GateSubject): GatedEvidence | undefined {\n const evidence: GatedEvidence = {}\n // `payloadIsPopulated`, not an inline key count: a plain-JavaScript caller can\n // hand any value here, and a non-record `metrics` is still reward-derived\n // payload. See that function for the primitives an `Object.keys` test missed.\n const metrics = field(subject, 'metrics')\n if (payloadIsPopulated(metrics)) evidence.metrics = metrics as Record<string, unknown>\n const verdict = field(subject, 'verdict')\n if (verdict !== null && verdict !== undefined) evidence.verdict = verdict\n const steps = undeclaredStepPayload(subject.steps)\n if (steps !== undefined) evidence.steps = steps\n return evidence.metrics === undefined && evidence.verdict === undefined && steps === undefined\n ? undefined\n : evidence\n}\n\n/**\n * The invariant is about the OUTCOME, not about one field of it.\n *\n * `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`\n * with no gate, so a gated run exported `reward: 0` (correct) while the\n * deterministic per-layer scores that reward was COMPUTED FROM — the `layer.*`\n * keys `rl/verifiable-reward.ts` calls the RL training signal — shipped at 1.0,\n * in the top-level `metrics` dict of the Prime Intellect verifiers format, which\n * IS that format's per-rubric score dict. `verdict` leaks the same way into\n * `toRftItem`'s `reference.verdict`, where a grader author reads\n * `resolved: true` off a run that faked it.\n */\nconst gatedEvidence: GateCheck = {\n id: 'gated-evidence',\n refuses: 'the numbers a fabricated reward was computed from, riding along at reward 0',\n tripwires: [\n { outcome: { reward: 0, realness_gated: true, metrics: { 'layer.tests': 1 } } },\n // A PRIMITIVE `metrics`: `Object.keys(0.95)` is `[]`, so the emptiness test\n // this check used to run reported it clean and the verifiers format shipped\n // it as its per-rubric score dict.\n { outcome: { reward: 0, realness_gated: true, metrics: 0.95 as unknown as never } },\n ],\n errors: (subject) => {\n if (field(subject, 'realness_gated') !== true) return []\n const evidence = gatedEvidenceOf(subject)\n if (evidence === undefined) return []\n // Each check reports only the fields it owns; `steps` is `undeclared-step-payload`.\n if (evidence.metrics === undefined && evidence.verdict === undefined) return []\n const path = evidence.metrics !== undefined ? 'outcome.metrics' : 'outcome.verdict'\n return [\n `${path} is populated with outcome.realness_gated: true — a run flagged as gamed may not ` +\n 'ship the numbers its fabricated reward was computed from, even at reward 0. The ' +\n 'per-layer scores in `metrics` ARE the reward signal in the verifiers format, and ' +\n '`verdict` is the record that claimed the success. Mint through `assertMinted`, which ' +\n 'relocates both to `provenance.gated_evidence` (see `gateGamedOutcome`), or drop the line.',\n ]\n },\n}\n\n/**\n * A reward-bearing field hung on `steps[]`, where the gate was not looking.\n *\n * The first three checks all read `outcome`, so the whole apparatus was blind to\n * anything a producer wrote elsewhere on the line — and `toRewardRows` emits\n * `steps` verbatim beside the scalar it just forced to 0. A generated corpus\n * found it in twenty cases: a gated line carrying\n * `steps: [{kind, name, reward: 0.86}]` exported `{\"reward\": 0, \"steps\":\n * [{\"reward\": 0.86}]}`, which is the per-step credit assignment of a run that\n * faked its success, at full value, through the MINTED door.\n *\n * `RolloutStep` declares ten keys and none of them is a reward, so the partition\n * needs no judgement call: whatever the format does not name is unclassified\n * payload, and on a gated line unclassified payload is exactly what the gate\n * refuses. `assertMinted` relocates it rather than rejecting, like the sibling\n * check, so an already-published ledger stays readable.\n */\nconst undeclaredStepPayloadCheck: GateCheck = {\n id: 'undeclared-step-payload',\n refuses: 'per-step reward signal hung on a gated line’s steps, outside every field the gate read',\n tripwires: [\n {\n outcome: { reward: 0, realness_gated: true },\n steps: [{ kind: 'tool', name: 'edit', reward: 0.9 }],\n },\n ],\n errors: (subject) => {\n if (field(subject, 'realness_gated') !== true) return []\n const extra = undeclaredStepPayload(subject.steps)\n if (extra === undefined) return []\n return [\n `steps[] carries fields \\`${ROLLOUT_SCHEMA_NAME}\\` does not declare (${JSON.stringify(extra).slice(0, 120)}) ` +\n 'with outcome.realness_gated: true — a per-step reward is training signal exactly like ' +\n 'the scalar, and the exporters copy `steps` through verbatim, so zeroing `outcome.reward` ' +\n 'while leaving it in place ships the gamed run’s step-level credit assignment at full ' +\n 'value. Mint through `assertMinted`, which projects the steps down to the declared keys ' +\n 'and relocates the rest to `provenance.gated_evidence.steps` (see `gateGamedOutcome`).',\n ]\n },\n}\n\n/** Named here rather than imported, so `gate-checks` stays free of schema cycles. */\nconst ROLLOUT_SCHEMA_NAME = 'tangle.rollout.v1'\n\n/**\n * A positive reward whose producer DECLARED that no authenticity screen exists.\n *\n * `realness_gated: false` is the screen's VERDICT, so writing it with no screen\n * behind it claims \"we looked and nothing fired\" about a reward nobody looked\n * at. `realness_screened: false` is the producer saying so out loud, and a\n * positive reward carrying it is exactly the signal the gate exists to qualify\n * with nothing having qualified it.\n */\nconst unscreenedReward: GateCheck = {\n id: 'unscreened-reward',\n refuses: 'a positive reward whose producer declares that no authenticity screen ever ran',\n tripwires: [\n { outcome: { reward: 1, realness_gated: false, realness_screened: false } },\n {\n outcome: {\n reward: '2' as unknown as number,\n realness_gated: false,\n realness_screened: false,\n },\n },\n ],\n errors: (subject) => {\n if (field(subject, 'realness_screened') !== false) return []\n const reading = readReward(field(subject, 'reward'))\n if (reading.kind === 'absent' || reading.kind === 'cleared') return []\n return [\n `outcome.reward: ${reading.shown} with outcome.realness_screened: false — this producer declares ` +\n 'that NO authenticity screen ran on this reward, so nothing has established the success ' +\n 'is real, and an unscreened positive reward is exactly the signal the anti-Goodhart gate ' +\n 'exists to qualify. Screen the run and write the verdict (`rolloutRewardFields` from a ' +\n '`RunRecord` carrying `outcome.realness`), or emit `reward: null` — an unqualified ' +\n 'verdict is a labeled gap, not a measured success.',\n ]\n },\n}\n\n/**\n * The registry. Total over `GateCheckId`, so an id with no check does not\n * compile, and `GATE_CHECK_IDS` stays the single enumeration everything\n * iterates.\n */\nexport const GATE_CHECKS: { readonly [K in GateCheckId]: GateCheck } = {\n 'reward-relationship': rewardRelationship,\n 'gated-evidence': gatedEvidence,\n 'undeclared-step-payload': undeclaredStepPayloadCheck,\n 'unscreened-reward': unscreenedReward,\n}\n\n/**\n * What ONE entry point does about ONE check.\n *\n * `repair` and `omit` both carry a mandatory sentence, which is the mechanism\n * that keeps a legitimate omission distinguishable from a forgotten one: you\n * cannot skip a check without writing down why, and the reasons are readable\n * side by side in `GATE_POLICIES`.\n */\nexport type GateCheckDisposition =\n | { readonly kind: 'enforce' }\n /** Resolved by TRANSFORMING the line instead of rejecting it; `by` names the function. */\n | { readonly kind: 'repair'; readonly by: string }\n /** Deliberately not applied here; `because` states the reason. */\n | { readonly kind: 'omit'; readonly because: string }\n\nexport const enforced: GateCheckDisposition = { kind: 'enforce' }\nexport const repairedBy = (by: string): GateCheckDisposition => ({ kind: 'repair', by })\nexport const omittedBecause = (because: string): GateCheckDisposition => ({\n kind: 'omit',\n because,\n})\n\n/** Total over `GateCheckId`: a new check makes every policy literal a type error. */\nexport type GatePolicy = { readonly [K in GateCheckId]: GateCheckDisposition }\n\n/**\n * Every entry point that decides about the gate, and what it decides.\n *\n * Read this as the package's gate policy in one screen. The four entry points\n * are not interchangeable — a validator that rejects, a mint funnel that\n * repairs, a runtime backstop for untyped callers, and a release certifier over\n * emitted rows — and the dispositions say which is which.\n */\nexport const GATE_POLICIES = {\n /**\n * The schema validator. Rejects the reward relationship and NOTHING ELSE, on\n * purpose: it runs on every line read off disk, and the other two conditions\n * describe artifacts that already exist.\n */\n validateRolloutLine: {\n // Stays a REJECTION rather than a transformation. A caller emitting\n // `{reward: 0.95, realness_gated: true}` is a producer defect and has to\n // fail loudly; laundering it into `reward: 0` here would hide the producer,\n // which is the actual bug.\n 'reward-relationship': enforced,\n 'gated-evidence': omittedBecause(\n 'a ledger written before the relocation existed carries `metrics` on its gated lines; ' +\n 'rejecting those would make every already-published artifact unreadable. The condition ' +\n 'is REPAIRED at `assertMinted` (`gateGamedOutcome`) instead, which is the only door into ' +\n 'the training path, so the artifact stays readable and the leak still closes.',\n ),\n 'undeclared-step-payload': omittedBecause(\n 'identical reasoning to `gated-evidence`, one field over: a foreign or pre-unification ' +\n 'ledger may carry producer-invented keys on its steps, and refusing to READ those files ' +\n 'buys nothing the relocation at `assertMinted` does not already buy. Promotion into ' +\n 'training is where it closes.',\n ),\n 'unscreened-reward': omittedBecause(\n 'supervision-journal rows legitimately carry an unscreened positive reward (there is no ' +\n '`RunRecord.outcome.realness` behind them) and must stay writable, readable and ' +\n 'reportable. Only PROMOTION into training is closed, at `assertMinted`.',\n ),\n },\n /**\n * The mint funnel — the single door every `MintedRolloutLine` passes. Validates\n * first (so the reward relationship has already been rejected), then refuses\n * what cannot be repaired, then repairs what can.\n */\n assertMinted: {\n // Re-run after `assertRolloutLine`, which already rejected it. Cheap, and it\n // makes this entry point independently complete rather than complete only\n // because of what it happens to call first.\n 'reward-relationship': enforced,\n 'gated-evidence': repairedBy('gateGamedOutcome'),\n 'undeclared-step-payload': repairedBy('gateGamedOutcome'),\n 'unscreened-reward': enforced,\n },\n /**\n * The runtime backstop, and the one entry point with no license to omit\n * anything: it exists for callers the type system never saw (plain JavaScript\n * handing an object literal to a published exporter), so a check it skips is a\n * check that does not run at all for them. This is where the fourth leak was.\n */\n assertRewardGate: {\n 'reward-relationship': enforced,\n 'gated-evidence': enforced,\n 'undeclared-step-payload': enforced,\n 'unscreened-reward': enforced,\n },\n /**\n * The release certifier. Same checks, measured over the rows a release is\n * ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`\n * in `release/gate-report.ts`, which is the second total map this policy\n * drives.\n */\n assertGateReport: {\n 'reward-relationship': enforced,\n 'gated-evidence': enforced,\n 'undeclared-step-payload': enforced,\n 'unscreened-reward': enforced,\n },\n} as const satisfies Record<string, GatePolicy>\n\n/** Every entry point that declares a gate policy. */\nexport type GateEntryPoint = keyof typeof GATE_POLICIES\n\n/**\n * Run the checks one entry point enforces. The ONLY way an entry point should\n * obtain gate errors — hand-composing two of the three is the bug this module\n * exists to remove.\n */\nexport function gateErrors(subject: GateSubject, policy: GatePolicy): string[] {\n const errors: string[] = []\n for (const id of GATE_CHECK_IDS) {\n if (policy[id].kind !== 'enforce') continue\n errors.push(...GATE_CHECKS[id].errors(subject))\n }\n return errors\n}\n","/**\n * `tangle.rollout.v1` — THE canonical rollout serialization, owned by\n * agent-eval. One JSONL line per agent invocation (a solo eval run, a\n * supervisor episode, a worker session, a proposer shot, a judge call, an\n * analyst pass), labeled with its task/split coordinates and a single\n * scalar reward, carrying the FULL message transcript inline.\n *\n * This schema is the reconciliation of two prior producers:\n * - agent-eval's RunRecord-joined rollout rows (PR #410): identity,\n * provenance hashes, the realness gate travelling into the reward,\n * trace-derived steps.\n * - the bench rollout-ledger (agent-runtime PR #591): the wire shape —\n * role, task.split/rep, parent_rollout_id, policy provenance, capture\n * provenance, inline canonical chat-with-tools messages.\n * Where the two conflicted, RunRecord-derived semantics won; the wire\n * field names follow the ledger (snake_case). See `docs/rollout.md` for\n * the field-by-field decision table.\n *\n * Messages are inlined — never referenced — because every harness store a\n * rollout can be recovered from is mutable or garbage-collected. A line\n * must stay a complete training/eval example on its own.\n *\n * `outcome.reward` is THE single scalar (null = no verdict exists — a\n * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart\n * flag: a gated line must never export as a positive training example.\n *\n * That last sentence is enforced here, by `validateRolloutLine`, not merely\n * documented. Validating `reward` and `realness_gated` independently — each a\n * well-typed field, their COMBINATION unchecked — is what let a line claiming\n * `{reward: 0.95, realness_gated: true}` validate clean and walk into every\n * training export. The relationship between the two IS the invariant, so it is\n * checked where every other structural claim about a line is checked.\n *\n * The invariant is about the OUTCOME, not about one field of it. Zeroing\n * `reward` while `outcome.metrics` still carried the per-layer scores that\n * reward was computed from exported the gamed signal anyway, in the dict the\n * verifiers format reads as its per-rubric scores. So `gateGamedOutcome`\n * transforms the whole outcome once, at `assertMinted` — the funnel every\n * minted line passes — and the reward-bearing components are relocated to\n * `provenance.gated_evidence`, which no exporter projects.\n *\n * WHICH checks each door applies is not decided in this file. `./gate-checks`\n * owns the canonical list and the total per-entry-point policy; the three doors\n * below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call\n * `gateErrors` with their declared policy, so a check added to that list applies\n * here without anyone editing this file, and a check deliberately skipped has to\n * name itself there.\n */\n\nimport { declaredSteps, GATE_POLICIES, gatedEvidenceOf, gateErrors } from './gate-checks'\n\nexport const ROLLOUT_SCHEMA = 'tangle.rollout.v1'\n\n/** `agent` = a solo evaluation run (no multi-agent topology). */\nexport type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst'\nexport const ROLLOUT_ROLES: readonly RolloutRole[] = [\n 'agent',\n 'supervisor',\n 'worker',\n 'proposer',\n 'judge',\n 'analyst',\n]\n\n/** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */\nexport type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary'\nexport const ROLLOUT_SPLITS: readonly RolloutSplit[] = ['search', 'dev', 'holdout', 'canary']\n/** Splits that may ship in training exports. Everything else is fail-closed excluded. */\nconst TRAINABLE_SPLITS: readonly RolloutSplit[] = ['search']\n\nexport function isTrainableSplit(split: RolloutSplit): boolean {\n return TRAINABLE_SPLITS.includes(split)\n}\n\n/** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */\nexport type RolloutCapture = 'mint' | 'settle-time' | 'backfill'\nexport const ROLLOUT_CAPTURES: readonly RolloutCapture[] = ['mint', 'settle-time', 'backfill']\n\n// ---------------------------------------------------------------------------\n// Canonical message format — OpenAI chat-with-tools, full fidelity.\n// ---------------------------------------------------------------------------\n\nexport type ChatRole = 'system' | 'user' | 'assistant' | 'tool'\nconst CHAT_ROLES: readonly ChatRole[] = ['system', 'user', 'assistant', 'tool']\n\nexport interface ChatToolCall {\n id: string\n type: 'function'\n function: {\n name: string\n /** JSON-encoded argument object, exactly as the model emitted it. */\n arguments: string\n }\n}\n\nexport interface ChatMessage {\n role: ChatRole\n content: string | null\n /** Reasoning/thinking channel where the harness captured it (full fidelity). */\n reasoning_content?: string\n tool_calls?: ChatToolCall[]\n /** Required on role:\"tool\" — the ChatToolCall this result answers. */\n tool_call_id?: string\n name?: string\n /**\n * Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN\n * from another trajectory's context, not produced by the agent on this line.\n * The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training\n * on it teaches the model to author text it never authored, and credits this\n * run for another one's work. Absent = false (authored here).\n */\n is_copied_context?: boolean\n}\n\nexport interface ToolDef {\n type: 'function'\n function: {\n name: string\n description?: string\n parameters?: Record<string, unknown>\n }\n}\n\n/**\n * Compact trace-span projection (llm/tool step) carried alongside the\n * conversation when the line was minted from a trace. Optional: lines\n * recovered from harness stores have no span structure.\n */\nexport interface RolloutStep {\n kind: string\n name: string\n /** llm: last-message summary · tool: stringified args. Scrubbed. */\n input?: string\n /** llm: output text · tool: stringified result. Scrubbed. */\n output?: string\n status?: 'ok' | 'error'\n durationMs?: number\n // The four fields below are lifted verbatim from Harbor ATIF's per-step\n // `metrics` (see `interchange/harbor.ts`); the wire names stay snake_case to\n // match that spec exactly, which is why they differ from `durationMs`.\n // All optional and never back-filled: an absent field means \"not captured\",\n // which is not the same claim as an empty array.\n /**\n * LLM inferences this span represents. 0 = deterministic dispatch with no\n * model call — distinct from absent, which means the producer did not track it.\n */\n llm_call_count?: number\n /** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */\n prompt_token_ids?: number[]\n /** Exact completion tokenization; aligns index-wise with `logprobs`. */\n completion_token_ids?: number[]\n /**\n * Per-completion-token log probabilities under the sampling policy. Required\n * for off-policy correction (importance weighting) when the rollout was\n * generated by a policy other than the one being trained.\n */\n logprobs?: number[]\n}\n\n// ---------------------------------------------------------------------------\n// Ledger line sections.\n// ---------------------------------------------------------------------------\n\nexport interface RolloutTask {\n /** Benchmark/suite id (e.g. \"swe-bench-verified\") or the experiment id. */\n suite: string\n instance_id: string\n split: RolloutSplit\n /** Sampling seed the campaign pinned; null = not recorded. */\n seed: number | null\n /** Replicate index (0-based). */\n rep: number\n}\n\nexport interface RolloutPolicy {\n /** Harness that drove the invocation (e.g. \"opencode\", \"claude\", \"pi-loops\"). */\n harness: string | null\n harness_version: string | null\n model: string | null\n provider: string | null\n /** Commit of the agent profile / candidate under evaluation. */\n profile_commit: string | null\n /** sha256 of the effective prompt (post-steering), when recorded. */\n prompt_hash?: string | null\n /** sha256 of the effective run config, when recorded. */\n config_hash?: string | null\n /** Canonical agent-profile cell identity, when the run carries one. */\n agent_profile_cell_id?: string | null\n /** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */\n sampling: Record<string, unknown> | null\n}\n\nexport interface RolloutOutcome {\n /**\n * THE single scalar training signal — the official verdict.\n * null = no verdict exists for this invocation (a labeled gap, never 0).\n */\n reward: number | null\n /** Where the reward came from (judge id; \"/inherited\" = parent episode's). */\n reward_source: string | null\n /** Raw judge verdict record, verbatim. */\n verdict: unknown\n /** Everything that is NOT the scalar reward. */\n metrics: Record<string, unknown>\n is_completed: boolean\n is_truncated: boolean\n error: string | null\n /**\n * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked\n * its success signal. `true` requires `reward` to be 0 or null — the\n * validator rejects the line otherwise — and the line never qualifies for\n * SFT. Required on the wire: a line that does not state the flag does not\n * validate, so no producer can dodge the gate by omitting it.\n *\n * `true` ALSO requires `metrics` to be empty and `verdict` to be null: the\n * numbers the reward was computed from are relocated to\n * `provenance.gated_evidence` by `gateGamedOutcome`. See that function for\n * why zeroing the scalar alone was not enough.\n */\n realness_gated: boolean\n /**\n * Whether an authenticity SCREEN ever RAN on this reward — a different claim\n * from `realness_gated`, which is the screen's VERDICT.\n *\n * `realness_gated: false` reads as \"we looked and nothing fired\". A producer\n * with no screen at all was emitting exactly that, so a never-screened reward\n * was indistinguishable on the wire from a screened-clean one, and the whole\n * anti-Goodhart apparatus silently treated the first as the second. The two\n * claims are now separable:\n *\n * - `true` — a screen ran; `realness_gated` is its verdict.\n * - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).\n * `assertMinted` REFUSES such a line when its reward is above\n * zero: an unscreened positive reward is precisely the signal\n * the gate exists to qualify, and nothing has qualified it.\n * - absent — not stated. Pre-unification ledgers land here, as does a\n * `RunRecord` carrying no `outcome.realness` at all. Absent is\n * read as \"unknown\", never as `false` (which would refuse most\n * of the existing corpus) and never as `true`.\n */\n realness_screened?: boolean\n}\n\nexport interface RolloutCostBlock {\n usd: number | null\n tokens_in: number | null\n tokens_out: number | null\n tokens_reasoning: number | null\n cache_read: number | null\n cache_write: number | null\n wall_s: number | null\n /**\n * Total LLM inferences across the invocation (ATIF `llm_call_count`,\n * aggregated). Optional and additive: absent = not tracked, never 0.\n */\n llm_call_count?: number | null\n}\n\nexport interface RolloutArtifacts {\n patch_path: string | null\n run_dir: string | null\n /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */\n transcript_ref: string | null\n}\n\n/**\n * The reward-bearing half of a GATED line's outcome, moved off `outcome` and\n * parked here verbatim. Diagnostics, never training input — see\n * `gateGamedOutcome`.\n */\nexport interface GatedEvidence {\n /** `outcome.metrics` exactly as the producer measured it. */\n metrics?: Record<string, unknown>\n /** `outcome.verdict` verbatim — the judge record that claimed the success. */\n verdict?: unknown\n /**\n * The per-step fields `tangle.rollout.v1` does not declare, parked here when\n * the gate projected `steps[]` down to the schema's own key set.\n *\n * A per-step reward is training signal exactly like the scalar, and `steps`\n * rides through `toRewardRows` verbatim — so a gated line was shipping its\n * step-level credit assignment at full value beside a `reward` of 0.\n */\n steps?: unknown\n}\n\nexport interface RolloutProvenance {\n captured_at: string\n capture: RolloutCapture\n /**\n * Why this line is incomplete. Required when `messages` is empty (the\n * transcript could not be recovered); also set by interchange importers to\n * name a MISSING LABEL — an imported trajectory carries no verdict, so\n * `outcome.reward` is null and this says why.\n */\n gap?: string\n /**\n * Present only on a realness-gated line: the outcome fields the gate\n * relocated, kept so an auditor can still see WHY the run was gated and what\n * it claimed. Deliberately OUTSIDE `outcome`, because every training exporter\n * reads `outcome` and none reads `provenance`.\n */\n gated_evidence?: GatedEvidence\n}\n\nexport interface RolloutLine {\n schema: typeof ROLLOUT_SCHEMA\n rollout_id: string\n /** Spawning invocation within the same episode (worker → supervisor). */\n parent_rollout_id: string | null\n run_id: string\n /** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */\n experiment_id: string | null\n /** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */\n candidate_id: string | null\n /** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */\n generation: number | null\n /** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */\n candidate_index: number | null\n role: RolloutRole\n task: RolloutTask\n policy: RolloutPolicy\n /** Full transcript, inline. [] = gap line (see provenance.gap). */\n messages: ChatMessage[]\n tool_defs: ToolDef[]\n /** Trace-span projections, when minted from a trace. */\n steps?: RolloutStep[]\n outcome: RolloutOutcome\n cost: RolloutCostBlock\n artifacts: RolloutArtifacts\n provenance: RolloutProvenance\n}\n\n// ---------------------------------------------------------------------------\n// Validation — pure TS, no runtime schema dependency, mirroring the\n// run-record validator's fail-loud discipline. Returns [] when the value\n// is a valid RolloutLine; otherwise one dotted-path error per defect.\n// ---------------------------------------------------------------------------\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v)\n\nconst isNumberOrNull = (v: unknown): boolean => v === null || typeof v === 'number'\nconst isNumberArray = (v: unknown): boolean =>\n Array.isArray(v) && v.every((n) => typeof n === 'number' && Number.isFinite(n))\nconst isIntegerArray = (v: unknown): boolean =>\n Array.isArray(v) && v.every((n) => Number.isInteger(n))\nconst isStringOrNull = (v: unknown): boolean => v === null || typeof v === 'string'\nconst isIntegerOrNull = (v: unknown): boolean => v === null || Number.isInteger(v)\n\nfunction validateChatMessage(value: unknown, path: string, errors: string[]): void {\n if (!isRecord(value)) {\n errors.push(`${path}: not an object`)\n return\n }\n if (!CHAT_ROLES.includes(value.role as ChatRole))\n errors.push(`${path}.role: invalid role ${String(value.role)}`)\n if (!isStringOrNull(value.content)) errors.push(`${path}.content: must be string|null`)\n if (value.reasoning_content !== undefined && typeof value.reasoning_content !== 'string') {\n errors.push(`${path}.reasoning_content: must be string when present`)\n }\n if (value.tool_call_id !== undefined && typeof value.tool_call_id !== 'string') {\n errors.push(`${path}.tool_call_id: must be string when present`)\n }\n if (value.role === 'tool' && typeof value.tool_call_id !== 'string') {\n errors.push(`${path}.tool_call_id: required on role:\"tool\"`)\n }\n if (value.is_copied_context !== undefined && typeof value.is_copied_context !== 'boolean') {\n errors.push(`${path}.is_copied_context: must be boolean when present`)\n }\n if (value.tool_calls !== undefined) {\n if (!Array.isArray(value.tool_calls)) {\n errors.push(`${path}.tool_calls: must be an array when present`)\n } else {\n value.tool_calls.forEach((call, i) => {\n if (!isRecord(call) || typeof call.id !== 'string' || call.type !== 'function') {\n errors.push(`${path}.tool_calls[${i}]: must be {id, type:\"function\", function}`)\n return\n }\n const fn = call.function\n if (!isRecord(fn) || typeof fn.name !== 'string' || typeof fn.arguments !== 'string') {\n errors.push(\n `${path}.tool_calls[${i}].function: must be {name: string, arguments: string}`,\n )\n }\n })\n }\n }\n}\n\nfunction validateSection(\n value: unknown,\n path: string,\n fields: Array<[name: string, check: (v: unknown) => boolean, expect: string]>,\n errors: string[],\n): void {\n if (!isRecord(value)) {\n errors.push(`${path}: not an object`)\n return\n }\n for (const [name, check, expect] of fields) {\n if (!check(value[name])) errors.push(`${path}.${name}: expected ${expect}`)\n }\n}\n\n/**\n * THE anti-Goodhart gate, applied to the WHOLE outcome as a TRANSFORMATION.\n *\n * Two prior rounds enforced the gate as a CHECK ON ONE FIELD at N call sites,\n * and each round the next reward-bearing field leaked. The one that shipped:\n * `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`\n * with no gate, so a gated run exported `reward: 0` (correct) while the\n * deterministic per-layer scores that reward was COMPUTED FROM — the\n * `layer.*` keys `rl/verifiable-reward.ts` calls the RL training signal —\n * shipped at 1.0, in the top-level `metrics` dict of the Prime Intellect\n * verifiers format, which IS that format's per-rubric score dict. `verdict`\n * leaks the same way into `toRftItem`'s `reference.verdict`, where a grader\n * author reads `resolved: true` off a run that faked it.\n *\n * So the rule is no longer \"zero the field we remembered\". It is: if the gate\n * fired, the outcome that leaves here carries NOTHING positive that was derived\n * from the reward, whichever field a present or future exporter decides to\n * read. `reward` is already forced to 0 upstream (`trainingReward`) and\n * REJECTED here if it is not; `metrics` and `verdict` are relocated.\n *\n * WHERE they go, and why relocation rather than deletion: zeroing destroys the\n * audit trail that shows why the run was gated and what it claimed, which is\n * the row an auditor most wants and the labeled example a gaming DETECTOR\n * trains on. `provenance.gated_evidence` keeps every byte, at a path no\n * training exporter reads — all four release configs and every `rl/exporters`\n * shape project from `outcome`, `messages`, `cost` and `task`; none projects\n * `provenance`. Auditability preserved, training signal removed, and a future\n * exporter that reads a field nobody thought of is safe by construction because\n * the field is empty rather than because the exporter remembered to check.\n *\n * Idempotent: a second application finds nothing left to move and returns the\n * line unchanged, so re-minting a line read back off a ledger cannot clobber\n * the evidence it already carries.\n */\nexport function gateGamedOutcome(line: RolloutLine): RolloutLine {\n if (line.outcome.realness_gated !== true) return line\n const moved = gatedEvidenceOf(line)\n if (moved === undefined) return line\n const kept = line.provenance.gated_evidence\n const evidence: GatedEvidence = {}\n const metrics = { ...kept?.metrics, ...moved.metrics }\n if (Object.keys(metrics).length > 0) evidence.metrics = metrics\n if (moved.verdict !== undefined) evidence.verdict = moved.verdict\n else if (kept?.verdict !== undefined) evidence.verdict = kept.verdict\n if (moved.steps !== undefined) evidence.steps = moved.steps\n else if (kept?.steps !== undefined) evidence.steps = kept.steps\n // The trajectory itself STAYS: a gamed transcript is the labeled example a\n // gaming detector trains on, and the exporters are meant to ship it. Only the\n // fields the wire format does not declare come off, which is the same\n // partition `gatedEvidenceOf` applies to `metrics` — the producer's own,\n // never a key-name guess.\n const steps = moved.steps === undefined ? line.steps : declaredSteps(line.steps)\n return {\n ...line,\n ...(steps === undefined ? {} : { steps }),\n outcome: { ...line.outcome, metrics: {}, verdict: null },\n provenance: { ...line.provenance, gated_evidence: evidence },\n }\n}\n\n/**\n * EVERY gate check, for the export path — the runtime backstop.\n *\n * `validateRolloutLine` checks the reward relationship too, but an exporter\n * cannot afford to re-validate every field of every line, and more importantly\n * it is not the exporter's job to re-check the schema — it is its job never to\n * emit a signal it was told is fabricated. Thrown rather than filtered: an\n * exporter silently dropping a poisoned line would hide the producer that made\n * it, and the producer is the actual defect.\n *\n * This is the third layer, and it exists for exactly one caller: JavaScript.\n * The brand stops TypeScript callers at compile time and `assertMinted` fixes\n * data arriving from disk, but neither is present for a plain JS consumer of\n * the published package handing an object literal to `toRewardRows`. Which is\n * exactly why this entry point's policy omits nothing: a check it skips is a\n * check that never runs for those callers at all. It composed two of the three\n * for one round, and a never-screened positive reward walked through all four\n * waist exporters at full value. `GATE_POLICIES.assertRewardGate` is now the\n * only place that list is written down.\n */\nexport function assertRewardGate(line: RolloutLine, context: string): void {\n const errors = gateErrors(line, GATE_POLICIES.assertRewardGate)\n if (errors.length > 0) {\n throw new Error(`${context}: rollout ${line.rollout_id} — ${errors[0]}`)\n }\n}\n\nexport function validateRolloutLine(value: unknown): string[] {\n const errors: string[] = []\n if (!isRecord(value)) return ['line: not an object']\n\n if (value.schema !== ROLLOUT_SCHEMA) errors.push(`schema: expected \"${ROLLOUT_SCHEMA}\"`)\n if (typeof value.rollout_id !== 'string' || value.rollout_id.length === 0)\n errors.push('rollout_id: expected non-empty string')\n if (!isStringOrNull(value.parent_rollout_id))\n errors.push('parent_rollout_id: expected string|null')\n if (typeof value.run_id !== 'string' || value.run_id.length === 0)\n errors.push('run_id: expected non-empty string')\n if (!isStringOrNull(value.experiment_id)) errors.push('experiment_id: expected string|null')\n if (!isStringOrNull(value.candidate_id)) errors.push('candidate_id: expected string|null')\n if (!isIntegerOrNull(value.generation)) errors.push('generation: expected integer|null')\n if (!isIntegerOrNull(value.candidate_index)) errors.push('candidate_index: expected integer|null')\n if (!ROLLOUT_ROLES.includes(value.role as RolloutRole))\n errors.push(`role: invalid role ${String(value.role)}`)\n\n validateSection(\n value.task,\n 'task',\n [\n ['suite', (v) => typeof v === 'string' && v.length > 0, 'non-empty string'],\n ['instance_id', (v) => typeof v === 'string' && v.length > 0, 'non-empty string'],\n [\n 'split',\n (v) => ROLLOUT_SPLITS.includes(v as RolloutSplit),\n `one of ${ROLLOUT_SPLITS.join('|')}`,\n ],\n ['seed', isNumberOrNull, 'number|null'],\n ['rep', (v) => Number.isInteger(v), 'integer'],\n ],\n errors,\n )\n\n validateSection(\n value.policy,\n 'policy',\n [\n ['harness', isStringOrNull, 'string|null'],\n ['harness_version', isStringOrNull, 'string|null'],\n ['model', isStringOrNull, 'string|null'],\n ['provider', isStringOrNull, 'string|null'],\n ['profile_commit', isStringOrNull, 'string|null'],\n ['sampling', (v) => v === null || isRecord(v), 'object|null'],\n ],\n errors,\n )\n if (isRecord(value.policy)) {\n for (const key of ['prompt_hash', 'config_hash', 'agent_profile_cell_id'] as const) {\n if (value.policy[key] !== undefined && !isStringOrNull(value.policy[key])) {\n errors.push(`policy.${key}: expected string|null when present`)\n }\n }\n }\n\n if (!Array.isArray(value.messages)) {\n errors.push('messages: expected array')\n } else {\n for (const [i, m] of value.messages.entries()) validateChatMessage(m, `messages[${i}]`, errors)\n }\n\n if (!Array.isArray(value.tool_defs)) {\n errors.push('tool_defs: expected array')\n } else {\n value.tool_defs.forEach((d, i) => {\n if (\n !isRecord(d) ||\n d.type !== 'function' ||\n !isRecord(d.function) ||\n typeof d.function.name !== 'string'\n ) {\n errors.push(`tool_defs[${i}]: must be {type:\"function\", function:{name}}`)\n }\n })\n }\n\n if (value.steps !== undefined) {\n if (!Array.isArray(value.steps)) {\n errors.push('steps: expected array when present')\n } else {\n value.steps.forEach((s, i) => {\n if (!isRecord(s) || typeof s.kind !== 'string' || typeof s.name !== 'string') {\n errors.push(`steps[${i}]: must be {kind: string, name: string, …}`)\n return\n }\n // ATIF-lifted optionals: absent on every line written before they\n // existed, so they are checked only when present — old ledgers stay valid.\n if (s.llm_call_count !== undefined && !Number.isInteger(s.llm_call_count)) {\n errors.push(`steps[${i}].llm_call_count: expected integer when present`)\n }\n for (const key of ['prompt_token_ids', 'completion_token_ids'] as const) {\n if (s[key] !== undefined && !isIntegerArray(s[key])) {\n errors.push(`steps[${i}].${key}: expected integer[] when present`)\n }\n }\n if (s.logprobs !== undefined && !isNumberArray(s.logprobs)) {\n errors.push(`steps[${i}].logprobs: expected number[] when present`)\n }\n })\n }\n }\n\n validateSection(\n value.outcome,\n 'outcome',\n [\n ['reward', isNumberOrNull, 'number|null'],\n ['reward_source', isStringOrNull, 'string|null'],\n ['metrics', isRecord, 'object'],\n ['is_completed', (v) => typeof v === 'boolean', 'boolean'],\n ['is_truncated', (v) => typeof v === 'boolean', 'boolean'],\n ['error', isStringOrNull, 'string|null'],\n ],\n errors,\n )\n if (isRecord(value.outcome)) {\n if (!('verdict' in value.outcome)) errors.push('outcome.verdict: field required (may be null)')\n if (typeof value.outcome.realness_gated !== 'boolean') {\n errors.push('outcome.realness_gated: expected boolean')\n }\n if (\n value.outcome.realness_screened !== undefined &&\n typeof value.outcome.realness_screened !== 'boolean'\n ) {\n errors.push('outcome.realness_screened: expected boolean when present')\n }\n errors.push(\n ...gateErrors(\n { outcome: value.outcome, steps: value.steps },\n GATE_POLICIES.validateRolloutLine,\n ),\n )\n }\n\n validateSection(\n value.cost,\n 'cost',\n [\n ['usd', isNumberOrNull, 'number|null'],\n ['tokens_in', isNumberOrNull, 'number|null'],\n ['tokens_out', isNumberOrNull, 'number|null'],\n ['tokens_reasoning', isNumberOrNull, 'number|null'],\n ['cache_read', isNumberOrNull, 'number|null'],\n ['cache_write', isNumberOrNull, 'number|null'],\n ['wall_s', isNumberOrNull, 'number|null'],\n ],\n errors,\n )\n if (\n isRecord(value.cost) &&\n value.cost.llm_call_count !== undefined &&\n !(value.cost.llm_call_count === null || Number.isInteger(value.cost.llm_call_count))\n ) {\n errors.push('cost.llm_call_count: expected integer|null when present')\n }\n\n validateSection(\n value.artifacts,\n 'artifacts',\n [\n ['patch_path', isStringOrNull, 'string|null'],\n ['run_dir', isStringOrNull, 'string|null'],\n ['transcript_ref', isStringOrNull, 'string|null'],\n ],\n errors,\n )\n\n validateSection(\n value.provenance,\n 'provenance',\n [\n [\n 'captured_at',\n (v) => typeof v === 'string' && !Number.isNaN(Date.parse(v)),\n 'ISO-8601 timestamp',\n ],\n [\n 'capture',\n (v) => ROLLOUT_CAPTURES.includes(v as RolloutCapture),\n `one of ${ROLLOUT_CAPTURES.join('|')}`,\n ],\n ],\n errors,\n )\n if (isRecord(value.provenance)) {\n if (value.provenance.gap !== undefined && typeof value.provenance.gap !== 'string') {\n errors.push('provenance.gap: must be string when present')\n }\n const evidence = value.provenance.gated_evidence\n if (evidence !== undefined) {\n if (!isRecord(evidence)) {\n errors.push('provenance.gated_evidence: must be an object when present')\n } else if (evidence.metrics !== undefined && !isRecord(evidence.metrics)) {\n errors.push('provenance.gated_evidence.metrics: must be an object when present')\n }\n }\n }\n\n // A gap line must say WHY it is a gap; a full line must not carry a gap note.\n if (Array.isArray(value.messages) && isRecord(value.provenance)) {\n if (value.messages.length === 0 && typeof value.provenance.gap !== 'string') {\n errors.push('provenance.gap: required when messages is empty')\n }\n }\n\n return errors\n}\n\nexport function assertRolloutLine(\n value: unknown,\n context = 'rollout line',\n): asserts value is RolloutLine {\n const errors = validateRolloutLine(value)\n if (errors.length > 0) {\n throw new Error(`invalid ${context}:\\n ${errors.join('\\n ')}`)\n }\n}\n\nexport function isRolloutLine(value: unknown): value is RolloutLine {\n return validateRolloutLine(value).length === 0\n}\n\n// ---------------------------------------------------------------------------\n// The minted brand — the COMPILE-TIME half of the anti-Goodhart gate.\n// ---------------------------------------------------------------------------\n\n/**\n * Phantom property. `declare const` means it exists only in the type system:\n * nothing is written at runtime, so a branded line still serializes to exactly\n * the same JSON as a plain one.\n */\ndeclare const MINTED_ROLLOUT: unique symbol\n\n/**\n * A minted outcome states the gate verdict — it is not allowed to stay silent —\n * and, when that verdict is `true`, carries nothing else the reward was derived\n * from (`gateGamedOutcome` has run).\n */\nexport interface MintedRolloutOutcome extends RolloutOutcome {\n realness_gated: boolean\n}\n\n/**\n * A `RolloutLine` whose reward has been checked against the anti-Goodhart\n * invariant. The type every training-data exporter takes.\n *\n * Why a brand and not just the interface: `RolloutLine` is structural, so any\n * hand-built object literal of the right shape IS one — which is how a line\n * declaring `{reward: 0.95, realness_gated: true}` reached the exporters\n * despite them \"only accepting a minted line\". The phantom symbol makes the\n * type nominal: it cannot be produced by writing an object literal, only by\n * `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which\n * validates every line off disk), or an explicit, greppable `assertMinted`.\n *\n * Belt and braces on purpose. The brand closes first-party call sites at\n * COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger\n * files, foreign imports, JSON from another process) where types are absent.\n * Neither alone is enough.\n *\n * Assignable to `RolloutLine` in one direction only: readers, analysis, and\n * the ledger writer keep taking the plain type.\n */\nexport type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {\n readonly [MINTED_ROLLOUT]: true\n outcome: MintedRolloutOutcome\n}\n\n/**\n * Promote a line to the type the training exporters accept, applying the\n * anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch\n * — grep `assertMinted` to enumerate every place a line enters the training\n * path without coming from mint or a ledger.\n *\n * The gate runs HERE, once, rather than at each producer, because this is the\n * single funnel every minted line passes: `mintRolloutRows` calls it,\n * `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the\n * way out of a release, and a hand-built line has no other door. One\n * transformation at the funnel means an already-published ledger holding a\n * gated line with populated `metrics` is RE-GATED when it is read, instead of\n * being rejected (which would make every such artifact unreadable) or trusted\n * (which is the leak). Three steps, in this order:\n *\n * 1. VALIDATE the schema.\n * 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today\n * the reward relationship (which stays a REJECTION: a caller claiming\n * `{reward: 0.95, realness_gated: true}` is a producer defect and must fail\n * loudly, since laundering it into `reward: 0` here would hide the\n * producer) and a positive reward the producer declared it never screened.\n * 3. TRANSFORM the one check that policy marks `repair` — relocate the\n * reward's components off `outcome` (`gateGamedOutcome`), so no exporter\n * can leak them whichever field it reads.\n *\n * Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced\n * here the moment its disposition in that policy says so.\n *\n * Also normalizes the optional wire flag to an explicit boolean.\n * `realness_gated` is absent on pre-unification ledgers and absent means \"not\n * flagged\" per the schema, so filling it in states a claim the line was already\n * making, and makes the flag readable on every published row instead of most of\n * them. `realness_screened` is NOT filled in: absent means \"unknown\", and\n * inventing either value there would be the same overclaim this round removed.\n */\nexport function assertMinted(value: unknown, context = 'rollout line'): MintedRolloutLine {\n assertRolloutLine(value, context)\n const refused = gateErrors(value, GATE_POLICIES.assertMinted)\n if (refused.length > 0) {\n throw new Error(`invalid ${context}: rollout ${value.rollout_id} — ${refused[0]}`)\n }\n const gated = gateGamedOutcome(value)\n if (gated.outcome.realness_gated === undefined) {\n return {\n ...gated,\n outcome: { ...gated.outcome, realness_gated: false },\n } as MintedRolloutLine\n }\n return gated as MintedRolloutLine\n}\n\n/** `assertMinted` over a batch, naming the offending index in the error. */\nexport function assertMintedLines(\n values: readonly unknown[],\n context = 'rollout line',\n): MintedRolloutLine[] {\n return values.map((value, i) => assertMinted(value, `${context} [${i}]`))\n}\n"],"mappings":";;;;;;;;AA2CA,MAAa,iBAAiB;CAC5B;CACA;CACA;CACA;AACF;;AAqCA,MAAM,SAAS,SAAsB,SAClC,QAAQ,UAAgD;AAkD3D,SAAgB,WAAW,OAA+B;CACxD,IAAI,UAAU,QAAQ,UAAU,KAAA,GAAW,OAAO,EAAE,MAAM,SAAS;CACnE,IAAI,OAAO,UAAU,UAAU;EAC7B,IAAI,OAAO,MAAM,KAAK,GAAG,OAAO;GAAE,MAAM;GAAc,OAAO;EAAM;EACnE,OAAO,QAAQ,IAAI;GAAE,MAAM;GAAY,OAAO,OAAO,KAAK;EAAE,IAAI,EAAE,MAAM,UAAU;CACpF;CACA,OAAO;EACL,MAAM;EACN,OAAO,GAAG,KAAK,UAAU,KAAK,KAAK,OAAO,KAAK,EAAE,IAAI,OAAO,MAAM;CACpE;AACF;AAEA,MAAM,iBAAiB,UACrB,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;;;;;;;;;;;;;AAcrE,SAAgB,mBAAmB,OAAyB;CAC1D,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM,OAAO;CAClD,IAAI,cAAc,KAAK,GAAG,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,SAAS;CAC7D,OAAO;AACT;;;;;;;;;;;AAYA,MAAM,qBAA4E;CAChF,MAAM;CACN,MAAM;CACN,OAAO;CACP,QAAQ;CACR,QAAQ;CACR,YAAY;CACZ,gBAAgB;CAChB,kBAAkB;CAClB,sBAAsB;CACtB,UAAU;AACZ;;;;;;;AAQA,SAAgB,sBAAsB,OAAyB;CAC7D,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM,OAAO,KAAA;CAElD,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO;CAClC,MAAM,SAAyC,CAAC;CAChD,IAAI,QAAQ;CACZ,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAAiC,CAAC;EACxC,IAAI,cAAc,IAAI,GACpB,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAAI,GAAG;GAC/C,IAAI,OAAO,oBAAoB;GAC/B,MAAM,OAAO;GACb,QAAQ;EACV;OACK,IAAI,SAAS,KAAA,KAAa,SAAS,MAAM;GAE9C,OAAO,KAAK,EAAE,KAAK,CAA4B;GAC/C,QAAQ;GACR;EACF;EACA,OAAO,KAAK,KAAK;CACnB;CACA,OAAO,QAAQ,SAAS,KAAA;AAC1B;;AAGA,SAAgB,cAAc,OAA2C;CACvE,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,KAAA;CAClC,MAAM,OAAsB,CAAC;CAC7B,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,cAAc,IAAI,GAAG;EAC1B,MAAM,YAAqC,CAAC;EAC5C,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAAI,GAC5C,IAAI,OAAO,oBAAoB,UAAU,OAAO;EAElD,KAAK,KAAK,SAAmC;CAC/C;CACA,OAAO;AACT;;;;;;;;;;;;;AAcA,MAAM,qBAAgC;CACpC,IAAI;CACJ,SAAS;CACT,WAAW,CACT,EAAE,SAAS;EAAE,QAAQ;EAAM,gBAAgB;CAAK,EAAE,GAGlD,EAAE,SAAS;EAAE,QAAQ;EAA6B,gBAAgB;CAAK,EAAE,CAC3E;CACA,SAAS,YAAY;EACnB,IAAI,MAAM,SAAS,gBAAgB,MAAM,MAAM,OAAO,CAAC;EACvD,MAAM,UAAU,WAAW,MAAM,SAAS,QAAQ,CAAC;EACnD,IAAI,QAAQ,SAAS,YAAY,QAAQ,SAAS,WAAW,OAAO,CAAC;EACrE,IAAI,QAAQ,SAAS,cACnB,OAAO,CACL,mBAAmB,QAAQ,MAAM,geAMnC;EAEF,OAAO,CACL,mBAAmB,QAAQ,MAAM,wWAKnC;CACF;AACF;;;;;;;;;;;;AAaA,SAAgB,gBAAgB,SAAiD;CAC/E,MAAM,WAA0B,CAAC;CAIjC,MAAM,UAAU,MAAM,SAAS,SAAS;CACxC,IAAI,mBAAmB,OAAO,GAAG,SAAS,UAAU;CACpD,MAAM,UAAU,MAAM,SAAS,SAAS;CACxC,IAAI,YAAY,QAAQ,YAAY,KAAA,GAAW,SAAS,UAAU;CAClE,MAAM,QAAQ,sBAAsB,QAAQ,KAAK;CACjD,IAAI,UAAU,KAAA,GAAW,SAAS,QAAQ;CAC1C,OAAO,SAAS,YAAY,KAAA,KAAa,SAAS,YAAY,KAAA,KAAa,UAAU,KAAA,IACjF,KAAA,IACA;AACN;;;;;;;;;;;;;AAcA,MAAM,gBAA2B;CAC/B,IAAI;CACJ,SAAS;CACT,WAAW,CACT,EAAE,SAAS;EAAE,QAAQ;EAAG,gBAAgB;EAAM,SAAS,EAAE,eAAe,EAAE;CAAE,EAAE,GAI9E,EAAE,SAAS;EAAE,QAAQ;EAAG,gBAAgB;EAAM,SAAS;CAAyB,EAAE,CACpF;CACA,SAAS,YAAY;EACnB,IAAI,MAAM,SAAS,gBAAgB,MAAM,MAAM,OAAO,CAAC;EACvD,MAAM,WAAW,gBAAgB,OAAO;EACxC,IAAI,aAAa,KAAA,GAAW,OAAO,CAAC;EAEpC,IAAI,SAAS,YAAY,KAAA,KAAa,SAAS,YAAY,KAAA,GAAW,OAAO,CAAC;EAE9E,OAAO,CACL,GAFW,SAAS,YAAY,KAAA,IAAY,oBAAoB,kBAExD,2aAKV;CACF;AACF;;;;;;;;;;;;;;;;;;AAmBA,MAAM,6BAAwC;CAC5C,IAAI;CACJ,SAAS;CACT,WAAW,CACT;EACE,SAAS;GAAE,QAAQ;GAAG,gBAAgB;EAAK;EAC3C,OAAO,CAAC;GAAE,MAAM;GAAQ,MAAM;GAAQ,QAAQ;EAAI,CAAC;CACrD,CACF;CACA,SAAS,YAAY;EACnB,IAAI,MAAM,SAAS,gBAAgB,MAAM,MAAM,OAAO,CAAC;EACvD,MAAM,QAAQ,sBAAsB,QAAQ,KAAK;EACjD,IAAI,UAAU,KAAA,GAAW,OAAO,CAAC;EACjC,OAAO,CACL,4BAA4B,oBAAoB,uBAAuB,KAAK,UAAU,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,EAAE,6bAM7G;CACF;AACF;;AAGA,MAAM,sBAAsB;;;;;;AA4C5B,MAAa,cAA0D;CACrE,uBAAuB;CACvB,kBAAkB;CAClB,2BAA2B;CAC3B,qBAAqB;EApCrB,IAAI;EACJ,SAAS;EACT,WAAW,CACT,EAAE,SAAS;GAAE,QAAQ;GAAG,gBAAgB;GAAO,mBAAmB;EAAM,EAAE,GAC1E,EACE,SAAS;GACP,QAAQ;GACR,gBAAgB;GAChB,mBAAmB;EACrB,EACF,CACF;EACA,SAAS,YAAY;GACnB,IAAI,MAAM,SAAS,mBAAmB,MAAM,OAAO,OAAO,CAAC;GAC3D,MAAM,UAAU,WAAW,MAAM,SAAS,QAAQ,CAAC;GACnD,IAAI,QAAQ,SAAS,YAAY,QAAQ,SAAS,WAAW,OAAO,CAAC;GACrE,OAAO,CACL,mBAAmB,QAAQ,MAAM,idAMnC;EACF;CAYqB;AACvB;AAiBA,MAAa,WAAiC,EAAE,MAAM,UAAU;AAChE,MAAa,cAAc,QAAsC;CAAE,MAAM;CAAU;AAAG;AACtF,MAAa,kBAAkB,aAA2C;CACxE,MAAM;CACN;AACF;;;;;;;;;AAaA,MAAa,gBAAgB;;;;;;CAM3B,qBAAqB;EAKnB,uBAAuB;EACvB,kBAAkB,eAChB,iVAIF;EACA,2BAA2B,eACzB,8RAIF;EACA,qBAAqB,eACnB,8OAGF;CACF;;;;;;CAMA,cAAc;EAIZ,uBAAuB;EACvB,kBAAkB,WAAW,kBAAkB;EAC/C,2BAA2B,WAAW,kBAAkB;EACxD,qBAAqB;CACvB;;;;;;;CAOA,kBAAkB;EAChB,uBAAuB;EACvB,kBAAkB;EAClB,2BAA2B;EAC3B,qBAAqB;CACvB;;;;;;;CAOA,kBAAkB;EAChB,uBAAuB;EACvB,kBAAkB;EAClB,2BAA2B;EAC3B,qBAAqB;CACvB;AACF;;;;;;AAUA,SAAgB,WAAW,SAAsB,QAA8B;CAC7E,MAAM,SAAmB,CAAC;CAC1B,KAAK,MAAM,MAAM,gBAAgB;EAC/B,IAAI,OAAO,GAAG,CAAC,SAAS,WAAW;EACnC,OAAO,KAAK,GAAG,YAAY,GAAG,CAAC,OAAO,OAAO,CAAC;CAChD;CACA,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC7fA,MAAa,iBAAiB;AAI9B,MAAa,gBAAwC;CACnD;CACA;CACA;CACA;CACA;CACA;AACF;AAIA,MAAa,iBAA0C;CAAC;CAAU;CAAO;CAAW;AAAQ;;AAE5F,MAAM,mBAA4C,CAAC,QAAQ;AAE3D,SAAgB,iBAAiB,OAA8B;CAC7D,OAAO,iBAAiB,SAAS,KAAK;AACxC;AAIA,MAAa,mBAA8C;CAAC;CAAQ;CAAe;AAAU;AAO7F,MAAM,aAAkC;CAAC;CAAU;CAAQ;CAAa;AAAM;AAgQ9E,MAAM,YAAY,MAChB,OAAO,MAAM,YAAY,MAAM,QAAQ,CAAC,MAAM,QAAQ,CAAC;AAEzD,MAAM,kBAAkB,MAAwB,MAAM,QAAQ,OAAO,MAAM;AAC3E,MAAM,iBAAiB,MACrB,MAAM,QAAQ,CAAC,KAAK,EAAE,OAAO,MAAM,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;AAChF,MAAM,kBAAkB,MACtB,MAAM,QAAQ,CAAC,KAAK,EAAE,OAAO,MAAM,OAAO,UAAU,CAAC,CAAC;AACxD,MAAM,kBAAkB,MAAwB,MAAM,QAAQ,OAAO,MAAM;AAC3E,MAAM,mBAAmB,MAAwB,MAAM,QAAQ,OAAO,UAAU,CAAC;AAEjF,SAAS,oBAAoB,OAAgB,MAAc,QAAwB;CACjF,IAAI,CAAC,SAAS,KAAK,GAAG;EACpB,OAAO,KAAK,GAAG,KAAK,gBAAgB;EACpC;CACF;CACA,IAAI,CAAC,WAAW,SAAS,MAAM,IAAgB,GAC7C,OAAO,KAAK,GAAG,KAAK,sBAAsB,OAAO,MAAM,IAAI,GAAG;CAChE,IAAI,CAAC,eAAe,MAAM,OAAO,GAAG,OAAO,KAAK,GAAG,KAAK,8BAA8B;CACtF,IAAI,MAAM,sBAAsB,KAAA,KAAa,OAAO,MAAM,sBAAsB,UAC9E,OAAO,KAAK,GAAG,KAAK,gDAAgD;CAEtE,IAAI,MAAM,iBAAiB,KAAA,KAAa,OAAO,MAAM,iBAAiB,UACpE,OAAO,KAAK,GAAG,KAAK,2CAA2C;CAEjE,IAAI,MAAM,SAAS,UAAU,OAAO,MAAM,iBAAiB,UACzD,OAAO,KAAK,GAAG,KAAK,uCAAuC;CAE7D,IAAI,MAAM,sBAAsB,KAAA,KAAa,OAAO,MAAM,sBAAsB,WAC9E,OAAO,KAAK,GAAG,KAAK,iDAAiD;CAEvE,IAAI,MAAM,eAAe,KAAA,GACvB,IAAI,CAAC,MAAM,QAAQ,MAAM,UAAU,GACjC,OAAO,KAAK,GAAG,KAAK,2CAA2C;MAE/D,MAAM,WAAW,SAAS,MAAM,MAAM;EACpC,IAAI,CAAC,SAAS,IAAI,KAAK,OAAO,KAAK,OAAO,YAAY,KAAK,SAAS,YAAY;GAC9E,OAAO,KAAK,GAAG,KAAK,cAAc,EAAE,2CAA2C;GAC/E;EACF;EACA,MAAM,KAAK,KAAK;EAChB,IAAI,CAAC,SAAS,EAAE,KAAK,OAAO,GAAG,SAAS,YAAY,OAAO,GAAG,cAAc,UAC1E,OAAO,KACL,GAAG,KAAK,cAAc,EAAE,sDAC1B;CAEJ,CAAC;AAGP;AAEA,SAAS,gBACP,OACA,MACA,QACA,QACM;CACN,IAAI,CAAC,SAAS,KAAK,GAAG;EACpB,OAAO,KAAK,GAAG,KAAK,gBAAgB;EACpC;CACF;CACA,KAAK,MAAM,CAAC,MAAM,OAAO,WAAW,QAClC,IAAI,CAAC,MAAM,MAAM,KAAK,GAAG,OAAO,KAAK,GAAG,KAAK,GAAG,KAAK,aAAa,QAAQ;AAE9E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,SAAgB,iBAAiB,MAAgC;CAC/D,IAAI,KAAK,QAAQ,mBAAmB,MAAM,OAAO;CACjD,MAAM,QAAQ,gBAAgB,IAAI;CAClC,IAAI,UAAU,KAAA,GAAW,OAAO;CAChC,MAAM,OAAO,KAAK,WAAW;CAC7B,MAAM,WAA0B,CAAC;CACjC,MAAM,UAAU;EAAE,GAAG,MAAM;EAAS,GAAG,MAAM;CAAQ;CACrD,IAAI,OAAO,KAAK,OAAO,CAAC,CAAC,SAAS,GAAG,SAAS,UAAU;CACxD,IAAI,MAAM,YAAY,KAAA,GAAW,SAAS,UAAU,MAAM;MACrD,IAAI,MAAM,YAAY,KAAA,GAAW,SAAS,UAAU,KAAK;CAC9D,IAAI,MAAM,UAAU,KAAA,GAAW,SAAS,QAAQ,MAAM;MACjD,IAAI,MAAM,UAAU,KAAA,GAAW,SAAS,QAAQ,KAAK;CAM1D,MAAM,QAAQ,MAAM,UAAU,KAAA,IAAY,KAAK,QAAQ,cAAc,KAAK,KAAK;CAC/E,OAAO;EACL,GAAG;EACH,GAAI,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM;EACvC,SAAS;GAAE,GAAG,KAAK;GAAS,SAAS,CAAC;GAAG,SAAS;EAAK;EACvD,YAAY;GAAE,GAAG,KAAK;GAAY,gBAAgB;EAAS;CAC7D;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,SAAgB,iBAAiB,MAAmB,SAAuB;CACzE,MAAM,SAAS,WAAW,MAAM,cAAc,gBAAgB;CAC9D,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,MAAM,GAAG,QAAQ,YAAY,KAAK,WAAW,KAAK,OAAO,IAAI;AAE3E;AAEA,SAAgB,oBAAoB,OAA0B;CAC5D,MAAM,SAAmB,CAAC;CAC1B,IAAI,CAAC,SAAS,KAAK,GAAG,OAAO,CAAC,qBAAqB;CAEnD,IAAI,MAAM,WAAA,qBAA2B,OAAO,KAAK,qBAAqB,eAAe,EAAE;CACvF,IAAI,OAAO,MAAM,eAAe,YAAY,MAAM,WAAW,WAAW,GACtE,OAAO,KAAK,uCAAuC;CACrD,IAAI,CAAC,eAAe,MAAM,iBAAiB,GACzC,OAAO,KAAK,yCAAyC;CACvD,IAAI,OAAO,MAAM,WAAW,YAAY,MAAM,OAAO,WAAW,GAC9D,OAAO,KAAK,mCAAmC;CACjD,IAAI,CAAC,eAAe,MAAM,aAAa,GAAG,OAAO,KAAK,qCAAqC;CAC3F,IAAI,CAAC,eAAe,MAAM,YAAY,GAAG,OAAO,KAAK,oCAAoC;CACzF,IAAI,CAAC,gBAAgB,MAAM,UAAU,GAAG,OAAO,KAAK,mCAAmC;CACvF,IAAI,CAAC,gBAAgB,MAAM,eAAe,GAAG,OAAO,KAAK,wCAAwC;CACjG,IAAI,CAAC,cAAc,SAAS,MAAM,IAAmB,GACnD,OAAO,KAAK,sBAAsB,OAAO,MAAM,IAAI,GAAG;CAExD,gBACE,MAAM,MACN,QACA;EACE;GAAC;IAAU,MAAM,OAAO,MAAM,YAAY,EAAE,SAAS;GAAG;EAAkB;EAC1E;GAAC;IAAgB,MAAM,OAAO,MAAM,YAAY,EAAE,SAAS;GAAG;EAAkB;EAChF;GACE;IACC,MAAM,eAAe,SAAS,CAAiB;GAChD,UAAU,eAAe,KAAK,GAAG;EACnC;EACA;GAAC;GAAQ;GAAgB;EAAa;EACtC;GAAC;IAAQ,MAAM,OAAO,UAAU,CAAC;GAAG;EAAS;CAC/C,GACA,MACF;CAEA,gBACE,MAAM,QACN,UACA;EACE;GAAC;GAAW;GAAgB;EAAa;EACzC;GAAC;GAAmB;GAAgB;EAAa;EACjD;GAAC;GAAS;GAAgB;EAAa;EACvC;GAAC;GAAY;GAAgB;EAAa;EAC1C;GAAC;GAAkB;GAAgB;EAAa;EAChD;GAAC;IAAa,MAAM,MAAM,QAAQ,SAAS,CAAC;GAAG;EAAa;CAC9D,GACA,MACF;CACA,IAAI,SAAS,MAAM,MAAM,GAClB;OAAA,MAAM,OAAO;GAAC;GAAe;GAAe;EAAuB,GACtE,IAAI,MAAM,OAAO,SAAS,KAAA,KAAa,CAAC,eAAe,MAAM,OAAO,IAAI,GACtE,OAAO,KAAK,UAAU,IAAI,oCAAoC;CAAA;CAKpE,IAAI,CAAC,MAAM,QAAQ,MAAM,QAAQ,GAC/B,OAAO,KAAK,0BAA0B;MAEtC,KAAK,MAAM,CAAC,GAAG,MAAM,MAAM,SAAS,QAAQ,GAAG,oBAAoB,GAAG,YAAY,EAAE,IAAI,MAAM;CAGhG,IAAI,CAAC,MAAM,QAAQ,MAAM,SAAS,GAChC,OAAO,KAAK,2BAA2B;MAEvC,MAAM,UAAU,SAAS,GAAG,MAAM;EAChC,IACE,CAAC,SAAS,CAAC,KACX,EAAE,SAAS,cACX,CAAC,SAAS,EAAE,QAAQ,KACpB,OAAO,EAAE,SAAS,SAAS,UAE3B,OAAO,KAAK,aAAa,EAAE,8CAA8C;CAE7E,CAAC;CAGH,IAAI,MAAM,UAAU,KAAA,GAClB,IAAI,CAAC,MAAM,QAAQ,MAAM,KAAK,GAC5B,OAAO,KAAK,oCAAoC;MAEhD,MAAM,MAAM,SAAS,GAAG,MAAM;EAC5B,IAAI,CAAC,SAAS,CAAC,KAAK,OAAO,EAAE,SAAS,YAAY,OAAO,EAAE,SAAS,UAAU;GAC5E,OAAO,KAAK,SAAS,EAAE,2CAA2C;GAClE;EACF;EAGA,IAAI,EAAE,mBAAmB,KAAA,KAAa,CAAC,OAAO,UAAU,EAAE,cAAc,GACtE,OAAO,KAAK,SAAS,EAAE,gDAAgD;EAEzE,KAAK,MAAM,OAAO,CAAC,oBAAoB,sBAAsB,GAC3D,IAAI,EAAE,SAAS,KAAA,KAAa,CAAC,eAAe,EAAE,IAAI,GAChD,OAAO,KAAK,SAAS,EAAE,IAAI,IAAI,kCAAkC;EAGrE,IAAI,EAAE,aAAa,KAAA,KAAa,CAAC,cAAc,EAAE,QAAQ,GACvD,OAAO,KAAK,SAAS,EAAE,2CAA2C;CAEtE,CAAC;CAIL,gBACE,MAAM,SACN,WACA;EACE;GAAC;GAAU;GAAgB;EAAa;EACxC;GAAC;GAAiB;GAAgB;EAAa;EAC/C;GAAC;GAAW;GAAU;EAAQ;EAC9B;GAAC;IAAiB,MAAM,OAAO,MAAM;GAAW;EAAS;EACzD;GAAC;IAAiB,MAAM,OAAO,MAAM;GAAW;EAAS;EACzD;GAAC;GAAS;GAAgB;EAAa;CACzC,GACA,MACF;CACA,IAAI,SAAS,MAAM,OAAO,GAAG;EAC3B,IAAI,EAAE,aAAa,MAAM,UAAU,OAAO,KAAK,+CAA+C;EAC9F,IAAI,OAAO,MAAM,QAAQ,mBAAmB,WAC1C,OAAO,KAAK,0CAA0C;EAExD,IACE,MAAM,QAAQ,sBAAsB,KAAA,KACpC,OAAO,MAAM,QAAQ,sBAAsB,WAE3C,OAAO,KAAK,0DAA0D;EAExE,OAAO,KACL,GAAG,WACD;GAAE,SAAS,MAAM;GAAS,OAAO,MAAM;EAAM,GAC7C,cAAc,mBAChB,CACF;CACF;CAEA,gBACE,MAAM,MACN,QACA;EACE;GAAC;GAAO;GAAgB;EAAa;EACrC;GAAC;GAAa;GAAgB;EAAa;EAC3C;GAAC;GAAc;GAAgB;EAAa;EAC5C;GAAC;GAAoB;GAAgB;EAAa;EAClD;GAAC;GAAc;GAAgB;EAAa;EAC5C;GAAC;GAAe;GAAgB;EAAa;EAC7C;GAAC;GAAU;GAAgB;EAAa;CAC1C,GACA,MACF;CACA,IACE,SAAS,MAAM,IAAI,KACnB,MAAM,KAAK,mBAAmB,KAAA,KAC9B,EAAE,MAAM,KAAK,mBAAmB,QAAQ,OAAO,UAAU,MAAM,KAAK,cAAc,IAElF,OAAO,KAAK,yDAAyD;CAGvE,gBACE,MAAM,WACN,aACA;EACE;GAAC;GAAc;GAAgB;EAAa;EAC5C;GAAC;GAAW;GAAgB;EAAa;EACzC;GAAC;GAAkB;GAAgB;EAAa;CAClD,GACA,MACF;CAEA,gBACE,MAAM,YACN,cACA,CACE;EACE;GACC,MAAM,OAAO,MAAM,YAAY,CAAC,OAAO,MAAM,KAAK,MAAM,CAAC,CAAC;EAC3D;CACF,GACA;EACE;GACC,MAAM,iBAAiB,SAAS,CAAmB;EACpD,UAAU,iBAAiB,KAAK,GAAG;CACrC,CACF,GACA,MACF;CACA,IAAI,SAAS,MAAM,UAAU,GAAG;EAC9B,IAAI,MAAM,WAAW,QAAQ,KAAA,KAAa,OAAO,MAAM,WAAW,QAAQ,UACxE,OAAO,KAAK,6CAA6C;EAE3D,MAAM,WAAW,MAAM,WAAW;EAClC,IAAI,aAAa,KAAA,GACX;OAAA,CAAC,SAAS,QAAQ,GACpB,OAAO,KAAK,2DAA2D;QAClE,IAAI,SAAS,YAAY,KAAA,KAAa,CAAC,SAAS,SAAS,OAAO,GACrE,OAAO,KAAK,mEAAmE;EAAA;CAGrF;CAGA,IAAI,MAAM,QAAQ,MAAM,QAAQ,KAAK,SAAS,MAAM,UAAU,GACxD;MAAA,MAAM,SAAS,WAAW,KAAK,OAAO,MAAM,WAAW,QAAQ,UACjE,OAAO,KAAK,iDAAiD;CAAA;CAIjE,OAAO;AACT;AAEA,SAAgB,kBACd,OACA,UAAU,gBACoB;CAC9B,MAAM,SAAS,oBAAoB,KAAK;CACxC,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,MAAM,WAAW,QAAQ,OAAO,OAAO,KAAK,MAAM,GAAG;AAEnE;AAEA,SAAgB,cAAc,OAAsC;CAClE,OAAO,oBAAoB,KAAK,CAAC,CAAC,WAAW;AAC/C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAkFA,SAAgB,aAAa,OAAgB,UAAU,gBAAmC;CACxF,kBAAkB,OAAO,OAAO;CAChC,MAAM,UAAU,WAAW,OAAO,cAAc,YAAY;CAC5D,IAAI,QAAQ,SAAS,GACnB,MAAM,IAAI,MAAM,WAAW,QAAQ,YAAY,MAAM,WAAW,KAAK,QAAQ,IAAI;CAEnF,MAAM,QAAQ,iBAAiB,KAAK;CACpC,IAAI,MAAM,QAAQ,mBAAmB,KAAA,GACnC,OAAO;EACL,GAAG;EACH,SAAS;GAAE,GAAG,MAAM;GAAS,gBAAgB;EAAM;CACrD;CAEF,OAAO;AACT;;AAGA,SAAgB,kBACd,QACA,UAAU,gBACW;CACrB,OAAO,OAAO,KAAK,OAAO,MAAM,aAAa,OAAO,GAAG,QAAQ,IAAI,EAAE,EAAE,CAAC;AAC1E"}
|
|
@@ -54,16 +54,10 @@ function isLlmSpan(s) {
|
|
|
54
54
|
function isToolSpan(s) {
|
|
55
55
|
return s.kind === "tool";
|
|
56
56
|
}
|
|
57
|
-
function isRetrievalSpan(s) {
|
|
58
|
-
return s.kind === "retrieval";
|
|
59
|
-
}
|
|
60
57
|
function isJudgeSpan(s) {
|
|
61
58
|
return s.kind === "judge";
|
|
62
59
|
}
|
|
63
|
-
function isSandboxSpan(s) {
|
|
64
|
-
return s.kind === "sandbox";
|
|
65
|
-
}
|
|
66
60
|
//#endregion
|
|
67
|
-
export {
|
|
61
|
+
export { isToolSpan as a, isLlmSpan as i, TRACE_SCHEMA_VERSION as n, isJudgeSpan as r, FAILURE_CLASSES as t };
|
|
68
62
|
|
|
69
|
-
//# sourceMappingURL=schema-
|
|
63
|
+
//# sourceMappingURL=schema-k6ZBftVv.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"schema-
|
|
1
|
+
{"version":3,"file":"schema-k6ZBftVv.js","names":[],"sources":["../src/trace/schema.ts"],"sourcesContent":["/**\n * TraceSchema v1 — the canonical data model for agent-eval.\n *\n * Every score, every failure class, every pipeline in the framework is\n * a view over this data. Shape it once, live with it.\n *\n * Wire-compatible with OpenTelemetry span semantics (see trace/otel.ts)\n * but extended with agent-specific span kinds (llm, tool, retrieval,\n * judge, sandbox) and first-class BudgetLedger / Artifact / JudgeVerdict\n * entities that OTEL leaves as free-form attributes.\n */\n\nexport const TRACE_SCHEMA_VERSION = '1.0.0'\n\n// ── Run ──────────────────────────────────────────────────────────────\n\nexport type RunStatus = 'running' | 'completed' | 'failed' | 'aborted'\n\nexport interface BudgetSpec {\n tokens?: number\n wallMs?: number\n calls?: number\n usd?: number\n}\n\nexport interface RunOutcome {\n score?: number\n pass?: boolean\n failureClass?: FailureClass\n notes?: string\n}\n\n/**\n * Layer — optional classification in a nested build workflow.\n * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).\n * `app-build`: sandbox harness that compiled + tested the generated scaffold.\n * `app-runtime`: a run of the generated agent against a domain scenario.\n * `meta`: any meta-eval (judge replay, correlation analysis).\n */\nexport type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom'\n\nexport interface Run {\n runId: string\n /**\n * Stable identifier of the scenario being executed.\n *\n * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts\n * input WITHOUT this field, substituting a sensible default\n * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no\n * curated scenario to anchor to (runtime / operator / meta-eval runs). This\n * keeps the persisted shape unambiguous for downstream filters + aggregations\n * while removing the boilerplate of inventing placeholder ids at the call site.\n */\n scenarioId: string\n variantId?: string\n datasetVersion?: string\n /** Git SHA of agent code at run time. */\n codeSha?: string\n /** Hash of the prompt template + any system prompt. */\n promptSha?: string\n /** Model id + date + system-prompt hash, concatenated. */\n modelFingerprint?: string\n seed?: number\n /** Arbitrary environment markers (shell, docker version, tz). */\n envFingerprint?: Record<string, string>\n /** Version of the redaction rules applied to this run. */\n redactionVersion?: string\n /** Parent run in a nested build workflow. A builder run's children are\n * app-build runs; those children are app-runtime runs. */\n parentRunId?: string\n /** Stable project identifier — groups runs across chats + sessions. */\n projectId?: string\n /** Chat/conversation identifier within a project. */\n chatId?: string\n /** Layer classification — hint for aggregation; not enforced. */\n layer?: RunLayer\n startedAt: number\n endedAt?: number\n status: RunStatus\n outcome?: RunOutcome\n budget?: BudgetSpec\n /** Free-form labels for downstream grouping. */\n tags?: Record<string, string>\n}\n\n// ── Spans (hierarchical work units) ──────────────────────────────────\n\nexport type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom'\n\nexport type SpanStatus = 'ok' | 'error'\n\nexport interface SpanBase {\n spanId: string\n parentSpanId?: string\n runId: string\n kind: SpanKind\n name: string\n startedAt: number\n endedAt?: number\n status?: SpanStatus\n error?: string\n /** Anything not covered by typed fields. Kept deliberately free-form. */\n attributes?: Record<string, unknown>\n}\n\nexport interface Message {\n role: 'system' | 'user' | 'assistant' | 'tool'\n content: string\n tokens?: number\n /** Multi-modal content descriptors; blobs themselves live in Artifacts. */\n images?: Array<{ artifactId?: string; url?: string; mime?: string }>\n}\n\nexport interface LlmSpan extends SpanBase {\n kind: 'llm'\n model: string\n messages: Message[]\n output?: string\n inputTokens?: number\n /** All generated tokens, including the reasoning subset when present. */\n outputTokens?: number\n cachedTokens?: number\n cacheWriteTokens?: number\n /** Reasoning-token subset of `outputTokens`. */\n reasoningTokens?: number\n costUsd?: number\n finishReason?: string\n}\n\nexport interface ToolSpan extends SpanBase {\n kind: 'tool'\n toolName: string\n args: unknown\n /** False when the source observed the call but did not capture its arguments. */\n argsCaptured?: boolean\n result?: unknown\n latencyMs?: number\n}\n\nexport interface RetrievalSpan extends SpanBase {\n kind: 'retrieval'\n query: string\n hits: Array<{ docId: string; score: number; content?: string }>\n}\n\nexport interface JudgeSpan extends SpanBase {\n kind: 'judge'\n judgeId: string\n /** Span this judgment applies to. */\n targetSpanId: string\n dimension: string\n /** Numeric score (free-range; interpretation up to the judge). */\n score: number\n rationale?: string\n evidence?: string\n}\n\nexport interface SandboxSpan extends SpanBase {\n kind: 'sandbox'\n image?: string\n command?: string\n exitCode?: number\n testsTotal?: number\n testsPassed?: number\n stdoutHash?: string\n stderrHash?: string\n /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */\n wallMs?: number\n}\n\nexport interface GenericSpan extends SpanBase {\n kind: 'agent' | 'custom'\n}\n\nexport type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan\n\n// ── Events (point-in-time occurrences within a span) ─────────────────\n\nexport type EventKind =\n | 'log'\n | 'error'\n | 'budget_decrement'\n | 'budget_breach'\n | 'state_mutation'\n | 'policy_violation'\n | 'redaction_applied'\n | 'custom'\n\nexport interface TraceEvent {\n eventId: string\n runId: string\n spanId?: string\n kind: EventKind\n timestamp: number\n payload: Record<string, unknown>\n}\n\n// ── Budget ledger (running token/wall/call/$ accounting) ─────────────\n\nexport interface BudgetLedgerEntry {\n runId: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n remaining: number\n timestamp: number\n breached: boolean\n /** Span that triggered this entry, if any. */\n spanId?: string\n}\n\n// ── Artifacts (blobs addressed by hash) ──────────────────────────────\n\nexport interface Artifact {\n artifactId: string\n runId: string\n spanId?: string\n contentType: string\n sizeBytes: number\n /** sha256 in hex. */\n hash: string\n /** External storage URL (R2, S3, filesystem path). */\n storageUrl?: string\n /** Inline content for small blobs — keep under ~64KB. */\n inlineContent?: string\n}\n\n// ── Failure taxonomy ─────────────────────────────────────────────────\n\nexport type FailureClass =\n | 'success'\n | 'reasoning_error'\n | 'tool_selection_error'\n | 'tool_argument_error'\n | 'tool_recovery_failure'\n | 'hallucination'\n | 'instruction_following'\n | 'safety_refusal_miss'\n | 'policy_violation'\n | 'budget_exceeded'\n | 'format_drift'\n | 'permission_escalation'\n | 'pii_leak'\n | 'cost_overrun'\n | 'timeout'\n | 'sandbox_failure'\n | 'missing_user_data'\n | 'missing_domain_data'\n | 'missing_codebase_context'\n | 'missing_runtime_context'\n | 'missing_credentials'\n | 'missing_integration_connection'\n | 'missing_integration_scope'\n | 'integration_approval_required'\n | 'integration_auth_expired'\n | 'integration_provider_failure'\n | 'bad_integration_manifest'\n | 'unsafe_integration_write_denied'\n | 'stale_external_data'\n | 'bad_retrieval'\n | 'insufficient_evidence'\n | 'contradictory_evidence'\n | 'ambiguous_user_intent'\n | 'knowledge_readiness_blocked'\n | 'unknown'\n\nexport const FAILURE_CLASSES: readonly FailureClass[] = [\n 'success',\n 'reasoning_error',\n 'tool_selection_error',\n 'tool_argument_error',\n 'tool_recovery_failure',\n 'hallucination',\n 'instruction_following',\n 'safety_refusal_miss',\n 'policy_violation',\n 'budget_exceeded',\n 'format_drift',\n 'permission_escalation',\n 'pii_leak',\n 'cost_overrun',\n 'timeout',\n 'sandbox_failure',\n 'missing_user_data',\n 'missing_domain_data',\n 'missing_codebase_context',\n 'missing_runtime_context',\n 'missing_credentials',\n 'missing_integration_connection',\n 'missing_integration_scope',\n 'integration_approval_required',\n 'integration_auth_expired',\n 'integration_provider_failure',\n 'bad_integration_manifest',\n 'unsafe_integration_write_denied',\n 'stale_external_data',\n 'bad_retrieval',\n 'insufficient_evidence',\n 'contradictory_evidence',\n 'ambiguous_user_intent',\n 'knowledge_readiness_blocked',\n 'unknown',\n] as const\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nexport function isLlmSpan(s: Span): s is LlmSpan {\n return s.kind === 'llm'\n}\nexport function isToolSpan(s: Span): s is ToolSpan {\n return s.kind === 'tool'\n}\nexport function isJudgeSpan(s: Span): s is JudgeSpan {\n return s.kind === 'judge'\n}\n"],"mappings":";;;;;;;;;;;;AAYA,MAAa,uBAAuB;AA8PpC,MAAa,kBAA2C;CACtD;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAIA,SAAgB,UAAU,GAAuB;CAC/C,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,WAAW,GAAwB;CACjD,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,YAAY,GAAyB;CACnD,OAAO,EAAE,SAAS;AACpB"}
|