@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"run-record-BmSPWXJR.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - `model` MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: bare model aliases are not paper-grade.\n if (!modelHasSnapshot(obj.model as string)) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;;;;;;;;AAkPA,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAG5C,IAAI,CAAC,iBAAiB,IAAI,KAAe,GACvC,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,wEACpB,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAEF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
1
+ {"version":3,"file":"run-record-BvHPVS-i.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - `model` MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: bare model aliases are not paper-grade.\n if (!modelHasSnapshot(obj.model as string)) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;;;;;;;;AAkPA,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAG5C,IAAI,CAAC,iBAAiB,IAAI,KAAe,GACvC,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,wEACpB,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAEF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
@@ -1,6 +1,6 @@
1
- import { l as ValidationError } from "./errors-CKPfb2aH.js";
2
- import { r as AgentProfileCell } from "./agent-profile-cell-BOP-iA9Q.js";
3
- import { p as CostProvenance } from "./cost-ledger-Bv_e8XHY.js";
1
+ import { c as ValidationError } from "./errors-DEE6u6ot.js";
2
+ import { r as AgentProfileCell } from "./agent-profile-cell-BkcRDikH.js";
3
+ import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
4
4
  import { o as FailureClass } from "./schema-BtVldJ3T.js";
5
5
  //#region src/run-record.d.ts
6
6
  /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
@@ -239,4 +239,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
239
239
  declare function modelHasSnapshot(model: string): boolean;
240
240
  //#endregion
241
241
  export { RunRecord as a, RunTaskFailure as c, isRunRecord as d, modelHasSnapshot as f, validateRunRecord as g, runTaskScore as h, RunOutcome as i, RunTerminalOutcome as l, roundTripRunRecord as m, RunCostProvenance as n, RunRecordValidationError as o, parseRunRecordSafe as p, RunJudgeMetadata as r, RunSplitTag as s, JudgeScoresRecord as t, RunTokenUsage as u };
242
- //# sourceMappingURL=run-record-CF4Dwpxr.d.ts.map
242
+ //# sourceMappingURL=run-record-CKiihE6f.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-record-CF4Dwpxr.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;UAEK;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;UAmB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAuOnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
1
+ {"version":3,"file":"run-record-CKiihE6f.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;UAEK;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;UAmB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAuOnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
@@ -8,35 +8,6 @@ function combineAbortSignals(...signals) {
8
8
  return AbortSignal.any(active);
9
9
  }
10
10
  //#endregion
11
- //#region src/run-score.ts
12
- const DEFAULT_RUN_SCORE_WEIGHTS = {
13
- success: 4,
14
- goalProgress: 2,
15
- repoGroundedness: 1.5,
16
- driftPenalty: -1.5,
17
- toolUseQuality: 1,
18
- patchQuality: 1.25,
19
- testReality: 1.5,
20
- finalGate: 3,
21
- reviewerBlockers: -2,
22
- costUsd: -.2,
23
- wallSeconds: -.1
24
- };
25
- function aggregateRunScore(score, weights = {}) {
26
- const w = {
27
- ...DEFAULT_RUN_SCORE_WEIGHTS,
28
- ...weights
29
- };
30
- return w.success * clamp01(score.success) + w.goalProgress * clamp01(score.goalProgress) + w.repoGroundedness * clamp01(score.repoGroundedness) + w.driftPenalty * clamp01(score.driftPenalty) + w.toolUseQuality * clamp01(score.toolUseQuality) + w.patchQuality * clamp01(score.patchQuality) + w.testReality * clamp01(score.testReality) + w.finalGate * clamp01(score.finalGate) + w.reviewerBlockers * clamp01(score.reviewerBlockers) + w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) + w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60);
31
- }
32
- function clamp01(value) {
33
- if (!Number.isFinite(value)) return 0;
34
- return Math.max(0, Math.min(1, value));
35
- }
36
- function finiteOrZero(value) {
37
- return Number.isFinite(value) ? value : 0;
38
- }
39
- //#endregion
40
11
  //#region src/analyst/proposal-findings.ts
41
12
  const ProposalFindingSchema = z.object({
42
13
  schema_version: z.literal("1.0.0"),
@@ -93,6 +64,35 @@ function findingLabel(finding, index) {
93
64
  return typeof id === "string" && id.length > 0 ? id : `index ${index}`;
94
65
  }
95
66
  //#endregion
96
- export { clamp01 as a, aggregateRunScore as i, isProposalFinding as n, combineAbortSignals as o, DEFAULT_RUN_SCORE_WEIGHTS as r, assertProposalFindings as t };
67
+ //#region src/run-score.ts
68
+ const DEFAULT_RUN_SCORE_WEIGHTS = {
69
+ success: 4,
70
+ goalProgress: 2,
71
+ repoGroundedness: 1.5,
72
+ driftPenalty: -1.5,
73
+ toolUseQuality: 1,
74
+ patchQuality: 1.25,
75
+ testReality: 1.5,
76
+ finalGate: 3,
77
+ reviewerBlockers: -2,
78
+ costUsd: -.2,
79
+ wallSeconds: -.1
80
+ };
81
+ function aggregateRunScore(score, weights = {}) {
82
+ const w = {
83
+ ...DEFAULT_RUN_SCORE_WEIGHTS,
84
+ ...weights
85
+ };
86
+ return w.success * clamp01(score.success) + w.goalProgress * clamp01(score.goalProgress) + w.repoGroundedness * clamp01(score.repoGroundedness) + w.driftPenalty * clamp01(score.driftPenalty) + w.toolUseQuality * clamp01(score.toolUseQuality) + w.patchQuality * clamp01(score.patchQuality) + w.testReality * clamp01(score.testReality) + w.finalGate * clamp01(score.finalGate) + w.reviewerBlockers * clamp01(score.reviewerBlockers) + w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) + w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60);
87
+ }
88
+ function clamp01(value) {
89
+ if (!Number.isFinite(value)) return 0;
90
+ return Math.max(0, Math.min(1, value));
91
+ }
92
+ function finiteOrZero(value) {
93
+ return Number.isFinite(value) ? value : 0;
94
+ }
95
+ //#endregion
96
+ export { combineAbortSignals as a, isProposalFinding as i, clamp01 as n, assertProposalFindings as r, aggregateRunScore as t };
97
97
 
98
- //# sourceMappingURL=proposal-findings-2GIUo1et.js.map
98
+ //# sourceMappingURL=run-score-lDzV0X8j.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-score-lDzV0X8j.js","names":[],"sources":["../src/abort-signal.ts","../src/analyst/proposal-findings.ts","../src/run-score.ts"],"sourcesContent":["/** Combine active cancellation sources without wrapping a single source. */\nexport function combineAbortSignals(\n ...signals: Array<AbortSignal | undefined>\n): AbortSignal | undefined {\n const active = [\n ...new Set(signals.filter((signal): signal is AbortSignal => signal !== undefined)),\n ]\n if (active.length === 0) return undefined\n if (active.length === 1) return active[0]\n return AbortSignal.any(active)\n}\n","import { z } from 'zod'\nimport type { ProposalFinding } from './types'\n\nconst ProposalFindingSchema = z\n .object({\n schema_version: z.literal('1.0.0'),\n finding_id: z.string().min(1),\n analyst_id: z.string().min(1),\n produced_at: z.string().min(1),\n severity: z.enum(['critical', 'high', 'medium', 'low', 'info']),\n area: z.string().min(1),\n claim: z.string().min(1),\n rationale: z.string().optional(),\n evidence_refs: z.array(\n z\n .object({\n kind: z.enum(['span', 'event', 'artifact', 'finding', 'metric']),\n uri: z.string().min(1),\n excerpt: z.string().optional(),\n })\n .strict(),\n ),\n recommended_action: z.string().optional(),\n validation_plan: z.string().optional(),\n confidence: z.number().min(0).max(1),\n subject: z.string().optional(),\n derived_from_judge: z.boolean().optional(),\n metadata: z.record(z.string(), z.unknown()).optional(),\n proposal_origin: z.enum(['search', 'production']),\n })\n .strict() satisfies z.ZodType<ProposalFinding>\n\n/** True when a finding names a source candidate generation may learn from. */\nexport function isProposalFinding(finding: unknown): finding is ProposalFinding {\n return ProposalFindingSchema.safeParse(finding).success\n}\n\n/**\n * Reject findings whose source has not been explicitly admitted for candidate\n * generation. Search feedback and observed production behavior are allowed;\n * final evaluation data has no allowed origin.\n */\nexport function assertProposalFindings(\n findings: unknown,\n context = 'proposal findings',\n): ReadonlyArray<ProposalFinding> {\n if (!Array.isArray(findings)) {\n throw new TypeError(`${context}: expected an array`)\n }\n const rejected = findings.flatMap((finding, index) =>\n isProposalFinding(finding) ? [] : [findingLabel(finding, index)],\n )\n if (rejected.length > 0) {\n throw new Error(\n `${context}: every finding must match AnalystFinding and declare ` +\n `proposal_origin as search or production. ` +\n `Rejected findings: [${rejected.join(', ')}].`,\n )\n }\n return findings as ReadonlyArray<ProposalFinding>\n}\n\nfunction findingLabel(finding: unknown, index: number): string {\n if (typeof finding !== 'object' || finding === null) return `index ${index}`\n const id = (finding as { finding_id?: unknown }).finding_id\n return typeof id === 'string' && id.length > 0 ? id : `index ${index}`\n}\n","export interface RunScore {\n success: number\n goalProgress: number\n repoGroundedness: number\n driftPenalty: number\n toolUseQuality: number\n patchQuality: number\n testReality: number\n finalGate: number\n reviewerBlockers: number\n costUsd: number\n wallSeconds: number\n notes?: string[]\n}\n\nexport interface RunScoreWeights {\n success: number\n goalProgress: number\n repoGroundedness: number\n driftPenalty: number\n toolUseQuality: number\n patchQuality: number\n testReality: number\n finalGate: number\n reviewerBlockers: number\n costUsd: number\n wallSeconds: number\n}\n\nexport const DEFAULT_RUN_SCORE_WEIGHTS: RunScoreWeights = {\n success: 4,\n goalProgress: 2,\n repoGroundedness: 1.5,\n driftPenalty: -1.5,\n toolUseQuality: 1,\n patchQuality: 1.25,\n testReality: 1.5,\n finalGate: 3,\n reviewerBlockers: -2,\n costUsd: -0.2,\n wallSeconds: -0.1,\n}\n\nexport function aggregateRunScore(score: RunScore, weights: Partial<RunScoreWeights> = {}): number {\n const w = { ...DEFAULT_RUN_SCORE_WEIGHTS, ...weights }\n return (\n w.success * clamp01(score.success) +\n w.goalProgress * clamp01(score.goalProgress) +\n w.repoGroundedness * clamp01(score.repoGroundedness) +\n w.driftPenalty * clamp01(score.driftPenalty) +\n w.toolUseQuality * clamp01(score.toolUseQuality) +\n w.patchQuality * clamp01(score.patchQuality) +\n w.testReality * clamp01(score.testReality) +\n w.finalGate * clamp01(score.finalGate) +\n w.reviewerBlockers * clamp01(score.reviewerBlockers) +\n w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) +\n w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60)\n )\n}\n\nexport function clamp01(value: number): number {\n if (!Number.isFinite(value)) return 0\n return Math.max(0, Math.min(1, value))\n}\n\nfunction finiteOrZero(value: number): number {\n return Number.isFinite(value) ? value : 0\n}\n"],"mappings":";;;AACA,SAAgB,oBACd,GAAG,SACsB;CACzB,MAAM,SAAS,CACb,GAAG,IAAI,IAAI,QAAQ,QAAQ,WAAkC,WAAW,KAAA,CAAS,CAAC,CACpF;CACA,IAAI,OAAO,WAAW,GAAG,OAAO,KAAA;CAChC,IAAI,OAAO,WAAW,GAAG,OAAO,OAAO;CACvC,OAAO,YAAY,IAAI,MAAM;AAC/B;;;ACPA,MAAM,wBAAwB,EAC3B,OAAO;CACN,gBAAgB,EAAE,QAAQ,OAAO;CACjC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,aAAa,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC7B,UAAU,EAAE,KAAK;EAAC;EAAY;EAAQ;EAAU;EAAO;CAAM,CAAC;CAC9D,MAAM,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACtB,OAAO,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACvB,WAAW,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,eAAe,EAAE,MACf,EACG,OAAO;EACN,MAAM,EAAE,KAAK;GAAC;GAAQ;GAAS;GAAY;GAAW;EAAQ,CAAC;EAC/D,KAAK,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;EACrB,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,CAAC,CAAC,CACD,OAAO,CACZ;CACA,oBAAoB,EAAE,OAAO,CAAC,CAAC,SAAS;CACxC,iBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS;CACrC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;CACnC,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC7B,oBAAoB,EAAE,QAAQ,CAAC,CAAC,SAAS;CACzC,UAAU,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,QAAQ,CAAC,CAAC,CAAC,SAAS;CACrD,iBAAiB,EAAE,KAAK,CAAC,UAAU,YAAY,CAAC;AAClD,CAAC,CAAC,CACD,OAAO;;AAGV,SAAgB,kBAAkB,SAA8C;CAC9E,OAAO,sBAAsB,UAAU,OAAO,CAAC,CAAC;AAClD;;;;;;AAOA,SAAgB,uBACd,UACA,UAAU,qBACsB;CAChC,IAAI,CAAC,MAAM,QAAQ,QAAQ,GACzB,MAAM,IAAI,UAAU,GAAG,QAAQ,oBAAoB;CAErD,MAAM,WAAW,SAAS,SAAS,SAAS,UAC1C,kBAAkB,OAAO,IAAI,CAAC,IAAI,CAAC,aAAa,SAAS,KAAK,CAAC,CACjE;CACA,IAAI,SAAS,SAAS,GACpB,MAAM,IAAI,MACR,GAAG,QAAQ,qHAEc,SAAS,KAAK,IAAI,EAAE,GAC/C;CAEF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,OAAuB;CAC7D,IAAI,OAAO,YAAY,YAAY,YAAY,MAAM,OAAO,SAAS;CACrE,MAAM,KAAM,QAAqC;CACjD,OAAO,OAAO,OAAO,YAAY,GAAG,SAAS,IAAI,KAAK,SAAS;AACjE;;;ACrCA,MAAa,4BAA6C;CACxD,SAAS;CACT,cAAc;CACd,kBAAkB;CAClB,cAAc;CACd,gBAAgB;CAChB,cAAc;CACd,aAAa;CACb,WAAW;CACX,kBAAkB;CAClB,SAAS;CACT,aAAa;AACf;AAEA,SAAgB,kBAAkB,OAAiB,UAAoC,CAAC,GAAW;CACjG,MAAM,IAAI;EAAE,GAAG;EAA2B,GAAG;CAAQ;CACrD,OACE,EAAE,UAAU,QAAQ,MAAM,OAAO,IACjC,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,mBAAmB,QAAQ,MAAM,gBAAgB,IACnD,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,iBAAiB,QAAQ,MAAM,cAAc,IAC/C,EAAE,eAAe,QAAQ,MAAM,YAAY,IAC3C,EAAE,cAAc,QAAQ,MAAM,WAAW,IACzC,EAAE,YAAY,QAAQ,MAAM,SAAS,IACrC,EAAE,mBAAmB,QAAQ,MAAM,gBAAgB,IACnD,EAAE,UAAU,KAAK,IAAI,GAAG,aAAa,MAAM,OAAO,CAAC,IACnD,EAAE,cAAc,KAAK,IAAI,GAAG,aAAa,MAAM,WAAW,IAAI,EAAE;AAEpE;AAEA,SAAgB,QAAQ,OAAuB;CAC7C,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG,OAAO;CACpC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,KAAK,CAAC;AACvC;AAEA,SAAS,aAAa,OAAuB;CAC3C,OAAO,OAAO,SAAS,KAAK,IAAI,QAAQ;AAC1C"}
@@ -0,0 +1,70 @@
1
+ //#region src/sandbox-harness.d.ts
2
+ interface HarnessConfig {
3
+ /** Setup command (e.g. "pnpm install"). Non-zero exit fails the run. */
4
+ setupCommand?: string;
5
+ /** Run command (e.g. "pnpm build"). */
6
+ runCommand?: string;
7
+ /** Test command (e.g. "pnpm test --run"). Drives the test count + pass count. */
8
+ testCommand?: string;
9
+ /** Absolute cwd for the subprocess driver. Ignored by docker driver. */
10
+ cwd?: string;
11
+ /** Max wall-clock per phase in ms. Default 10 minutes. */
12
+ timeoutMs?: number;
13
+ /**
14
+ * Cap on captured stdout+stderr bytes per phase. A runaway process can
15
+ * otherwise grow the in-memory buffer without bound. Once hit, further
16
+ * output is dropped and `outputTruncated` is set. Default 16 MiB.
17
+ */
18
+ maxOutputBytes?: number;
19
+ /** Image for the docker driver. */
20
+ image?: string;
21
+ /** Extra env vars (validated; shell-escaped). */
22
+ env?: Record<string, string>;
23
+ /** Parser for the test output — maps stdout/stderr/exit code → pass count. */
24
+ testParser?: TestOutputParser;
25
+ }
26
+ interface TestOutputParser {
27
+ id: string;
28
+ parse(stdout: string, stderr: string, exitCode: number): {
29
+ testsTotal: number;
30
+ testsPassed: number;
31
+ } | undefined;
32
+ }
33
+ interface SandboxResult {
34
+ phase: 'setup' | 'run' | 'test';
35
+ exitCode: number;
36
+ stdout: string;
37
+ stderr: string;
38
+ wallMs: number;
39
+ testsTotal?: number;
40
+ testsPassed?: number;
41
+ /**
42
+ * True when the process was killed because it exceeded `timeoutMs`. A
43
+ * SIGKILLed child can still close with exit code 0; callers MUST treat
44
+ * a timed-out phase as a hard failure regardless of `exitCode`, never
45
+ * as a pass. `undefined`/`false` means the process completed on its own.
46
+ */
47
+ killedByTimeout?: boolean;
48
+ /**
49
+ * True when captured stdout/stderr hit `maxOutputBytes` and further
50
+ * output was dropped. The result is still returned (the process was
51
+ * not killed for this), but downstream parsers see truncated text.
52
+ */
53
+ outputTruncated?: boolean;
54
+ }
55
+ interface SandboxDriver {
56
+ id: string;
57
+ exec(phase: SandboxResult['phase'], command: string, config: HarnessConfig): Promise<SandboxResult>;
58
+ }
59
+ interface SandboxHarnessResult {
60
+ passed: boolean;
61
+ setup?: SandboxResult;
62
+ run?: SandboxResult;
63
+ test?: SandboxResult;
64
+ totalWallMs: number;
65
+ /** Final score — 0 when no tests; otherwise testsPassed/testsTotal. */
66
+ score: number;
67
+ }
68
+ //#endregion
69
+ export { SandboxDriver as n, SandboxHarnessResult as r, HarnessConfig as t };
70
+ //# sourceMappingURL=sandbox-harness-BlSOu4LX.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sandbox-harness-BlSOu4LX.d.ts","names":[],"sources":["../src/sandbox-harness.ts"],"mappings":";UAkBiB;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;EAMA;;EAEA;;EAEA,MAAM;;EAEN,aAAa;;UAGE;EACf;EACA,MACE,gBACA,gBACA;IACG;IAAoB;;;UAGV;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;;;;;EAOA;;;;;;EAMA;;UAGe;EACf;EACA,KACE,OAAO,wBACP,iBACA,QAAQ,gBACP,QAAQ;;UA4PI;EACf;EACA,QAAQ;EACR,MAAM;EACN,OAAO;EACP;;EAEA"}
@@ -405,4 +405,4 @@ declare function assertMinted(value: unknown, context?: string): MintedRolloutLi
405
405
  declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
406
406
  //#endregion
407
407
  export { isTrainableSplit as A, TRAINABLE_SPLITS as C, assertRolloutLine as D, assertMintedLines as E, gateGamedOutcome as O, RolloutTask as S, assertMinted as T, RolloutPolicy as _, GatedEvidence as a, RolloutSplit as b, ROLLOUT_CAPTURES as c, ROLLOUT_SPLITS as d, RolloutArtifacts as f, RolloutOutcome as g, RolloutLine as h, ChatToolCall as i, validateRolloutLine as j, isRolloutLine as k, ROLLOUT_ROLES as l, RolloutCostBlock as m, ChatMessage as n, MintedRolloutLine as o, RolloutCapture as p, ChatRole as r, MintedRolloutOutcome as s, CHAT_ROLES as t, ROLLOUT_SCHEMA as u, RolloutProvenance as v, ToolDef as w, RolloutStep as x, RolloutRole as y };
408
- //# sourceMappingURL=schema-Cef2cFmb2.d.ts.map
408
+ //# sourceMappingURL=schema-Cef2cFmb.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"schema-Cef2cFmb.d.ts","names":[],"sources":["../src/rollout/schema.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cAmDa;;KAGD;cACC,wBAAwB;;KAUzB;cACC,yBAAyB;;cAEzB,2BAA2B;iBAExB,iBAAiB,OAAO;;KAK5B;cACC,2BAA2B;KAM5B;cACC,qBAAqB;UAEjB;EACf;EACA;EACA;IACE;;IAEA;;;UAIa;EACf,MAAM;EACN;;EAEA;EACA,aAAa;;EAEb;EACA;;;;;;;;EAQA;;UAGe;EACf;EACA;IACE;IACA;IACA,aAAa;;;;;;;;UASA;EACf;EACA;;EAEA;;EAEA;EACA;EACA;;;;;EAUA;;EAEA;;EAEA;;;;;;EAMA;;UAOe;;EAEf;EACA;EACA,OAAO;;EAEP;;EAEA;;UAGe;;EAEf;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;;;;EAKf;;EAEA;;EAEA;;EAEA,SAAS;EACT;EACA;EACA;;;;;;;;;;;;;EAaA;;;;;;;;;;;;;;;;;;;;;EAqBA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;;;EAKA;;UAGe;EACf;EACA;;EAEA;;;;;;;UAQe;;EAEf,UAAU;;EAEV;;;;;;;;;EASA;;UAGe;EACf;EACA,SAAS;;;;;;;EAOT;;;;;;;EAOA,iBAAiB;;UAGF;EACf,eAAe;EACf;;EAEA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,MAAM;EACN,MAAM;EACN,QAAQ;;EAER,UAAU;EACV,WAAW;;EAEX,QAAQ;EACR,SAAS;EACT,MAAM;EACN,WAAW;EACX,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6GE,iBAAiB,MAAM,cAAc;iBAqDrC,oBAAoB;iBAiNpB,kBACd,gBACA,2BACS,SAAS;iBAOJ,cAAc,iBAAiB,SAAS;;;;;;cAa1C;;;;;;UAOG,6BAA6B;EAC5C;;;;;;;;;;;;;;;;;;;;;;KAuBU,oBAAoB,KAAK;YACzB;EACV,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsCK,aAAa,gBAAgB,mBAA2B;;iBAiBxD,kBACd,4BACA,mBACC"}