@tangle-network/agent-eval 0.150.2 → 0.161.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (342) hide show
  1. package/CHANGELOG.md +149 -1
  2. package/README.md +7 -3
  3. package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
  4. package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
  5. package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
  6. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +19 -36
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
  12. package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
  13. package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
  14. package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
  15. package/dist/{benchmark-command-BU1Las59.js → benchmark-command-BVtaq_ve.js} +26 -31
  16. package/dist/benchmark-command-BVtaq_ve.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +6 -19
  18. package/dist/benchmarks/index.d.ts.map +1 -1
  19. package/dist/benchmarks/index.js +4 -4
  20. package/dist/benchmarks/index.js.map +1 -1
  21. package/dist/builder-eval/index.d.ts +3 -3
  22. package/dist/builder-eval/index.js +2 -2
  23. package/dist/campaign/index.d.ts +9 -9
  24. package/dist/campaign/index.js +7 -7
  25. package/dist/{campaign-la-gEYNz.js → campaign-BSmOwskD.js} +77 -795
  26. package/dist/campaign-BSmOwskD.js.map +1 -0
  27. package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
  28. package/dist/canonical-IL-Bu-14.js.map +1 -0
  29. package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
  30. package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
  31. package/dist/{chat-client-Bvmxedyv.js → chat-client-DlMlAeYI.js} +5 -55
  32. package/dist/{chat-client-Bvmxedyv.js.map → chat-client-DlMlAeYI.js.map} +1 -1
  33. package/dist/chat-json-call-6g5sJobJ.js +53 -0
  34. package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
  35. package/dist/cli.js +54 -19
  36. package/dist/cli.js.map +1 -1
  37. package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
  38. package/dist/client-CX7KqIdB.js.map +1 -0
  39. package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
  40. package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -27
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +14 -17
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
  46. package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
  47. package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
  48. package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
  49. package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
  50. package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
  51. package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
  52. package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
  53. package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
  54. package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
  55. package/dist/{define-agent-eval-C8V8sMqP.js → define-agent-eval-h-s-sI-v.js} +15 -9
  56. package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
  57. package/dist/{dspy-rlm-engine-CBYlPvNy.js → dspy-rlm-engine-DptEII26.js} +95 -15
  58. package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
  59. package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
  60. package/dist/emitter-BpYFQPj4.js.map +1 -0
  61. package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
  62. package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
  63. package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
  64. package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
  65. package/dist/{eval-campaign-CQuZrLR_.js → eval-campaign-BsXWL2-2.js} +17 -28
  66. package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
  67. package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
  68. package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
  69. package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
  70. package/dist/exec-D9WpA2p-.js.map +1 -0
  71. package/dist/experiment/index.d.ts +45 -11
  72. package/dist/experiment/index.d.ts.map +1 -1
  73. package/dist/experiment/index.js +30 -11
  74. package/dist/experiment/index.js.map +1 -1
  75. package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
  76. package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
  77. package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
  78. package/dist/exporters-Df7TgHFv.js.map +1 -0
  79. package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
  80. package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
  81. package/dist/{external-optimizer-process-x9oEXKsU.js → external-optimizer-process-WosTBChy.js} +4 -4
  82. package/dist/{external-optimizer-process-x9oEXKsU.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
  83. package/dist/{external-optimizer-subprocess-CKNb42oM.js → external-optimizer-subprocess-BIWbHpgD.js} +4 -4
  84. package/dist/{external-optimizer-subprocess-CKNb42oM.js.map → external-optimizer-subprocess-BIWbHpgD.js.map} +1 -1
  85. package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
  86. package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
  87. package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
  88. package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
  89. package/dist/fuzz.js +1 -1
  90. package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
  91. package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
  92. package/dist/hosted/index.d.ts +3 -14
  93. package/dist/hosted/index.d.ts.map +1 -1
  94. package/dist/hosted/index.js +2 -2
  95. package/dist/{index-Aj3WO3_a.d.ts → index-CGtH1piv.d.ts} +48 -26
  96. package/dist/index-CGtH1piv.d.ts.map +1 -0
  97. package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
  98. package/dist/index-D-IiQIBB.d.ts.map +1 -0
  99. package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
  100. package/dist/index-D-V8gCs_.d.ts.map +1 -0
  101. package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
  102. package/dist/index-lfaSeKSD.d.ts.map +1 -0
  103. package/dist/index-vrJugRal.d.ts +1 -0
  104. package/dist/index.d.ts +68 -56
  105. package/dist/index.d.ts.map +1 -1
  106. package/dist/index.js +59 -110
  107. package/dist/index.js.map +1 -1
  108. package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
  109. package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
  110. package/dist/{integrity-DL91tucI.js → integrity-CyWSSoQS.js} +2 -2
  111. package/dist/{integrity-DL91tucI.js.map → integrity-CyWSSoQS.js.map} +1 -1
  112. package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
  113. package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
  114. package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
  115. package/dist/kind-factory-DY8FdoXf.js.map +1 -0
  116. package/dist/ledger-core/index.d.ts +2 -2
  117. package/dist/ledger-core/index.js +3 -3
  118. package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
  119. package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
  120. package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
  121. package/dist/llm-client-hgDieDNN.js.map +1 -0
  122. package/dist/{llm-judge-BqqMS8t7.js → llm-judge-BhasIPFT.js} +1135 -77
  123. package/dist/llm-judge-BhasIPFT.js.map +1 -0
  124. package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
  125. package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
  126. package/dist/meta-eval/index.d.ts +8 -6
  127. package/dist/meta-eval/index.d.ts.map +1 -1
  128. package/dist/meta-eval/index.js +8 -6
  129. package/dist/meta-eval/index.js.map +1 -1
  130. package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
  131. package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
  132. package/dist/multishot/golden/index.d.ts +2 -8
  133. package/dist/multishot/golden/index.d.ts.map +1 -1
  134. package/dist/multishot/golden/index.js +56 -86
  135. package/dist/multishot/golden/index.js.map +1 -1
  136. package/dist/multishot/index.d.ts +11 -46
  137. package/dist/multishot/index.d.ts.map +1 -1
  138. package/dist/multishot/index.js +30 -83
  139. package/dist/multishot/index.js.map +1 -1
  140. package/dist/openapi.json +1 -1
  141. package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
  142. package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
  143. package/dist/pipelines/index.d.ts +5 -5
  144. package/dist/pipelines/index.js +3 -3
  145. package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
  146. package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
  147. package/dist/pre-registration-KN9jkh58.js +110 -0
  148. package/dist/pre-registration-KN9jkh58.js.map +1 -0
  149. package/dist/{produced-state-Bnq4FaDO.js → produced-state-DZ89riy5.js} +8 -8
  150. package/dist/produced-state-DZ89riy5.js.map +1 -0
  151. package/dist/profile-cell.d.ts +1 -1
  152. package/dist/profile-cell.js +31 -5
  153. package/dist/profile-cell.js.map +1 -1
  154. package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
  155. package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
  156. package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
  157. package/dist/query-CHmMP42p.js.map +1 -0
  158. package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
  159. package/dist/query-DxPYqpmT.d.ts.map +1 -0
  160. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  161. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  162. package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
  163. package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
  164. package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
  165. package/dist/release-confidence-DKfD2RYU.js.map +1 -0
  166. package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
  167. package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
  168. package/dist/reporting.d.ts +3 -3
  169. package/dist/reporting.js +3 -3
  170. package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
  171. package/dist/researcher-Cz565b7D.d.ts.map +1 -0
  172. package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
  173. package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
  174. package/dist/{reward-hacking-C0x0xihA.js → reward-hacking-t4lB1yt8.js} +2 -2
  175. package/dist/{reward-hacking-C0x0xihA.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
  176. package/dist/rl.d.ts +11 -42
  177. package/dist/rl.d.ts.map +1 -1
  178. package/dist/rl.js +41 -24
  179. package/dist/rl.js.map +1 -1
  180. package/dist/rollout/index.d.ts +3 -3
  181. package/dist/rollout/index.js +7 -7
  182. package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
  183. package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
  184. package/dist/{rubric-predictive-validity-CzxLoZge.js → rubric-predictive-validity-CK8SCOg-.js} +5 -15
  185. package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
  186. package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
  187. package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
  188. package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
  189. package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
  190. package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
  191. package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
  192. package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
  193. package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
  194. package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
  195. package/dist/schema-BzWDXhOR.d.ts.map +1 -0
  196. package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
  197. package/dist/schema-C1aaAxTf.js.map +1 -0
  198. package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
  199. package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
  200. package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
  201. package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
  202. package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
  203. package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
  204. package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
  205. package/dist/sequential-rYW-Ophm.js.map +1 -0
  206. package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
  207. package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
  208. package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
  209. package/dist/server-BtFd4uzB.js.map +1 -0
  210. package/dist/{skillopt-optimization-method-CPBlTcj5.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
  211. package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
  212. package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
  213. package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
  214. package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
  215. package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
  216. package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
  217. package/dist/steps-CiNVJry_.d.ts.map +1 -0
  218. package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
  219. package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
  220. package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
  221. package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
  222. package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
  223. package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
  224. package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
  225. package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
  226. package/dist/storyboard/index.d.ts +1 -1
  227. package/dist/{summary-report-DW2bEpdB.js → summary-report-BI5hUtvK.js} +6 -16
  228. package/dist/summary-report-BI5hUtvK.js.map +1 -0
  229. package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
  230. package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
  231. package/dist/supervisor-run/index.d.ts +3 -16
  232. package/dist/supervisor-run/index.d.ts.map +1 -1
  233. package/dist/supervisor-run/index.js +3 -3
  234. package/dist/supervisor-run/index.js.map +1 -1
  235. package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
  236. package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
  237. package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
  238. package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
  239. package/dist/{tool-waste-C-VHSRwF.js → tool-waste-BqzmVdJk.js} +2 -2
  240. package/dist/{tool-waste-C-VHSRwF.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
  241. package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
  242. package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
  243. package/dist/trace-repair/index.d.ts +4 -77
  244. package/dist/trace-repair/index.d.ts.map +1 -1
  245. package/dist/trace-repair/index.js +5 -15
  246. package/dist/trace-repair/index.js.map +1 -1
  247. package/dist/traces.d.ts +13 -23
  248. package/dist/traces.d.ts.map +1 -1
  249. package/dist/traces.js +9 -20
  250. package/dist/traces.js.map +1 -1
  251. package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
  252. package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
  253. package/dist/trajectory-replay/index.d.ts +5 -5
  254. package/dist/trajectory-replay/index.js +5 -5
  255. package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
  256. package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
  257. package/dist/types-B3jzCp0p.js.map +1 -1
  258. package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
  259. package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
  260. package/dist/types-Bfk0uxRj.d.ts +443 -0
  261. package/dist/types-Bfk0uxRj.d.ts.map +1 -0
  262. package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
  263. package/dist/types-D4s7Z6nq.d.ts.map +1 -0
  264. package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
  265. package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
  266. package/dist/{types-I5WwVzQ7.d.ts → types-DeIUdzNd.d.ts} +2 -2
  267. package/dist/{types-I5WwVzQ7.d.ts.map → types-DeIUdzNd.d.ts.map} +1 -1
  268. package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
  269. package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
  270. package/dist/verdict-cache-CdVVTVmn.js +88 -0
  271. package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
  272. package/dist/wire/index.d.ts +21 -111
  273. package/dist/wire/index.d.ts.map +1 -1
  274. package/dist/wire/index.js +2 -2
  275. package/docs/building-doctrine.md +3 -3
  276. package/docs/campaign-proposers.md +41 -10
  277. package/docs/design/statistics-decisions.md +89 -1
  278. package/docs/eval-surface-map.md +14 -0
  279. package/docs/experiment.md +19 -2
  280. package/docs/feedback-trajectories.md +1 -1
  281. package/docs/multishot-golden-records.md +4 -4
  282. package/docs/public-api.md +1616 -0
  283. package/docs/research-report-methodology.md +1 -1
  284. package/docs/search-history-receipts.md +39 -1
  285. package/docs/trace-analysis.md +1 -1
  286. package/docs/trace-repair-admission.md +1 -1
  287. package/docs/trace-repair-continuation.md +1 -1
  288. package/docs/verdicts.md +24 -0
  289. package/package.json +6 -2
  290. package/dist/active-curriculum-C4mk67HP.js.map +0 -1
  291. package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
  292. package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
  293. package/dist/benchmark-command-BU1Las59.js.map +0 -1
  294. package/dist/campaign-la-gEYNz.js.map +0 -1
  295. package/dist/canonical-D-XsTQ6_.js.map +0 -1
  296. package/dist/client-LIuo-KPv.js.map +0 -1
  297. package/dist/define-agent-eval-C8V8sMqP.js.map +0 -1
  298. package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
  299. package/dist/dspy-rlm-engine-CBYlPvNy.js.map +0 -1
  300. package/dist/emitter-CPBAhxum.js.map +0 -1
  301. package/dist/eval-campaign-CQuZrLR_.js.map +0 -1
  302. package/dist/exec-y-DCLqK7.js.map +0 -1
  303. package/dist/exporters-q9iL-2Jf.js.map +0 -1
  304. package/dist/index-Aj3WO3_a.d.ts.map +0 -1
  305. package/dist/index-B8Ui1mr1.d.ts.map +0 -1
  306. package/dist/index-BNPtkBPf.d.ts.map +0 -1
  307. package/dist/index-C1ravkGA.d.ts +0 -1
  308. package/dist/index-IQccV3Ou.d.ts.map +0 -1
  309. package/dist/kind-factory-DmAa0h3K.js.map +0 -1
  310. package/dist/llm-client-Bg32RW0j.js.map +0 -1
  311. package/dist/llm-judge-BqqMS8t7.js.map +0 -1
  312. package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
  313. package/dist/pre-registration-DakwTRXk.js +0 -96
  314. package/dist/pre-registration-DakwTRXk.js.map +0 -1
  315. package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
  316. package/dist/produced-state-Bnq4FaDO.js.map +0 -1
  317. package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
  318. package/dist/query-CJ_DX8vl.d.ts.map +0 -1
  319. package/dist/query-Di7eEQ79.js.map +0 -1
  320. package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
  321. package/dist/release-confidence-BknrpBnO.js.map +0 -1
  322. package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
  323. package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
  324. package/dist/rubric-predictive-validity-CzxLoZge.js.map +0 -1
  325. package/dist/schema-C6DW4ZHR.js.map +0 -1
  326. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  327. package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
  328. package/dist/sequential-C458DXNf.js.map +0 -1
  329. package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
  330. package/dist/server-dIWwF3j_.js.map +0 -1
  331. package/dist/skillopt-optimization-method-CPBlTcj5.js.map +0 -1
  332. package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
  333. package/dist/steps-AmkT-GIM.d.ts.map +0 -1
  334. package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
  335. package/dist/summary-report-DW2bEpdB.js.map +0 -1
  336. package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
  337. package/dist/types-CLAwnY-L.d.ts.map +0 -1
  338. package/dist/types-DdFNuyxQ.d.ts.map +0 -1
  339. package/dist/types-jUBXJ7Iz.d.ts +0 -884
  340. package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
  341. package/dist/verdict-cache-mZf5FEiY.js +0 -107
  342. package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"mint-BV6tLVWl.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
1
+ {"version":3,"file":"mint-DfODW1KW.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
@@ -1,4 +1,4 @@
1
- import { S as MultishotToolDefinition, T as MultishotTransportRequest, c as RunMultishotMatrixResult, f as RunMultishotOptions, s as RunMultishotMatrixOptions, v as MultishotPersona, y as MultishotResult } from "../../matrix-DrVnRp4G.js";
1
+ import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-eXKRMHnL.js";
2
2
  //#region src/multishot/golden/compare.d.ts
3
3
  interface CompareOptions {
4
4
  /** Stop after this many mismatches. A structural divergence high in the
@@ -109,9 +109,6 @@ interface MultishotMatrixGoldenCase {
109
109
  requests: MultishotRecordedRequest[];
110
110
  /** Judge calls, filled while the case runs. Sorted before comparison. */
111
111
  judgeRequests: RecordedJudgeRequest[];
112
- /** Installs the deterministic judge wire on `globalThis.fetch` and returns
113
- * the function that restores the previous one. */
114
- installJudgeWire: () => () => void;
115
112
  }
116
113
  interface MultishotMatrixGoldenScenario {
117
114
  readonly id: string;
@@ -190,9 +187,6 @@ declare function assertMultishotMatrixGoldenScenario(opts: {
190
187
  }): Promise<void>;
191
188
  //#endregion
192
189
  //#region src/multishot/golden/recording.d.ts
193
- /** Keys whose value is wall clock or run identity. Two runs never agree on
194
- * them, so they are removed before comparison instead of being compared. */
195
- declare const VOLATILE_KEYS: ReadonlySet<string>;
196
190
  declare function recordMessage(raw: unknown): MultishotRecordedMessage;
197
191
  declare function recordRequest(leg: 'agent' | 'driver', req: MultishotTransportRequest): MultishotRecordedRequest;
198
192
  /** Judge calls reach the wire as an OpenAI-compat body, not through a
@@ -226,5 +220,5 @@ declare const CURRENT_MULTISHOT_GOLDEN_VERSION = "v1";
226
220
  declare function multishotGoldenVersions(): string[];
227
221
  declare function goldenRecords(version?: string): MultishotGoldenRecordSet;
228
222
  //#endregion
229
- export { CURRENT_MULTISHOT_GOLDEN_VERSION, type CompareOptions, type MultishotGoldenCase, type MultishotGoldenEngine, MultishotGoldenMismatchError, type MultishotGoldenOutcome, type MultishotGoldenRecord, type MultishotGoldenRecordSet, type MultishotGoldenReport, type MultishotGoldenScenario, type MultishotGoldenScenarioReport, type MultishotMatrixGoldenCase, type MultishotMatrixGoldenEngine, type MultishotMatrixGoldenRecord, type MultishotMatrixGoldenScenario, type MultishotRecordedMessage, type MultishotRecordedRequest, type RecordedJudgeRequest, type RecordedMultishotError, type RecordedMultishotResult, VOLATILE_KEYS, assertMultishotGoldenScenario, assertMultishotMatrixGoldenScenario, checkMultishotGolden, checkMultishotGoldenScenario, checkMultishotMatrixGoldenScenario, compareJson, goldenRecords, maskVolatileMarkdown, multishotGoldenScenarios, multishotGoldenVersions, multishotMatrixGoldenScenarios, readRunDir, recordError, recordJudgeRequest, recordMessage, recordRequest, recordResult, sortJudgeRequests, stripVolatile };
223
+ export { CURRENT_MULTISHOT_GOLDEN_VERSION, type CompareOptions, type MultishotGoldenCase, type MultishotGoldenEngine, MultishotGoldenMismatchError, type MultishotGoldenOutcome, type MultishotGoldenRecord, type MultishotGoldenRecordSet, type MultishotGoldenReport, type MultishotGoldenScenario, type MultishotGoldenScenarioReport, type MultishotMatrixGoldenCase, type MultishotMatrixGoldenEngine, type MultishotMatrixGoldenRecord, type MultishotMatrixGoldenScenario, type MultishotRecordedMessage, type MultishotRecordedRequest, type RecordedJudgeRequest, type RecordedMultishotError, type RecordedMultishotResult, assertMultishotGoldenScenario, assertMultishotMatrixGoldenScenario, checkMultishotGolden, checkMultishotGoldenScenario, checkMultishotMatrixGoldenScenario, compareJson, goldenRecords, maskVolatileMarkdown, multishotGoldenScenarios, multishotGoldenVersions, multishotMatrixGoldenScenarios, readRunDir, recordError, recordJudgeRequest, recordMessage, recordRequest, recordResult, sortJudgeRequests, stripVolatile };
230
224
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../../src/multishot/golden/compare.ts","../../../src/multishot/golden/engine.ts","../../../src/multishot/golden/types.ts","../../../src/multishot/golden/matrix-scenarios.ts","../../../src/multishot/golden/scenarios.ts","../../../src/multishot/golden/harness.ts","../../../src/multishot/golden/recording.ts","../../../src/multishot/golden/records/index.ts"],"mappings":";;UAQiB;;;EAGf;;iBAGc,YACd,mBACA,iBACA,cACA,UAAS;;;;KCPC,yBACV,MAAM,oBAAoB,sBACvB,QAAQ;;KAGD,+BACV,MAAM,0BAA0B,sBAC7B,QAAQ;;;;;UCRI;EACf;EACA;;EAEA;;EAEA,YAAY;IAAQ;IAAY;IAAc;;;;;;;UAO/B;EACf;EACA;EACA;EACA;;;;;EAKA,OAAO;EACP,UAAU;;;KAIA,0BAA0B,KAAK;;;UAI1B;EACf;EACA;;;EAGA;IAAa;IAAiB;;;KAGpB;EACN;EAAgB,QAAQ;;EACxB;EAAe,OAAO;;UAEX;EACf;EACA;EACA,UAAU;EACV,SAAS;;;UAIM;EACf;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA,UAAU;EACV,eAAe;;EAEf;;;;EAIA,OAAO;;;;UAKQ;EACf;;EAEA;;EAEA;EACA;EACA,WAAW;EACX,iBAAiB;;;;UCpEF;EACf,SAAS,0BAA0B;EACnC,UAAU;;EAEV,eAAe;;;EAGf;;UAGe;WACN;WACA;WACA,QAAQ,mBAAmB;;iBAqJtB,kCAAkC;;;;UCzJjC;EACf,SAAS,oBAAoB;;EAE7B,UAAU;;UAGK;WACN;WACA;;;WAGA,aAAa;;;iBA4WR,4BAA4B;;;UCzX3B;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,WAAW;;cAGA,qCAAqC;WAErC;WACA;WACA;EAHX,YACW,oBACA,sBACA;;;iBAoCS,6BAA6B;EACjD,QAAQ;EACR,UAAU;EACV,UAAU;IACR,QAAQ;;iBAoCU,8BAA8B;EAClD,QAAQ;EACR,UAAU;EACV,UAAU;IACR;;iBASkB,qBAAqB;EACzC,QAAQ;EACR,UAAU;EACV;IACE,QAAQ;;iBA6BU,mCAAmC;EACvD,QAAQ;EACR,UAAU;;EAEV;EACA,UAAU;IACR,QAAQ;iBA0BU,oCAAoC;EACxD,QAAQ;EACR,UAAU;EACV;EACA,UAAU;IACR;;;;;cC1LS,eAAe;iBAOZ,cAAc,eAAe;iBA0B7B,cACd,yBACA,KAAK,4BACJ;;;iBAaa,mBAAmB,MAAM,0BAA0B;iBASnD,aAAa,QAAQ,kBAAkB;iBAKvC,YAAY,eAAe;;iBAU3B,cAAc;;;;;;;;iBAoBd,qBAAqB;;;;;iBAgBrB,kBACd,mBAAmB,yBAClB;;;;iBAWa,WAAW,cAAc;;;;cCpI5B;iBAoDG;iBAIA,cACd,mBACC"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../../src/multishot/golden/compare.ts","../../../src/multishot/golden/engine.ts","../../../src/multishot/golden/types.ts","../../../src/multishot/golden/matrix-scenarios.ts","../../../src/multishot/golden/scenarios.ts","../../../src/multishot/golden/harness.ts","../../../src/multishot/golden/recording.ts","../../../src/multishot/golden/records/index.ts"],"mappings":";;UAQiB;;;EAGf;;iBAGc,YACd,mBACA,iBACA,cACA,UAAS;;;;KCPC,yBACV,MAAM,oBAAoB,sBACvB,QAAQ;;KAGD,+BACV,MAAM,0BAA0B,sBAC7B,QAAQ;;;;;UCRI;EACf;EACA;;EAEA;;EAEA,YAAY;IAAQ;IAAY;IAAc;;;;;;;UAO/B;EACf;EACA;EACA;EACA;;;;;EAKA,OAAO;EACP,UAAU;;;KAIA,0BAA0B,KAAK;;;UAI1B;EACf;EACA;;;EAGA;IAAa;IAAiB;;;KAGpB;EACN;EAAgB,QAAQ;;EACxB;EAAe,OAAO;;UAEX;EACf;EACA;EACA,UAAU;EACV,SAAS;;;UAIM;EACf;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA,UAAU;EACV,eAAe;;EAEf;;;;EAIA,OAAO;;;;UAKQ;EACf;;EAEA;;EAEA;EACA;EACA,WAAW;EACX,iBAAiB;;;;UCpEF;EACf,SAAS,0BAA0B;EACnC,UAAU;;EAEV,eAAe;;UAGA;WACN;WACA;WACA,QAAQ,mBAAmB;;iBAkJtB,kCAAkC;;;;UCnJjC;EACf,SAAS,oBAAoB;;EAE7B,UAAU;;UAGK;WACN;WACA;;;WAGA,aAAa;;;iBAwWR,4BAA4B;;;UCrX3B;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,WAAW;;cAGA,qCAAqC;WAErC;WACA;WACA;EAHX,YACW,oBACA,sBACA;;;iBAoCS,6BAA6B;EACjD,QAAQ;EACR,UAAU;EACV,UAAU;IACR,QAAQ;;iBAoCU,8BAA8B;EAClD,QAAQ;EACR,UAAU;EACV,UAAU;IACR;;iBASkB,qBAAqB;EACzC,QAAQ;EACR,UAAU;EACV;IACE,QAAQ;;iBA6BU,mCAAmC;EACvD,QAAQ;EACR,UAAU;;EAEV;EACA,UAAU;IACR,QAAQ;iBAoBU,oCAAoC;EACxD,QAAQ;EACR,UAAU;EACV;EACA,UAAU;IACR;;;iBC7KY,cAAc,eAAe;iBA0B7B,cACd,yBACA,KAAK,4BACJ;;;iBAaa,mBAAmB,MAAM,0BAA0B;iBASnD,aAAa,QAAQ,kBAAkB;iBAKvC,YAAY,eAAe;;iBAU3B,cAAc;;;;;;;;iBAoBd,qBAAqB;;;;;iBAgBrB,kBACd,mBAAmB,yBAClB;;;;iBAWa,WAAW,cAAc;;;;cCpI5B;iBAoDG;iBAIA,cACd,mBACC"}
@@ -192,7 +192,6 @@ function walkFiles(dir) {
192
192
  }
193
193
  //#endregion
194
194
  //#region src/multishot/golden/matrix-scenarios.ts
195
- const JUDGE_BASE_URL = "http://router.invalid/v1";
196
195
  const personas = [{
197
196
  id: "retail-founder",
198
197
  ask: "a launch brief"
@@ -328,21 +327,17 @@ const dimensions = [{
328
327
  key: "specificity",
329
328
  description: "Was it specific? (0-10)"
330
329
  }];
331
- function judge(name, buildPrompt) {
330
+ function judge(name, buildPrompt, transport) {
332
331
  return {
333
332
  name,
333
+ transport,
334
334
  model: "test/judge-model",
335
335
  dimensions,
336
336
  systemPrompt: `JUDGE:${name}`,
337
- buildPrompt,
338
- apiKey: "golden-key",
339
- baseUrl: JUDGE_BASE_URL
337
+ buildPrompt
340
338
  };
341
339
  }
342
- /** True while some case holds `globalThis.fetch`. Module scope, because the
343
- * resource being guarded is the process's own fetch. */
344
- let judgeWireInstalled = false;
345
- /** Scores keyed by judge name, so the wire is a pure function of the request. */
340
+ /** Scores keyed by judge name, so the judge leg is a pure function of the request. */
346
341
  const JUDGE_SCORES = {
347
342
  conversation: {
348
343
  usefulness: 8,
@@ -367,75 +362,60 @@ function multishotMatrixGoldenScenarios() {
367
362
  function buildMatrixCase(runDir) {
368
363
  const requests = [];
369
364
  const judgeRequests = [];
370
- const options = {
371
- profiles,
372
- personas,
373
- shape,
374
- judges: {
375
- conversation: judge("conversation", (input) => `Score this conversation of ${input.transcript.length} messages.`),
376
- codeReview: judge("code-review", (input) => `Score this code: ${input.artifact.content}`),
377
- contentQuality: judge("content-quality", (input) => `Score this content: ${input.artifact.content}`)
378
- },
379
- tools,
380
- toolExecutors: toolExecutors(),
381
- artifactTypeFor,
382
- runDir,
383
- reps: 1,
384
- maxTurns: 3,
385
- maxConcurrency: 1,
386
- agentModel: "test/agent-model",
387
- driverModel: "test/driver-model",
388
- apiKey: "golden-key",
389
- baseUrl: JUDGE_BASE_URL,
390
- agentTransport: async (req) => {
391
- requests.push(recordRequest("agent", req));
392
- return agentTransport(req);
393
- },
394
- driverTransport: async (req) => {
395
- requests.push(recordRequest("driver", req));
396
- return driverTransport(req);
397
- }
398
- };
399
- const installJudgeWire = () => {
400
- if (judgeWireInstalled) throw new Error("multishot golden judge wire: another matrix check already holds globalThis.fetch — run matrix checks serially within one process");
401
- judgeWireInstalled = true;
402
- const previous = globalThis.fetch;
403
- globalThis.fetch = (async (url, init) => {
404
- if (String(url) !== `${JUDGE_BASE_URL}/chat/completions`) throw new Error(`multishot golden judge wire: unexpected request to ${String(url)}`);
405
- const body = JSON.parse(init?.body ?? "{}");
406
- judgeRequests.push(recordJudgeRequest(body));
407
- const system = (body.messages ?? []).find((m) => m.role === "system")?.content ?? "";
408
- const name = system.replace("JUDGE:", "");
409
- const score = JUDGE_SCORES[name];
410
- if (!score) throw new Error(`multishot golden judge wire: unknown judge system prompt ${system}`);
411
- return {
412
- ok: true,
413
- status: 200,
414
- json: async () => ({
415
- choices: [{ message: { content: JSON.stringify({
416
- ...score,
417
- notes: `${name} ok`
418
- }) } }],
419
- usage: {
420
- prompt_tokens: 300,
421
- completion_tokens: 25
422
- },
423
- model: "test/judge-model",
424
- _response_cost: 7e-4
425
- }),
426
- text: async () => ""
427
- };
428
- });
429
- return () => {
430
- globalThis.fetch = previous;
431
- judgeWireInstalled = false;
365
+ const judgeTransport = async (req) => {
366
+ judgeRequests.push(recordJudgeRequest({
367
+ model: req.model,
368
+ temperature: req.temperature,
369
+ max_tokens: req.maxTokens,
370
+ messages: req.messages
371
+ }));
372
+ const system = req.messages.find((m) => m.role === "system")?.content ?? "";
373
+ const name = system.replace("JUDGE:", "");
374
+ const score = JUDGE_SCORES[name];
375
+ if (!score) throw new Error(`multishot golden judge transport: unknown judge system prompt ${system}`);
376
+ return {
377
+ message: { content: JSON.stringify({
378
+ ...score,
379
+ notes: `${name} ok`
380
+ }) },
381
+ usage: {
382
+ prompt_tokens: 300,
383
+ completion_tokens: 25
384
+ },
385
+ model: "test/judge-model",
386
+ costUsd: 7e-4
432
387
  };
433
388
  };
434
389
  return {
435
- options,
390
+ options: {
391
+ profiles,
392
+ personas,
393
+ shape,
394
+ judges: {
395
+ conversation: judge("conversation", (input) => `Score this conversation of ${input.transcript.length} messages.`, judgeTransport),
396
+ codeReview: judge("code-review", (input) => `Score this code: ${input.artifact.content}`, judgeTransport),
397
+ contentQuality: judge("content-quality", (input) => `Score this content: ${input.artifact.content}`, judgeTransport)
398
+ },
399
+ tools,
400
+ toolExecutors: toolExecutors(),
401
+ artifactTypeFor,
402
+ runDir,
403
+ reps: 1,
404
+ maxTurns: 3,
405
+ maxConcurrency: 1,
406
+ agentModel: "test/agent-model",
407
+ driverModel: "test/driver-model",
408
+ agentTransport: async (req) => {
409
+ requests.push(recordRequest("agent", req));
410
+ return agentTransport(req);
411
+ },
412
+ driverTransport: async (req) => {
413
+ requests.push(recordRequest("driver", req));
414
+ return driverTransport(req);
415
+ }
416
+ },
436
417
  requests,
437
- judgeRequests,
438
- installJudgeWire
418
+ judgeRequests
439
419
  };
440
420
  }
441
421
  //#endregion
@@ -7220,8 +7200,6 @@ function delegationCase(overrides = {}) {
7220
7200
  maxTurns: overrides.maxTurns ?? 3,
7221
7201
  agentModel: "test/agent-model",
7222
7202
  driverModel: "test/driver-model",
7223
- apiKey: "golden-key",
7224
- baseUrl: "http://router.invalid",
7225
7203
  agentTransport: ledgerTransport(requests, "agent", overrides.agent ?? delegationAgent),
7226
7204
  driverTransport: ledgerTransport(requests, "driver", overrides.driver ?? delegationDriver)
7227
7205
  };
@@ -7440,8 +7418,6 @@ function samplingCase(overrides = {}) {
7440
7418
  agentModel: "scripted/agent",
7441
7419
  driverModel: "primary/driver",
7442
7420
  driverFallbackModels: ["fallback/driver"],
7443
- apiKey: "golden-key",
7444
- baseUrl: "http://router.invalid",
7445
7421
  agentTransport: ledgerTransport(requests, "agent", scriptedTransport(overrides.agentScript ?? samplingAgentScript(), "agent transport")),
7446
7422
  driverTransport: ledgerTransport(requests, "driver", scriptedTransport(overrides.driverScript ?? samplingDriverScript(), "driver transport"))
7447
7423
  },
@@ -7648,13 +7624,7 @@ async function checkMultishotGolden(opts) {
7648
7624
  async function checkMultishotMatrixGoldenScenario(opts) {
7649
7625
  const record = requireMatrixRecord(opts.records ?? goldenRecords(), opts.scenario.id);
7650
7626
  const runCase = opts.scenario.build(opts.runDir);
7651
- const restore = runCase.installJudgeWire();
7652
- let matrix;
7653
- try {
7654
- matrix = await opts.engine(runCase.options);
7655
- } finally {
7656
- restore();
7657
- }
7627
+ const matrix = await opts.engine(runCase.options);
7658
7628
  const mismatches = [
7659
7629
  ...compareJson(record.matrix, stripVolatile(matrix.matrix), "matrix"),
7660
7630
  ...compareJson(record.requests, runCase.requests, "requests"),
@@ -7681,6 +7651,6 @@ function isUsableDuration(value) {
7681
7651
  return typeof value === "number" && Number.isFinite(value) && value >= 0;
7682
7652
  }
7683
7653
  //#endregion
7684
- export { CURRENT_MULTISHOT_GOLDEN_VERSION, MultishotGoldenMismatchError, VOLATILE_KEYS, assertMultishotGoldenScenario, assertMultishotMatrixGoldenScenario, checkMultishotGolden, checkMultishotGoldenScenario, checkMultishotMatrixGoldenScenario, compareJson, goldenRecords, maskVolatileMarkdown, multishotGoldenScenarios, multishotGoldenVersions, multishotMatrixGoldenScenarios, readRunDir, recordError, recordJudgeRequest, recordMessage, recordRequest, recordResult, sortJudgeRequests, stripVolatile };
7654
+ export { CURRENT_MULTISHOT_GOLDEN_VERSION, MultishotGoldenMismatchError, assertMultishotGoldenScenario, assertMultishotMatrixGoldenScenario, checkMultishotGolden, checkMultishotGoldenScenario, checkMultishotMatrixGoldenScenario, compareJson, goldenRecords, maskVolatileMarkdown, multishotGoldenScenarios, multishotGoldenVersions, multishotMatrixGoldenScenarios, readRunDir, recordError, recordJudgeRequest, recordMessage, recordRequest, recordResult, sortJudgeRequests, stripVolatile };
7685
7655
 
7686
7656
  //# sourceMappingURL=index.js.map