@tangle-network/agent-eval 0.150.1 → 0.161.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (353) hide show
  1. package/CHANGELOG.md +178 -1
  2. package/README.md +7 -3
  3. package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
  4. package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
  5. package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
  6. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +19 -36
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
  12. package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
  13. package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
  14. package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
  15. package/dist/{benchmark-command-CAFwbH0L.js → benchmark-command-BVtaq_ve.js} +26 -31
  16. package/dist/benchmark-command-BVtaq_ve.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +6 -19
  18. package/dist/benchmarks/index.d.ts.map +1 -1
  19. package/dist/benchmarks/index.js +4 -4
  20. package/dist/benchmarks/index.js.map +1 -1
  21. package/dist/builder-eval/index.d.ts +3 -3
  22. package/dist/builder-eval/index.js +3 -3
  23. package/dist/campaign/index.d.ts +9 -9
  24. package/dist/campaign/index.js +7 -7
  25. package/dist/{campaign-CN_7xJdV.js → campaign-BSmOwskD.js} +77 -795
  26. package/dist/campaign-BSmOwskD.js.map +1 -0
  27. package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
  28. package/dist/canonical-IL-Bu-14.js.map +1 -0
  29. package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
  30. package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
  31. package/dist/{chat-client-2bVfrzhN.js → chat-client-DlMlAeYI.js} +5 -55
  32. package/dist/{chat-client-2bVfrzhN.js.map → chat-client-DlMlAeYI.js.map} +1 -1
  33. package/dist/chat-json-call-6g5sJobJ.js +53 -0
  34. package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
  35. package/dist/cli.js +54 -19
  36. package/dist/cli.js.map +1 -1
  37. package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
  38. package/dist/client-CX7KqIdB.js.map +1 -0
  39. package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
  40. package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -27
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +14 -17
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
  46. package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
  47. package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
  48. package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
  49. package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
  50. package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
  51. package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
  52. package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
  53. package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
  54. package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
  55. package/dist/{define-agent-eval-rqNyVhVV.js → define-agent-eval-h-s-sI-v.js} +17 -11
  56. package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
  57. package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
  58. package/dist/descriptive-jDOuI6mz.js.map +1 -0
  59. package/dist/{dspy-rlm-engine-DHI0WrUU.js → dspy-rlm-engine-DptEII26.js} +95 -15
  60. package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
  61. package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
  62. package/dist/emitter-BpYFQPj4.js.map +1 -0
  63. package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
  64. package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
  65. package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
  66. package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
  67. package/dist/{eval-campaign-C4jmuM-b.js → eval-campaign-BsXWL2-2.js} +17 -28
  68. package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
  69. package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
  70. package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
  71. package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
  72. package/dist/exec-D9WpA2p-.js.map +1 -0
  73. package/dist/experiment/index.d.ts +45 -11
  74. package/dist/experiment/index.d.ts.map +1 -1
  75. package/dist/experiment/index.js +30 -11
  76. package/dist/experiment/index.js.map +1 -1
  77. package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
  78. package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
  79. package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
  80. package/dist/exporters-Df7TgHFv.js.map +1 -0
  81. package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
  82. package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
  83. package/dist/{external-optimizer-process-Bhmzngf-.js → external-optimizer-process-WosTBChy.js} +4 -4
  84. package/dist/{external-optimizer-process-Bhmzngf-.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
  85. package/dist/{external-optimizer-subprocess-Dn90UqN2.js → external-optimizer-subprocess-BIWbHpgD.js} +10 -6
  86. package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +1 -0
  87. package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
  88. package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
  89. package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
  90. package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
  91. package/dist/fuzz.js +1 -1
  92. package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
  93. package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
  94. package/dist/hosted/index.d.ts +3 -14
  95. package/dist/hosted/index.d.ts.map +1 -1
  96. package/dist/hosted/index.js +2 -2
  97. package/dist/{index-CTKpu9ry.d.ts → index-CGtH1piv.d.ts} +48 -26
  98. package/dist/index-CGtH1piv.d.ts.map +1 -0
  99. package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
  100. package/dist/index-D-IiQIBB.d.ts.map +1 -0
  101. package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
  102. package/dist/index-D-V8gCs_.d.ts.map +1 -0
  103. package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
  104. package/dist/index-lfaSeKSD.d.ts.map +1 -0
  105. package/dist/index-vrJugRal.d.ts +1 -0
  106. package/dist/index.d.ts +68 -125
  107. package/dist/index.d.ts.map +1 -1
  108. package/dist/index.js +61 -112
  109. package/dist/index.js.map +1 -1
  110. package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
  111. package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
  112. package/dist/{integrity-DysDBWDu.js → integrity-CyWSSoQS.js} +17 -4
  113. package/dist/integrity-CyWSSoQS.js.map +1 -0
  114. package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
  115. package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
  116. package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
  117. package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
  118. package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
  119. package/dist/kind-factory-DY8FdoXf.js.map +1 -0
  120. package/dist/ledger-core/index.d.ts +2 -2
  121. package/dist/ledger-core/index.js +3 -3
  122. package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
  123. package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
  124. package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
  125. package/dist/llm-client-hgDieDNN.js.map +1 -0
  126. package/dist/{llm-judge-CVq33oz1.js → llm-judge-BhasIPFT.js} +1136 -78
  127. package/dist/llm-judge-BhasIPFT.js.map +1 -0
  128. package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
  129. package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
  130. package/dist/meta-eval/index.d.ts +8 -6
  131. package/dist/meta-eval/index.d.ts.map +1 -1
  132. package/dist/meta-eval/index.js +9 -7
  133. package/dist/meta-eval/index.js.map +1 -1
  134. package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
  135. package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
  136. package/dist/multishot/golden/index.d.ts +2 -8
  137. package/dist/multishot/golden/index.d.ts.map +1 -1
  138. package/dist/multishot/golden/index.js +56 -86
  139. package/dist/multishot/golden/index.js.map +1 -1
  140. package/dist/multishot/index.d.ts +11 -46
  141. package/dist/multishot/index.d.ts.map +1 -1
  142. package/dist/multishot/index.js +30 -83
  143. package/dist/multishot/index.js.map +1 -1
  144. package/dist/openapi.json +1 -1
  145. package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
  146. package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
  147. package/dist/pipelines/index.d.ts +5 -5
  148. package/dist/pipelines/index.js +3 -3
  149. package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
  150. package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
  151. package/dist/pre-registration-KN9jkh58.js +110 -0
  152. package/dist/pre-registration-KN9jkh58.js.map +1 -0
  153. package/dist/{produced-state-jfk8Du3b.js → produced-state-DZ89riy5.js} +8 -8
  154. package/dist/produced-state-DZ89riy5.js.map +1 -0
  155. package/dist/profile-cell.d.ts +1 -1
  156. package/dist/profile-cell.js +31 -5
  157. package/dist/profile-cell.js.map +1 -1
  158. package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
  159. package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
  160. package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
  161. package/dist/query-CHmMP42p.js.map +1 -0
  162. package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
  163. package/dist/query-DxPYqpmT.d.ts.map +1 -0
  164. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  165. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  166. package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
  167. package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
  168. package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
  169. package/dist/release-confidence-DKfD2RYU.js.map +1 -0
  170. package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
  171. package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
  172. package/dist/reporting.d.ts +3 -3
  173. package/dist/reporting.js +3 -3
  174. package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
  175. package/dist/researcher-Cz565b7D.d.ts.map +1 -0
  176. package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
  177. package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
  178. package/dist/{reward-hacking-DKI9T52l.js → reward-hacking-t4lB1yt8.js} +3 -3
  179. package/dist/{reward-hacking-DKI9T52l.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
  180. package/dist/rl.d.ts +11 -42
  181. package/dist/rl.d.ts.map +1 -1
  182. package/dist/rl.js +41 -24
  183. package/dist/rl.js.map +1 -1
  184. package/dist/rollout/index.d.ts +3 -3
  185. package/dist/rollout/index.js +7 -7
  186. package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
  187. package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
  188. package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CK8SCOg-.js} +6 -16
  189. package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
  190. package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
  191. package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
  192. package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
  193. package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
  194. package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
  195. package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
  196. package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
  197. package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
  198. package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
  199. package/dist/schema-BzWDXhOR.d.ts.map +1 -0
  200. package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
  201. package/dist/schema-C1aaAxTf.js.map +1 -0
  202. package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
  203. package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
  204. package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
  205. package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
  206. package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
  207. package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
  208. package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
  209. package/dist/sequential-rYW-Ophm.js.map +1 -0
  210. package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
  211. package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
  212. package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
  213. package/dist/server-BtFd4uzB.js.map +1 -0
  214. package/dist/{skillopt-optimization-method-DLeUcK-K.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
  215. package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
  216. package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
  217. package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
  218. package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
  219. package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
  220. package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
  221. package/dist/steps-CiNVJry_.d.ts.map +1 -0
  222. package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
  223. package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
  224. package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
  225. package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
  226. package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
  227. package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
  228. package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
  229. package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
  230. package/dist/storyboard/index.d.ts +1 -1
  231. package/dist/{summary-report-Blysd6Z2.js → summary-report-BI5hUtvK.js} +7 -17
  232. package/dist/summary-report-BI5hUtvK.js.map +1 -0
  233. package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
  234. package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
  235. package/dist/supervisor-run/index.d.ts +27 -18
  236. package/dist/supervisor-run/index.d.ts.map +1 -1
  237. package/dist/supervisor-run/index.js +104 -26
  238. package/dist/supervisor-run/index.js.map +1 -1
  239. package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
  240. package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
  241. package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
  242. package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
  243. package/dist/{tool-waste-BDdBZG1F.js → tool-waste-BqzmVdJk.js} +4 -4
  244. package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
  245. package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
  246. package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
  247. package/dist/trace-repair/index.d.ts +4 -77
  248. package/dist/trace-repair/index.d.ts.map +1 -1
  249. package/dist/trace-repair/index.js +5 -15
  250. package/dist/trace-repair/index.js.map +1 -1
  251. package/dist/traces.d.ts +13 -23
  252. package/dist/traces.d.ts.map +1 -1
  253. package/dist/traces.js +9 -20
  254. package/dist/traces.js.map +1 -1
  255. package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
  256. package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
  257. package/dist/trajectory-replay/index.d.ts +5 -5
  258. package/dist/trajectory-replay/index.js +5 -5
  259. package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
  260. package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
  261. package/dist/types-B3jzCp0p.js.map +1 -1
  262. package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
  263. package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
  264. package/dist/types-Bfk0uxRj.d.ts +443 -0
  265. package/dist/types-Bfk0uxRj.d.ts.map +1 -0
  266. package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
  267. package/dist/types-D4s7Z6nq.d.ts.map +1 -0
  268. package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
  269. package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
  270. package/dist/{types-yLK8gXE9.d.ts → types-DeIUdzNd.d.ts} +160 -10
  271. package/dist/types-DeIUdzNd.d.ts.map +1 -0
  272. package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
  273. package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
  274. package/dist/verdict-cache-CdVVTVmn.js +88 -0
  275. package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
  276. package/dist/wire/index.d.ts +21 -111
  277. package/dist/wire/index.d.ts.map +1 -1
  278. package/dist/wire/index.js +2 -2
  279. package/docs/adapters-observability.md +9 -23
  280. package/docs/building-doctrine.md +3 -3
  281. package/docs/campaign-proposers.md +54 -15
  282. package/docs/concepts.md +3 -4
  283. package/docs/design/statistics-decisions.md +89 -1
  284. package/docs/eval-surface-map.md +14 -0
  285. package/docs/experiment.md +19 -2
  286. package/docs/feedback-trajectories.md +1 -1
  287. package/docs/multishot-golden-records.md +4 -4
  288. package/docs/public-api.md +1616 -0
  289. package/docs/research-report-methodology.md +1 -1
  290. package/docs/search-history-receipts.md +39 -1
  291. package/docs/trace-analysis.md +1 -1
  292. package/docs/trace-repair-admission.md +1 -1
  293. package/docs/trace-repair-continuation.md +1 -1
  294. package/docs/verdicts.md +24 -0
  295. package/docs/wire-protocol.md +1 -1
  296. package/package.json +6 -2
  297. package/dist/active-curriculum-C4mk67HP.js.map +0 -1
  298. package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
  299. package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
  300. package/dist/benchmark-command-CAFwbH0L.js.map +0 -1
  301. package/dist/campaign-CN_7xJdV.js.map +0 -1
  302. package/dist/canonical-D-XsTQ6_.js.map +0 -1
  303. package/dist/client-LIuo-KPv.js.map +0 -1
  304. package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
  305. package/dist/define-agent-eval-rqNyVhVV.js.map +0 -1
  306. package/dist/descriptive-B5MwKfbf.js.map +0 -1
  307. package/dist/dspy-rlm-engine-DHI0WrUU.js.map +0 -1
  308. package/dist/emitter-CPBAhxum.js.map +0 -1
  309. package/dist/eval-campaign-C4jmuM-b.js.map +0 -1
  310. package/dist/exec-y-DCLqK7.js.map +0 -1
  311. package/dist/exporters-q9iL-2Jf.js.map +0 -1
  312. package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +0 -1
  313. package/dist/index-B8Ui1mr1.d.ts.map +0 -1
  314. package/dist/index-BNPtkBPf.d.ts.map +0 -1
  315. package/dist/index-C1ravkGA.d.ts +0 -1
  316. package/dist/index-CTKpu9ry.d.ts.map +0 -1
  317. package/dist/index-IQccV3Ou.d.ts.map +0 -1
  318. package/dist/integrity-DysDBWDu.js.map +0 -1
  319. package/dist/kind-factory-DmAa0h3K.js.map +0 -1
  320. package/dist/llm-client-Bg32RW0j.js.map +0 -1
  321. package/dist/llm-judge-CVq33oz1.js.map +0 -1
  322. package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
  323. package/dist/pre-registration-DakwTRXk.js +0 -96
  324. package/dist/pre-registration-DakwTRXk.js.map +0 -1
  325. package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
  326. package/dist/produced-state-jfk8Du3b.js.map +0 -1
  327. package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
  328. package/dist/query-CJ_DX8vl.d.ts.map +0 -1
  329. package/dist/query-Di7eEQ79.js.map +0 -1
  330. package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
  331. package/dist/release-confidence-BknrpBnO.js.map +0 -1
  332. package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
  333. package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
  334. package/dist/rubric-predictive-validity-Cwwyd7ah.js.map +0 -1
  335. package/dist/schema-C6DW4ZHR.js.map +0 -1
  336. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  337. package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
  338. package/dist/sequential-C458DXNf.js.map +0 -1
  339. package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
  340. package/dist/server-dIWwF3j_.js.map +0 -1
  341. package/dist/skillopt-optimization-method-DLeUcK-K.js.map +0 -1
  342. package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
  343. package/dist/steps-AmkT-GIM.d.ts.map +0 -1
  344. package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
  345. package/dist/summary-report-Blysd6Z2.js.map +0 -1
  346. package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
  347. package/dist/types-CLAwnY-L.d.ts.map +0 -1
  348. package/dist/types-DdFNuyxQ.d.ts.map +0 -1
  349. package/dist/types-jUBXJ7Iz.d.ts +0 -884
  350. package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
  351. package/dist/types-yLK8gXE9.d.ts.map +0 -1
  352. package/dist/verdict-cache-mZf5FEiY.js +0 -107
  353. package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
@@ -1 +0,0 @@
1
- {"version":3,"file":"dspy-rlm-engine-DHI0WrUU.js","names":["isRecord"],"sources":["../src/analyst/trace-tool-callback.ts","../src/analyst/dspy-rlm-engine.ts"],"sourcesContent":["import { randomBytes } from 'node:crypto'\nimport { createServer, type IncomingMessage, type ServerResponse } from 'node:http'\nimport {\n type ExternalOptimizerCallbackLimits,\n resolveExternalOptimizerCallbackLimits,\n} from '../campaign/external-optimizer-contracts'\nimport { closeServer, listenLocal, sendJson } from '../campaign/external-optimizer-http'\nimport type { TraceAnalysisToolDescriptor } from '../trace-analyst/tools'\n\nexport interface TraceToolCallback {\n url: string\n token: string\n calls: () => number\n close: () => Promise<void>\n}\n\nexport type TraceToolCallbackLimits = ExternalOptimizerCallbackLimits\n\n/** Expose one bounded trace-tool set only on an authenticated loopback socket. */\nexport async function startTraceToolCallback(args: {\n tools: readonly TraceAnalysisToolDescriptor[]\n maxCalls: number\n /** Trace-tool request/response byte limits. Omitted fields use finite defaults. */\n limits?: Partial<TraceToolCallbackLimits>\n signal?: AbortSignal\n}): Promise<TraceToolCallback> {\n if (!Number.isSafeInteger(args.maxCalls) || args.maxCalls <= 0) {\n throw new TypeError('trace tool callback maxCalls must be a positive safe integer')\n }\n const limits = resolveExternalOptimizerCallbackLimits(args.limits, 'trace tool callback limits')\n args.signal?.throwIfAborted()\n const byName = new Map(args.tools.map((tool) => [tool.name, tool]))\n if (byName.size !== args.tools.length) {\n throw new Error('trace tool callback received duplicate tool names')\n }\n\n const token = randomBytes(32).toString('hex')\n let calls = 0\n let accepting = true\n let closePromise: Promise<void> | undefined\n const activeControllers = new Set<AbortController>()\n const activeHandlers = new Set<Promise<void>>()\n const server = createServer((request, response) => {\n if (!accepting) {\n sendJsonIfOpen(response, 503, { error: 'trace tool callback is closing' })\n return\n }\n const controller = new AbortController()\n const abortRequest = (): void => {\n request.destroy()\n response.destroy()\n }\n activeControllers.add(controller)\n controller.signal.addEventListener('abort', abortRequest, { once: true })\n\n let handler!: Promise<void>\n handler = handleRequest(request, response, controller.signal).finally(() => {\n controller.signal.removeEventListener('abort', abortRequest)\n activeControllers.delete(controller)\n activeHandlers.delete(handler)\n })\n activeHandlers.add(handler)\n void handler.catch(() => undefined)\n })\n const port = await listenLocal(server)\n const close = (): Promise<void> => {\n closePromise ??= closeCallback()\n return closePromise\n }\n const onAbort = (): void => {\n void close().catch(() => undefined)\n }\n args.signal?.addEventListener('abort', onAbort, { once: true })\n if (args.signal?.aborted) onAbort()\n\n return {\n url: `http://127.0.0.1:${port}/call`,\n token,\n calls: () => calls,\n close,\n }\n\n async function handleRequest(\n request: IncomingMessage,\n response: ServerResponse,\n signal: AbortSignal,\n ): Promise<void> {\n try {\n if (request.method !== 'POST' || request.url !== '/call') {\n sendJsonIfOpen(response, 404, { error: 'not found' })\n return\n }\n if (request.headers.authorization !== `Bearer ${token}`) {\n sendJsonIfOpen(response, 401, { error: 'unauthorized' })\n return\n }\n if (calls >= args.maxCalls) {\n sendJsonIfOpen(response, 429, { error: 'trace tool call limit reached' })\n return\n }\n const body = await readJson(request, limits.maxRequestBytes)\n if (!isRecord(body) || typeof body.name !== 'string' || !('args' in body)) {\n sendJsonIfOpen(response, 400, { error: 'name and args are required' })\n return\n }\n const tool = byName.get(body.name)\n if (!tool) {\n sendJsonIfOpen(response, 404, { error: `unknown trace tool '${body.name}'` })\n return\n }\n calls += 1\n const result = await tool.handler(body.args, { signal })\n const encoded = JSON.stringify({ result })\n if (Buffer.byteLength(encoded) > limits.maxResponseBytes) {\n sendJsonIfOpen(response, 413, { error: 'trace tool response too large' })\n return\n }\n response.writeHead(200, {\n 'content-type': 'application/json; charset=utf-8',\n 'content-length': String(Buffer.byteLength(encoded)),\n })\n response.end(encoded)\n } catch (error) {\n sendJsonIfOpen(response, signal.aborted ? 499 : 400, {\n error: error instanceof Error ? error.message : String(error),\n })\n }\n }\n\n async function closeCallback(): Promise<void> {\n args.signal?.removeEventListener('abort', onAbort)\n accepting = false\n const closingServer = closeServer(server)\n server.closeIdleConnections?.()\n for (const controller of activeControllers) controller.abort()\n const [serverResult] = await Promise.allSettled([\n closingServer,\n waitForActiveHandlers(activeHandlers),\n ])\n if (activeControllers.size !== 0 || activeHandlers.size !== 0) {\n throw new Error('trace tool callback closed with active requests')\n }\n if (serverResult?.status === 'rejected') throw serverResult.reason\n }\n}\n\nasync function waitForActiveHandlers(activeHandlers: Set<Promise<void>>): Promise<void> {\n while (activeHandlers.size > 0) {\n await Promise.allSettled([...activeHandlers])\n }\n}\n\nfunction readJson(request: IncomingMessage, maxRequestBytes: number): Promise<unknown> {\n return new Promise((resolve, reject) => {\n let size = 0\n const chunks: Buffer[] = []\n request.on('data', (chunk: Buffer) => {\n size += chunk.length\n if (size > maxRequestBytes) {\n reject(new Error('trace tool request too large'))\n request.destroy()\n return\n }\n chunks.push(chunk)\n })\n request.on('error', reject)\n request.on('end', () => {\n try {\n resolve(JSON.parse(Buffer.concat(chunks).toString('utf8')))\n } catch (error) {\n reject(error)\n }\n })\n })\n}\n\nfunction sendJsonIfOpen(response: ServerResponse, status: number, body: unknown): void {\n if (response.destroyed || response.writableEnded) return\n sendJson(response, status, body)\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n","import {\n assertExternalOptimizerModelBudget,\n type ExternalOptimizerModelCall,\n type ExternalOptimizerModelExecutionObservation,\n type ExternalOptimizerModelProxy,\n type ExternalOptimizerRunnerCommand,\n removeCredentialEnvironment,\n resolveExternalOptimizerCallbackLimits,\n resolveExternalOptimizerProcessLimits,\n} from '../campaign/external-optimizer-contracts'\nimport { startExternalOptimizerModelProxy } from '../campaign/external-optimizer-model-proxy'\nimport { runWithCleanup } from '../campaign/external-optimizer-resources'\nimport { runExternalOptimizerProcess } from '../campaign/external-optimizer-subprocess'\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport { resolveModelPricing } from '../metrics'\nimport type { TraceAnalysisEngine, TraceAnalysisEngineResult } from './engine'\nimport { type RawAnalystFinding, RawAnalystFindingSchema } from './finding-signature'\nimport { startTraceToolCallback, type TraceToolCallbackLimits } from './trace-tool-callback'\n\nconst DEFAULT_TIMEOUT_MS = 10 * 60_000\n// 4096 is below what current coding models emit for a full findings array:\n// glm-5.2 through an OpenAI-compatible gateway returns 8192 and the request is\n// rejected outright (`502 — provider reported 8192 completion tokens,\n// exceeding requested limit 4096`), so the default failed 2/2 smoke cases\n// before any analysis ran. The cap exists to bound spend, and `maxCostUsd`\n// already does that directly, so it starts above what a real completion needs.\nconst DEFAULT_MODEL_OUTPUT_TOKENS = 16_384\nconst DEFAULT_MAX_COST_USD = 1\nconst DEFAULT_MAX_MODEL_REQUEST_BYTES = 16 * 1024 * 1024\nconst DEFAULT_MAX_MODEL_RESPONSE_BYTES = 4 * 1024 * 1024\nconst DEFAULT_TRACE_TOOL_TIMEOUT_MS = 60_000\nconst MAX_TIMER_DELAY_MS = 2_147_483_647\nconst BRIDGE_MODULE = 'agent_eval_rpc.dspy_rlm_bridge'\n/** Bumped whenever this engine's execution behavior changes. */\nconst DSPY_RLM_ENGINE_VERSION = '1.0.0'\n\nexport interface DspyRlmTraceEngineOptions {\n /** Caller-owned execution path. Agent Eval never receives provider credentials. */\n call: ExternalOptimizerModelCall\n /** Stable public identity for the caller-owned path, such as an AgentProfile digest. */\n callRef: string\n /** Persist the finite execution record returned for every admitted call. */\n recordExecution: (observation: ExternalOptimizerModelExecutionObservation) => void\n model: string\n /** Exact provider rates. Required when the model is absent from the pricing table. */\n pricing?: CustomTokenPricing\n /** Maximum provider spend for one investigation. Default: 1 USD. */\n maxCostUsd?: number\n /** Controller response cap. Default: 16384. */\n maxOutputTokens?: number\n /**\n * Thinking tokens one controller turn may bill on top of its completion.\n * A reasoning model bills these beyond `maxOutputTokens`, so the cost\n * reservation must cover them. Default: four times the completion cap.\n */\n maxReasoningTokens?: number\n /** Maximum caller-owned model invocations. Default derives from the analysis limits. */\n maxModelRequests?: number\n /** Maximum model request bytes. Default: 16 MiB. */\n maxModelRequestBytes?: number\n /** Maximum model response bytes. Default: 4 MiB. */\n maxModelResponseBytes?: number\n /** Deadline for one caller-owned model invocation. Default: the whole analysis deadline. */\n modelRequestTimeoutMs?: number\n /** Trace-tool loopback request/response byte limits. */\n traceToolLimits?: Partial<TraceToolCallbackLimits>\n /** Deadline for one Python-to-Node trace-tool call. Default: 60 seconds. */\n traceToolTimeoutMs?: number\n /**\n * How the controller's reasoning and code fields are obtained.\n *\n * `tolerant` parses marker output strictly first, then recovers the fields\n * deterministically from prose plus a fenced code block — the shape coding\n * models naturally emit — at no extra model cost. `two-step` extracts with a\n * second call per turn. `chat` accepts marker output only. Default:\n * `tolerant`.\n */\n controlAdapter?: 'chat' | 'two-step' | 'tolerant'\n /** Python command used to load agent-eval-rpc[dspy]. Default: python. */\n runner?: ExternalOptimizerRunnerCommand\n /** Whole investigation deadline. Default: 10 minutes. */\n timeoutMs?: number\n}\n\n/** Use the official DSPy RLM as a bounded recursive trace-analysis engine. */\nexport function createDspyRlmTraceEngine(options: DspyRlmTraceEngineOptions): TraceAnalysisEngine {\n assertOptions(options)\n const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_MODEL_OUTPUT_TOKENS\n if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) {\n throw new TypeError('DSPy RLM maxOutputTokens must be a positive safe integer')\n }\n const controlAdapter = options.controlAdapter ?? 'tolerant'\n const maxReasoningTokens = options.maxReasoningTokens ?? maxOutputTokens * 4\n if (!Number.isSafeInteger(maxReasoningTokens) || maxReasoningTokens < 0) {\n throw new TypeError('DSPy RLM maxReasoningTokens must be a non-negative safe integer')\n }\n const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS\n assertTimerDelay(timeoutMs, 'timeoutMs')\n const maxCostUsd = options.maxCostUsd ?? DEFAULT_MAX_COST_USD\n if (!Number.isFinite(maxCostUsd) || maxCostUsd <= 0) {\n throw new TypeError('DSPy RLM maxCostUsd must be positive and finite')\n }\n const pricing = options.pricing ?? pricingForModel(options.model)\n const maxModelRequestBytes = options.maxModelRequestBytes ?? DEFAULT_MAX_MODEL_REQUEST_BYTES\n const maxModelResponseBytes = options.maxModelResponseBytes ?? DEFAULT_MAX_MODEL_RESPONSE_BYTES\n const modelRequestTimeoutMs = options.modelRequestTimeoutMs ?? timeoutMs\n const traceToolTimeoutMs = options.traceToolTimeoutMs ?? DEFAULT_TRACE_TOOL_TIMEOUT_MS\n assertExternalOptimizerModelBudget(\n {\n maxCostUsd,\n maxRequests: options.maxModelRequests ?? 1,\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n pricing,\n requestTimeoutMs: modelRequestTimeoutMs,\n },\n 'DSPy RLM model limits',\n )\n resolveExternalOptimizerCallbackLimits(options.traceToolLimits, 'DSPy RLM trace tool limits')\n assertTimerDelay(traceToolTimeoutMs, 'traceToolTimeoutMs')\n\n const runner = sanitizedRunner(options.runner)\n const processLimits = resolveExternalOptimizerProcessLimits(runner?.limits)\n return {\n id: 'dspy-rlm',\n description: 'Official DSPy RLM with bounded trace tools and metered model calls.',\n model: options.model,\n version: DSPY_RLM_ENGINE_VERSION,\n executionConfig: {\n bridge_module: BRIDGE_MODULE,\n call_ref: options.callRef,\n model: options.model,\n pricing: { ...pricing },\n max_cost_usd: maxCostUsd,\n max_output_tokens: maxOutputTokens,\n max_reasoning_tokens: maxReasoningTokens,\n control_adapter: controlAdapter,\n timeout_ms: timeoutMs,\n max_model_requests: options.maxModelRequests ?? null,\n max_request_bytes: maxModelRequestBytes,\n max_response_bytes: maxModelResponseBytes,\n model_request_timeout_ms: modelRequestTimeoutMs,\n trace_tool_limits: options.traceToolLimits ?? null,\n trace_tool_timeout_ms: traceToolTimeoutMs,\n process_limits: processLimits,\n runner: runner ? 'caller-supplied' : 'default',\n runner_command: runner?.command ?? null,\n },\n async analyze(request) {\n const callback = await startTraceToolCallback({\n tools: request.tools,\n maxCalls: request.limits.maxToolCalls,\n ...(options.traceToolLimits ? { limits: options.traceToolLimits } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n let modelProxy: ExternalOptimizerModelProxy | undefined\n const modelExecutions: ExternalOptimizerModelExecutionObservation[] = []\n const result = await runWithCleanup({\n label: 'DSPy RLM trace-analysis resources',\n run: async () => {\n modelProxy = await startExternalOptimizerModelProxy({\n call: options.call,\n callRef: options.callRef,\n recordExecution: (observation) => {\n modelExecutions.push(structuredClone(observation))\n options.recordExecution(observation)\n },\n model: options.model,\n budget: {\n maxCostUsd,\n maxRequests: resolveMaxModelRequests(options.maxModelRequests, request.limits),\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n requestTimeoutMs: modelRequestTimeoutMs,\n pricing,\n },\n costLedger: request.costLedger,\n channel: 'analyst',\n phase: request.costPhase,\n actor: request.analystId,\n ...(request.costTags ? { tags: request.costTags } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n request.log?.('trace analyst engine started', {\n engine: 'dspy-rlm',\n model: options.model,\n tools: request.tools.map((tool) => tool.name),\n limits: request.limits,\n })\n const raw = await runExternalOptimizerProcess<unknown>({\n label: 'DSPy RLM trace analysis',\n tempPrefix: 'agent-eval-dspy-rlm-',\n module: BRIDGE_MODULE,\n input: {\n operation: 'analyze',\n question: request.question,\n instructions: request.instructions,\n modelProxy: {\n baseUrl: modelProxy.baseUrl,\n apiKey: modelProxy.apiKey,\n model: options.model,\n maxOutputTokens,\n },\n toolCallback: {\n url: callback.url,\n token: callback.token,\n timeoutMs: traceToolTimeoutMs,\n },\n toolSpecs: request.tools.map(({ name, description, parameters }) => ({\n name,\n description,\n parameters,\n })),\n controlAdapter,\n limits: {\n maxIterations: request.limits.maxIterations,\n maxLlmCalls: request.limits.maxLlmCalls,\n maxOutputChars: request.limits.maxOutputChars,\n },\n // Omitted entirely when the caller supplied none, so a request\n // without structured inputs sends the payload it always sent.\n ...(request.taskInputs ? { taskInputs: request.taskInputs } : {}),\n },\n ...(runner ? { runner } : {}),\n additionalArgs: ['--max-input-bytes', String(processLimits.maxInputBytes)],\n timeoutMs,\n ...(request.signal ? { signal: request.signal } : {}),\n })\n const parsed = parseBridgeOutput(raw, (index, reason) => {\n request.log?.('finding rejected: bridge row failed schema validation', {\n engine: 'dspy-rlm',\n index,\n reason,\n })\n })\n const successfulCompletions = modelProxy.successfulCompletions()\n const requestAttempts = modelProxy.requestAttempts()\n if (parsed.modelCalls !== successfulCompletions) {\n throw new Error(\n `DSPy RLM reported ${parsed.modelCalls} model calls, but the provider proxy recorded ${successfulCompletions}`,\n )\n }\n modelProxy.assertExecutionComplete()\n return {\n ...parsed,\n toolCalls: callback.calls(),\n runtime: {\n ...parsed.runtime,\n modelRequestAttempts: requestAttempts,\n modelSuccessfulCompletions: successfulCompletions,\n modelExecutions,\n },\n } satisfies TraceAnalysisEngineResult\n },\n cleanup: async () => {\n const results = await Promise.allSettled([\n ...(modelProxy ? [modelProxy.close()] : []),\n callback.close(),\n ])\n const errors = results.flatMap((entry) =>\n entry.status === 'rejected' ? [entry.reason] : [],\n )\n if (errors.length > 0) {\n throw new AggregateError(errors, 'DSPy RLM resource cleanup failed')\n }\n },\n })\n request.log?.('trace analyst engine completed', {\n engine: 'dspy-rlm',\n model_calls: result.modelCalls,\n model_request_attempts: result.runtime.modelRequestAttempts,\n tool_calls: result.toolCalls,\n findings: result.findings.length,\n })\n return result\n },\n }\n}\n\nfunction parseBridgeOutput(\n value: unknown,\n onRejectedFinding: (index: number, reason: string) => void,\n): Omit<TraceAnalysisEngineResult, 'toolCalls'> {\n if (!isRecord(value)) throw new Error('DSPy RLM bridge output must be an object')\n if (typeof value.answer !== 'string' || !value.answer.trim()) {\n throw new Error('DSPy RLM bridge returned no answer')\n }\n if (!Array.isArray(value.findings)) {\n throw new Error('DSPy RLM bridge findings must be an array')\n }\n // Findings are model output: one malformed row is model noise, not a bridge\n // fault, and the rest of the paid investigation must survive it. Rejected\n // rows are logged per row and counted in runtime.rejectedFindings.\n let rejectedFindings = 0\n const findings: RawAnalystFinding[] = []\n value.findings.forEach((finding, index) => {\n const parsed = RawAnalystFindingSchema.safeParse(finding)\n if (!parsed.success) {\n rejectedFindings += 1\n onRejectedFinding(\n index,\n parsed.error.issues.map((issue) => `${issue.path.join('.')}: ${issue.message}`).join('; '),\n )\n return\n }\n findings.push(parsed.data)\n })\n if (!Array.isArray(value.trajectory)) {\n throw new Error('DSPy RLM bridge trajectory must be an array')\n }\n if (!Number.isSafeInteger(value.modelCalls) || (value.modelCalls as number) <= 0) {\n throw new Error('DSPy RLM bridge modelCalls must be a positive safe integer')\n }\n if (!isRecord(value.runtime)) {\n throw new Error('DSPy RLM bridge runtime must be an object')\n }\n return {\n answer: value.answer,\n findings,\n trajectory: value.trajectory,\n modelCalls: value.modelCalls as number,\n runtime: { ...value.runtime, rejectedFindings },\n }\n}\n\nfunction assertOptions(options: DspyRlmTraceEngineOptions): void {\n for (const [name, value] of [\n ['callRef', options.callRef],\n ['model', options.model],\n ] as const) {\n if (typeof value !== 'string' || !value.trim() || value !== value.trim()) {\n throw new TypeError(`DSPy RLM ${name} must be a trimmed non-empty string`)\n }\n }\n if (typeof options.call !== 'function') {\n throw new TypeError('DSPy RLM call must be a function')\n }\n if (typeof options.recordExecution !== 'function') {\n throw new TypeError('DSPy RLM recordExecution must be a function')\n }\n if (\n options.maxModelRequests !== undefined &&\n (!Number.isSafeInteger(options.maxModelRequests) || options.maxModelRequests <= 0)\n ) {\n throw new TypeError('DSPy RLM maxModelRequests must be a positive safe integer')\n }\n}\n\nfunction resolveMaxModelRequests(\n configured: number | undefined,\n limits: { maxIterations: number; maxLlmCalls: number },\n): number {\n if (configured !== undefined) return configured\n const derived = limits.maxIterations + limits.maxLlmCalls + 1\n if (!Number.isSafeInteger(derived) || derived <= 0) {\n throw new Error('DSPy RLM analysis limits produce an invalid model request limit')\n }\n return derived\n}\n\nfunction assertTimerDelay(value: number, field: string): void {\n if (!Number.isSafeInteger(value) || value <= 0 || value > MAX_TIMER_DELAY_MS) {\n throw new TypeError(`DSPy RLM ${field} must be between 1 and ${MAX_TIMER_DELAY_MS}`)\n }\n}\n\nfunction pricingForModel(model: string): CustomTokenPricing {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n throw new Error(\n `no pricing is configured for '${model}'; provide DspyRlmTraceEngineOptions.pricing`,\n )\n }\n return {\n inputUsdPerMillion: pricing.input * 1_000,\n outputUsdPerMillion: pricing.output * 1_000,\n }\n}\n\nfunction sanitizedRunner(\n runner: ExternalOptimizerRunnerCommand | undefined,\n): ExternalOptimizerRunnerCommand | undefined {\n if (!runner) return undefined\n return {\n ...(runner.command ? { command: runner.command } : {}),\n ...(runner.args ? { args: runner.args } : {}),\n ...(runner.env ? { env: removeCredentialEnvironment(runner.env) } : {}),\n ...(runner.limits ? { limits: { ...runner.limits } } : {}),\n }\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n"],"mappings":";;;;;;;AAmBA,eAAsB,uBAAuB,MAMd;CAC7B,IAAI,CAAC,OAAO,cAAc,KAAK,QAAQ,KAAK,KAAK,YAAY,GAC3D,MAAM,IAAI,UAAU,8DAA8D;CAEpF,MAAM,SAAS,uCAAuC,KAAK,QAAQ,4BAA4B;CAC/F,KAAK,QAAQ,eAAe;CAC5B,MAAM,SAAS,IAAI,IAAI,KAAK,MAAM,KAAK,SAAS,CAAC,KAAK,MAAM,IAAI,CAAC,CAAC;CAClE,IAAI,OAAO,SAAS,KAAK,MAAM,QAC7B,MAAM,IAAI,MAAM,mDAAmD;CAGrE,MAAM,QAAQ,YAAY,EAAE,CAAC,CAAC,SAAS,KAAK;CAC5C,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI;CACJ,MAAM,oCAAoB,IAAI,IAAqB;CACnD,MAAM,iCAAiB,IAAI,IAAmB;CAC9C,MAAM,SAAS,cAAc,SAAS,aAAa;EACjD,IAAI,CAAC,WAAW;GACd,eAAe,UAAU,KAAK,EAAE,OAAO,iCAAiC,CAAC;GACzE;EACF;EACA,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,qBAA2B;GAC/B,QAAQ,QAAQ;GAChB,SAAS,QAAQ;EACnB;EACA,kBAAkB,IAAI,UAAU;EAChC,WAAW,OAAO,iBAAiB,SAAS,cAAc,EAAE,MAAM,KAAK,CAAC;EAExE,IAAI;EACJ,UAAU,cAAc,SAAS,UAAU,WAAW,MAAM,CAAC,CAAC,cAAc;GAC1E,WAAW,OAAO,oBAAoB,SAAS,YAAY;GAC3D,kBAAkB,OAAO,UAAU;GACnC,eAAe,OAAO,OAAO;EAC/B,CAAC;EACD,eAAe,IAAI,OAAO;EAC1B,QAAa,YAAY,KAAA,CAAS;CACpC,CAAC;CACD,MAAM,OAAO,MAAM,YAAY,MAAM;CACrC,MAAM,cAA6B;EACjC,iBAAiB,cAAc;EAC/B,OAAO;CACT;CACA,MAAM,gBAAsB;EAC1B,MAAW,CAAC,CAAC,YAAY,KAAA,CAAS;CACpC;CACA,KAAK,QAAQ,iBAAiB,SAAS,SAAS,EAAE,MAAM,KAAK,CAAC;CAC9D,IAAI,KAAK,QAAQ,SAAS,QAAQ;CAElC,OAAO;EACL,KAAK,oBAAoB,KAAK;EAC9B;EACA,aAAa;EACb;CACF;CAEA,eAAe,cACb,SACA,UACA,QACe;EACf,IAAI;GACF,IAAI,QAAQ,WAAW,UAAU,QAAQ,QAAQ,SAAS;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,YAAY,CAAC;IACpD;GACF;GACA,IAAI,QAAQ,QAAQ,kBAAkB,UAAU,SAAS;IACvD,eAAe,UAAU,KAAK,EAAE,OAAO,eAAe,CAAC;IACvD;GACF;GACA,IAAI,SAAS,KAAK,UAAU;IAC1B,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,MAAM,OAAO,MAAM,SAAS,SAAS,OAAO,eAAe;GAC3D,IAAI,CAACA,WAAS,IAAI,KAAK,OAAO,KAAK,SAAS,YAAY,EAAE,UAAU,OAAO;IACzE,eAAe,UAAU,KAAK,EAAE,OAAO,6BAA6B,CAAC;IACrE;GACF;GACA,MAAM,OAAO,OAAO,IAAI,KAAK,IAAI;GACjC,IAAI,CAAC,MAAM;IACT,eAAe,UAAU,KAAK,EAAE,OAAO,uBAAuB,KAAK,KAAK,GAAG,CAAC;IAC5E;GACF;GACA,SAAS;GACT,MAAM,SAAS,MAAM,KAAK,QAAQ,KAAK,MAAM,EAAE,OAAO,CAAC;GACvD,MAAM,UAAU,KAAK,UAAU,EAAE,OAAO,CAAC;GACzC,IAAI,OAAO,WAAW,OAAO,IAAI,OAAO,kBAAkB;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,SAAS,UAAU,KAAK;IACtB,gBAAgB;IAChB,kBAAkB,OAAO,OAAO,WAAW,OAAO,CAAC;GACrD,CAAC;GACD,SAAS,IAAI,OAAO;EACtB,SAAS,OAAO;GACd,eAAe,UAAU,OAAO,UAAU,MAAM,KAAK,EACnD,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAC9D,CAAC;EACH;CACF;CAEA,eAAe,gBAA+B;EAC5C,KAAK,QAAQ,oBAAoB,SAAS,OAAO;EACjD,YAAY;EACZ,MAAM,gBAAgB,YAAY,MAAM;EACxC,OAAO,uBAAuB;EAC9B,KAAK,MAAM,cAAc,mBAAmB,WAAW,MAAM;EAC7D,MAAM,CAAC,gBAAgB,MAAM,QAAQ,WAAW,CAC9C,eACA,sBAAsB,cAAc,CACtC,CAAC;EACD,IAAI,kBAAkB,SAAS,KAAK,eAAe,SAAS,GAC1D,MAAM,IAAI,MAAM,iDAAiD;EAEnE,IAAI,cAAc,WAAW,YAAY,MAAM,aAAa;CAC9D;AACF;AAEA,eAAe,sBAAsB,gBAAmD;CACtF,OAAO,eAAe,OAAO,GAC3B,MAAM,QAAQ,WAAW,CAAC,GAAG,cAAc,CAAC;AAEhD;AAEA,SAAS,SAAS,SAA0B,iBAA2C;CACrF,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,IAAI,OAAO;EACX,MAAM,SAAmB,CAAC;EAC1B,QAAQ,GAAG,SAAS,UAAkB;GACpC,QAAQ,MAAM;GACd,IAAI,OAAO,iBAAiB;IAC1B,uBAAO,IAAI,MAAM,8BAA8B,CAAC;IAChD,QAAQ,QAAQ;IAChB;GACF;GACA,OAAO,KAAK,KAAK;EACnB,CAAC;EACD,QAAQ,GAAG,SAAS,MAAM;EAC1B,QAAQ,GAAG,aAAa;GACtB,IAAI;IACF,QAAQ,KAAK,MAAM,OAAO,OAAO,MAAM,CAAC,CAAC,SAAS,MAAM,CAAC,CAAC;GAC5D,SAAS,OAAO;IACd,OAAO,KAAK;GACd;EACF,CAAC;CACH,CAAC;AACH;AAEA,SAAS,eAAe,UAA0B,QAAgB,MAAqB;CACrF,IAAI,SAAS,aAAa,SAAS,eAAe;CAClD,SAAS,UAAU,QAAQ,IAAI;AACjC;AAEA,SAASA,WAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E;;;ACpKA,MAAM,qBAAqB,KAAK;AAOhC,MAAM,8BAA8B;AACpC,MAAM,uBAAuB;AAC7B,MAAM,kCAAkC,KAAK,OAAO;AACpD,MAAM,mCAAmC,IAAI,OAAO;AACpD,MAAM,gCAAgC;AACtC,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;;AAEtB,MAAM,0BAA0B;;AAmDhC,SAAgB,yBAAyB,SAAyD;CAChG,cAAc,OAAO;CACrB,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,IAAI,CAAC,OAAO,cAAc,eAAe,KAAK,mBAAmB,GAC/D,MAAM,IAAI,UAAU,0DAA0D;CAEhF,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,qBAAqB,QAAQ,sBAAsB,kBAAkB;CAC3E,IAAI,CAAC,OAAO,cAAc,kBAAkB,KAAK,qBAAqB,GACpE,MAAM,IAAI,UAAU,iEAAiE;CAEvF,MAAM,YAAY,QAAQ,aAAa;CACvC,iBAAiB,WAAW,WAAW;CACvC,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,GAChD,MAAM,IAAI,UAAU,iDAAiD;CAEvE,MAAM,UAAU,QAAQ,WAAW,gBAAgB,QAAQ,KAAK;CAChE,MAAM,uBAAuB,QAAQ,wBAAwB;CAC7D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,qBAAqB,QAAQ,sBAAsB;CACzD,mCACE;EACE;EACA,aAAa,QAAQ,oBAAoB;EACzC,iBAAiB;EACjB,kBAAkB;EAClB,2BAA2B;EAC3B,8BAA8B;EAC9B;EACA,kBAAkB;CACpB,GACA,uBACF;CACA,uCAAuC,QAAQ,iBAAiB,4BAA4B;CAC5F,iBAAiB,oBAAoB,oBAAoB;CAEzD,MAAM,SAAS,gBAAgB,QAAQ,MAAM;CAC7C,MAAM,gBAAgB,sCAAsC,QAAQ,MAAM;CAC1E,OAAO;EACL,IAAI;EACJ,aAAa;EACb,OAAO,QAAQ;EACf,SAAS;EACT,iBAAiB;GACf,eAAe;GACf,UAAU,QAAQ;GAClB,OAAO,QAAQ;GACf,SAAS,EAAE,GAAG,QAAQ;GACtB,cAAc;GACd,mBAAmB;GACnB,sBAAsB;GACtB,iBAAiB;GACjB,YAAY;GACZ,oBAAoB,QAAQ,oBAAoB;GAChD,mBAAmB;GACnB,oBAAoB;GACpB,0BAA0B;GAC1B,mBAAmB,QAAQ,mBAAmB;GAC9C,uBAAuB;GACvB,gBAAgB;GAChB,QAAQ,SAAS,oBAAoB;GACrC,gBAAgB,QAAQ,WAAW;EACrC;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,WAAW,MAAM,uBAAuB;IAC5C,OAAO,QAAQ;IACf,UAAU,QAAQ,OAAO;IACzB,GAAI,QAAQ,kBAAkB,EAAE,QAAQ,QAAQ,gBAAgB,IAAI,CAAC;IACrE,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;GACrD,CAAC;GACD,IAAI;GACJ,MAAM,kBAAgE,CAAC;GACvE,MAAM,SAAS,MAAM,eAAe;IAClC,OAAO;IACP,KAAK,YAAY;KACf,aAAa,MAAM,iCAAiC;MAClD,MAAM,QAAQ;MACd,SAAS,QAAQ;MACjB,kBAAkB,gBAAgB;OAChC,gBAAgB,KAAK,gBAAgB,WAAW,CAAC;OACjD,QAAQ,gBAAgB,WAAW;MACrC;MACA,OAAO,QAAQ;MACf,QAAQ;OACN;OACA,aAAa,wBAAwB,QAAQ,kBAAkB,QAAQ,MAAM;OAC7E,iBAAiB;OACjB,kBAAkB;OAClB,2BAA2B;OAC3B,8BAA8B;OAC9B,kBAAkB;OAClB;MACF;MACA,YAAY,QAAQ;MACpB,SAAS;MACT,OAAO,QAAQ;MACf,OAAO,QAAQ;MACf,GAAI,QAAQ,WAAW,EAAE,MAAM,QAAQ,SAAS,IAAI,CAAC;MACrD,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC;KACD,QAAQ,MAAM,gCAAgC;MAC5C,QAAQ;MACR,OAAO,QAAQ;MACf,OAAO,QAAQ,MAAM,KAAK,SAAS,KAAK,IAAI;MAC5C,QAAQ,QAAQ;KAClB,CAAC;KAwCD,MAAM,SAAS,kBAAkB,MAvCf,4BAAqC;MACrD,OAAO;MACP,YAAY;MACZ,QAAQ;MACR,OAAO;OACL,WAAW;OACX,UAAU,QAAQ;OAClB,cAAc,QAAQ;OACtB,YAAY;QACV,SAAS,WAAW;QACpB,QAAQ,WAAW;QACnB,OAAO,QAAQ;QACf;OACF;OACA,cAAc;QACZ,KAAK,SAAS;QACd,OAAO,SAAS;QAChB,WAAW;OACb;OACA,WAAW,QAAQ,MAAM,KAAK,EAAE,MAAM,aAAa,kBAAkB;QACnE;QACA;QACA;OACF,EAAE;OACF;OACA,QAAQ;QACN,eAAe,QAAQ,OAAO;QAC9B,aAAa,QAAQ,OAAO;QAC5B,gBAAgB,QAAQ,OAAO;OACjC;OAGA,GAAI,QAAQ,aAAa,EAAE,YAAY,QAAQ,WAAW,IAAI,CAAC;MACjE;MACA,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;MAC3B,gBAAgB,CAAC,qBAAqB,OAAO,cAAc,aAAa,CAAC;MACzE;MACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC,IACsC,OAAO,WAAW;MACvD,QAAQ,MAAM,yDAAyD;OACrE,QAAQ;OACR;OACA;MACF,CAAC;KACH,CAAC;KACD,MAAM,wBAAwB,WAAW,sBAAsB;KAC/D,MAAM,kBAAkB,WAAW,gBAAgB;KACnD,IAAI,OAAO,eAAe,uBACxB,MAAM,IAAI,MACR,qBAAqB,OAAO,WAAW,gDAAgD,uBACzF;KAEF,WAAW,wBAAwB;KACnC,OAAO;MACL,GAAG;MACH,WAAW,SAAS,MAAM;MAC1B,SAAS;OACP,GAAG,OAAO;OACV,sBAAsB;OACtB,4BAA4B;OAC5B;MACF;KACF;IACF;IACA,SAAS,YAAY;KAKnB,MAAM,UAAS,MAJO,QAAQ,WAAW,CACvC,GAAI,aAAa,CAAC,WAAW,MAAM,CAAC,IAAI,CAAC,GACzC,SAAS,MAAM,CACjB,CAAC,EAAA,CACsB,SAAS,UAC9B,MAAM,WAAW,aAAa,CAAC,MAAM,MAAM,IAAI,CAAC,CAClD;KACA,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,eAAe,QAAQ,kCAAkC;IAEvE;GACF,CAAC;GACD,QAAQ,MAAM,kCAAkC;IAC9C,QAAQ;IACR,aAAa,OAAO;IACpB,wBAAwB,OAAO,QAAQ;IACvC,YAAY,OAAO;IACnB,UAAU,OAAO,SAAS;GAC5B,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,kBACP,OACA,mBAC8C;CAC9C,IAAI,CAAC,SAAS,KAAK,GAAG,MAAM,IAAI,MAAM,0CAA0C;CAChF,IAAI,OAAO,MAAM,WAAW,YAAY,CAAC,MAAM,OAAO,KAAK,GACzD,MAAM,IAAI,MAAM,oCAAoC;CAEtD,IAAI,CAAC,MAAM,QAAQ,MAAM,QAAQ,GAC/B,MAAM,IAAI,MAAM,2CAA2C;CAK7D,IAAI,mBAAmB;CACvB,MAAM,WAAgC,CAAC;CACvC,MAAM,SAAS,SAAS,SAAS,UAAU;EACzC,MAAM,SAAS,wBAAwB,UAAU,OAAO;EACxD,IAAI,CAAC,OAAO,SAAS;GACnB,oBAAoB;GACpB,kBACE,OACA,OAAO,MAAM,OAAO,KAAK,UAAU,GAAG,MAAM,KAAK,KAAK,GAAG,EAAE,IAAI,MAAM,SAAS,CAAC,CAAC,KAAK,IAAI,CAC3F;GACA;EACF;EACA,SAAS,KAAK,OAAO,IAAI;CAC3B,CAAC;CACD,IAAI,CAAC,MAAM,QAAQ,MAAM,UAAU,GACjC,MAAM,IAAI,MAAM,6CAA6C;CAE/D,IAAI,CAAC,OAAO,cAAc,MAAM,UAAU,KAAM,MAAM,cAAyB,GAC7E,MAAM,IAAI,MAAM,4DAA4D;CAE9E,IAAI,CAAC,SAAS,MAAM,OAAO,GACzB,MAAM,IAAI,MAAM,2CAA2C;CAE7D,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM;EAClB,YAAY,MAAM;EAClB,SAAS;GAAE,GAAG,MAAM;GAAS;EAAiB;CAChD;AACF;AAEA,SAAS,cAAc,SAA0C;CAC/D,KAAK,MAAM,CAAC,MAAM,UAAU,CAC1B,CAAC,WAAW,QAAQ,OAAO,GAC3B,CAAC,SAAS,QAAQ,KAAK,CACzB,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,KAAK,UAAU,MAAM,KAAK,GACrE,MAAM,IAAI,UAAU,YAAY,KAAK,oCAAoC;CAG7E,IAAI,OAAO,QAAQ,SAAS,YAC1B,MAAM,IAAI,UAAU,kCAAkC;CAExD,IAAI,OAAO,QAAQ,oBAAoB,YACrC,MAAM,IAAI,UAAU,6CAA6C;CAEnE,IACE,QAAQ,qBAAqB,KAAA,MAC5B,CAAC,OAAO,cAAc,QAAQ,gBAAgB,KAAK,QAAQ,oBAAoB,IAEhF,MAAM,IAAI,UAAU,2DAA2D;AAEnF;AAEA,SAAS,wBACP,YACA,QACQ;CACR,IAAI,eAAe,KAAA,GAAW,OAAO;CACrC,MAAM,UAAU,OAAO,gBAAgB,OAAO,cAAc;CAC5D,IAAI,CAAC,OAAO,cAAc,OAAO,KAAK,WAAW,GAC/C,MAAM,IAAI,MAAM,iEAAiE;CAEnF,OAAO;AACT;AAEA,SAAS,iBAAiB,OAAe,OAAqB;CAC5D,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,SAAS,KAAK,QAAQ,oBACxD,MAAM,IAAI,UAAU,YAAY,MAAM,yBAAyB,oBAAoB;AAEvF;AAEA,SAAS,gBAAgB,OAAmC;CAC1D,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SACH,MAAM,IAAI,MACR,iCAAiC,MAAM,6CACzC;CAEF,OAAO;EACL,oBAAoB,QAAQ,QAAQ;EACpC,qBAAqB,QAAQ,SAAS;CACxC;AACF;AAEA,SAAS,gBACP,QAC4C;CAC5C,IAAI,CAAC,QAAQ,OAAO,KAAA;CACpB,OAAO;EACL,GAAI,OAAO,UAAU,EAAE,SAAS,OAAO,QAAQ,IAAI,CAAC;EACpD,GAAI,OAAO,OAAO,EAAE,MAAM,OAAO,KAAK,IAAI,CAAC;EAC3C,GAAI,OAAO,MAAM,EAAE,KAAK,4BAA4B,OAAO,GAAG,EAAE,IAAI,CAAC;EACrE,GAAI,OAAO,SAAS,EAAE,QAAQ,EAAE,GAAG,OAAO,OAAO,EAAE,IAAI,CAAC;CAC1D;AACF;AAEA,SAAS,SAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"emitter-CPBAhxum.js","names":[],"sources":["../src/trace/emitter.ts"],"sourcesContent":["/**\n * TraceEmitter — hierarchical span builder that auto-parents using an\n * internal stack. One emitter per Run; emitters do NOT share state.\n *\n * Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)\n * return a `SpanHandle` with `.end()` / `.fail()` so callers don't\n * have to thread spanIds manually. For async workflows that can't use\n * the stack (e.g. fan-out parallel calls), pass `parentSpanId`\n * explicitly.\n */\n\nimport type {\n Artifact,\n BudgetLedgerEntry,\n EventKind,\n JudgeSpan,\n LlmSpan,\n Message,\n RetrievalSpan,\n Run,\n RunOutcome,\n SandboxSpan,\n Span,\n SpanKind,\n ToolSpan,\n TraceEvent,\n} from './schema'\nimport type { TraceStore } from './store'\n\nexport interface SpanHandle<S extends Span = Span> {\n span: S\n end(patch?: Partial<S>): Promise<void>\n fail(error: string | Error, patch?: Partial<S>): Promise<void>\n}\n\nexport interface RunCompleteHookContext {\n runId: string\n emitter: TraceEmitter\n store: TraceStore\n /** Outcome the caller passed to `endRun` (undefined for `abortRun`). */\n outcome?: RunOutcome\n /** Final run status. */\n status: 'completed' | 'failed' | 'aborted'\n}\n\nexport type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void\n\nexport interface TraceEmitterOptions {\n runId?: string\n /** Inject a clock for deterministic tests. */\n now?: () => number\n /** Inject an id generator for deterministic tests. */\n id?: () => string\n /**\n * Hooks fired after `endRun` / `abortRun` writes the final run state.\n * Designed for trace-analyst auto-execution, integrity assertions, and\n * outbound notifications. Hooks run sequentially in the order supplied.\n *\n * By default a hook that throws is swallowed and logged as a `note` event\n * on the run — auto-orchestration must not crash the underlying flow.\n * Set `hookErrors: 'throw'` to propagate.\n */\n onRunComplete?: RunCompleteHook[]\n /** `'swallow'` (default) | `'throw'`. */\n hookErrors?: 'swallow' | 'throw'\n}\n\nexport class TraceEmitter {\n private store: TraceStore\n private stack: string[] = []\n private _runId: string\n private now: () => number\n private id: () => string\n private hooks: RunCompleteHook[]\n private hookErrors: 'swallow' | 'throw'\n\n constructor(store: TraceStore, options: TraceEmitterOptions = {}) {\n this.store = store\n this.now = options.now ?? (() => Date.now())\n this.id = options.id ?? (() => cryptoRandomId())\n this._runId = options.runId ?? this.id()\n this.hooks = options.onRunComplete ?? []\n this.hookErrors = options.hookErrors ?? 'swallow'\n }\n\n get runId(): string {\n return this._runId\n }\n\n get traceStore(): TraceStore {\n return this.store\n }\n\n /** Append a hook after construction (e.g. attach the trace analyst). */\n addRunCompleteHook(hook: RunCompleteHook): void {\n this.hooks.push(hook)\n }\n\n // ── Run lifecycle ──────────────────────────────────────────────────\n\n /**\n * Begin a Run.\n *\n * `scenarioId` is required on the persisted Run shape — every Run downstream\n * gets a non-empty scenarioId so filters and aggregations stay simple — but\n * the INPUT here accepts it as optional. When omitted, startRun substitutes\n * a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so\n * runtime / operator / meta-eval runs that have no curated-scenario corpus\n * to anchor to don't have to invent placeholder strings at the call site.\n */\n async startRun(\n run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & { scenarioId?: string },\n ): Promise<Run> {\n const scenarioId = run.scenarioId ?? run.layer ?? run.tags?.kind ?? 'runtime'\n const full: Run = {\n ...run,\n scenarioId,\n runId: this._runId,\n startedAt: this.now(),\n status: 'running',\n }\n await this.store.appendRun(full)\n return full\n }\n\n async endRun(outcome?: RunOutcome): Promise<void> {\n const status: 'completed' | 'failed' = outcome?.pass === false ? 'failed' : 'completed'\n await this.store.updateRun(this._runId, { endedAt: this.now(), status, outcome })\n await this.runHooks({ runId: this._runId, emitter: this, store: this.store, outcome, status })\n }\n\n async abortRun(reason: string): Promise<void> {\n const outcome = { pass: false, notes: reason }\n await this.store.updateRun(this._runId, {\n endedAt: this.now(),\n status: 'aborted',\n outcome,\n })\n await this.runHooks({\n runId: this._runId,\n emitter: this,\n store: this.store,\n outcome,\n status: 'aborted',\n })\n }\n\n private async runHooks(ctx: RunCompleteHookContext): Promise<void> {\n for (const hook of this.hooks) {\n try {\n await hook(ctx)\n } catch (err) {\n if (this.hookErrors === 'throw') throw err\n try {\n await this.store.appendEvent({\n eventId: this.id(),\n runId: this._runId,\n kind: 'log',\n timestamp: this.now(),\n payload: {\n source: 'run_complete_hook',\n error: err instanceof Error ? err.message : String(err),\n },\n })\n } catch {\n // best-effort\n }\n }\n }\n }\n\n // ── Generic span ───────────────────────────────────────────────────\n\n async span<S extends Span = Span>(\n init: {\n kind: SpanKind\n name: string\n parentSpanId?: string\n attributes?: Record<string, unknown>\n } & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>,\n ): Promise<SpanHandle<S>> {\n const spanId = this.id()\n const parent = init.parentSpanId ?? this.stack[this.stack.length - 1]\n const span = {\n spanId,\n parentSpanId: parent,\n runId: this._runId,\n startedAt: this.now(),\n ...init,\n } as unknown as S\n await this.store.appendSpan(span)\n this.stack.push(spanId)\n return this.handle<S>(span)\n }\n\n private handle<S extends Span>(span: S): SpanHandle<S> {\n return {\n span,\n end: async (patch?: Partial<S>) => {\n const endedAt = this.now()\n await this.store.updateSpan(span.spanId, {\n endedAt,\n status: 'ok',\n ...patch,\n } as Partial<Span>)\n this.pop(span.spanId)\n },\n fail: async (error: string | Error, patch?: Partial<S>) => {\n const endedAt = this.now()\n const errStr = error instanceof Error ? error.message : error\n await this.store.updateSpan(span.spanId, {\n endedAt,\n status: 'error',\n error: errStr,\n ...patch,\n } as Partial<Span>)\n this.pop(span.spanId)\n },\n }\n }\n\n private pop(spanId: string): void {\n const idx = this.stack.lastIndexOf(spanId)\n if (idx >= 0) this.stack.splice(idx, 1)\n }\n\n // ── Typed span conveniences ────────────────────────────────────────\n\n llm(\n init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>,\n ): Promise<SpanHandle<LlmSpan>> {\n return this.span<LlmSpan>({ kind: 'llm', ...init })\n }\n\n tool(\n init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>,\n ): Promise<SpanHandle<ToolSpan>> {\n return this.span<ToolSpan>({ kind: 'tool', ...init })\n }\n\n retrieval(\n init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>,\n ): Promise<SpanHandle<RetrievalSpan>> {\n return this.span<RetrievalSpan>({ kind: 'retrieval', ...init })\n }\n\n async recordJudge(\n verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>,\n ): Promise<JudgeSpan> {\n const spanId = this.id()\n const now = this.now()\n const full: JudgeSpan = {\n spanId,\n runId: this._runId,\n kind: 'judge',\n startedAt: now,\n endedAt: now,\n status: 'ok',\n ...verdict,\n }\n await this.store.appendSpan(full)\n return full\n }\n\n sandbox(\n init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>,\n ): Promise<SpanHandle<SandboxSpan>> {\n return this.span<SandboxSpan>({ kind: 'sandbox', ...init })\n }\n\n // ── Events ─────────────────────────────────────────────────────────\n\n async emit(event: {\n kind: EventKind\n spanId?: string\n payload?: Record<string, unknown>\n }): Promise<TraceEvent> {\n const full: TraceEvent = {\n eventId: this.id(),\n runId: this._runId,\n spanId: event.spanId ?? this.stack[this.stack.length - 1],\n kind: event.kind,\n timestamp: this.now(),\n payload: event.payload ?? {},\n }\n await this.store.appendEvent(full)\n return full\n }\n\n // ── Budget ledger ──────────────────────────────────────────────────\n\n async recordBudget(\n entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & { timestamp?: number },\n ): Promise<BudgetLedgerEntry> {\n const full: BudgetLedgerEntry = {\n runId: this._runId,\n timestamp: entry.timestamp ?? this.now(),\n dimension: entry.dimension,\n limit: entry.limit,\n consumed: entry.consumed,\n remaining: entry.remaining,\n breached: entry.breached,\n spanId: entry.spanId ?? this.stack[this.stack.length - 1],\n }\n await this.store.appendBudgetEntry(full)\n if (full.breached) {\n await this.emit({\n kind: 'budget_breach',\n spanId: full.spanId,\n payload: { dimension: full.dimension, limit: full.limit, consumed: full.consumed },\n })\n }\n return full\n }\n\n // ── Artifacts ──────────────────────────────────────────────────────\n\n async recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact> {\n const full: Artifact = { artifactId: this.id(), runId: this._runId, ...artifact }\n await this.store.appendArtifact(full)\n return full\n }\n\n // ── Nested composition ─────────────────────────────────────────────\n\n /**\n * Runs `fn` inside a span; auto-ends on success, auto-fails on throw.\n * Returns the fn's return value. Use this for the 95% case.\n */\n async within<T>(\n init: Parameters<TraceEmitter['span']>[0],\n fn: (handle: SpanHandle) => Promise<T>,\n ): Promise<T> {\n const handle = await this.span(init)\n try {\n const result = await fn(handle)\n await handle.end()\n return result\n } catch (err) {\n await handle.fail(err instanceof Error ? err : String(err))\n throw err\n }\n }\n}\n\n// Helpers -------------------------------------------------------------\n\nfunction cryptoRandomId(): string {\n if (typeof globalThis.crypto?.randomUUID === 'function') return globalThis.crypto.randomUUID()\n return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`\n}\n\n/** Helper to build an LLM span handle args object from a provider-shaped response. */\nexport function llmSpanFromProvider(args: {\n name?: string\n model: string\n messages: Message[]\n output: string\n usage?: {\n inputTokens?: number\n outputTokens?: number\n cachedTokens?: number\n cacheWriteTokens?: number\n reasoningTokens?: number\n }\n costUsd?: number\n finishReason?: string\n}): Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'> {\n return {\n name: args.name ?? args.model,\n model: args.model,\n messages: args.messages,\n output: args.output,\n inputTokens: args.usage?.inputTokens,\n outputTokens: args.usage?.outputTokens,\n cachedTokens: args.usage?.cachedTokens,\n cacheWriteTokens: args.usage?.cacheWriteTokens,\n reasoningTokens: args.usage?.reasoningTokens,\n costUsd: args.costUsd,\n finishReason: args.finishReason,\n }\n}\n"],"mappings":";AAmEA,IAAa,eAAb,MAA0B;CACxB;CACA,QAA0B,CAAC;CAC3B;CACA;CACA;CACA;CACA;CAEA,YAAY,OAAmB,UAA+B,CAAC,GAAG;EAChE,KAAK,QAAQ;EACb,KAAK,MAAM,QAAQ,cAAc,KAAK,IAAI;EAC1C,KAAK,KAAK,QAAQ,aAAa,eAAe;EAC9C,KAAK,SAAS,QAAQ,SAAS,KAAK,GAAG;EACvC,KAAK,QAAQ,QAAQ,iBAAiB,CAAC;EACvC,KAAK,aAAa,QAAQ,cAAc;CAC1C;CAEA,IAAI,QAAgB;EAClB,OAAO,KAAK;CACd;CAEA,IAAI,aAAyB;EAC3B,OAAO,KAAK;CACd;;CAGA,mBAAmB,MAA6B;EAC9C,KAAK,MAAM,KAAK,IAAI;CACtB;;;;;;;;;;;CAcA,MAAM,SACJ,KACc;EACd,MAAM,aAAa,IAAI,cAAc,IAAI,SAAS,IAAI,MAAM,QAAQ;EACpE,MAAM,OAAY;GAChB,GAAG;GACH;GACA,OAAO,KAAK;GACZ,WAAW,KAAK,IAAI;GACpB,QAAQ;EACV;EACA,MAAM,KAAK,MAAM,UAAU,IAAI;EAC/B,OAAO;CACT;CAEA,MAAM,OAAO,SAAqC;EAChD,MAAM,SAAiC,SAAS,SAAS,QAAQ,WAAW;EAC5E,MAAM,KAAK,MAAM,UAAU,KAAK,QAAQ;GAAE,SAAS,KAAK,IAAI;GAAG;GAAQ;EAAQ,CAAC;EAChF,MAAM,KAAK,SAAS;GAAE,OAAO,KAAK;GAAQ,SAAS;GAAM,OAAO,KAAK;GAAO;GAAS;EAAO,CAAC;CAC/F;CAEA,MAAM,SAAS,QAA+B;EAC5C,MAAM,UAAU;GAAE,MAAM;GAAO,OAAO;EAAO;EAC7C,MAAM,KAAK,MAAM,UAAU,KAAK,QAAQ;GACtC,SAAS,KAAK,IAAI;GAClB,QAAQ;GACR;EACF,CAAC;EACD,MAAM,KAAK,SAAS;GAClB,OAAO,KAAK;GACZ,SAAS;GACT,OAAO,KAAK;GACZ;GACA,QAAQ;EACV,CAAC;CACH;CAEA,MAAc,SAAS,KAA4C;EACjE,KAAK,MAAM,QAAQ,KAAK,OACtB,IAAI;GACF,MAAM,KAAK,GAAG;EAChB,SAAS,KAAK;GACZ,IAAI,KAAK,eAAe,SAAS,MAAM;GACvC,IAAI;IACF,MAAM,KAAK,MAAM,YAAY;KAC3B,SAAS,KAAK,GAAG;KACjB,OAAO,KAAK;KACZ,MAAM;KACN,WAAW,KAAK,IAAI;KACpB,SAAS;MACP,QAAQ;MACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KACxD;IACF,CAAC;GACH,QAAQ,CAER;EACF;CAEJ;CAIA,MAAM,KACJ,MAMwB;EACxB,MAAM,SAAS,KAAK,GAAG;EAEvB,MAAM,OAAO;GACX;GACA,cAHa,KAAK,gBAAgB,KAAK,MAAM,KAAK,MAAM,SAAS;GAIjE,OAAO,KAAK;GACZ,WAAW,KAAK,IAAI;GACpB,GAAG;EACL;EACA,MAAM,KAAK,MAAM,WAAW,IAAI;EAChC,KAAK,MAAM,KAAK,MAAM;EACtB,OAAO,KAAK,OAAU,IAAI;CAC5B;CAEA,OAA+B,MAAwB;EACrD,OAAO;GACL;GACA,KAAK,OAAO,UAAuB;IACjC,MAAM,UAAU,KAAK,IAAI;IACzB,MAAM,KAAK,MAAM,WAAW,KAAK,QAAQ;KACvC;KACA,QAAQ;KACR,GAAG;IACL,CAAkB;IAClB,KAAK,IAAI,KAAK,MAAM;GACtB;GACA,MAAM,OAAO,OAAuB,UAAuB;IACzD,MAAM,UAAU,KAAK,IAAI;IACzB,MAAM,SAAS,iBAAiB,QAAQ,MAAM,UAAU;IACxD,MAAM,KAAK,MAAM,WAAW,KAAK,QAAQ;KACvC;KACA,QAAQ;KACR,OAAO;KACP,GAAG;IACL,CAAkB;IAClB,KAAK,IAAI,KAAK,MAAM;GACtB;EACF;CACF;CAEA,IAAY,QAAsB;EAChC,MAAM,MAAM,KAAK,MAAM,YAAY,MAAM;EACzC,IAAI,OAAO,GAAG,KAAK,MAAM,OAAO,KAAK,CAAC;CACxC;CAIA,IACE,MAC8B;EAC9B,OAAO,KAAK,KAAc;GAAE,MAAM;GAAO,GAAG;EAAK,CAAC;CACpD;CAEA,KACE,MAC+B;EAC/B,OAAO,KAAK,KAAe;GAAE,MAAM;GAAQ,GAAG;EAAK,CAAC;CACtD;CAEA,UACE,MACoC;EACpC,OAAO,KAAK,KAAoB;GAAE,MAAM;GAAa,GAAG;EAAK,CAAC;CAChE;CAEA,MAAM,YACJ,SACoB;EACpB,MAAM,SAAS,KAAK,GAAG;EACvB,MAAM,MAAM,KAAK,IAAI;EACrB,MAAM,OAAkB;GACtB;GACA,OAAO,KAAK;GACZ,MAAM;GACN,WAAW;GACX,SAAS;GACT,QAAQ;GACR,GAAG;EACL;EACA,MAAM,KAAK,MAAM,WAAW,IAAI;EAChC,OAAO;CACT;CAEA,QACE,MACkC;EAClC,OAAO,KAAK,KAAkB;GAAE,MAAM;GAAW,GAAG;EAAK,CAAC;CAC5D;CAIA,MAAM,KAAK,OAIa;EACtB,MAAM,OAAmB;GACvB,SAAS,KAAK,GAAG;GACjB,OAAO,KAAK;GACZ,QAAQ,MAAM,UAAU,KAAK,MAAM,KAAK,MAAM,SAAS;GACvD,MAAM,MAAM;GACZ,WAAW,KAAK,IAAI;GACpB,SAAS,MAAM,WAAW,CAAC;EAC7B;EACA,MAAM,KAAK,MAAM,YAAY,IAAI;EACjC,OAAO;CACT;CAIA,MAAM,aACJ,OAC4B;EAC5B,MAAM,OAA0B;GAC9B,OAAO,KAAK;GACZ,WAAW,MAAM,aAAa,KAAK,IAAI;GACvC,WAAW,MAAM;GACjB,OAAO,MAAM;GACb,UAAU,MAAM;GAChB,WAAW,MAAM;GACjB,UAAU,MAAM;GAChB,QAAQ,MAAM,UAAU,KAAK,MAAM,KAAK,MAAM,SAAS;EACzD;EACA,MAAM,KAAK,MAAM,kBAAkB,IAAI;EACvC,IAAI,KAAK,UACP,MAAM,KAAK,KAAK;GACd,MAAM;GACN,QAAQ,KAAK;GACb,SAAS;IAAE,WAAW,KAAK;IAAW,OAAO,KAAK;IAAO,UAAU,KAAK;GAAS;EACnF,CAAC;EAEH,OAAO;CACT;CAIA,MAAM,eAAe,UAAqE;EACxF,MAAM,OAAiB;GAAE,YAAY,KAAK,GAAG;GAAG,OAAO,KAAK;GAAQ,GAAG;EAAS;EAChF,MAAM,KAAK,MAAM,eAAe,IAAI;EACpC,OAAO;CACT;;;;;CAQA,MAAM,OACJ,MACA,IACY;EACZ,MAAM,SAAS,MAAM,KAAK,KAAK,IAAI;EACnC,IAAI;GACF,MAAM,SAAS,MAAM,GAAG,MAAM;GAC9B,MAAM,OAAO,IAAI;GACjB,OAAO;EACT,SAAS,KAAK;GACZ,MAAM,OAAO,KAAK,eAAe,QAAQ,MAAM,OAAO,GAAG,CAAC;GAC1D,MAAM;EACR;CACF;AACF;AAIA,SAAS,iBAAyB;CAChC,IAAI,OAAO,WAAW,QAAQ,eAAe,YAAY,OAAO,WAAW,OAAO,WAAW;CAC7F,OAAO,GAAG,KAAK,IAAI,CAAC,CAAC,SAAS,EAAE,EAAE,GAAG,KAAK,OAAO,CAAC,CAAC,SAAS,EAAE,CAAC,CAAC,MAAM,GAAG,EAAE;AAC7E;;AAGA,SAAgB,oBAAoB,MAcyB;CAC3D,OAAO;EACL,MAAM,KAAK,QAAQ,KAAK;EACxB,OAAO,KAAK;EACZ,UAAU,KAAK;EACf,QAAQ,KAAK;EACb,aAAa,KAAK,OAAO;EACzB,cAAc,KAAK,OAAO;EAC1B,cAAc,KAAK,OAAO;EAC1B,kBAAkB,KAAK,OAAO;EAC9B,iBAAiB,KAAK,OAAO;EAC7B,SAAS,KAAK;EACd,cAAc,KAAK;CACrB;AACF"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"eval-campaign-C4jmuM-b.js","names":[],"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - calls `assertLlmRoute` once at preflight before any work runs\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`\n * - hands the runner an `LlmClientOptions` pre-wired with the sink and\n * trace context — the runner can't accidentally call an LLM without\n * capturing the raw HTTP envelope\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what `LlmClient` already does\n */\n\nimport {\n type AgentProfileCell,\n type AgentProfileCellInput,\n buildAgentProfileCell,\n verifyAgentProfileCell,\n} from './agent-profile-cell'\nimport { assertLlmRoute, type LlmClientOptions, type LlmRouteRequirements } from './llm-client'\nimport { canonicalize, hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunCostProvenance,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTaskFailure,\n RunTokenUsage,\n} from './run-record'\nimport { validateRunRecord } from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * Pre-wired LLM client options — `rawSink` and `traceContext` are populated\n * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The\n * runner can spread additional fields if needed.\n */\n llmOpts: LlmClientOptions\n}\n\ninterface CampaignRunOutcomeFields {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Cost in USD, or null when the runner could not capture it. */\n costUsd: number | null\n /** Source of the cost amount. */\n costProvenance: RunCostProvenance\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n /**\n * Agent profile cell observed by the runner. When supplied, it overrides\n * `EvalCampaignOptions.agentProfile` for this run and must match the\n * outcome's `model` and `promptHash`.\n */\n agentProfile?: AgentProfileCell | AgentProfileCellInput\n}\n\n/** Campaign result with the same task-failure invariant as `RunRecord`. */\nexport type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * LLM client config. Augmented per-run with `rawSink` and `traceContext`\n * before being passed to the runner. The campaign asserts this config\n * matches `routeRequirements` once at preflight.\n */\n llmOpts: LlmClientOptions\n /**\n * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail\n * loud if the campaign would silently fall back to the public router or\n * run unauthenticated. Override with an empty object to disable.\n */\n routeRequirements?: LlmRouteRequirements\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n /**\n * Agent profile cell for campaign runs. Static profiles can pass an object;\n * routers or variant-specific harnesses can pass a factory. The campaign\n * stamps the built cell onto every `RunRecord` and rejects profile/model or\n * profile/prompt contradictions.\n */\n agentProfile?:\n | AgentProfileCell\n | AgentProfileCellInput\n | ((\n params: CampaignFactoryParams & {\n variant: V\n scenarioTags: Record<string, string>\n },\n ) =>\n | AgentProfileCell\n | AgentProfileCellInput\n | Promise<AgentProfileCell | AgentProfileCellInput>)\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nconst DEFAULT_ROUTE: LlmRouteRequirements = {\n requireExplicitBaseUrl: true,\n requireAuth: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n assertLlmRoute(opts.llmOpts, opts.routeRequirements ?? DEFAULT_ROUTE)\n\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const baseUrl = (opts.llmOpts.baseUrl ?? '').replace(/\\/+$/, '')\n const provider = opts.llmOpts.provider ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson(\n canonicalize({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n baseUrl,\n provider,\n preregistrationHash,\n }),\n )\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n // A genuine (non-CellExecutionError) error from any worker is a bug, not a\n // run-level failure. We must NOT let `Promise.all` reject mid-flight and\n // orphan the other workers' in-progress runs (emitters never finalized,\n // sinks/handles leak, partial work silently discarded). Instead: capture the\n // first genuine error, stop dispatching new cells so in-flight workers wind\n // down, finalize every still-open run, then re-throw the aggregated error.\n let cursor = 0\n let aborting = false\n const genuineErrors: unknown[] = []\n // Emitters for runs that are currently mid-flight, keyed by runId. A worker\n // registers its emitter before invoking the runner and removes it once the\n // run is finalized (via endRun/abortRun, success or failure). Anything left\n // here after the pool settles is an orphan we must abort.\n const openRuns = new Map<string, TraceEmitter>()\n\n async function worker(): Promise<void> {\n while (!aborting) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Capture it and stop dispatching so peers wind down gracefully;\n // re-thrown after the pool settles. Do not surface here — that would\n // reject Promise.all and orphan the other workers' runs.\n genuineErrors.push(err)\n aborting = true\n return\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n // Track this run as open so a genuine error elsewhere in the pool can\n // finalize it instead of orphaning it. Removed in the finally below.\n openRuns.set(runId, emitter)\n\n const llmOpts: LlmClientOptions = {\n ...opts.llmOpts,\n rawSink,\n traceContext: { runId },\n }\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n llmOpts,\n }\n\n try {\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution. Abort the run so the emitter\n // finalizes. The only benign abortRun failure is \"the runner never\n // started the run\" (nothing to finalize). A store-write failure (disk\n // full, FS error) is a genuine diagnostic — surface it rather than\n // masking it as a plain runner failure.\n await finalizeAbort(emitter, runId, message)\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n costProvenance: outcome.costProvenance,\n tokenUsage: outcome.tokenUsage,\n terminalOutcome: 'succeeded',\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n ...(outcome.failureClass ? { failureClass: outcome.failureClass } : {}),\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n const profileSource =\n outcome.agentProfile ??\n (typeof opts.agentProfile === 'function'\n ? await opts.agentProfile({\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n variant: cell.variant.payload,\n scenarioTags: cell.scenario.tags ?? {},\n })\n : opts.agentProfile)\n if (profileSource !== undefined) {\n const agentProfile = await resolveAgentProfileCell(profileSource)\n assertAgentProfileMatchesRun(agentProfile, outcome.model, outcome.promptHash)\n record.agentProfile = agentProfile\n }\n return { record: validateRunRecord(record), integrity: integrityReport }\n } finally {\n // This run's worker has finished with it (success, run-level failure, or\n // genuine error). It is no longer the pool's job to finalize — drop it so\n // the post-settle sweep doesn't double-abort a finalized run.\n openRuns.delete(runId)\n }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n // allSettled (not all): a genuine error in one worker must not reject the\n // pool mid-flight and orphan the others. Each worker captures its own\n // genuine error into `genuineErrors` and returns; we re-throw below.\n await Promise.allSettled(workers)\n\n // Finalize any run still open after the pool wound down. With the\n // stop-dispatch flag these are the runs that were mid-flight in peer workers\n // when the first genuine error fired — abort them so their emitters finalize\n // (hooks fire, store records a terminal status) instead of leaking.\n for (const [runId, emitter] of openRuns) {\n await finalizeAbort(emitter, runId, 'campaign aborted: genuine error in a sibling run')\n }\n openRuns.clear()\n\n if (genuineErrors.length > 0) {\n throw genuineErrors.length === 1\n ? genuineErrors[0]\n : new AggregateError(\n genuineErrors,\n `runEvalCampaign: ${genuineErrors.length} runs failed with genuine (non-run-level) errors`,\n )\n }\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\n/**\n * Abort a run whose owning work threw or was orphaned by a sibling's genuine\n * error, finalizing the emitter so hooks fire and the store records a terminal\n * status. Safe to call unconditionally: it only aborts runs that are still\n * `running`. Two no-op cases are intentional and benign:\n *\n * - the run was never started (absent from the store) — nothing to finalize\n * - the run is already terminal (`completed` / `failed` / `aborted`) —\n * finalizing again would overwrite the real outcome (e.g. flip a passed run\n * to `aborted` with `{ pass: false, notes: reason }`), so we leave it alone\n *\n * Any OTHER failure of the store read or the abort write (disk full, FS fault,\n * backend down) is a genuine diagnostic and propagates rather than being\n * swallowed.\n */\nexport async function finalizeAbort(\n emitter: TraceEmitter,\n runId: string,\n reason: string,\n): Promise<void> {\n const existing = await emitter.traceStore.getRun(runId)\n if (existing === undefined) return // run never started; nothing to abort\n if (existing.status !== 'running') return // already finalized; never overwrite a real outcome\n await emitter.abortRun(reason)\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nasync function resolveAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n if (isAgentProfileCell(input)) {\n if (!(await verifyAgentProfileCell(input))) {\n throw new Error(`runEvalCampaign: agentProfile.cellId does not match its content`)\n }\n return input\n }\n return buildAgentProfileCell(input)\n}\n\nfunction isAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): input is AgentProfileCell {\n return 'schemaVersion' in input && 'cellId' in input\n}\n\nfunction assertAgentProfileMatchesRun(\n profile: AgentProfileCell,\n model: string,\n promptHash: string,\n): void {\n if (profile.model !== undefined && profile.model !== model) {\n throw new Error(\n `runEvalCampaign: agentProfile.model \"${profile.model}\" does not match outcome.model \"${model}\"`,\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== promptHash) {\n throw new Error(\n `runEvalCampaign: agentProfile.promptHash \"${profile.promptHash}\" does not match outcome.promptHash \"${promptHash}\"`,\n )\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // Lightweight hex: we don't need crypto-grade here, just stability + uniqueness.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmSA,MAAM,oBAA8C;CAClD,aAAa;CACb,8BAA8B;CAC9B,gBAAgB;AAClB;AAEA,MAAM,gBAAsC;CAC1C,wBAAwB;CACxB,aAAa;AACf;AAEA,eAAsB,gBACpB,MAC6B;CAE7B,eAAe,KAAK,SAAS,KAAK,qBAAqB,aAAa;CAEpE,IAAI,KAAK,SAAS,WAAW,GAC3B,MAAM,IAAI,MAAM,8CAA8C;CAEhE,IAAI,KAAK,UAAU,WAAW,GAC5B,MAAM,IAAI,MAAM,+CAA+C;CAEjE,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,KAAK,UAAU;EAC7B,IAAI,WAAW,IAAI,EAAE,EAAE,GACrB,MAAM,IAAI,MAAM,0CAA0C,EAAE,GAAG,GAAG;EAEpE,WAAW,IAAI,EAAE,EAAE;CACrB;CACA,MAAM,8BAAc,IAAI,IAAY;CACpC,KAAK,MAAM,KAAK,KAAK,WAAW;EAC9B,IAAI,YAAY,IAAI,EAAE,UAAU,GAC9B,MAAM,IAAI,MAAM,0CAA0C,EAAE,WAAW,GAAG;EAE5E,YAAY,IAAI,EAAE,UAAU;CAC9B;CACA,IAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GACnE,MAAM,IAAI,MACR,uCAAuC,KAAK,OAAO,WAAW,iCAChE;CAEF,IAAI,CAAC,KAAK,WACR,MAAM,IAAI,MAAM,oEAAoE;CAGtF,MAAM,QAAQ,KAAK,SAAS;EAAC;EAAG;EAAG;CAAC;CACpC,MAAM,WAAwB,KAAK,YAAY;CAC/C,MAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;CACrD,MAAM,YAAY;EAAE,GAAG;EAAmB,GAAI,KAAK,aAAa,CAAC;CAAG;CACpE,MAAM,qBAA8C,KAAK,sBAAsB;CAC/E,MAAM,MAAM,KAAK,cAAc,KAAK,IAAI;CACxC,MAAM,WAAW,KAAK,QAAQ,WAAW,GAAA,CAAI,QAAQ,QAAQ,EAAE;CAC/D,MAAM,WAAW,KAAK,QAAQ,YAAY;CAC1C,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,MAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;CAGhF,MAAM,sBAAsB,MAAM,SAChC,aAAa;EACX,YAAY,KAAK;EACjB,UAAU,KAAK,SAAS,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC,KAAK;EAC9C,WAAW,KAAK,UAAU,KAAK,MAAM,EAAE,UAAU,CAAC,CAAC,KAAK;EACxD,OAAO,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EACtC;EACA,YAAY,KAAK,QAAQ,cAAc;EACvC;EACA;EACA;CACF,CAAC,CACH;CAIA,MAAM,QAAgB,CAAC;CACvB,KAAK,MAAM,WAAW,KAAK,UACzB,KAAK,MAAM,YAAY,KAAK,WAC1B,KAAK,MAAM,QAAQ,OACjB,MAAM,KAAK;EAAE;EAAS;EAAU;CAAK,CAAC;CAK5C,MAAM,YAAY,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAC9C,MAAM,OAAoB,CAAC;CAC3B,MAAM,mBAAyC,CAAC;CAChD,MAAM,aAA0B,CAAC;CASjC,IAAI,SAAS;CACb,IAAI,WAAW;CACf,MAAM,gBAA2B,CAAC;CAKlC,MAAM,2BAAW,IAAI,IAA0B;CAE/C,eAAe,SAAwB;EACrC,OAAO,CAAC,UAAU;GAChB,MAAM,IAAI;GACV,IAAI,KAAK,MAAM,QAAQ;GACvB,MAAM,OAAO,MAAM;GACnB,IAAI;IACF,MAAM,SAAS,MAAM,WAAW,IAAI;IACpC,KAAK,KAAK,OAAO,MAAM;IACvB,iBAAiB,KAAK,OAAO,SAAS;GACxC,SAAS,KAAK;IACZ,IAAI,eAAe,oBAAoB;KACrC,WAAW,KAAK,IAAI,MAAM;KAC1B,IAAI,IAAI,WAAW,iBAAiB,KAAK,IAAI,SAAS;IACxD,OAAO;KAKL,cAAc,KAAK,GAAG;KACtB,WAAW;KACX;IACF;GACF;EACF;CACF;CAEA,eAAe,WACb,MAC+D;EAC/D,MAAM,SAAS,KAAK,SAAS,aAAA,CAAc;GACzC,YAAY,KAAK;GACjB,OAAO;GACP,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb,CAAC;EACD,MAAM,gBAAuC;GAC3C,YAAY,KAAK;GACjB;GACA,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb;EACA,MAAM,QAAQ,KAAK,aAAa,aAAa;EAC7C,MAAM,UAAU,eAAe,aAAa;EAE5C,MAAM,UAAU,IAAI,aAAa,OAAO;GACtC;GACA,KAAK,KAAK;GACV,eAAe,KAAK;EACtB,CAAC;EAGD,SAAS,IAAI,OAAO,OAAO;EAE3B,MAAM,UAA4B;GAChC,GAAG,KAAK;GACR;GACA,cAAc,EAAE,MAAM;EACxB;EAEA,MAAM,MAA6B;GACjC;GACA,cAAc,KAAK;GACnB,SAAS,KAAK,QAAQ;GACtB,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;GACrC,MAAM,KAAK;GACX;GACA;GACA;GACA;GACA;EACF;EAEA,IAAI;GACF,MAAM,YAAY,IAAI;GACtB,IAAI;GACJ,IAAI;IACF,UAAU,MAAM,KAAK,OAAO,GAAG;GACjC,SAAS,KAAK;IACZ,MAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAM/D,MAAM,cAAc,SAAS,OAAO,OAAO;IAC3C,MAAM,IAAI,mBAAmB;KAC3B;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO;IACT,CAAC;GACH;GACA,MAAM,SAAS,IAAI,IAAI;GAEvB,MAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO;IAAE,GAAG;IAAW;GAAQ,CAAC;GACvF,IAAI,CAAC,gBAAgB,IACnB,QAAQ,oBAAR;IACE,KAAK,SACH,MAAM,IAAI,kBAAkB,eAAe;IAC7C,KAAK,eACH,MAAM,IAAI,mBACR;KACE;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO,gBAAgB,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,IAAI;IAC5D,GACA,eACF;IACF,KAAK,OAEH;GACJ;GAGF,MAAM,gBAA4B,EAChC,KAAK,QAAQ,OAAO,CAAC,EACvB;GACA,IAAI,aAAa,WAAW,cAAc,eAAe,QAAQ;QAC5D,cAAc,cAAc,QAAQ;GACzC,IAAI,QAAQ,gBAAgB,KAAA,GAAW,cAAc,cAAc,QAAQ;GAE3E,MAAM,SAAoB;IACxB;IACA,cAAc,KAAK;IACnB,aAAa,KAAK,QAAQ;IAC1B,MAAM,KAAK;IACX,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,YAAY,QAAQ;IACpB,WAAW,KAAK;IAChB;IACA,SAAS,QAAQ;IACjB,gBAAgB,QAAQ;IACxB,YAAY,QAAQ;IACpB,iBAAiB;IACjB,eAAe,QAAQ;IACvB,SAAS;IACT,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;IACrE,aAAa,QAAQ;IACrB;IACA,YAAY,KAAK,SAAS;GAC5B;GACA,MAAM,gBACJ,QAAQ,iBACP,OAAO,KAAK,iBAAiB,aAC1B,MAAM,KAAK,aAAa;IACtB,YAAY,KAAK;IACjB;IACA,WAAW,KAAK,QAAQ;IACxB,YAAY,KAAK,SAAS;IAC1B,MAAM,KAAK;IACX,SAAS,KAAK,QAAQ;IACtB,cAAc,KAAK,SAAS,QAAQ,CAAC;GACvC,CAAC,IACD,KAAK;GACX,IAAI,kBAAkB,KAAA,GAAW;IAC/B,MAAM,eAAe,MAAM,wBAAwB,aAAa;IAChE,6BAA6B,cAAc,QAAQ,OAAO,QAAQ,UAAU;IAC5E,OAAO,eAAe;GACxB;GACA,OAAO;IAAE,QAAQ,kBAAkB,MAAM;IAAG,WAAW;GAAgB;EACzE,UAAU;GAIR,SAAS,OAAO,KAAK;EACvB;CACF;CAEA,MAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,SAAS,OAAO,CAAC;CAI1F,MAAM,QAAQ,WAAW,OAAO;CAMhC,KAAK,MAAM,CAAC,OAAO,YAAY,UAC7B,MAAM,cAAc,SAAS,OAAO,kDAAkD;CAExF,SAAS,MAAM;CAEf,IAAI,cAAc,SAAS,GACzB,MAAM,cAAc,WAAW,IAC3B,cAAc,KACd,IAAI,eACF,eACA,oBAAoB,cAAc,OAAO,iDAC3C;CAIN,IAAI;CACJ,IAAI,KAAK,QAQP,SAAS,MAAM,eAAe,MAAM;EANlC,GAAG,KAAK;EACR,YAAY,KAAK,OAAO;EACxB,OAAO,aAAa,QAAQ,WAAW;EACvC,aAAa,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EACzC,qBAAqB,uBAAuB,KAAA;CAED,CAAC;CAGhD,MAAM,UAAU,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAE5C,OAAO;EACL,YAAY,KAAK;EACjB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;CACrC;CACA;CACA,YAAY,QAAmB,WAAgC;EAC7D,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,WAAW,GAAG,OAAO,KAAK,WAAW,OAAO,QAAQ;EAC7F,KAAK,SAAS;EACd,KAAK,YAAY;CACnB;AACF;;;;;;;;;;;;;;;;AAiBA,eAAsB,cACpB,SACA,OACA,QACe;CACf,MAAM,WAAW,MAAM,QAAQ,WAAW,OAAO,KAAK;CACtD,IAAI,aAAa,KAAA,GAAW;CAC5B,IAAI,SAAS,WAAW,WAAW;CACnC,MAAM,QAAQ,SAAS,MAAM;AAC/B;AAEA,SAAS,sBAAsB,SAA6B;CAC1D,QAAQ,WAAmD;EACzD,IAAI,CAAC,SACH,MAAM,IAAI,MACR,6LACF;EAEF,OAAO,IAAI,0BAA0B,EACnC,KAAK,GAAG,QAAQ,cAAc,OAAO,QACvC,CAAC;CACH;AACF;AAEA,eAAe,wBACb,OAC2B;CAC3B,IAAI,mBAAmB,KAAK,GAAG;EAC7B,IAAI,CAAE,MAAM,uBAAuB,KAAK,GACtC,MAAM,IAAI,MAAM,iEAAiE;EAEnF,OAAO;CACT;CACA,OAAO,sBAAsB,KAAK;AACpC;AAEA,SAAS,mBACP,OAC2B;CAC3B,OAAO,mBAAmB,SAAS,YAAY;AACjD;AAEA,SAAS,6BACP,SACA,OACA,YACM;CACN,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OACnD,MAAM,IAAI,MACR,wCAAwC,QAAQ,MAAM,kCAAkC,MAAM,EAChG;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,YAC7D,MAAM,IAAI,MACR,6CAA6C,QAAQ,WAAW,uCAAuC,WAAW,EACpH;AAEJ;AAEA,SAAS,aAAa,QAAuC;CAG3D,MAAM,OAAO,GAAG,OAAO,WAAW,IAAI,OAAO,UAAU,IAAI,OAAO,WAAW,IAAI,OAAO;CAExF,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,KAAK,WAAW,CAAC;EAC3B,KAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;EACvC,KAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;CACzC;CACA,OAAO,OAAO,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,IAAI,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClF"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"exec-y-DCLqK7.js","names":[],"sources":["../src/trajectory-replay/steps.ts","../src/trajectory-replay/exec.ts"],"sourcesContent":["/**\n * Recorded shell-trajectory steps and the observation grammar they carry.\n *\n * A recorded trajectory is the action/observation sequence an agent actually\n * ran. Scaffolds that execute one shell command per step (mini-SWE and the\n * CodeTracer-normalized corpora built from it) tag each observation with the\n * command's returncode and its combined output:\n *\n * <returncode>2</returncode>\n * <output>\n * …command output…\n * </output>\n *\n * That is one of four shapes a recorded turn carries, and the other three are\n * not command results at all:\n *\n * - a timeout notice, when the environment killed the command at its bound;\n * - a format-error notice, when the scaffold rejected the turn and ran\n * nothing;\n * - an elision marker `$<hex>`, when the published dump dropped the string.\n *\n * A turn also carries no observation when it is the run's last turn, because\n * the scaffold records an observation only when it hands one back to the model.\n *\n * The parsers here are the only place that grammar is decoded. Everything\n * downstream — replay verdicts, corpus enumeration, admission funnels, fix\n * prompts — reads the returncode, the output, and the failure signature through\n * these functions. A second decoder elsewhere is how a corpus reads as\n * unreplayable when it is not.\n */\n\n/**\n * One step of a recorded shell trajectory. Structural: any richer step record\n * (file refs, thinking text, tool type) satisfies it.\n */\nexport interface RecordedTrajectoryStep {\n /** 1-based position in the trajectory. */\n readonly step_id: number\n readonly action: string\n /** Null when the step recorded no observation (terminal submit steps). */\n readonly observation: string | null\n}\n\n/** Recorded returncode of a step, or null when the observation carries none. */\nexport function parseRecordedReturncode(observation: string | null): number | null {\n if (!observation) return null\n const m = /<returncode>(-?\\d+)<\\/returncode>/.exec(observation)\n return m ? Number(m[1]) : null\n}\n\n/** Text between the observation's <output> tags, or the raw observation when\n * the tags are absent. */\nexport function parseObservationOutput(observation: string | null): string {\n if (!observation) return ''\n const m = /<output>\\n?([\\s\\S]*?)\\n?<\\/output>/.exec(observation)\n return m ? m[1]! : observation\n}\n\n/**\n * Stable failure-signature candidate: the first line of the recorded output\n * that contains the word \"error\". Null when no such line exists — a verdict\n * then falls back to returncode-only matching and says so.\n * Pass an explicit signature to override (compiler quote glyphs vary with\n * locale, so a hand-picked ASCII substring is often more robust).\n */\nexport function deriveFailureSignature(observation: string | null): string | null {\n const line = parseObservationOutput(observation)\n .split('\\n')\n .find((l) => /\\berror\\b/i.test(l))\n return line ? line.trim().slice(0, 200) : null\n}\n\n/** mini-SWE's end-of-run submit convention: the agent echoes this sentinel\n * and dumps the diff. A label on this step marks a bad SUBMIT DECISION, not a\n * failed command — there is no executable failure to reproduce, so it is\n * never a counterfactual replay target. */\nexport const SUBMIT_ACTION_SIGNATURE = 'COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT'\n\nexport function isSubmitAction(action: string): boolean {\n return action.includes(SUBMIT_ACTION_SIGNATURE)\n}\n\n/**\n * True when the action is the sentinel echo and nothing else.\n *\n * The distinction decides whether a step may be dropped. An agent is told to\n * issue the sentinel alone, and 5.7% of recorded runs end on a command that\n * writes files or edits them and then echoes it. Dropping such a step because\n * it holds the sentinel would remove the run's last state change from the\n * replay, so the recorded end state and the replayed one would differ.\n */\nexport function isSubmitOnlyAction(action: string): boolean {\n return /^echo\\s+(?:\"COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT\"|'COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT'|COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT)$/.test(\n action.trim(),\n )\n}\n\n// ── Fields the published dump dropped ────────────────────────────────\n\n/**\n * A field the dump replaced with a counter instead of its text.\n *\n * The counter is hexadecimal and rises by one per dropped string in document\n * order, so a decimal-only reading (`$12`) accepts every marker that carries a\n * letter (`$3a`) as if it were real text. A command read that way replays as\n * the literal two-to-four characters `$3a`, which is not the command the run\n * executed.\n *\n * Nothing in the dump maps a marker back to its text: the same marker carries\n * different content in different rows, so there is no dictionary to read. A\n * field that matches is unrecoverable, so a caller rejects the row that holds\n * it rather than replaying the marker.\n */\nexport const RECORDED_ELISION_PATTERN = /^\\$[0-9a-f]+$/\n\nexport function isElidedField(value: string | null | undefined): boolean {\n return typeof value === 'string' && RECORDED_ELISION_PATTERN.test(value)\n}\n\n// ── Observation grammar ──────────────────────────────────────────────\n\n/** Substring the timeout notice always carries, whatever the command was. */\nexport const TIMEOUT_OBSERVATION_MARKER = 'timed out and has been killed'\n\n/** Opening of the notice the scaffold writes when a turn held no single action. */\nexport const FORMAT_ERROR_OBSERVATION_PREFIX =\n 'Please always provide EXACTLY ONE action in triple backticks, found '\n\n/**\n * What a recorded observation is.\n *\n * `command-result` is the only kind that carries an exit status. `timeout` and\n * `format-error` are the scaffold speaking rather than a command: the first\n * says the environment killed the command, the second says no command ran at\n * all. `elided` and `absent` carry no information about the step, and\n * `unreadable` is a shape this grammar does not know — never assumed to be\n * anything.\n */\nexport type RecordedObservationKind =\n | 'command-result'\n | 'timeout'\n | 'format-error'\n | 'elided'\n | 'absent'\n | 'unreadable'\n\nexport function classifyObservation(observation: string | null): RecordedObservationKind {\n if (observation === null) return 'absent'\n if (isElidedField(observation)) return 'elided'\n if (parseRecordedReturncode(observation) !== null) return 'command-result'\n if (observation.startsWith(FORMAT_ERROR_OBSERVATION_PREFIX)) return 'format-error'\n if (observation.includes(TIMEOUT_OBSERVATION_MARKER)) return 'timeout'\n return 'unreadable'\n}\n\n/** True when the recording shows the environment killed this step at its\n * wall-clock bound. Such a step carries no returncode, so no replay can\n * confirm or contradict it. */\nexport function isRecordedTimeout(observation: string | null): boolean {\n return classifyObservation(observation) === 'timeout'\n}\n\n// ── Turns to steps ───────────────────────────────────────────────────\n\n/**\n * One turn as the published trajectory dump holds it.\n *\n * A turn is not a step: the system prompt, the task statement and every turn\n * the scaffold rejected are turns that executed nothing.\n */\nexport interface RecordedTrajectoryTurn {\n readonly src?: string | null\n readonly msg?: string | null\n readonly tools?: readonly { readonly cmd?: string | null }[] | null\n readonly obs?: string | null\n}\n\nexport interface DecodedTrajectory {\n /** Executed commands in recorded order, each holding its OWN observation. */\n readonly steps: readonly RecordedTrajectoryStep[]\n /** Turns the scaffold rejected before anything ran. */\n readonly formatErrorTurns: number\n /**\n * Executed commands whose text the dump dropped.\n *\n * The step stays in `steps` carrying the marker, because the run did execute\n * a command there and a shorter list would misreport the trajectory. The\n * marker is not a command, so any count above zero means this trajectory\n * cannot be replayed: gate on it, or call `assertReplayableTrajectory`.\n */\n readonly elidedCommands: number\n /** True when the run's last turn was the submit sentinel with no observation. */\n readonly endedOnSubmitSentinel: boolean\n /** Turns carrying an observation this grammar cannot read. */\n readonly unreadableTurns: number\n}\n\n/**\n * Pair every recorded command with the observation of its own turn.\n *\n * Collecting commands and observations into two lists and zipping them is the\n * decode that looks right and is not: a turn the scaffold rejected carries an\n * observation of its own, so from the first such turn onward every observation\n * belongs to a different command than the one it is read against.\n *\n * A rejected turn executed nothing, so it is never a step — including when the\n * dump recorded a command for it. The scaffold rejects a turn holding several\n * bash blocks and runs none of them, while the dump keeps one of the blocks in\n * the command field. Replaying that field would execute a command the recorded\n * run did not, which is a worse corpus than a smaller one.\n *\n * A rejected turn is recognised by its observation, so a rejected turn whose\n * observation the dump elided is indistinguishable from an executed command\n * whose observation it elided. Both read as a step. Nothing in the dump\n * separates them, and the row-level defence is the share of unreadable exits a\n * caller admits: a row with no elided observation cannot hold this case at all.\n *\n * A trailing step that echoes the sentinel and nothing else, with no\n * observation, is dropped from `steps` and reported as `endedOnSubmitSentinel`.\n * The scaffold records an observation only when it hands one to the model, and\n * the sentinel ends the run, so the missing observation is the end of the\n * transcript rather than a gap in it. Echoing the sentinel changes no state, so\n * the recorded end state is the state the step before it left.\n *\n * A step that DID get an observation stays a step: the run continued past it.\n * So does a step that echoes the sentinel after doing real work — its state\n * change is part of the recorded end state, and with no observation its exit is\n * unknown, which `finalRecordedOutcome` reports rather than hides.\n */\nexport function decodeRecordedTurns(turns: readonly RecordedTrajectoryTurn[]): DecodedTrajectory {\n const steps: RecordedTrajectoryStep[] = []\n let formatErrorTurns = 0\n let unreadableTurns = 0\n for (const turn of turns) {\n const command = turn.tools?.[0]?.cmd ?? null\n const observation = turn.obs ?? null\n const kind = classifyObservation(observation)\n if (kind === 'format-error') {\n formatErrorTurns += 1\n continue\n }\n if (command === null) {\n if (kind === 'unreadable' || kind === 'elided') unreadableTurns += 1\n continue\n }\n steps.push({ step_id: steps.length + 1, action: command, observation })\n }\n const last = steps[steps.length - 1]\n const endedOnSubmitSentinel =\n last !== undefined && last.observation === null && isSubmitOnlyAction(last.action)\n if (endedOnSubmitSentinel) steps.pop()\n return {\n steps,\n formatErrorTurns,\n elidedCommands: steps.filter((step) => isElidedField(step.action)).length,\n endedOnSubmitSentinel,\n unreadableTurns,\n }\n}\n\n// ── The state a trajectory ended in ──────────────────────────────────\n\n/**\n * How the recorded run's last executed command ended.\n *\n * `killed` is a measured outcome, not a missing one: the environment stopped\n * the command at its bound and wrote a notice instead of an exit status.\n * `unreadable` names the observation kind that blocked the read, so a funnel\n * can report which shape cost it the row.\n */\nexport type RecordedFinalOutcome =\n | { readonly kind: 'returncode'; readonly value: number }\n | { readonly kind: 'killed' }\n | { readonly kind: 'unreadable'; readonly reason: RecordedObservationKind }\n\n/**\n * The outcome of the last step, or `null` when the trajectory has no steps.\n *\n * Reads only the last step. `decodeRecordedTurns` has already removed the turns\n * that executed nothing, so the last step is the last command the run ran.\n */\nexport function finalRecordedOutcome(\n steps: readonly RecordedTrajectoryStep[],\n): RecordedFinalOutcome | null {\n const last = steps[steps.length - 1]\n if (last === undefined) return null\n const kind = classifyObservation(last.observation)\n if (kind === 'command-result') {\n return { kind: 'returncode', value: parseRecordedReturncode(last.observation)! }\n }\n if (kind === 'timeout') return { kind: 'killed' }\n return { kind: 'unreadable', reason: kind }\n}\n\n/**\n * Steps whose recorded exit a replay cannot check.\n *\n * A killed step counts: the recording holds no exit status to compare a replay\n * against, so agreement on it cannot be measured either way.\n */\nexport function unreadableExitCount(steps: readonly RecordedTrajectoryStep[]): number {\n return steps.filter((step) => classifyObservation(step.observation) !== 'command-result').length\n}\n\n/**\n * Throw unless every recorded command survived the dump.\n *\n * A replay executes `steps` verbatim, so one elision marker in an action means\n * the replay runs the literal two-to-four characters of the marker instead of\n * the command the run executed. The guard is here so a replayer needs one call\n * rather than a field check it can forget.\n */\nexport function assertReplayableTrajectory(decoded: DecodedTrajectory): void {\n if (decoded.elidedCommands === 0) return\n const markers = decoded.steps\n .filter((step) => isElidedField(step.action))\n .map((step) => step.action)\n throw new Error(\n `trajectory holds ${decoded.elidedCommands} command(s) the dump elided (${markers.slice(0, 5).join(', ')}); ` +\n 'no dictionary maps a marker back to its text, so this trajectory cannot be replayed',\n )\n}\n","/**\n * The execution boundary replay runs across.\n *\n * A replay needs one thing from its environment: a session that runs a shell\n * command inside the trajectory's own image and reports the exit code and\n * output. That is the whole contract. Concrete backends — a sandbox platform\n * client, a docker exec, an SSH shell — live with the consumer that owns the\n * infrastructure, so this package depends on no sandbox client.\n */\n\nexport interface ReplayExecResult {\n exitCode: number\n stdout: string\n stderr: string\n}\n\nexport interface ReplayExecSession {\n exec(command: string, timeoutMs: number): Promise<ReplayExecResult>\n close(): Promise<void>\n}\n\nexport interface ReplayExecBackend {\n /** One fresh execution environment per call; the caller closes it. */\n open(): Promise<ReplayExecSession>\n}\n\n/** Builds a backend pinned to one image. Callers that resolve images\n * internally (batch, corpus wire, finding verification) take this instead of\n * a backend, so every case runs against its own image. */\nexport type ReplayExecBackendFactory = (image: string) => ReplayExecBackend\n\n/**\n * mini-SWE runs every action as a fresh /bin/sh subshell from a fixed\n * workdir. Reproduce that exactly — and stay quote-proof for arbitrary\n * recorded actions — by piping the base64 of the action into `sh` after\n * cd-ing to the workdir. Exit code is sh's, i.e. the action's.\n */\nexport function wrapActionForExec(action: string, cwd: string): string {\n const b64 = Buffer.from(action, 'utf8').toString('base64')\n const quotedCwd = `'${cwd.replaceAll(\"'\", `'\\\\''`)}'`\n return `cd ${quotedCwd} && printf %s ${b64} | base64 -d | sh`\n}\n"],"mappings":";;AA4CA,SAAgB,wBAAwB,aAA2C;CACjF,IAAI,CAAC,aAAa,OAAO;CACzB,MAAM,IAAI,oCAAoC,KAAK,WAAW;CAC9D,OAAO,IAAI,OAAO,EAAE,EAAE,IAAI;AAC5B;;;AAIA,SAAgB,uBAAuB,aAAoC;CACzE,IAAI,CAAC,aAAa,OAAO;CACzB,MAAM,IAAI,qCAAqC,KAAK,WAAW;CAC/D,OAAO,IAAI,EAAE,KAAM;AACrB;;;;;;;;AASA,SAAgB,uBAAuB,aAA2C;CAChF,MAAM,OAAO,uBAAuB,WAAW,CAAC,CAC7C,MAAM,IAAI,CAAC,CACX,MAAM,MAAM,aAAa,KAAK,CAAC,CAAC;CACnC,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,IAAI;AAC5C;;;;;AAMA,MAAa,0BAA0B;AAEvC,SAAgB,eAAe,QAAyB;CACtD,OAAO,OAAO,SAAS,uBAAuB;AAChD;;;;;;;;;;AAWA,SAAgB,mBAAmB,QAAyB;CAC1D,OAAO,qIAAqI,KAC1I,OAAO,KAAK,CACd;AACF;;;;;;;;;;;;;;;AAkBA,MAAa,2BAA2B;AAExC,SAAgB,cAAc,OAA2C;CACvE,OAAO,OAAO,UAAU,YAAY,yBAAyB,KAAK,KAAK;AACzE;;AAKA,MAAa,6BAA6B;;AAG1C,MAAa,kCACX;AAoBF,SAAgB,oBAAoB,aAAqD;CACvF,IAAI,gBAAgB,MAAM,OAAO;CACjC,IAAI,cAAc,WAAW,GAAG,OAAO;CACvC,IAAI,wBAAwB,WAAW,MAAM,MAAM,OAAO;CAC1D,IAAI,YAAY,WAAA,sEAA0C,GAAG,OAAO;CACpE,IAAI,YAAY,SAAA,+BAAmC,GAAG,OAAO;CAC7D,OAAO;AACT;;;;AAKA,SAAgB,kBAAkB,aAAqC;CACrE,OAAO,oBAAoB,WAAW,MAAM;AAC9C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,SAAgB,oBAAoB,OAA6D;CAC/F,MAAM,QAAkC,CAAC;CACzC,IAAI,mBAAmB;CACvB,IAAI,kBAAkB;CACtB,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,KAAK,QAAQ,EAAE,EAAE,OAAO;EACxC,MAAM,cAAc,KAAK,OAAO;EAChC,MAAM,OAAO,oBAAoB,WAAW;EAC5C,IAAI,SAAS,gBAAgB;GAC3B,oBAAoB;GACpB;EACF;EACA,IAAI,YAAY,MAAM;GACpB,IAAI,SAAS,gBAAgB,SAAS,UAAU,mBAAmB;GACnE;EACF;EACA,MAAM,KAAK;GAAE,SAAS,MAAM,SAAS;GAAG,QAAQ;GAAS;EAAY,CAAC;CACxE;CACA,MAAM,OAAO,MAAM,MAAM,SAAS;CAClC,MAAM,wBACJ,SAAS,KAAA,KAAa,KAAK,gBAAgB,QAAQ,mBAAmB,KAAK,MAAM;CACnF,IAAI,uBAAuB,MAAM,IAAI;CACrC,OAAO;EACL;EACA;EACA,gBAAgB,MAAM,QAAQ,SAAS,cAAc,KAAK,MAAM,CAAC,CAAC,CAAC;EACnE;EACA;CACF;AACF;;;;;;;AAuBA,SAAgB,qBACd,OAC6B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;CAClC,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,MAAM,OAAO,oBAAoB,KAAK,WAAW;CACjD,IAAI,SAAS,kBACX,OAAO;EAAE,MAAM;EAAc,OAAO,wBAAwB,KAAK,WAAW;CAAG;CAEjF,IAAI,SAAS,WAAW,OAAO,EAAE,MAAM,SAAS;CAChD,OAAO;EAAE,MAAM;EAAc,QAAQ;CAAK;AAC5C;;;;;;;AAQA,SAAgB,oBAAoB,OAAkD;CACpF,OAAO,MAAM,QAAQ,SAAS,oBAAoB,KAAK,WAAW,MAAM,gBAAgB,CAAC,CAAC;AAC5F;;;;;;;;;AAUA,SAAgB,2BAA2B,SAAkC;CAC3E,IAAI,QAAQ,mBAAmB,GAAG;CAClC,MAAM,UAAU,QAAQ,MACrB,QAAQ,SAAS,cAAc,KAAK,MAAM,CAAC,CAAC,CAC5C,KAAK,SAAS,KAAK,MAAM;CAC5B,MAAM,IAAI,MACR,oBAAoB,QAAQ,eAAe,+BAA+B,QAAQ,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,IAAI,EAAE,uFAE3G;AACF;;;;;;;;;AC5RA,SAAgB,kBAAkB,QAAgB,KAAqB;CACrE,MAAM,MAAM,OAAO,KAAK,QAAQ,MAAM,CAAC,CAAC,SAAS,QAAQ;CAEzD,OAAO,MAAM,IADS,IAAI,WAAW,KAAK,OAAO,EAAE,GAC5B,gBAAgB,IAAI;AAC7C"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"exporters-q9iL-2Jf.js","names":[],"sources":["../src/rollout/exporters.ts"],"sourcesContent":["/**\n * Pure exporters over `tangle.rollout.v1` lines → the training-data shapes\n * the improvement loops feed:\n * - SFT chat JSONL (clean trainable successes, {messages, metadata})\n * - reward rows (every scored line, success or failure, with steps)\n * - Prime Intellect verifiers RolloutOutput (prompt/completion split + reward)\n * - OpenAI RFT items (prompt turns + verdict reference fields)\n *\n * All exporters are pure functions of the lines — filtering (never train on\n * holdout, reward thresholds, the realness gate) happens HERE, on inline\n * labels, no joins.\n *\n * Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward\n * on a minted line has been checked against the anti-Goodhart invariant, and\n * the brand is what stops a hand-built object literal claiming a positive\n * reward on a gamed run from being handed to an exporter that copies it\n * verbatim into training data.\n */\n\nimport type {\n ChatMessage,\n MintedRolloutLine,\n RolloutLine,\n RolloutSplit,\n RolloutStep,\n ToolDef,\n} from './schema'\nimport { assertRewardGate } from './schema'\n\n// ---------------------------------------------------------------------------\n// The realness claims every emitted row carries.\n// ---------------------------------------------------------------------------\n\n/**\n * The gate's two claims, which travel TOGETHER on every emitted row.\n *\n * `realness_gated` alone is ambiguous, and the ambiguity is exploitable:\n * `false` reads as \"we screened it and nothing fired\", so a producer that has no\n * screen at all emitted rows indistinguishable from screened-clean ones, and\n * every consumer of the published dataset read them as clean. The second field\n * is what separates the two claims, and it only removes the ambiguity if it\n * reaches the WIRE — for a round it existed on `RolloutOutcome` and on no\n * exported row shape at all, which left the published rows exactly as ambiguous\n * as before.\n *\n * So there is one helper and every row shape spreads it. A row that states one\n * claim without the other is not constructible by copying the pattern, and\n * `exporters.test.ts` walks every emitted shape to prove none does.\n */\nexport interface RealnessLabels {\n /** The screen's VERDICT: the run faked its success signal. */\n realness_gated: boolean\n /**\n * Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its\n * verdict. `false` = the producer declares it has none. `null` = not stated\n * (pre-unification producers), which is \"unknown\" and never \"clean\".\n */\n realness_screened: boolean | null\n}\n\nexport function realnessLabels(line: MintedRolloutLine): RealnessLabels {\n return {\n realness_gated: line.outcome.realness_gated === true,\n // `?? null` rather than `?? false`: absent means the producer never stated\n // it, and `false` is a load-bearing claim (\"no screen exists\") that would be\n // an invention here.\n realness_screened: line.outcome.realness_screened ?? null,\n }\n}\n\n// ---------------------------------------------------------------------------\n// (a) SFT chat JSONL\n// ---------------------------------------------------------------------------\n\nexport interface TrainingExportOptions {\n /** Include held-out evaluation data in training output. Default false. */\n allowHeldOutTrainingData?: boolean\n /** Require reward to be strictly greater than this value. Default 0. */\n minimumQualityExclusive?: number\n}\n\n/**\n * What a signed-signal exporter (verifiers, RFT) does with lines that are not\n * clean trainable successes — realness-gated lines above all.\n *\n * - 'exclude' — the default, the same fail-closed policy as every\n * other training export: positive, completed,\n * non-gated rows on a trainable split.\n * - 'zero-and-flag' — keep them, at their non-positive (or null) reward,\n * with `RealnessLabels` on the row. The dataset release\n * sets this per `FORMAT_GATE_DISPOSITION`: in these\n * formats the reward is a signed learning signal, so a\n * gamed trajectory at reward 0 is a correct negative,\n * and dropping it would bias the negative population\n * toward honest failures and leave a trainer no example\n * of gaming being penalized. The split policy is NOT\n * relaxed: held-out lines still need the named opt-in.\n *\n * SFT deliberately has no such option — an SFT row is an imitation target and\n * a gamed trajectory must never appear in one at any weight.\n */\nexport type GatedLineDisposition = 'exclude' | 'zero-and-flag'\n\nexport interface SignedSignalExportOptions extends TrainingExportOptions {\n /** Disposition for non-trainable lines. Default 'exclude'. */\n gatedLines?: GatedLineDisposition\n}\n\nexport type SftExportOptions = TrainingExportOptions\n\nexport interface SftRow {\n messages: ChatMessage[]\n metadata: {\n rollout_id: string\n run_id: string\n candidate_id: string | null\n instance_id: string\n reward: number\n } & RealnessLabels\n}\n\n/**\n * Supervised fine-tune rows: the completed conversation of each qualifying\n * line. Fail-closed filters: trainable split only (never holdout/canary),\n * reward strictly above `minimumQualityExclusive` (default 0), realness-gated\n * lines never qualify, gap lines carry no trainable content, and\n * copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001\n * rule 7 — see `ChatMessage.is_copied_context`).\n *\n * `realness_gated` is therefore always `false` on an emitted row. It is carried\n * anyway: an SFT row is a pure imitation target, so the row states its realness\n * claims instead of making the reader know the format's policy, and carrying\n * both flags on all four shapes is what lets the release accounting measure\n * every config with one rule rather than skipping the one whose row shape\n * happened to omit the field.\n */\nexport function toSftRows(lines: MintedRolloutLine[], options: SftExportOptions = {}): SftRow[] {\n for (const line of lines) assertRewardGate(line, 'SFT export')\n return lines\n .filter((line) => isTrainingLineEligible(line, options) && line.messages.length > 0)\n .map((line) => ({\n // Dropped, not kept-and-masked: a copied-context turn was authored by\n // another agent, and an SFT trainer has no notion of \"present but not a\n // target\" — every message it is handed is something to imitate.\n messages: line.messages.filter((message) => message.is_copied_context !== true),\n metadata: {\n rollout_id: line.rollout_id,\n run_id: line.run_id,\n candidate_id: line.candidate_id ?? null,\n instance_id: line.task.instance_id,\n reward: line.outcome.reward as number,\n ...realnessLabels(line),\n },\n }))\n .filter((row) => row.messages.length > 0)\n}\n\n// ---------------------------------------------------------------------------\n// (b) Reward rows\n// ---------------------------------------------------------------------------\n\nexport interface RewardRow {\n /** First user turn — the task prompt. */\n prompt: string\n steps: RolloutStep[]\n reward: number\n metadata: {\n rollout_id: string\n run_id: string\n candidate_id: string | null\n instance_id: string\n split: RolloutSplit\n /**\n * The anti-Goodhart verdict AND whether a screen produced it, carried on the\n * row rather than left implicit in a reward of 0. Zeroing alone is lossy: it\n * makes a run that FAKED its success indistinguishable from one that\n * honestly failed, so a consumer can neither drop the gamed population nor\n * mine it (a labeled gamed trajectory is the training signal for a gaming\n * detector). `realness_gated` alone is lossy the same way in the other\n * direction — see `RealnessLabels`.\n */\n } & RealnessLabels\n}\n\n/**\n * Reward-labeled rows for completed, positive-quality training runs.\n */\nexport function toRewardRows(\n lines: MintedRolloutLine[],\n options: TrainingExportOptions = {},\n): RewardRow[] {\n for (const line of lines) assertRewardGate(line, 'reward-row export')\n return lines\n .filter(\n (line) =>\n isTrainingLineEligible(line, options) &&\n line.messages.some(\n (message) =>\n message.role === 'user' &&\n typeof message.content === 'string' &&\n message.content.length > 0,\n ),\n )\n .map((line) => ({\n prompt: line.messages.find((m) => m.role === 'user')?.content ?? '',\n steps: line.steps ?? [],\n reward: line.outcome.reward as number,\n metadata: {\n rollout_id: line.rollout_id,\n run_id: line.run_id,\n candidate_id: line.candidate_id ?? null,\n instance_id: line.task.instance_id,\n split: line.task.split,\n ...realnessLabels(line),\n },\n }))\n}\n\n// ---------------------------------------------------------------------------\n// (c) Prime Intellect verifiers RolloutOutput\n// ---------------------------------------------------------------------------\n\nexport interface VerifiersTokenUsage {\n input_tokens: number | null\n output_tokens: number | null\n reasoning_tokens: number | null\n cache_read_tokens: number | null\n cache_write_tokens: number | null\n}\n\nexport interface VerifiersRolloutOutput {\n /** Messages through the last turn BEFORE the first assistant turn. */\n prompt: ChatMessage[]\n /** The first assistant turn onward — what the policy produced. */\n completion: ChatMessage[]\n reward: number | null\n metrics: Record<string, unknown>\n tool_defs: ToolDef[]\n token_usage: VerifiersTokenUsage\n info: {\n task: RolloutLine['task']\n policy: RolloutLine['policy']\n rollout_id: string\n run_id: string\n experiment_id: string | null\n candidate_id: string | null\n generation: number | null\n candidate_index: number | null\n role: RolloutLine['role']\n } & RealnessLabels\n}\n\n/** Index of the first assistant turn; messages.length when none exists. */\nfunction firstAssistantIndex(messages: ChatMessage[]): number {\n const index = messages.findIndex((m) => m.role === 'assistant')\n return index === -1 ? messages.length : index\n}\n\nexport function toVerifiersRolloutOutput(line: MintedRolloutLine): VerifiersRolloutOutput {\n assertRewardGate(line, 'verifiers export')\n const split = firstAssistantIndex(line.messages)\n return {\n prompt: line.messages.slice(0, split),\n completion: line.messages.slice(split),\n reward: line.outcome.reward,\n metrics: line.outcome.metrics,\n tool_defs: line.tool_defs,\n token_usage: {\n input_tokens: line.cost.tokens_in,\n output_tokens: line.cost.tokens_out,\n reasoning_tokens: line.cost.tokens_reasoning,\n cache_read_tokens: line.cost.cache_read,\n cache_write_tokens: line.cost.cache_write,\n },\n info: {\n task: line.task,\n policy: line.policy,\n rollout_id: line.rollout_id,\n run_id: line.run_id,\n experiment_id: line.experiment_id ?? null,\n candidate_id: line.candidate_id ?? null,\n generation: line.generation,\n candidate_index: line.candidate_index,\n role: line.role,\n ...realnessLabels(line),\n },\n }\n}\n\nexport function toVerifiersRolloutOutputs(\n lines: MintedRolloutLine[],\n options: SignedSignalExportOptions = {},\n): VerifiersRolloutOutput[] {\n if (options.gatedLines === 'zero-and-flag') {\n return lines\n .filter((line) => isSplitEligible(line, options) && line.messages.length > 0)\n .map(toVerifiersRolloutOutput)\n }\n return lines\n .filter(\n (line) =>\n isTrainingLineEligible(line, options) &&\n firstAssistantIndex(line.messages) > 0 &&\n firstAssistantIndex(line.messages) < line.messages.length,\n )\n .map(toVerifiersRolloutOutput)\n}\n\n// ---------------------------------------------------------------------------\n// (d) OpenAI RFT items\n// ---------------------------------------------------------------------------\n\nexport interface RftItem {\n /** Prompt turns only — the graded completion is re-sampled during RFT. */\n messages: ChatMessage[]\n /** Verdict/label fields the grader references as item.reference.* */\n reference: {\n reward: number | null\n reward_source: string | null\n verdict: unknown\n instance_id: string\n suite: string\n split: RolloutSplit\n rollout_id: string\n } & RealnessLabels\n}\n\nexport function toRftItem(line: MintedRolloutLine): RftItem {\n assertRewardGate(line, 'RFT export')\n const split = firstAssistantIndex(line.messages)\n return {\n messages: line.messages.slice(0, split),\n reference: {\n reward: line.outcome.reward,\n reward_source: line.outcome.reward_source,\n verdict: line.outcome.verdict,\n instance_id: line.task.instance_id,\n suite: line.task.suite,\n split: line.task.split,\n rollout_id: line.rollout_id,\n ...realnessLabels(line),\n },\n }\n}\n\n/** RFT needs a real prompt: lines whose transcript starts with prompt turns. */\nexport function toRftItems(\n lines: MintedRolloutLine[],\n options: SignedSignalExportOptions = {},\n): RftItem[] {\n return lines\n .filter(\n (line) =>\n (options.gatedLines === 'zero-and-flag'\n ? isSplitEligible(line, options)\n : isTrainingLineEligible(line, options)) &&\n line.messages.length > 0 &&\n firstAssistantIndex(line.messages) > 0,\n )\n .map(toRftItem)\n}\n\n// ---------------------------------------------------------------------------\n// Serialization — one JSON object per line, the interchange format for\n// every export. `tangle.rollout.v1` lines and export rows alike.\n// ---------------------------------------------------------------------------\n\nexport function toJsonl(rows: ReadonlyArray<unknown>): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length ? '\\n' : '')\n}\n\n/**\n * The split half of the training policy alone: `search` is trainable, held-out\n * needs the named opt-in, `dev` and `canary` never ship. This is the ONE check\n * `'zero-and-flag'` does not relax — a gated line is shipped as a labeled\n * negative, not as a licence to train on evaluation data.\n *\n * Exported as the single implementation of that rule: `rl/exporters` applies\n * it on its line paths too, so the two waists cannot drift on which splits are\n * trainable.\n */\nexport function isSplitEligible(\n line: RolloutLine,\n options: Pick<TrainingExportOptions, 'allowHeldOutTrainingData'>,\n): boolean {\n if (line.task.split === 'search') return true\n return line.task.split === 'holdout' && options.allowHeldOutTrainingData === true\n}\n\nfunction isTrainingLineEligible(\n line: RolloutLine,\n options: TrainingExportOptions,\n): line is RolloutLine & { outcome: RolloutLine['outcome'] & { reward: number } } {\n const minimumQualityExclusive = options.minimumQualityExclusive ?? 0\n if (!Number.isFinite(minimumQualityExclusive)) {\n throw new Error('minimumQualityExclusive must be finite')\n }\n\n const reward = line.outcome.reward\n if (reward === null) return false\n if (!Number.isFinite(reward)) {\n throw new Error(`training reward for rollout \"${line.rollout_id}\" must be finite`)\n }\n if (reward <= minimumQualityExclusive) return false\n if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {\n return false\n }\n if (line.outcome.realness_gated === true) return false\n return isSplitEligible(line, options)\n}\n"],"mappings":";;AA4DA,SAAgB,eAAe,MAAyC;CACtE,OAAO;EACL,gBAAgB,KAAK,QAAQ,mBAAmB;EAIhD,mBAAmB,KAAK,QAAQ,qBAAqB;CACvD;AACF;;;;;;;;;;;;;;;;AAoEA,SAAgB,UAAU,OAA4B,UAA4B,CAAC,GAAa;CAC9F,KAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,YAAY;CAC7D,OAAO,MACJ,QAAQ,SAAS,uBAAuB,MAAM,OAAO,KAAK,KAAK,SAAS,SAAS,CAAC,CAAC,CACnF,KAAK,UAAU;EAId,UAAU,KAAK,SAAS,QAAQ,YAAY,QAAQ,sBAAsB,IAAI;EAC9E,UAAU;GACR,YAAY,KAAK;GACjB,QAAQ,KAAK;GACb,cAAc,KAAK,gBAAgB;GACnC,aAAa,KAAK,KAAK;GACvB,QAAQ,KAAK,QAAQ;GACrB,GAAG,eAAe,IAAI;EACxB;CACF,EAAE,CAAC,CACF,QAAQ,QAAQ,IAAI,SAAS,SAAS,CAAC;AAC5C;;;;AAgCA,SAAgB,aACd,OACA,UAAiC,CAAC,GACrB;CACb,KAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,mBAAmB;CACpE,OAAO,MACJ,QACE,SACC,uBAAuB,MAAM,OAAO,KACpC,KAAK,SAAS,MACX,YACC,QAAQ,SAAS,UACjB,OAAO,QAAQ,YAAY,YAC3B,QAAQ,QAAQ,SAAS,CAC7B,CACJ,CAAC,CACA,KAAK,UAAU;EACd,QAAQ,KAAK,SAAS,MAAM,MAAM,EAAE,SAAS,MAAM,CAAC,EAAE,WAAW;EACjE,OAAO,KAAK,SAAS,CAAC;EACtB,QAAQ,KAAK,QAAQ;EACrB,UAAU;GACR,YAAY,KAAK;GACjB,QAAQ,KAAK;GACb,cAAc,KAAK,gBAAgB;GACnC,aAAa,KAAK,KAAK;GACvB,OAAO,KAAK,KAAK;GACjB,GAAG,eAAe,IAAI;EACxB;CACF,EAAE;AACN;;AAqCA,SAAS,oBAAoB,UAAiC;CAC5D,MAAM,QAAQ,SAAS,WAAW,MAAM,EAAE,SAAS,WAAW;CAC9D,OAAO,UAAU,KAAK,SAAS,SAAS;AAC1C;AAEA,SAAgB,yBAAyB,MAAiD;CACxF,iBAAiB,MAAM,kBAAkB;CACzC,MAAM,QAAQ,oBAAoB,KAAK,QAAQ;CAC/C,OAAO;EACL,QAAQ,KAAK,SAAS,MAAM,GAAG,KAAK;EACpC,YAAY,KAAK,SAAS,MAAM,KAAK;EACrC,QAAQ,KAAK,QAAQ;EACrB,SAAS,KAAK,QAAQ;EACtB,WAAW,KAAK;EAChB,aAAa;GACX,cAAc,KAAK,KAAK;GACxB,eAAe,KAAK,KAAK;GACzB,kBAAkB,KAAK,KAAK;GAC5B,mBAAmB,KAAK,KAAK;GAC7B,oBAAoB,KAAK,KAAK;EAChC;EACA,MAAM;GACJ,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,QAAQ,KAAK;GACb,eAAe,KAAK,iBAAiB;GACrC,cAAc,KAAK,gBAAgB;GACnC,YAAY,KAAK;GACjB,iBAAiB,KAAK;GACtB,MAAM,KAAK;GACX,GAAG,eAAe,IAAI;EACxB;CACF;AACF;AAEA,SAAgB,0BACd,OACA,UAAqC,CAAC,GACZ;CAC1B,IAAI,QAAQ,eAAe,iBACzB,OAAO,MACJ,QAAQ,SAAS,gBAAgB,MAAM,OAAO,KAAK,KAAK,SAAS,SAAS,CAAC,CAAC,CAC5E,IAAI,wBAAwB;CAEjC,OAAO,MACJ,QACE,SACC,uBAAuB,MAAM,OAAO,KACpC,oBAAoB,KAAK,QAAQ,IAAI,KACrC,oBAAoB,KAAK,QAAQ,IAAI,KAAK,SAAS,MACvD,CAAC,CACA,IAAI,wBAAwB;AACjC;AAqBA,SAAgB,UAAU,MAAkC;CAC1D,iBAAiB,MAAM,YAAY;CACnC,MAAM,QAAQ,oBAAoB,KAAK,QAAQ;CAC/C,OAAO;EACL,UAAU,KAAK,SAAS,MAAM,GAAG,KAAK;EACtC,WAAW;GACT,QAAQ,KAAK,QAAQ;GACrB,eAAe,KAAK,QAAQ;GAC5B,SAAS,KAAK,QAAQ;GACtB,aAAa,KAAK,KAAK;GACvB,OAAO,KAAK,KAAK;GACjB,OAAO,KAAK,KAAK;GACjB,YAAY,KAAK;GACjB,GAAG,eAAe,IAAI;EACxB;CACF;AACF;;AAGA,SAAgB,WACd,OACA,UAAqC,CAAC,GAC3B;CACX,OAAO,MACJ,QACE,UACE,QAAQ,eAAe,kBACpB,gBAAgB,MAAM,OAAO,IAC7B,uBAAuB,MAAM,OAAO,MACxC,KAAK,SAAS,SAAS,KACvB,oBAAoB,KAAK,QAAQ,IAAI,CACzC,CAAC,CACA,IAAI,SAAS;AAClB;AAOA,SAAgB,QAAQ,MAAsC;CAC5D,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,OAAO;AAC/E;;;;;;;;;;;AAYA,SAAgB,gBACd,MACA,SACS;CACT,IAAI,KAAK,KAAK,UAAU,UAAU,OAAO;CACzC,OAAO,KAAK,KAAK,UAAU,aAAa,QAAQ,6BAA6B;AAC/E;AAEA,SAAS,uBACP,MACA,SACgF;CAChF,MAAM,0BAA0B,QAAQ,2BAA2B;CACnE,IAAI,CAAC,OAAO,SAAS,uBAAuB,GAC1C,MAAM,IAAI,MAAM,wCAAwC;CAG1D,MAAM,SAAS,KAAK,QAAQ;CAC5B,IAAI,WAAW,MAAM,OAAO;CAC5B,IAAI,CAAC,OAAO,SAAS,MAAM,GACzB,MAAM,IAAI,MAAM,gCAAgC,KAAK,WAAW,iBAAiB;CAEnF,IAAI,UAAU,yBAAyB,OAAO;CAC9C,IAAI,CAAC,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,UAAU,MACpF,OAAO;CAET,IAAI,KAAK,QAAQ,mBAAmB,MAAM,OAAO;CACjD,OAAO,gBAAgB,MAAM,OAAO;AACtC"}