@tangle-network/agent-eval 0.163.2 → 0.170.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/CHANGELOG.md +114 -0
  2. package/README.md +2 -0
  3. package/dist/adapters/http.d.ts +108 -0
  4. package/dist/adapters/http.d.ts.map +1 -0
  5. package/dist/adapters/http.js +208 -0
  6. package/dist/adapters/http.js.map +1 -0
  7. package/dist/analyst/index.d.ts +40 -70
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +18 -311
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
  12. package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
  13. package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
  14. package/dist/benchmark-C4wk_Sjr.js.map +1 -0
  15. package/dist/{benchmark-command-CF-4GEWZ.js → benchmark-command-BA7qOdWw.js} +236 -251
  16. package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
  17. package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
  18. package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
  19. package/dist/benchmarks/index.d.ts +5 -5
  20. package/dist/benchmarks/index.js +3 -3
  21. package/dist/builder-eval/index.d.ts +3 -3
  22. package/dist/builder-eval/index.js +1 -1
  23. package/dist/campaign/index.d.ts +8 -8
  24. package/dist/campaign/index.js +7 -7
  25. package/dist/{campaign-DQZmc2Dq.js → campaign-BeCbxFqs.js} +15 -14
  26. package/dist/campaign-BeCbxFqs.js.map +1 -0
  27. package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
  28. package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
  29. package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
  30. package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
  31. package/dist/cli.js +2 -2
  32. package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
  33. package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
  34. package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
  35. package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
  36. package/dist/contract/index.d.ts +12 -703
  37. package/dist/contract/index.js +11 -11
  38. package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
  39. package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
  40. package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
  41. package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
  42. package/dist/{define-agent-eval-D08pWIJb.js → define-agent-eval-Clj-8igZ.js} +20 -8
  43. package/dist/{define-agent-eval-D08pWIJb.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
  44. package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
  45. package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
  46. package/dist/{dspy-rlm-engine-DhA9qKIm.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
  47. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
  48. package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
  49. package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
  50. package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
  51. package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
  52. package/dist/{eval-campaign-BfohKmzx.js → eval-campaign-JDTeE6Pl.js} +4 -4
  53. package/dist/{eval-campaign-BfohKmzx.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
  54. package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
  55. package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
  56. package/dist/experiment/index.d.ts +5 -5
  57. package/dist/experiment/index.js +4 -4
  58. package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
  59. package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
  60. package/dist/{external-optimizer-process-BFmh36vW.js → external-optimizer-process-CQxylYeG.js} +4 -11
  61. package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
  62. package/dist/{external-optimizer-subprocess-CqLMW3nh.js → external-optimizer-subprocess-Cex8Da2i.js} +25 -11
  63. package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
  64. package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
  65. package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
  66. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
  67. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
  68. package/dist/fuzz.d.ts +1 -1
  69. package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
  70. package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/index-Bfs5aufo.d.ts +704 -0
  74. package/dist/index-Bfs5aufo.d.ts.map +1 -0
  75. package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
  76. package/dist/index-CM-SM00y.d.ts.map +1 -0
  77. package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
  78. package/dist/index-DBbivBNs.d.ts.map +1 -0
  79. package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
  80. package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
  81. package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
  82. package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
  83. package/dist/index.d.ts +67 -37
  84. package/dist/index.d.ts.map +1 -1
  85. package/dist/index.js +32 -26
  86. package/dist/index.js.map +1 -1
  87. package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
  88. package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
  89. package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
  90. package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
  91. package/dist/internal-BMFSR8Ns.js.map +1 -1
  92. package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
  93. package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
  94. package/dist/ledger-core/index.js +2 -2
  95. package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
  96. package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
  97. package/dist/{llm-judge-Du7WQPh7.js → llm-judge-DbJdo8Nj.js} +97 -19
  98. package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
  99. package/dist/matrix/index.d.ts +2 -2
  100. package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
  101. package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
  102. package/dist/meta-eval/index.d.ts +3 -3
  103. package/dist/meta-eval/index.js +1 -1
  104. package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
  105. package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
  106. package/dist/multishot/golden/index.d.ts +1 -1
  107. package/dist/multishot/index.d.ts +2 -2
  108. package/dist/openapi.json +4 -4
  109. package/dist/pipelines/index.d.ts +5 -5
  110. package/dist/pipelines/index.js +3 -3
  111. package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
  112. package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
  113. package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
  114. package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
  115. package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
  116. package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
  117. package/dist/{produced-state-Be0BK3RN.js → produced-state-CtSIp5cQ.js} +4 -4
  118. package/dist/{produced-state-Be0BK3RN.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
  119. package/dist/profile-cell.js +1 -1
  120. package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
  121. package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
  122. package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
  123. package/dist/proposal-findings-bko3GGy-.js.map +1 -0
  124. package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
  125. package/dist/provenance-CIRUardl.d.ts.map +1 -0
  126. package/dist/{query-_5g6re3_.js → query-BPGMVlbM.js} +3 -3
  127. package/dist/{query-_5g6re3_.js.map → query-BPGMVlbM.js.map} +1 -1
  128. package/dist/{query-CwnHlu5p.d.ts → query-Na5gEIGd.d.ts} +3 -3
  129. package/dist/{query-CwnHlu5p.d.ts.map → query-Na5gEIGd.d.ts.map} +1 -1
  130. package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
  131. package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
  132. package/dist/{release-confidence-nGDJiiwc.js → release-confidence-CzUHc4z4.js} +3 -3
  133. package/dist/{release-confidence-nGDJiiwc.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
  134. package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
  135. package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
  136. package/dist/reporting.d.ts +3 -3
  137. package/dist/reporting.js +2 -2
  138. package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
  139. package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
  140. package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
  141. package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-O5zKDANP.js → reward-hacking-SkxYgT0x.js} +2 -2
  143. package/dist/{reward-hacking-O5zKDANP.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
  144. package/dist/rl.d.ts +8 -8
  145. package/dist/rl.d.ts.map +1 -1
  146. package/dist/rl.js +6 -5
  147. package/dist/rl.js.map +1 -1
  148. package/dist/rollout/index.d.ts +1 -1
  149. package/dist/rollout/index.js +2 -2
  150. package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
  151. package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
  152. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
  153. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
  154. package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
  155. package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
  156. package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
  157. package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
  158. package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
  159. package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
  160. package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
  161. package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
  162. package/dist/{semantic-concept-judge-BsDMOwJr.js → semantic-concept-judge-I36eejJx.js} +2 -2
  163. package/dist/{semantic-concept-judge-BsDMOwJr.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
  164. package/dist/{sequential-BLMbdrD7.js → sequential-B51qAYE4.js} +2 -2
  165. package/dist/{sequential-BLMbdrD7.js.map → sequential-B51qAYE4.js.map} +1 -1
  166. package/dist/{server-BjYiJHoJ.js → server-CCEnywOR.js} +26 -18
  167. package/dist/server-CCEnywOR.js.map +1 -0
  168. package/dist/{skillopt-optimization-method-UArRo-nr.js → skillopt-optimization-method-B2R9C5aG.js} +9 -9
  169. package/dist/{skillopt-optimization-method-UArRo-nr.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
  170. package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
  171. package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
  172. package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
  173. package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
  174. package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
  175. package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
  176. package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
  177. package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
  178. package/dist/{store-tool-spans-BVga3c37.js → store-tool-spans-B9o6tU8f.js} +3 -3
  179. package/dist/{store-tool-spans-BVga3c37.js.map → store-tool-spans-B9o6tU8f.js.map} +1 -1
  180. package/dist/storyboard/index.d.ts +1 -1
  181. package/dist/{summary-report-BXeQ5Ues.js → summary-report-Bgh8CpNK.js} +2 -2
  182. package/dist/{summary-report-BXeQ5Ues.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
  183. package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
  184. package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
  185. package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
  186. package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
  187. package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
  188. package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
  189. package/dist/{tool-waste-CKc7bYIg.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
  190. package/dist/{tool-waste-CKc7bYIg.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
  191. package/dist/{tool-waste-8BQiUc8K.js → tool-waste-CwGHzBzX.js} +2 -2
  192. package/dist/{tool-waste-8BQiUc8K.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
  193. package/dist/trace-repair/index.d.ts +3 -3
  194. package/dist/trace-repair/index.d.ts.map +1 -1
  195. package/dist/trace-repair/index.js +4 -3
  196. package/dist/trace-repair/index.js.map +1 -1
  197. package/dist/traces.d.ts +59 -61
  198. package/dist/traces.d.ts.map +1 -1
  199. package/dist/traces.js +7 -7
  200. package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
  201. package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
  202. package/dist/trajectory-replay/index.d.ts +3 -3
  203. package/dist/trajectory-replay/index.js +1 -1
  204. package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
  205. package/dist/types-CiWITkGo.js.map +1 -0
  206. package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
  207. package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
  208. package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
  209. package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
  210. package/dist/{types-BPb2Kf_C.d.ts → types-i21ccEkr.d.ts} +3 -3
  211. package/dist/{types-BPb2Kf_C.d.ts.map → types-i21ccEkr.d.ts.map} +1 -1
  212. package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
  213. package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
  214. package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
  215. package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
  216. package/dist/wire/index.d.ts +23 -8
  217. package/dist/wire/index.d.ts.map +1 -1
  218. package/dist/wire/index.js +2 -2
  219. package/docs/code-agent-intake.md +64 -0
  220. package/docs/concepts.md +1 -1
  221. package/docs/design/statistics-decisions.md +1 -1
  222. package/docs/distributed-driver.md +3 -6
  223. package/docs/public-api.md +122 -106
  224. package/docs/wire-protocol.md +5 -3
  225. package/package.json +9 -2
  226. package/dist/benchmark-BhT16ep9.js.map +0 -1
  227. package/dist/benchmark-command-CF-4GEWZ.js.map +0 -1
  228. package/dist/campaign-DQZmc2Dq.js.map +0 -1
  229. package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
  230. package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
  231. package/dist/contract/index.d.ts.map +0 -1
  232. package/dist/dspy-rlm-engine-DhA9qKIm.js.map +0 -1
  233. package/dist/external-optimizer-process-BFmh36vW.js.map +0 -1
  234. package/dist/external-optimizer-subprocess-CqLMW3nh.js.map +0 -1
  235. package/dist/index-CGtH1piv.d.ts.map +0 -1
  236. package/dist/index-D-V8gCs_.d.ts.map +0 -1
  237. package/dist/index-vrJugRal.d.ts +0 -1
  238. package/dist/kind-factory-DY8FdoXf.js.map +0 -1
  239. package/dist/llm-judge-Du7WQPh7.js.map +0 -1
  240. package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
  241. package/dist/run-score-lDzV0X8j.js.map +0 -1
  242. package/dist/server-BjYiJHoJ.js.map +0 -1
  243. package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
  244. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
  245. package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
  246. package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
  247. package/dist/types-BI4fT3HN.js.map +0 -1
@@ -1,4 +1,4 @@
1
- import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
2
1
  import { p as CostProvenance } from "../cost-ledger-DbQdN3nO.js";
3
- import { a as MatrixCell, c as CellSpend, i as MatrixAxis, l as readCellSpend, n as AxisSummary, o as MatrixResult, r as CellResult, s as RunAgentMatrixOptions, t as runAgentMatrix, u as withCellSpend } from "../index-D_P7Ye43.js";
2
+ import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
3
+ import { a as MatrixCell, c as CellSpend, i as MatrixAxis, l as readCellSpend, n as AxisSummary, o as MatrixResult, r as CellResult, s as RunAgentMatrixOptions, t as runAgentMatrix, u as withCellSpend } from "../index-DNgf5gyG.js";
4
4
  export { type AxisSummary, type CellResult, type CellSpend, type CostProvenance, type DefaultVerdict, type MatrixAxis, type MatrixCell, type MatrixResult, type RunAgentMatrixOptions, readCellSpend, runAgentMatrix, withCellSpend };
@@ -1,6 +1,6 @@
1
1
  import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
2
- import { w as JudgeScore } from "./types-D4s7Z6nq.js";
3
- import { o as MatrixResult } from "./index-D_P7Ye43.js";
2
+ import { w as JudgeScore } from "./types-Dy237wiH.js";
3
+ import { o as MatrixResult } from "./index-DNgf5gyG.js";
4
4
  import { AgentProfile } from "@tangle-network/agent-interface";
5
5
  //#region src/multishot/types.d.ts
6
6
  interface MultishotMessage {
@@ -398,4 +398,4 @@ interface RunMultishotMatrixResult {
398
398
  declare function runMultishotMatrix<TPersona extends MultishotPersona>(opts: RunMultishotMatrixOptions<TPersona>): Promise<RunMultishotMatrixResult>;
399
399
  //#endregion
400
400
  export { MultishotToolExecutor as A, MultishotFatalToolError as C, MultishotShape as D, MultishotResult as E, assertMultishotShotResult as F, MultishotTransportRequest as M, MultishotTransportResponse as N, MultishotShotResultError as O, MultishotTransportToolCall as P, MultishotDriverEmptyError as S, MultishotPersona as T, JudgeRunResult as _, MultishotCellOutput as a, runJudge as b, RunMultishotMatrixResult as c, MultishotShot as d, RunMultishotOptions as f, JudgeDimension as g, JudgeConfig as h, ConversationJudgeInput as i, MultishotTransport as j, MultishotToolDefinition as k, computeCellComposite as l, DEFAULT_JUDGE_MODEL as m, CellCompositeInput as n, MultishotJudges as o, runMultishot as p, CellCompositeScore as r, RunMultishotMatrixOptions as s, ArtifactJudgeInput as t, runMultishotMatrix as u, renderDimensions as v, MultishotMessage as w, MultishotArtifact as x, renderJsonFooter as y };
401
- //# sourceMappingURL=matrix-eXKRMHnL.d.ts.map
401
+ //# sourceMappingURL=matrix-BpI5Trmo.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"matrix-eXKRMHnL.d.ts","names":[],"sources":["../src/multishot/types.ts","../src/multishot/judges.ts","../src/multishot/multishot.ts","../src/multishot/matrix.ts"],"mappings":";;;;;UAIiB;EACf;EACA;EACA;EACA,YAAY;IAAQ;IAAY;IAAc,MAAM;;;UAGrC;EACf;EACA;EACA;IAAc;IAAc,MAAM;;EAClC;;UAGe;EACf,YAAY;EACZ,WAAW;EACX;EACA;;;EAGA;;;;;;;;EAQA,iBAAiB;;UAGF;EACf;EACA;IACE;IACA;IACA,YAAY;;;;;;UAOC;EACf;EACA,UAAU,MAAM;EAChB,QAAQ;EACR;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;IAAY;IAAc;;;UAGX;EACf;IAAW;IAAyB,aAAa;;EACjD;IAAU;IAAwB;;;;EAGlC;;;;EAIA;;;;;;;;KASU,sBACV,KAAK,8BACF,QAAQ;KAED,yBACV,MAAM,yBACN;;EAEE,WAAW;EACX,SAAS;MAER;EAAU;EAAiB;;UAEf;;EAEf;;GAEC;;;;;;;;UASc,eAAe,iBAAiB;;EAE/C,eAAe,SAAS;;;EAGxB,2BAA2B,SAAS;;cAGzB,kCAAkC;WACjB;EAA5B,YAA4B;;cAMjB,gCAAgC;EAC3C,YAAY;;cAMD,iCAAiC;EAC5C,YAAY;;;;;;;;;;;;;;;;;;;iBAyBE,0BAA0B,yBAAyB,SAAS;;;cCvI/D;UAEI;;EAEf;;EAEA;;UAGe,YAAY;;EAE3B;;;EAGA,WAAW;;EAEX;;EAEA,YAAY;;EAEZ;;;EAGA,cAAc,OAAO;;EAErB;;UAGe;;EAEf,OAAO;;EAEP,MAAM;;iBAGc,SAAS,QAC7B,OAAO,YAAY,SACnB,OAAO,SACN,QAAQ;;;iBAsIK,iBAAiB,eAAe;;iBAKhC,iBAAiB,eAAe;;;UCxK/B,oBAAoB,iBAAiB;EACpD,SAAS;EACT,SAAS;;;EAGT,QAAQ,eAAe;;EAEvB,QAAQ;;EAER,gBAAgB,eAAe;;;;EAI/B,mBAAmB;EACnB;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,gBAAgB;;;;EAIhB,iBAAiB;;;EAGjB,gBAAgB;EAChB,SAAS;;;;;;;;;;;KAYC,cAAc,iBAAiB,qBACzC,MAAM,oBAAoB,cACvB,QAAQ;;;;;;;;;;;;;;;;;iBA2BS,aAAa,iBAAiB,kBAClD,MAAM,oBAAoB,YACzB,QAAQ;;;UClFM,uBAAuB,iBAAiB;EACvD,YAAY;EACZ,SAAS;;UAGM,mBAAmB,iBAAiB;EACnD,UAAU;EACV,SAAS;;UAGM,gBAAgB,iBAAiB;;EAEhD,cAAc,YAAY,uBAAuB;;EAEjD,aAAa,YAAY,mBAAmB;;EAE5C,iBAAiB,YAAY,mBAAmB;;EAEhD;;EAEA;;UAGe;EACf;EACA,cAAc;EACd;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;EAEF;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;;UAIa,0BAA0B,iBAAiB;;EAE1D,UAAU;IAAQ;IAAY,OAAO;;;EAErC,UAAU;;;EAGV,QAAQ,eAAe;;EAEvB,QAAQ,gBAAgB;;EAExB,QAAQ;;EAER,gBAAgB,eAAe;;EAE/B,mBAAmB;;EAEnB;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,gBAAgB;;EAEhB,iBAAiB;;;EAGjB,gBAAgB;;;;;;;;;;;;;;;EAehB,UAAU,cAAc;;;;;UAMT;EACf;EACA;EACA;;UAoBe;EACf,cAAc;;EAEd,cAAc,cAAc;;EAE5B,iBAAiB,cAAc;;;;;;;iBAQjB,qBAAqB,OAAO;EAC1C;EACA;EACA;EACA;;UAuBe;EACf,QAAQ,aAAa;;iBAGD,mBAAmB,iBAAiB,kBACxD,MAAM,0BAA0B,YAC/B,QAAQ"}
1
+ {"version":3,"file":"matrix-BpI5Trmo.d.ts","names":[],"sources":["../src/multishot/types.ts","../src/multishot/judges.ts","../src/multishot/multishot.ts","../src/multishot/matrix.ts"],"mappings":";;;;;UAIiB;EACf;EACA;EACA;EACA,YAAY;IAAQ;IAAY;IAAc,MAAM;;;UAGrC;EACf;EACA;EACA;IAAc;IAAc,MAAM;;EAClC;;UAGe;EACf,YAAY;EACZ,WAAW;EACX;EACA;;;EAGA;;;;;;;;EAQA,iBAAiB;;UAGF;EACf;EACA;IACE;IACA;IACA,YAAY;;;;;;UAOC;EACf;EACA,UAAU,MAAM;EAChB,QAAQ;EACR;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;IAAY;IAAc;;;UAGX;EACf;IAAW;IAAyB,aAAa;;EACjD;IAAU;IAAwB;;;;EAGlC;;;;EAIA;;;;;;;;KASU,sBACV,KAAK,8BACF,QAAQ;KAED,yBACV,MAAM,yBACN;;EAEE,WAAW;EACX,SAAS;MAER;EAAU;EAAiB;;UAEf;;EAEf;;GAEC;;;;;;;;UASc,eAAe,iBAAiB;;EAE/C,eAAe,SAAS;;;EAGxB,2BAA2B,SAAS;;cAGzB,kCAAkC;WACjB;EAA5B,YAA4B;;cAMjB,gCAAgC;EAC3C,YAAY;;cAMD,iCAAiC;EAC5C,YAAY;;;;;;;;;;;;;;;;;;;iBAyBE,0BAA0B,yBAAyB,SAAS;;;cCvI/D;UAEI;;EAEf;;EAEA;;UAGe,YAAY;;EAE3B;;;EAGA,WAAW;;EAEX;;EAEA,YAAY;;EAEZ;;;EAGA,cAAc,OAAO;;EAErB;;UAGe;;EAEf,OAAO;;EAEP,MAAM;;iBAGc,SAAS,QAC7B,OAAO,YAAY,SACnB,OAAO,SACN,QAAQ;;;iBAsIK,iBAAiB,eAAe;;iBAKhC,iBAAiB,eAAe;;;UCxK/B,oBAAoB,iBAAiB;EACpD,SAAS;EACT,SAAS;;;EAGT,QAAQ,eAAe;;EAEvB,QAAQ;;EAER,gBAAgB,eAAe;;;;EAI/B,mBAAmB;EACnB;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,gBAAgB;;;;EAIhB,iBAAiB;;;EAGjB,gBAAgB;EAChB,SAAS;;;;;;;;;;;KAYC,cAAc,iBAAiB,qBACzC,MAAM,oBAAoB,cACvB,QAAQ;;;;;;;;;;;;;;;;;iBA2BS,aAAa,iBAAiB,kBAClD,MAAM,oBAAoB,YACzB,QAAQ;;;UClFM,uBAAuB,iBAAiB;EACvD,YAAY;EACZ,SAAS;;UAGM,mBAAmB,iBAAiB;EACnD,UAAU;EACV,SAAS;;UAGM,gBAAgB,iBAAiB;;EAEhD,cAAc,YAAY,uBAAuB;;EAEjD,aAAa,YAAY,mBAAmB;;EAE5C,iBAAiB,YAAY,mBAAmB;;EAEhD;;EAEA;;UAGe;EACf;EACA,cAAc;EACd;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;EAEF;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;;UAIa,0BAA0B,iBAAiB;;EAE1D,UAAU;IAAQ;IAAY,OAAO;;;EAErC,UAAU;;;EAGV,QAAQ,eAAe;;EAEvB,QAAQ,gBAAgB;;EAExB,QAAQ;;EAER,gBAAgB,eAAe;;EAE/B,mBAAmB;;EAEnB;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,gBAAgB;;EAEhB,iBAAiB;;;EAGjB,gBAAgB;;;;;;;;;;;;;;;EAehB,UAAU,cAAc;;;;;UAMT;EACf;EACA;EACA;;UAoBe;EACf,cAAc;;EAEd,cAAc,cAAc;;EAE5B,iBAAiB,cAAc;;;;;;;iBAQjB,qBAAqB,OAAO;EAC1C;EACA;EACA;EACA;;UAuBe;EACf,QAAQ,aAAa;;iBAGD,mBAAmB,iBAAiB,kBACxD,MAAM,0BAA0B,YAC/B,QAAQ"}
@@ -1,9 +1,9 @@
1
- import { f as Run } from "../schema-Bjgdsn73.js";
2
- import { s as TraceStore } from "../store-B06JdC56.js";
1
+ import { f as Run } from "../schema-DID1Cqct.js";
2
+ import { s as TraceStore } from "../store-Cq9oOrI1.js";
3
3
  import { a as ContinuousCalibrationResult, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, t as CalibrationResult } from "../judge-calibration-C5CbMYce.js";
4
4
  import { n as SeriesConvergenceResult, o as CorpusAgreementReport, t as SeriesConvergenceOptions } from "../series-convergence-D9WgpXGi.js";
5
5
  import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
6
- import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-CxycqzX5.js";
6
+ import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-DluJLCKQ.js";
7
7
  //#region src/meta-eval/correlation-study.d.ts
8
8
  interface EvalMetricSpec {
9
9
  id: string;
@@ -1,7 +1,7 @@
1
1
  import { s as ValidationError } from "../errors-Dngq5h35.js";
2
2
  import { i as makeRng } from "../internal-BMFSR8Ns.js";
3
3
  import { a as spearmanR, r as pearsonR } from "../descriptive-1V17A-qa.js";
4
- import { u as runMetricExtractor } from "../query-_5g6re3_.js";
4
+ import { u as runMetricExtractor } from "../query-BPGMVlbM.js";
5
5
  import { t as analyzeSeries } from "../series-convergence-CjO2QdRW.js";
6
6
  import { t as rubricPredictiveValidity } from "../rubric-predictive-validity-2D5Gw9z9.js";
7
7
  import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
3
- import { s as runTaskScore } from "./run-record-BC0ebuRP.js";
3
+ import { s as runTaskScore } from "./run-record-DLORoL7t.js";
4
4
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
5
5
  import { a as assertMinted, r as ROLLOUT_SCHEMA } from "./schema-C1aaAxTf.js";
6
6
  //#region src/rollout/mint.ts
@@ -315,4 +315,4 @@ async function mintRolloutRows(records, store, options = {}) {
315
315
  //#endregion
316
316
  export { unmintableReasons as n, mintRolloutRows as t };
317
317
 
318
- //# sourceMappingURL=mint-DfODW1KW.js.map
318
+ //# sourceMappingURL=mint-DjfDUMHr.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"mint-DfODW1KW.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
1
+ {"version":3,"file":"mint-DjfDUMHr.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
@@ -1,4 +1,4 @@
1
- import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-eXKRMHnL.js";
1
+ import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-BpI5Trmo.js";
2
2
  //#region src/multishot/golden/compare.d.ts
3
3
  interface CompareOptions {
4
4
  /** Stop after this many mismatches. A structural divergence high in the
@@ -1,5 +1,5 @@
1
- import { w as JudgeScore } from "../types-D4s7Z6nq.js";
2
- import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-eXKRMHnL.js";
1
+ import { w as JudgeScore } from "../types-Dy237wiH.js";
2
+ import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-BpI5Trmo.js";
3
3
  import { AgentProfile } from "@tangle-network/agent-interface";
4
4
  //#region src/multishot/cost.d.ts
5
5
  /**
package/dist/openapi.json CHANGED
@@ -2,8 +2,8 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.163.2",
6
- "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
5
+ "version": "0.170.0",
6
+ "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.1.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",
9
9
  "url": "https://github.com/tangle-network/agent-eval"
@@ -217,7 +217,7 @@
217
217
  },
218
218
  "rubricVersion": {
219
219
  "type": "string",
220
- "description": "Stable hash of the rubric used. Scores are only comparable across runs when this matches."
220
+ "description": "Stable identity of the rubric used, as `<name>@sha256-rfc8785:<hex>`. Scores are only comparable across runs when this matches exactly."
221
221
  },
222
222
  "model": {
223
223
  "type": "string",
@@ -296,7 +296,7 @@
296
296
  },
297
297
  "rubricVersion": {
298
298
  "type": "string",
299
- "description": "Stable hash — match this to compare scores across runs."
299
+ "description": "Stable identity, as `<name>@sha256-rfc8785:<hex>` — match this to compare scores across runs."
300
300
  }
301
301
  },
302
302
  "required": [
@@ -1,8 +1,8 @@
1
- import { f as Run } from "../schema-Bjgdsn73.js";
2
- import { a as RunFilter, s as TraceStore } from "../store-B06JdC56.js";
3
- import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-CXL8NbEw.js";
4
- import { n as TrajectoryStep } from "../trajectory-Bi157Gun.js";
5
- import { c as computeToolUseMetrics, d as judgeAgreementView, f as BudgetBreachFinding, g as BaselineReport, h as BaselineOptions, i as toolWasteView, l as JudgeAgreementReport, m as budgetBreachView, n as ToolWasteOptions, p as BudgetBreachReport, r as ToolWasteReport, t as ToolWasteFinding, u as JudgePair } from "../tool-waste-CKc7bYIg.js";
1
+ import { f as Run } from "../schema-DID1Cqct.js";
2
+ import { a as RunFilter, s as TraceStore } from "../store-Cq9oOrI1.js";
3
+ import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-6YSvsKlp.js";
4
+ import { n as TrajectoryStep } from "../trajectory-r1bQqvBQ.js";
5
+ import { c as computeToolUseMetrics, d as judgeAgreementView, f as BudgetBreachFinding, g as BaselineReport, h as BaselineOptions, i as toolWasteView, l as JudgeAgreementReport, m as budgetBreachView, n as ToolWasteOptions, p as BudgetBreachReport, r as ToolWasteReport, t as ToolWasteFinding, u as JudgePair } from "../tool-waste-BrmLKxMw.js";
6
6
  //#region src/pipelines/first-divergence.d.ts
7
7
  interface DivergenceReport {
8
8
  runA: string;
@@ -1,7 +1,7 @@
1
- import { a as budgetBreachView, i as failureClusterView, n as computeToolUseMetrics, r as judgeAgreementView, t as toolWasteView } from "../tool-waste-8BQiUc8K.js";
1
+ import { a as budgetBreachView, i as failureClusterView, n as computeToolUseMetrics, r as judgeAgreementView, t as toolWasteView } from "../tool-waste-CwGHzBzX.js";
2
2
  import { t as compareToBaseline } from "../baseline-BC-eBZ7U.js";
3
- import { a as isToolSpan } from "../schema-k6ZBftVv.js";
4
- import { a as hasCapturedToolArgs, r as argHash, u as runMetricExtractor } from "../query-_5g6re3_.js";
3
+ import { a as isToolSpan } from "../schema-CdIX2aHu.js";
4
+ import { a as hasCapturedToolArgs, r as argHash, u as runMetricExtractor } from "../query-BPGMVlbM.js";
5
5
  import { t as buildTrajectory } from "../trajectory-D_7rLrvE.js";
6
6
  import { t as executionTrackByLane } from "../execution-tracks-CpgFPpS5.js";
7
7
  //#region src/pipelines/first-divergence.ts
@@ -1,3 +1,44 @@
1
+ //#region src/pareto.d.ts
2
+ /**
3
+ * Pareto frontier — multi-objective optimization over candidate runs.
4
+ *
5
+ * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
6
+ * trading off (cost, latency, quality) or (passRate, tokenBudget,
7
+ * ttfb), you rarely have a single "winner" — you have a set of
8
+ * non-dominated candidates. This module exposes:
9
+ *
10
+ * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
11
+ * - `dominates`: does A dominate B across all objectives?
12
+ *
13
+ * Each objective is declared with a direction: 'maximize' (higher=better)
14
+ * or 'minimize' (lower=better). Candidates are any object; pass an
15
+ * `objective(candidate)` accessor.
16
+ */
17
+ type Direction = 'maximize' | 'minimize';
18
+ interface Objective<T> {
19
+ /** Stable label used in reports. */
20
+ name: string;
21
+ direction: Direction;
22
+ value: (candidate: T) => number;
23
+ }
24
+ interface ParetoResult<T> {
25
+ frontier: T[];
26
+ dominated: T[];
27
+ /** Index map: frontier[i] dominates each of dominatedBy[i]. */
28
+ dominanceMap: Array<{
29
+ dominator: T;
30
+ dominated: T[];
31
+ }>;
32
+ }
33
+ /** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
34
+ declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
35
+ /**
36
+ * Compute the non-dominated frontier. Candidates with NaN/Infinity on any
37
+ * objective are excluded (can't rank them). A candidate enters the frontier
38
+ * iff no other candidate dominates it.
39
+ */
40
+ declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
41
+ //#endregion
1
42
  //#region src/campaign/gates/power-preflight.d.ts
2
43
  /**
3
44
  * Power preflight — "can this budget detect the effect you are hunting?"
@@ -72,46 +113,5 @@ interface PowerPreflight {
72
113
  * observable at this holdout size and worker variance. */
73
114
  declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
74
115
  //#endregion
75
- //#region src/pareto.d.ts
76
- /**
77
- * Pareto frontier — multi-objective optimization over candidate runs.
78
- *
79
- * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
80
- * trading off (cost, latency, quality) or (passRate, tokenBudget,
81
- * ttfb), you rarely have a single "winner" — you have a set of
82
- * non-dominated candidates. This module exposes:
83
- *
84
- * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
85
- * - `dominates`: does A dominate B across all objectives?
86
- *
87
- * Each objective is declared with a direction: 'maximize' (higher=better)
88
- * or 'minimize' (lower=better). Candidates are any object; pass an
89
- * `objective(candidate)` accessor.
90
- */
91
- type Direction = 'maximize' | 'minimize';
92
- interface Objective<T> {
93
- /** Stable label used in reports. */
94
- name: string;
95
- direction: Direction;
96
- value: (candidate: T) => number;
97
- }
98
- interface ParetoResult<T> {
99
- frontier: T[];
100
- dominated: T[];
101
- /** Index map: frontier[i] dominates each of dominatedBy[i]. */
102
- dominanceMap: Array<{
103
- dominator: T;
104
- dominated: T[];
105
- }>;
106
- }
107
- /** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
108
- declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
109
- /**
110
- * Compute the non-dominated frontier. Candidates with NaN/Infinity on any
111
- * objective are excluded (can't rank them). A candidate enters the frontier
112
- * iff no other candidate dominates it.
113
- */
114
- declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
115
- //#endregion
116
- export { paretoFrontier as a, powerPreflight as c, dominates as i, Objective as n, PowerPreflight as o, ParetoResult as r, PowerPreflightOptions as s, Direction as t };
117
- //# sourceMappingURL=pareto-BqNW3LJR.d.ts.map
116
+ export { Objective as a, paretoFrontier as c, Direction as i, PowerPreflightOptions as n, ParetoResult as o, powerPreflight as r, dominates as s, PowerPreflight as t };
117
+ //# sourceMappingURL=power-preflight-Ptse_Kq7.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"power-preflight-Ptse_Kq7.d.ts","names":[],"sources":["../src/pareto.ts","../src/campaign/gates/power-preflight.ts"],"mappings":";;;;;;;;;;;;;;;;KAgBY;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa;;;;;;;;;;;;;;;;;;;;;;;;;;UC5B5E;;EAEf;;;EAGA;;EAEA;;EAEA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;;EAGA;EACA;EACA;;;EAGA;;EAEA;;;;;;iBAec,eAAe,MAAM,wBAAwB"}
@@ -1,4 +1,4 @@
1
- import { i as hashCanonical, r as canonicalString } from "./canonical-IL-Bu-14.js";
1
+ import { a as hashCanonical, r as canonicalString } from "./canonical-DPyQ_rpt.js";
2
2
  import { createHash } from "node:crypto";
3
3
  //#region src/pre-registration.ts
4
4
  /**
@@ -107,4 +107,4 @@ async function evaluateHypothesis(manifest, observed) {
107
107
  //#endregion
108
108
  export { verifyManifest as a, signManifest as i, hashJson as n, manifestContentDigest as r, evaluateHypothesis as t };
109
109
 
110
- //# sourceMappingURL=pre-registration-KN9jkh58.js.map
110
+ //# sourceMappingURL=pre-registration-D94b7Of5.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"pre-registration-KN9jkh58.js","names":[],"sources":["../src/pre-registration.ts"],"sourcesContent":["/**\n * Pre-registered hypotheses — declare what you're testing BEFORE the\n * run, check it AFTER. Prevents p-hacking, optional stopping, and the\n * \"we ran until it looked good\" failure mode.\n *\n * Manifest is a plain JSON-friendly object. Sign it with a content hash\n * + timestamp; the registered record becomes immutable. Post-run,\n * evaluate the manifest against observed results — the library refuses\n * to let you re-interpret a different metric as the declared one.\n *\n * A signed manifest is a portable record: it is written once and verified\n * later, possibly by a different release. `algo` names the digest scheme it\n * was signed under, and verification selects the encoder by that field, so a\n * manifest signed by an earlier release still verifies.\n */\n\nimport { createHash } from 'node:crypto'\nimport { canonicalString, hashCanonical } from './ledger-core/canonical'\n\nexport interface HypothesisManifest {\n id: string\n /** Human prose — goes into the audit trail. */\n hypothesis: string\n /** Metric the hypothesis claims to move. */\n metric: string\n /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */\n direction: 'increase' | 'decrease'\n /** Minimum effect size to count (same units as the metric). */\n minEffect: number\n /** Alpha threshold. */\n alpha: number\n /** Target statistical power at which sample size was pre-computed. */\n power: number\n /** Declared N per arm before running. */\n preRegisteredN: number\n /** ISO8601 timestamp the manifest was registered. */\n registeredAt: string\n /** Optional identifiers to tie into the trace corpus. */\n baselineLabel?: string\n candidateLabel?: string\n}\n\n/**\n * Identifier for the hashing scheme used to produce `contentHash`.\n *\n * Both schemes are sha256 hex over the manifest with `contentHash` and `algo`\n * stripped, and differ only in how that manifest is serialized:\n *\n * - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}\n * emits.\n * - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests\n * signed by an earlier release carry it, or carry no `algo` at all, and\n * {@link verifyManifest} still verifies them.\n */\nexport type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785'\n\nexport interface SignedManifest extends HypothesisManifest {\n /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */\n contentHash: string\n /**\n * Algorithm string describing how `contentHash` was produced.\n *\n * Optional on the type so serialized manifests without it still parse,\n * but ALWAYS populated by {@link signManifest}. Consumers that want to\n * enforce a known algorithm should reject manifests where this field\n * is missing or unrecognized.\n */\n algo?: SignedManifestAlgo\n}\n\nexport interface HypothesisResult {\n manifest: SignedManifest\n observedN: number\n observedEffect: number\n observedPValue: number\n /** True iff the observed effect hits the pre-declared direction with\n * magnitude ≥ minEffect AND p < alpha. */\n confirmed: boolean\n /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */\n rejectionReasons: Array<\n 'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'\n >\n notes?: string\n}\n\n/**\n * SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of\n * `obj` — the package's one identity scheme, shared with `ledger-core`.\n *\n * Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,\n * class instances, cycles — are refused rather than coerced, because a\n * coercion maps two distinct records onto one digest.\n *\n * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,\n * which takes a string input and returns a truncated 12-char prompt id.\n *\n * @example\n * const hash = await hashJson({ id: '1', kind: 'spec' })\n * // 'a3f1...' (64 hex chars)\n */\nexport async function hashJson<T>(obj: T): Promise<string> {\n return hashCanonical(obj).slice('sha256:'.length)\n}\n\n/**\n * Key-sorted `JSON.stringify` digest. Private and read-only: it exists so a\n * manifest signed under `'sha256-content'` still verifies, and nothing that\n * WRITES a digest may call it.\n */\nfunction legacyContentDigest(value: unknown): string {\n return createHash('sha256')\n .update(JSON.stringify(sortKeysDeep(value)), 'utf8')\n .digest('hex')\n}\n\nfunction sortKeysDeep(value: unknown): unknown {\n if (value === null || typeof value !== 'object') return value\n if (Array.isArray(value)) return value.map(sortKeysDeep)\n const out: Record<string, unknown> = {}\n for (const key of Object.keys(value as Record<string, unknown>).sort()) {\n out[key] = sortKeysDeep((value as Record<string, unknown>)[key])\n }\n return out\n}\n\n/**\n * Digest of a manifest under its own declared scheme, with `contentHash` and\n * `algo` stripped. Synchronous, so a caller that must fail before consuming an\n * observation does not have to await. Throws on an `algo` this release does\n * not know — an unverifiable manifest must not read as a valid one.\n */\nexport function manifestContentDigest(manifest: SignedManifest): string {\n const { contentHash: _contentHash, algo, ...rest } = manifest\n void _contentHash\n if (algo === undefined || algo === 'sha256-content') return legacyContentDigest(rest)\n if (algo === 'sha256-rfc8785') {\n return createHash('sha256').update(canonicalString(rest), 'utf8').digest('hex')\n }\n throw new Error(`pre-registration: unrecognized manifest hash algo '${String(algo)}'`)\n}\n\n/**\n * Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical\n * JSON, with `contentHash` and `algo` stripped, and stamp the scheme in\n * `algo` so a later reader knows which encoder to verify with.\n */\nexport async function signManifest(m: HypothesisManifest): Promise<SignedManifest> {\n const signed: SignedManifest = { ...m, contentHash: '', algo: 'sha256-rfc8785' }\n return { ...signed, contentHash: manifestContentDigest(signed) }\n}\n\n/**\n * Verify that a signed manifest has not been tampered with, under the scheme\n * the manifest itself declares.\n */\nexport async function verifyManifest(m: SignedManifest): Promise<boolean> {\n return manifestContentDigest(m) === m.contentHash\n}\n\n/**\n * Evaluate a pre-registered hypothesis against observed results.\n * Mechanical — no re-interpretation permitted.\n */\nexport async function evaluateHypothesis(\n manifest: SignedManifest,\n observed: { n: number; effect: number; pValue: number },\n): Promise<HypothesisResult> {\n if (!(await verifyManifest(manifest))) {\n throw new Error('evaluateHypothesis: manifest content hash mismatch (tampered)')\n }\n const reasons: HypothesisResult['rejectionReasons'] = []\n const directionOk = manifest.direction === 'increase' ? observed.effect > 0 : observed.effect < 0\n if (!directionOk) reasons.push('wrong_direction')\n if (Math.abs(observed.effect) < manifest.minEffect) reasons.push('effect_too_small')\n if (observed.pValue >= manifest.alpha) reasons.push('not_significant')\n if (observed.n < manifest.preRegisteredN) reasons.push('undersampled')\n return {\n manifest,\n observedN: observed.n,\n observedEffect: observed.effect,\n observedPValue: observed.pValue,\n confirmed: reasons.length === 0,\n rejectionReasons: reasons,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoGA,eAAsB,SAAY,KAAyB;CACzD,OAAO,cAAc,GAAG,CAAC,CAAC,MAAM,CAAgB;AAClD;;;;;;AAOA,SAAS,oBAAoB,OAAwB;CACnD,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAU,aAAa,KAAK,CAAC,GAAG,MAAM,CAAC,CACnD,OAAO,KAAK;AACjB;AAEA,SAAS,aAAa,OAAyB;CAC7C,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU,OAAO;CACxD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,IAAI,YAAY;CACvD,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,OAAO,OAAO,KAAK,KAAgC,CAAC,CAAC,KAAK,GACnE,IAAI,OAAO,aAAc,MAAkC,IAAI;CAEjE,OAAO;AACT;;;;;;;AAQA,SAAgB,sBAAsB,UAAkC;CACtE,MAAM,EAAE,aAAa,cAAc,MAAM,GAAG,SAAS;CAErD,IAAI,SAAS,KAAA,KAAa,SAAS,kBAAkB,OAAO,oBAAoB,IAAI;CACpF,IAAI,SAAS,kBACX,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,IAAI,GAAG,MAAM,CAAC,CAAC,OAAO,KAAK;CAEhF,MAAM,IAAI,MAAM,sDAAsD,OAAO,IAAI,EAAE,EAAE;AACvF;;;;;;AAOA,eAAsB,aAAa,GAAgD;CACjF,MAAM,SAAyB;EAAE,GAAG;EAAG,aAAa;EAAI,MAAM;CAAiB;CAC/E,OAAO;EAAE,GAAG;EAAQ,aAAa,sBAAsB,MAAM;CAAE;AACjE;;;;;AAMA,eAAsB,eAAe,GAAqC;CACxE,OAAO,sBAAsB,CAAC,MAAM,EAAE;AACxC;;;;;AAMA,eAAsB,mBACpB,UACA,UAC2B;CAC3B,IAAI,CAAE,MAAM,eAAe,QAAQ,GACjC,MAAM,IAAI,MAAM,+DAA+D;CAEjF,MAAM,UAAgD,CAAC;CAEvD,IAAI,EADgB,SAAS,cAAc,aAAa,SAAS,SAAS,IAAI,SAAS,SAAS,IAC9E,QAAQ,KAAK,iBAAiB;CAChD,IAAI,KAAK,IAAI,SAAS,MAAM,IAAI,SAAS,WAAW,QAAQ,KAAK,kBAAkB;CACnF,IAAI,SAAS,UAAU,SAAS,OAAO,QAAQ,KAAK,iBAAiB;CACrE,IAAI,SAAS,IAAI,SAAS,gBAAgB,QAAQ,KAAK,cAAc;CACrE,OAAO;EACL;EACA,WAAW,SAAS;EACpB,gBAAgB,SAAS;EACzB,gBAAgB,SAAS;EACzB,WAAW,QAAQ,WAAW;EAC9B,kBAAkB;CACpB;AACF"}
1
+ {"version":3,"file":"pre-registration-D94b7Of5.js","names":[],"sources":["../src/pre-registration.ts"],"sourcesContent":["/**\n * Pre-registered hypotheses — declare what you're testing BEFORE the\n * run, check it AFTER. Prevents p-hacking, optional stopping, and the\n * \"we ran until it looked good\" failure mode.\n *\n * Manifest is a plain JSON-friendly object. Sign it with a content hash\n * + timestamp; the registered record becomes immutable. Post-run,\n * evaluate the manifest against observed results — the library refuses\n * to let you re-interpret a different metric as the declared one.\n *\n * A signed manifest is a portable record: it is written once and verified\n * later, possibly by a different release. `algo` names the digest scheme it\n * was signed under, and verification selects the encoder by that field, so a\n * manifest signed by an earlier release still verifies.\n */\n\nimport { createHash } from 'node:crypto'\nimport { canonicalString, hashCanonical } from './ledger-core/canonical'\n\nexport interface HypothesisManifest {\n id: string\n /** Human prose — goes into the audit trail. */\n hypothesis: string\n /** Metric the hypothesis claims to move. */\n metric: string\n /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */\n direction: 'increase' | 'decrease'\n /** Minimum effect size to count (same units as the metric). */\n minEffect: number\n /** Alpha threshold. */\n alpha: number\n /** Target statistical power at which sample size was pre-computed. */\n power: number\n /** Declared N per arm before running. */\n preRegisteredN: number\n /** ISO8601 timestamp the manifest was registered. */\n registeredAt: string\n /** Optional identifiers to tie into the trace corpus. */\n baselineLabel?: string\n candidateLabel?: string\n}\n\n/**\n * Identifier for the hashing scheme used to produce `contentHash`.\n *\n * Both schemes are sha256 hex over the manifest with `contentHash` and `algo`\n * stripped, and differ only in how that manifest is serialized:\n *\n * - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}\n * emits.\n * - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests\n * signed by an earlier release carry it, or carry no `algo` at all, and\n * {@link verifyManifest} still verifies them.\n */\nexport type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785'\n\nexport interface SignedManifest extends HypothesisManifest {\n /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */\n contentHash: string\n /**\n * Algorithm string describing how `contentHash` was produced.\n *\n * Optional on the type so serialized manifests without it still parse,\n * but ALWAYS populated by {@link signManifest}. Consumers that want to\n * enforce a known algorithm should reject manifests where this field\n * is missing or unrecognized.\n */\n algo?: SignedManifestAlgo\n}\n\nexport interface HypothesisResult {\n manifest: SignedManifest\n observedN: number\n observedEffect: number\n observedPValue: number\n /** True iff the observed effect hits the pre-declared direction with\n * magnitude ≥ minEffect AND p < alpha. */\n confirmed: boolean\n /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */\n rejectionReasons: Array<\n 'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'\n >\n notes?: string\n}\n\n/**\n * SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of\n * `obj` — the package's one identity scheme, shared with `ledger-core`.\n *\n * Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,\n * class instances, cycles — are refused rather than coerced, because a\n * coercion maps two distinct records onto one digest.\n *\n * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,\n * which takes a string input and returns a truncated 12-char prompt id.\n *\n * @example\n * const hash = await hashJson({ id: '1', kind: 'spec' })\n * // 'a3f1...' (64 hex chars)\n */\nexport async function hashJson<T>(obj: T): Promise<string> {\n return hashCanonical(obj).slice('sha256:'.length)\n}\n\n/**\n * Key-sorted `JSON.stringify` digest. Private and read-only: it exists so a\n * manifest signed under `'sha256-content'` still verifies, and nothing that\n * WRITES a digest may call it.\n */\nfunction legacyContentDigest(value: unknown): string {\n return createHash('sha256')\n .update(JSON.stringify(sortKeysDeep(value)), 'utf8')\n .digest('hex')\n}\n\nfunction sortKeysDeep(value: unknown): unknown {\n if (value === null || typeof value !== 'object') return value\n if (Array.isArray(value)) return value.map(sortKeysDeep)\n const out: Record<string, unknown> = {}\n for (const key of Object.keys(value as Record<string, unknown>).sort()) {\n out[key] = sortKeysDeep((value as Record<string, unknown>)[key])\n }\n return out\n}\n\n/**\n * Digest of a manifest under its own declared scheme, with `contentHash` and\n * `algo` stripped. Synchronous, so a caller that must fail before consuming an\n * observation does not have to await. Throws on an `algo` this release does\n * not know — an unverifiable manifest must not read as a valid one.\n */\nexport function manifestContentDigest(manifest: SignedManifest): string {\n const { contentHash: _contentHash, algo, ...rest } = manifest\n void _contentHash\n if (algo === undefined || algo === 'sha256-content') return legacyContentDigest(rest)\n if (algo === 'sha256-rfc8785') {\n return createHash('sha256').update(canonicalString(rest), 'utf8').digest('hex')\n }\n throw new Error(`pre-registration: unrecognized manifest hash algo '${String(algo)}'`)\n}\n\n/**\n * Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical\n * JSON, with `contentHash` and `algo` stripped, and stamp the scheme in\n * `algo` so a later reader knows which encoder to verify with.\n */\nexport async function signManifest(m: HypothesisManifest): Promise<SignedManifest> {\n const signed: SignedManifest = { ...m, contentHash: '', algo: 'sha256-rfc8785' }\n return { ...signed, contentHash: manifestContentDigest(signed) }\n}\n\n/**\n * Verify that a signed manifest has not been tampered with, under the scheme\n * the manifest itself declares.\n */\nexport async function verifyManifest(m: SignedManifest): Promise<boolean> {\n return manifestContentDigest(m) === m.contentHash\n}\n\n/**\n * Evaluate a pre-registered hypothesis against observed results.\n * Mechanical — no re-interpretation permitted.\n */\nexport async function evaluateHypothesis(\n manifest: SignedManifest,\n observed: { n: number; effect: number; pValue: number },\n): Promise<HypothesisResult> {\n if (!(await verifyManifest(manifest))) {\n throw new Error('evaluateHypothesis: manifest content hash mismatch (tampered)')\n }\n const reasons: HypothesisResult['rejectionReasons'] = []\n const directionOk = manifest.direction === 'increase' ? observed.effect > 0 : observed.effect < 0\n if (!directionOk) reasons.push('wrong_direction')\n if (Math.abs(observed.effect) < manifest.minEffect) reasons.push('effect_too_small')\n if (observed.pValue >= manifest.alpha) reasons.push('not_significant')\n if (observed.n < manifest.preRegisteredN) reasons.push('undersampled')\n return {\n manifest,\n observedN: observed.n,\n observedEffect: observed.effect,\n observedPValue: observed.pValue,\n confirmed: reasons.length === 0,\n rejectionReasons: reasons,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoGA,eAAsB,SAAY,KAAyB;CACzD,OAAO,cAAc,GAAG,CAAC,CAAC,MAAM,CAAgB;AAClD;;;;;;AAOA,SAAS,oBAAoB,OAAwB;CACnD,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAU,aAAa,KAAK,CAAC,GAAG,MAAM,CAAC,CACnD,OAAO,KAAK;AACjB;AAEA,SAAS,aAAa,OAAyB;CAC7C,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU,OAAO;CACxD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,IAAI,YAAY;CACvD,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,OAAO,OAAO,KAAK,KAAgC,CAAC,CAAC,KAAK,GACnE,IAAI,OAAO,aAAc,MAAkC,IAAI;CAEjE,OAAO;AACT;;;;;;;AAQA,SAAgB,sBAAsB,UAAkC;CACtE,MAAM,EAAE,aAAa,cAAc,MAAM,GAAG,SAAS;CAErD,IAAI,SAAS,KAAA,KAAa,SAAS,kBAAkB,OAAO,oBAAoB,IAAI;CACpF,IAAI,SAAS,kBACX,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,IAAI,GAAG,MAAM,CAAC,CAAC,OAAO,KAAK;CAEhF,MAAM,IAAI,MAAM,sDAAsD,OAAO,IAAI,EAAE,EAAE;AACvF;;;;;;AAOA,eAAsB,aAAa,GAAgD;CACjF,MAAM,SAAyB;EAAE,GAAG;EAAG,aAAa;EAAI,MAAM;CAAiB;CAC/E,OAAO;EAAE,GAAG;EAAQ,aAAa,sBAAsB,MAAM;CAAE;AACjE;;;;;AAMA,eAAsB,eAAe,GAAqC;CACxE,OAAO,sBAAsB,CAAC,MAAM,EAAE;AACxC;;;;;AAMA,eAAsB,mBACpB,UACA,UAC2B;CAC3B,IAAI,CAAE,MAAM,eAAe,QAAQ,GACjC,MAAM,IAAI,MAAM,+DAA+D;CAEjF,MAAM,UAAgD,CAAC;CAEvD,IAAI,EADgB,SAAS,cAAc,aAAa,SAAS,SAAS,IAAI,SAAS,SAAS,IAC9E,QAAQ,KAAK,iBAAiB;CAChD,IAAI,KAAK,IAAI,SAAS,MAAM,IAAI,SAAS,WAAW,QAAQ,KAAK,kBAAkB;CACnF,IAAI,SAAS,UAAU,SAAS,OAAO,QAAQ,KAAK,iBAAiB;CACrE,IAAI,SAAS,IAAI,SAAS,gBAAgB,QAAQ,KAAK,cAAc;CACrE,OAAO;EACL;EACA,WAAW,SAAS;EACpB,gBAAgB,SAAS;EACzB,gBAAgB,SAAS;EACzB,WAAW,QAAQ,WAAW;EAC9B,kBAAkB;CACpB;AACF"}
@@ -1,4 +1,4 @@
1
- import { a as RunRecord } from "./run-record-VVy4T9OW.js";
1
+ import { a as RunRecord } from "./run-record-DQjRcYwA.js";
2
2
  import { d as PairedBootstrapResult, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
3
3
  //#region src/statistics/paired-binary.d.ts
4
4
  /** A binomial proportion estimate with a confidence interval. */
@@ -589,4 +589,4 @@ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
589
589
  }): Promise<HypothesisResult>;
590
590
  //#endregion
591
591
  export { ProportionInterval as A, wilson as B, EProcess as C, eProcess as D, EProcessStep as E, pairedBinaryScale as F, pairedRiskDifference as I, pairedRiskDifferenceExact as L, ScoreRiskDifferenceResult as M, isBinaryOutcomeVector as N, ExactRiskDifferenceResult as O, mcnemar as P, pairedRiskDifferenceScore as R, pairRunRecords as S, EProcessState as T, PairedArmsComparison as _, evaluateHypothesis as a, comparePairedArms as b, signManifest as c, MatchedPair as d, MatchedRunRecordPair as f, PairedArmRow as g, PairRunRecordsResult as h, SignedManifestAlgo as i, RiskDifferenceResult as j, McNemarResult as k, verifyManifest as l, PairArmsResult as m, HypothesisResult as n, hashJson as o, PairArmsOptions as p, SignedManifest as r, manifestContentDigest as s, HypothesisManifest as t, ComparePairedArmsOptions as u, PairedCorrectness as v, EProcessOptions as w, pairArms as x, PairedMetricDelta as y, passAtK as z };
592
- //# sourceMappingURL=pre-registration-CzFCcwYk.d.ts.map
592
+ //# sourceMappingURL=pre-registration-DHz6P_6f.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"pre-registration-CzFCcwYk.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;;;;;;;EAQA,SAAS;;UAGM;;EAEf;;EAEA;;EAEA;;;;;;;;;;UAWe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UCpDrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;;;;;;UClWc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;;;;;KAeU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;;;;;;;iBAkBoB,SAAS,GAAG,KAAK,IAAI;;;;;;;iBA+B3B,sBAAsB,UAAU;;;;;;iBAe1B,aAAa,GAAG,qBAAqB,QAAQ;;;;;iBAS7C,eAAe,GAAG,iBAAiB;;;;;iBAQnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}
1
+ {"version":3,"file":"pre-registration-DHz6P_6f.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;;;;;;;EAQA,SAAS;;UAGM;;EAEf;;EAEA;;EAEA;;;;;;;;;;UAWe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UCpDrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;;;;;;UClWc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;;;;;KAeU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;;;;;;;iBAkBoB,SAAS,GAAG,KAAK,IAAI;;;;;;;iBA+B3B,sBAAsB,UAAU;;;;;;iBAe1B,aAAa,GAAG,qBAAqB,QAAQ;;;;;iBAS7C,eAAe,GAAG,iBAAiB;;;;;iBAQnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}
@@ -1,8 +1,8 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { r as canonicalString } from "./canonical-IL-Bu-14.js";
3
- import { K as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-Du7WQPh7.js";
2
+ import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
+ import { J as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-DbJdo8Nj.js";
4
4
  import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
5
- import { t as certificationEvidenceDigest } from "./verdict-B0xltqu6.js";
5
+ import { t as certificationEvidenceDigest } from "./verdict-BQ3pCFf8.js";
6
6
  import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-BFMRpmqb.js";
7
7
  import { createHash, randomUUID } from "node:crypto";
8
8
  import { harnessSupportsModel } from "@tangle-network/agent-interface";
@@ -583,4 +583,4 @@ function extractProducedState(events) {
583
583
  //#endregion
584
584
  export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
585
585
 
586
- //# sourceMappingURL=produced-state-Be0BK3RN.js.map
586
+ //# sourceMappingURL=produced-state-CtSIp5cQ.js.map