@tangle-network/agent-eval 0.163.2 → 0.171.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (249) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/README.md +2 -0
  3. package/dist/adapters/http.d.ts +108 -0
  4. package/dist/adapters/http.d.ts.map +1 -0
  5. package/dist/adapters/http.js +208 -0
  6. package/dist/adapters/http.js.map +1 -0
  7. package/dist/analyst/index.d.ts +40 -70
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +18 -311
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
  12. package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
  13. package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
  14. package/dist/benchmark-C4wk_Sjr.js.map +1 -0
  15. package/dist/{benchmark-command-CF-4GEWZ.js → benchmark-command-D8k3Gf0J.js} +236 -251
  16. package/dist/benchmark-command-D8k3Gf0J.js.map +1 -0
  17. package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
  18. package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
  19. package/dist/benchmarks/index.d.ts +5 -5
  20. package/dist/benchmarks/index.js +3 -3
  21. package/dist/builder-eval/index.d.ts +3 -3
  22. package/dist/builder-eval/index.js +1 -1
  23. package/dist/campaign/index.d.ts +8 -8
  24. package/dist/campaign/index.js +7 -7
  25. package/dist/{campaign-DQZmc2Dq.js → campaign-B72njjHj.js} +15 -14
  26. package/dist/campaign-B72njjHj.js.map +1 -0
  27. package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
  28. package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
  29. package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
  30. package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
  31. package/dist/cli.js +2 -2
  32. package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
  33. package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
  34. package/dist/{client-L9VVPkim.d.ts → client-CDtcZ3p9.d.ts} +4 -4
  35. package/dist/{client-L9VVPkim.d.ts.map → client-CDtcZ3p9.d.ts.map} +1 -1
  36. package/dist/contract/index.d.ts +12 -703
  37. package/dist/contract/index.js +11 -11
  38. package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
  39. package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
  40. package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-B0s2zU-s.d.ts} +6 -6
  41. package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-B0s2zU-s.d.ts.map} +1 -1
  42. package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-Cjy2yhqP.d.ts} +7 -7
  43. package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-Cjy2yhqP.d.ts.map} +1 -1
  44. package/dist/{define-agent-eval-D08pWIJb.js → define-agent-eval-Dy8QgxAI.js} +20 -8
  45. package/dist/{define-agent-eval-D08pWIJb.js.map → define-agent-eval-Dy8QgxAI.js.map} +1 -1
  46. package/dist/{dspy-rlm-engine-DhA9qKIm.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
  47. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
  48. package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
  49. package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
  50. package/dist/{engine-Cu5qD5Fc.d.ts → engine-CAmTUk52.d.ts} +7 -7
  51. package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-CAmTUk52.d.ts.map} +1 -1
  52. package/dist/{eval-campaign-BfohKmzx.js → eval-campaign-JDTeE6Pl.js} +4 -4
  53. package/dist/{eval-campaign-BfohKmzx.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
  54. package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
  55. package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
  56. package/dist/experiment/index.d.ts +5 -5
  57. package/dist/experiment/index.js +4 -4
  58. package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
  59. package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
  60. package/dist/{external-optimizer-process-BFmh36vW.js → external-optimizer-process-CQxylYeG.js} +4 -11
  61. package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
  62. package/dist/{external-optimizer-subprocess-CqLMW3nh.js → external-optimizer-subprocess-Cex8Da2i.js} +25 -11
  63. package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
  64. package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
  65. package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
  66. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
  67. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
  68. package/dist/fuzz.d.ts +1 -1
  69. package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Dh2b62w8.d.ts} +111 -111
  70. package/dist/heldout-gate-Dh2b62w8.d.ts.map +1 -0
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/{index-D-V8gCs_.d.ts → index-8VIogTyS.d.ts} +30 -22
  74. package/dist/index-8VIogTyS.d.ts.map +1 -0
  75. package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
  76. package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
  77. package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
  78. package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
  79. package/dist/{index-CGtH1piv.d.ts → index-DT73JraI.d.ts} +7 -38
  80. package/dist/index-DT73JraI.d.ts.map +1 -0
  81. package/dist/index-fNXZMCzX.d.ts +704 -0
  82. package/dist/index-fNXZMCzX.d.ts.map +1 -0
  83. package/dist/index.d.ts +197 -38
  84. package/dist/index.d.ts.map +1 -1
  85. package/dist/index.js +736 -28
  86. package/dist/index.js.map +1 -1
  87. package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
  88. package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
  89. package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
  90. package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
  91. package/dist/internal-BMFSR8Ns.js.map +1 -1
  92. package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
  93. package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
  94. package/dist/ledger-core/index.js +2 -2
  95. package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
  96. package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
  97. package/dist/{llm-judge-Du7WQPh7.js → llm-judge-BtJ2Sfk_.js} +101 -20
  98. package/dist/llm-judge-BtJ2Sfk_.js.map +1 -0
  99. package/dist/matrix/index.d.ts +2 -2
  100. package/dist/{matrix-eXKRMHnL.d.ts → matrix-DiHmUobV.d.ts} +3 -3
  101. package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-DiHmUobV.d.ts.map} +1 -1
  102. package/dist/meta-eval/index.d.ts +3 -3
  103. package/dist/meta-eval/index.js +1 -1
  104. package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
  105. package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
  106. package/dist/multishot/golden/index.d.ts +1 -1
  107. package/dist/multishot/index.d.ts +2 -2
  108. package/dist/openapi.json +4 -4
  109. package/dist/pipelines/index.d.ts +5 -5
  110. package/dist/pipelines/index.js +3 -3
  111. package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
  112. package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
  113. package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
  114. package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
  115. package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
  116. package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
  117. package/dist/{produced-state-Be0BK3RN.js → produced-state-Cm6DU_Ao.js} +4 -4
  118. package/dist/{produced-state-Be0BK3RN.js.map → produced-state-Cm6DU_Ao.js.map} +1 -1
  119. package/dist/profile-cell.js +1 -1
  120. package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-WSXtBgBb.d.ts} +3 -3
  121. package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-WSXtBgBb.d.ts.map} +1 -1
  122. package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
  123. package/dist/proposal-findings-bko3GGy-.js.map +1 -0
  124. package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CafMdZKM.d.ts} +895 -891
  125. package/dist/provenance-CafMdZKM.d.ts.map +1 -0
  126. package/dist/{query-_5g6re3_.js → query-BPGMVlbM.js} +3 -3
  127. package/dist/{query-_5g6re3_.js.map → query-BPGMVlbM.js.map} +1 -1
  128. package/dist/{query-CwnHlu5p.d.ts → query-Na5gEIGd.d.ts} +3 -3
  129. package/dist/{query-CwnHlu5p.d.ts.map → query-Na5gEIGd.d.ts.map} +1 -1
  130. package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
  131. package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
  132. package/dist/{release-confidence-nGDJiiwc.js → release-confidence-CzUHc4z4.js} +3 -3
  133. package/dist/{release-confidence-nGDJiiwc.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
  134. package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
  135. package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
  136. package/dist/reporting.d.ts +3 -3
  137. package/dist/reporting.js +2 -2
  138. package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
  139. package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
  140. package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
  141. package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-O5zKDANP.js → reward-hacking-SkxYgT0x.js} +2 -2
  143. package/dist/{reward-hacking-O5zKDANP.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
  144. package/dist/rl.d.ts +8 -8
  145. package/dist/rl.d.ts.map +1 -1
  146. package/dist/rl.js +6 -5
  147. package/dist/rl.js.map +1 -1
  148. package/dist/rollout/index.d.ts +1 -1
  149. package/dist/rollout/index.js +2 -2
  150. package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
  151. package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
  152. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
  153. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
  154. package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
  155. package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
  156. package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
  157. package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
  158. package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
  159. package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
  160. package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
  161. package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
  162. package/dist/{semantic-concept-judge-BsDMOwJr.js → semantic-concept-judge-I36eejJx.js} +2 -2
  163. package/dist/{semantic-concept-judge-BsDMOwJr.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
  164. package/dist/{sequential-BLMbdrD7.js → sequential-B51qAYE4.js} +2 -2
  165. package/dist/{sequential-BLMbdrD7.js.map → sequential-B51qAYE4.js.map} +1 -1
  166. package/dist/{server-BjYiJHoJ.js → server-CCEnywOR.js} +26 -18
  167. package/dist/server-CCEnywOR.js.map +1 -0
  168. package/dist/{skillopt-optimization-method-UArRo-nr.js → skillopt-optimization-method-DzlF2RM7.js} +9 -9
  169. package/dist/{skillopt-optimization-method-UArRo-nr.js.map → skillopt-optimization-method-DzlF2RM7.js.map} +1 -1
  170. package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-UhiexnjU.d.ts} +3 -3
  171. package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-UhiexnjU.d.ts.map} +1 -1
  172. package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
  173. package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
  174. package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
  175. package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
  176. package/dist/{store-tool-spans-BVga3c37.js → store-tool-spans-B9o6tU8f.js} +3 -3
  177. package/dist/{store-tool-spans-BVga3c37.js.map → store-tool-spans-B9o6tU8f.js.map} +1 -1
  178. package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-D_qMl2__.d.ts} +102 -36
  179. package/dist/store-tool-spans-D_qMl2__.d.ts.map +1 -0
  180. package/dist/storyboard/index.d.ts +1 -1
  181. package/dist/{summary-report-BXeQ5Ues.js → summary-report-Bgh8CpNK.js} +2 -2
  182. package/dist/{summary-report-BXeQ5Ues.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
  183. package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
  184. package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
  185. package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
  186. package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
  187. package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-RGYfVWpc.d.ts} +3 -3
  188. package/dist/tool-groups-RGYfVWpc.d.ts.map +1 -0
  189. package/dist/{tool-waste-CKc7bYIg.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
  190. package/dist/{tool-waste-CKc7bYIg.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
  191. package/dist/{tool-waste-8BQiUc8K.js → tool-waste-CwGHzBzX.js} +2 -2
  192. package/dist/{tool-waste-8BQiUc8K.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
  193. package/dist/trace-repair/index.d.ts +3 -3
  194. package/dist/trace-repair/index.d.ts.map +1 -1
  195. package/dist/trace-repair/index.js +4 -3
  196. package/dist/trace-repair/index.js.map +1 -1
  197. package/dist/traces.d.ts +59 -61
  198. package/dist/traces.d.ts.map +1 -1
  199. package/dist/traces.js +7 -7
  200. package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
  201. package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
  202. package/dist/trajectory-replay/index.d.ts +3 -3
  203. package/dist/trajectory-replay/index.js +1 -1
  204. package/dist/{types-BPb2Kf_C.d.ts → types-CMyW4GnH.d.ts} +3 -3
  205. package/dist/{types-BPb2Kf_C.d.ts.map → types-CMyW4GnH.d.ts.map} +1 -1
  206. package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
  207. package/dist/types-CiWITkGo.js.map +1 -0
  208. package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
  209. package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
  210. package/dist/{types-D4s7Z6nq.d.ts → types-JHMOqZI4.d.ts} +13 -3
  211. package/dist/types-JHMOqZI4.d.ts.map +1 -0
  212. package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
  213. package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
  214. package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
  215. package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
  216. package/dist/wire/index.d.ts +26 -11
  217. package/dist/wire/index.d.ts.map +1 -1
  218. package/dist/wire/index.js +2 -2
  219. package/docs/campaign-proposers.md +4 -0
  220. package/docs/code-agent-intake.md +64 -0
  221. package/docs/concepts.md +1 -1
  222. package/docs/design/statistics-decisions.md +1 -1
  223. package/docs/distributed-driver.md +3 -6
  224. package/docs/public-api.md +122 -106
  225. package/docs/wire-protocol.md +5 -3
  226. package/package.json +27 -19
  227. package/dist/benchmark-BhT16ep9.js.map +0 -1
  228. package/dist/benchmark-command-CF-4GEWZ.js.map +0 -1
  229. package/dist/campaign-DQZmc2Dq.js.map +0 -1
  230. package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
  231. package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
  232. package/dist/contract/index.d.ts.map +0 -1
  233. package/dist/dspy-rlm-engine-DhA9qKIm.js.map +0 -1
  234. package/dist/external-optimizer-process-BFmh36vW.js.map +0 -1
  235. package/dist/external-optimizer-subprocess-CqLMW3nh.js.map +0 -1
  236. package/dist/index-CGtH1piv.d.ts.map +0 -1
  237. package/dist/index-D-V8gCs_.d.ts.map +0 -1
  238. package/dist/index-vrJugRal.d.ts +0 -1
  239. package/dist/kind-factory-DY8FdoXf.js.map +0 -1
  240. package/dist/llm-judge-Du7WQPh7.js.map +0 -1
  241. package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
  242. package/dist/run-score-lDzV0X8j.js.map +0 -1
  243. package/dist/server-BjYiJHoJ.js.map +0 -1
  244. package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
  245. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
  246. package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
  247. package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
  248. package/dist/types-BI4fT3HN.js.map +0 -1
  249. package/dist/types-D4s7Z6nq.d.ts.map +0 -1
@@ -0,0 +1,704 @@
1
+ import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
2
+ import { a as RunRecord, n as RunCostProvenance, s as RunSplitTag } from "./run-record-DQjRcYwA.js";
3
+ import "./types-DMoNFDWi.js";
4
+ import "./provenance-CafMdZKM.js";
5
+ import { _ as GateDecision } from "./types-JHMOqZI4.js";
6
+ import "./heldout-gate-Dh2b62w8.js";
7
+ import "./promotion-policy-WSXtBgBb.js";
8
+ import { o as InsightReport } from "./insight-report-08F022xN.js";
9
+ import { c as EvalRunGenerationSnapshot, g as TraceSpanEvent, o as EvalRunCellScore, s as EvalRunEvent } from "./client-CDtcZ3p9.js";
10
+ import "./define-agent-eval-Cjy2yhqP.js";
11
+ import "./default-registry-B0s2zU-s.js";
12
+ import { _ as AnalyzeRunsOptions } from "./engine-CAmTUk52.js";
13
+ import { AgentCandidateBenchmarkCellRef, AgentCandidateBenchmarkSuiteInputs, AgentCandidateBenchmarkTask, AgentCandidateBenchmarkTaskMaterial, AgentCandidateBundle, AgentCandidateEvaluationPolicy, AgentCandidateExperiment, AgentCandidateExperimentMaterial, AgentCandidateExperimentMeasurement, AgentImprovementCost, AgentImprovementMeasuredComparison, AgentProfileImprovementExperiment, AgentProfileImprovementExperimentMaterial, AgentProfileImprovementMeasuredComparison, AgentProfileImprovementMeasurement, AgentProfileImprovementRunCell, AgentProfileImprovementRunReceipt, AgentProfileImprovementSuiteInputs, AgentProfileImprovementTask, AgentProfileImprovementTaskMaterial, CandidateExecutionEvidence, Sha256Digest } from "@tangle-network/agent-interface";
14
+ //#region src/contract/measured-comparison.d.ts
15
+ interface SealCandidateBenchmarkSuiteOptions {
16
+ tasks: [AgentCandidateBenchmarkTask, ...AgentCandidateBenchmarkTask[]];
17
+ reps: number;
18
+ seeds: [number, ...number[]];
19
+ }
20
+ interface CandidateExperimentExecutionInput {
21
+ experiment: AgentCandidateExperiment;
22
+ arm: 'baseline' | 'candidate';
23
+ bundle: AgentCandidateBundle;
24
+ task: AgentCandidateBenchmarkTask;
25
+ benchmarkCell: AgentCandidateBenchmarkCellRef;
26
+ seed: number;
27
+ signal?: AbortSignal;
28
+ }
29
+ interface RunCandidateExperimentOptions {
30
+ experiment: AgentCandidateExperiment;
31
+ execute(input: CandidateExperimentExecutionInput): Promise<CandidateExecutionEvidence>;
32
+ /** Maximum number of simultaneous execute calls across both arms. */
33
+ maxConcurrency?: number;
34
+ /** Shared run budget that also accounts for analysis and candidate search. */
35
+ costLedger?: CostLedgerHandle;
36
+ signal?: AbortSignal;
37
+ }
38
+ interface CandidateExperimentRun {
39
+ measurements: AgentCandidateExperimentMeasurement[];
40
+ measurement: {
41
+ wallDurationMs: number;
42
+ cost: AgentImprovementCost;
43
+ };
44
+ }
45
+ interface CompareCandidateExperimentOptions {
46
+ experiment: AgentCandidateExperiment;
47
+ measurements: AgentCandidateExperimentMeasurement[];
48
+ preparation: {
49
+ wallDurationMs: number;
50
+ cost: AgentImprovementCost;
51
+ };
52
+ measurement: CandidateExperimentRun['measurement'];
53
+ runId: string;
54
+ candidate?: AgentImprovementMeasuredComparison['candidate'];
55
+ generationsExplored?: number;
56
+ metadata?: AgentImprovementMeasuredComparison['metadata'];
57
+ }
58
+ /** One exact baseline/candidate observation of the same held-out cell. */
59
+ interface PairedMeasurement<TRun> {
60
+ cellId: string;
61
+ baseline: TRun;
62
+ candidate: TRun;
63
+ }
64
+ /** Maps a product-owned run receipt into the measurements required for a fair paired decision. */
65
+ interface PairedMeasurementAdapter<TRun> {
66
+ score(run: TRun): number;
67
+ dimensions(run: TRun): readonly {
68
+ name: string;
69
+ score: number;
70
+ }[];
71
+ costUsd(run: TRun): number;
72
+ costProvenance(run: TRun): AgentImprovementCost['provenance'];
73
+ latencyMs(run: TRun): number;
74
+ completed(run: TRun): boolean;
75
+ passed(run: TRun): boolean;
76
+ }
77
+ interface EvaluatePairedMeasurementsOptions<TRun> {
78
+ measurements: readonly PairedMeasurement<TRun>[];
79
+ policy: AgentCandidateEvaluationPolicy;
80
+ adapter: PairedMeasurementAdapter<TRun>;
81
+ /** Whether both arms use the same scorer family as the promotion decision. */
82
+ sharedScorerChannel: boolean;
83
+ /** Analysis and candidate-search spend that belongs to the same frozen budget. */
84
+ preparationCost?: AgentImprovementCost;
85
+ /** Settled aggregate receipt for this exact paired suite, when an executor provides one. */
86
+ measurementCost?: AgentImprovementCost;
87
+ }
88
+ /** Statistical and operational result derived from complete paired receipts. */
89
+ type PairedMeasurementEvaluation = Pick<AgentImprovementMeasuredComparison, 'overall' | 'objectives' | 'decision' | 'power'> & {
90
+ measurementCost: AgentImprovementCost;
91
+ totalCost: AgentImprovementCost;
92
+ measurementWorkDurationMs: number;
93
+ };
94
+ /** Content-address one task before any measured execution can see it. */
95
+ declare function sealCandidateBenchmarkTask(material: AgentCandidateBenchmarkTaskMaterial): AgentCandidateBenchmarkTask;
96
+ /** Freeze task order, repetitions, and every seed before either arm runs. */
97
+ declare function sealCandidateBenchmarkSuite(options: SealCandidateBenchmarkSuiteOptions): AgentCandidateBenchmarkSuiteInputs;
98
+ /** Freeze both complete agent states and their exact held-out work. */
99
+ declare function sealCandidateExperiment(material: AgentCandidateExperimentMaterial): AgentCandidateExperiment;
100
+ declare function verifyCandidateExperiment(input: unknown): AgentCandidateExperiment;
101
+ /** Execute each signed cell for both arms. The callback is Runtime's one executor. */
102
+ declare function runCandidateExperiment(options: RunCandidateExperimentOptions): Promise<CandidateExperimentRun>;
103
+ /**
104
+ * Calculate the shared paired decision from any complete receipt shape.
105
+ *
106
+ * Callers still own sealing their tasks, verifying each receipt against its
107
+ * expected arm and state, and proving every expected cell exists. This function
108
+ * only validates the projected measurements and derives their shared decision.
109
+ */
110
+ declare function evaluatePairedMeasurements<TRun>(options: EvaluatePairedMeasurementsOptions<TRun>): PairedMeasurementEvaluation;
111
+ /** Build the only publishable comparison: paired statistics over Runtime receipts. */
112
+ declare function measuredComparisonFromCandidateExperiment(options: CompareCandidateExperimentOptions): AgentImprovementMeasuredComparison;
113
+ /** Recompute every statistic and decision from the signed experiment receipts. */
114
+ declare function verifyCandidateExperimentComparison(input: unknown): AgentImprovementMeasuredComparison;
115
+ //#endregion
116
+ //#region src/contract/profile-measured-comparison.d.ts
117
+ interface SealAgentProfileImprovementSuiteOptions {
118
+ splitDigest: Sha256Digest;
119
+ tasks: [AgentProfileImprovementTask, ...AgentProfileImprovementTask[]];
120
+ reps: number;
121
+ seeds: [number, ...number[]];
122
+ }
123
+ interface AgentProfileImprovementExperimentExecutionInput {
124
+ experiment: AgentProfileImprovementExperiment;
125
+ arm: 'baseline' | 'candidate';
126
+ stateDigest: Sha256Digest;
127
+ task: AgentProfileImprovementTask;
128
+ runCell: AgentProfileImprovementRunCell;
129
+ seed: number;
130
+ signal?: AbortSignal;
131
+ }
132
+ interface RunAgentProfileImprovementExperimentOptions {
133
+ experiment: AgentProfileImprovementExperiment;
134
+ execute(input: AgentProfileImprovementExperimentExecutionInput): Promise<AgentProfileImprovementRunReceipt>;
135
+ /** Maximum number of simultaneous execute calls across both arms. */
136
+ maxConcurrency?: number;
137
+ /** Shared run budget that also accounts for analysis and candidate search. */
138
+ costLedger?: CostLedgerHandle;
139
+ signal?: AbortSignal;
140
+ }
141
+ interface AgentProfileImprovementExperimentRun {
142
+ measurements: AgentProfileImprovementMeasurement[];
143
+ measurement: {
144
+ wallDurationMs: number;
145
+ cost: AgentImprovementCost;
146
+ };
147
+ }
148
+ interface CompareAgentProfileImprovementExperimentOptions {
149
+ experiment: AgentProfileImprovementExperiment;
150
+ measurements: AgentProfileImprovementMeasurement[];
151
+ preparation: {
152
+ wallDurationMs: number;
153
+ cost: AgentImprovementCost;
154
+ };
155
+ measurement: AgentProfileImprovementExperimentRun['measurement'];
156
+ runId: string;
157
+ candidate?: AgentProfileImprovementMeasuredComparison['candidate'];
158
+ generationsExplored?: number;
159
+ metadata?: AgentProfileImprovementMeasuredComparison['metadata'];
160
+ }
161
+ /** Content-address one held-out profile task before either state can execute it. */
162
+ declare function sealAgentProfileImprovementTask(material: AgentProfileImprovementTaskMaterial): AgentProfileImprovementTask;
163
+ /** Freeze profile task order, repetitions, seeds, and the held-out split. */
164
+ declare function sealAgentProfileImprovementSuite(options: SealAgentProfileImprovementSuiteOptions): AgentProfileImprovementSuiteInputs;
165
+ /** Freeze the two host-owned profile states and their exact held-out work. */
166
+ declare function sealAgentProfileImprovementExperiment(material: AgentProfileImprovementExperimentMaterial): AgentProfileImprovementExperiment;
167
+ /**
168
+ * Execute each signed profile cell through the host's one exact-state executor.
169
+ * Eval owns only the cell schedule and receipt checks; the host resolves each
170
+ * state digest and captures its own run, billing, trace, and grader evidence.
171
+ */
172
+ declare function runAgentProfileImprovementExperiment(options: RunAgentProfileImprovementExperimentOptions): Promise<AgentProfileImprovementExperimentRun>;
173
+ /** Build the only publishable profile comparison from complete host receipts. */
174
+ declare function measuredComparisonFromAgentProfileImprovementExperiment(options: CompareAgentProfileImprovementExperimentOptions): AgentProfileImprovementMeasuredComparison;
175
+ /** Recompute a profile comparison from the exact sealed experiment and receipts. */
176
+ declare function verifyAgentProfileImprovementExperimentComparison(input: unknown): AgentProfileImprovementMeasuredComparison;
177
+ //#endregion
178
+ //#region src/contract/intake/run-record-dir.d.ts
179
+ /** A record that failed boundary validation, with enough context to fix it. */
180
+ interface RunRecordRejection {
181
+ /** Absolute or caller-relative path to the file the record came from. */
182
+ file: string;
183
+ /** Zero-based position within the file (array index or JSONL line number). */
184
+ index: number;
185
+ /** The validator's message. */
186
+ reason: string;
187
+ }
188
+ interface FromRunRecordDirOptions {
189
+ /**
190
+ * How to treat a record that fails `parseRunRecordSafe`:
191
+ * - `'throw'` (default) — fail loud on the first invalid record.
192
+ * - `'collect'` — drop it, keep the rest, and return it under `rejected`.
193
+ */
194
+ onInvalid?: 'throw' | 'collect';
195
+ /**
196
+ * When the input is a directory, only files matching this predicate are
197
+ * read. Default: any file ending in `.json` or `.jsonl`. The `analysis.json`
198
+ * artifact `evalReportingSuite` writes is always skipped so a re-run never
199
+ * ingests its own output.
200
+ */
201
+ include?: (fileName: string) => boolean;
202
+ /**
203
+ * Recurse into subdirectories when the input is a directory. Default false —
204
+ * a flat run directory is the common case and recursion can silently pull in
205
+ * unrelated corpora.
206
+ */
207
+ recursive?: boolean;
208
+ }
209
+ interface FromRunRecordDirResult {
210
+ /** Records that passed boundary validation, in file-then-index order. */
211
+ runs: RunRecord[];
212
+ /** Records that failed validation. Empty unless `onInvalid: 'collect'`. */
213
+ rejected: RunRecordRejection[];
214
+ /** The files that were read, in the order they were processed. */
215
+ files: string[];
216
+ }
217
+ /**
218
+ * Resolve a file or directory path into validated `RunRecord[]`.
219
+ *
220
+ * A `.json` file must parse to a top-level array; a `.jsonl` file is one
221
+ * record per non-empty line. Directories are read shallowly by default
222
+ * (set `recursive` to descend); the `analysis.json` output artifact is
223
+ * always excluded.
224
+ */
225
+ declare function fromRunRecordDir(path: string, options?: FromRunRecordDirOptions): Promise<FromRunRecordDirResult>;
226
+ //#endregion
227
+ //#region src/contract/eval-reporting-suite.d.ts
228
+ /** Either records in hand or a path to a `.json` / `.jsonl` file or a
229
+ * directory of them. */
230
+ type EvalReportingSuiteInput = RunRecord[] | string;
231
+ interface EvalReportingSuiteOptions {
232
+ /** Forwarded verbatim to `analyzeRuns` (everything except `runs`, which the
233
+ * suite supplies from the resolved input). Use this for split selection,
234
+ * baseline/candidate ids, canaries, prior-period runs, the analyst registry,
235
+ * etc. */
236
+ analyze?: Omit<AnalyzeRunsOptions, 'runs'>;
237
+ /** Loader options used only when the input is a path. */
238
+ load?: FromRunRecordDirOptions;
239
+ /**
240
+ * Write the suite result as a single `analysis.json`.
241
+ * - `true` — write to `<dir>/analysis.json` when the input is a directory,
242
+ * or alongside the input file; throws if the input is in-memory records
243
+ * (no directory to anchor to — pass an explicit path instead).
244
+ * - a string — write to exactly this path (a directory path gets
245
+ * `analysis.json` appended; any other path is used verbatim).
246
+ * - omitted / false — do not write.
247
+ */
248
+ write?: boolean | string;
249
+ }
250
+ /** The suite artifact — the `analyzeRuns` report plus provenance. This is the
251
+ * exact shape serialized to `analysis.json`. */
252
+ interface EvalReportingSuiteResult {
253
+ /** The analysis itself — distributions, paired stats/lift, failure rollup,
254
+ * recommendations. Produced by `analyzeRuns`. */
255
+ report: InsightReport;
256
+ /** How the suite was run, so a reader can verify provenance. */
257
+ provenance: {
258
+ /** ISO timestamp the suite ran. */
259
+ generatedAt: string;
260
+ /** Number of records analyzed (mirrors `report.n`). */
261
+ runCount: number;
262
+ /** The source path when the input was a directory/file; null for
263
+ * in-memory records. */
264
+ sourcePath: string | null;
265
+ /** Files read when loading from disk; empty for in-memory input. */
266
+ files: string[];
267
+ /** Records dropped at the validation boundary. Always empty unless
268
+ * `load.onInvalid` was set to `'collect'`. */
269
+ rejected: FromRunRecordDirResult['rejected'];
270
+ };
271
+ /** The path `analysis.json` was written to, or null when `write` was unset. */
272
+ writtenTo: string | null;
273
+ }
274
+ /**
275
+ * Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
276
+ * single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
277
+ * function is composition + I/O.
278
+ */
279
+ declare function evalReportingSuite(input: EvalReportingSuiteInput, options?: EvalReportingSuiteOptions): Promise<EvalReportingSuiteResult>;
280
+ //#endregion
281
+ //#region src/contract/diff.d.ts
282
+ /** Per-dimension delta. `before` / `after` are null when the judge did not
283
+ * emit a value for that side. `delta` is `after - before`; null when
284
+ * either side is null. */
285
+ interface EvalDimensionDelta {
286
+ before: number | null;
287
+ after: number | null;
288
+ delta: number | null;
289
+ }
290
+ /** Per-cell delta, keyed on `(scenarioId, rep)`. */
291
+ interface EvalCellScoreDelta {
292
+ scenarioId: string;
293
+ rep: number;
294
+ compositeBefore: number | null;
295
+ compositeAfter: number | null;
296
+ compositeDelta: number | null;
297
+ /** Per-judge → per-dimension deltas. Outer key = judge name from
298
+ * `EvalRunCellScore.dimensions`; inner key = dimension name. */
299
+ dimensions: Record<string, Record<string, EvalDimensionDelta>>;
300
+ }
301
+ /** Diff between two generation snapshots — the unit the dashboard renders
302
+ * for a single "v3 vs v4" comparison. */
303
+ interface EvalGenerationDiff {
304
+ beforeIndex: number;
305
+ afterIndex: number;
306
+ beforeSurfaceHash: string;
307
+ afterSurfaceHash: string;
308
+ surfaceChanged: boolean;
309
+ /** Cells present in both snapshots, matched on `(scenarioId, rep)`. */
310
+ matched: EvalCellScoreDelta[];
311
+ /** Cells present in `before` but missing from `after`. */
312
+ removed: EvalRunCellScore[];
313
+ /** Cells present in `after` but missing from `before`. */
314
+ added: EvalRunCellScore[];
315
+ /** Aggregate composite mean, null when that snapshot was unscored. */
316
+ compositeBefore: number | null;
317
+ compositeAfter: number | null;
318
+ compositeDelta: number | null;
319
+ costUsdBefore: number;
320
+ costUsdAfter: number;
321
+ costUsdDelta: number;
322
+ durationMsBefore: number;
323
+ durationMsAfter: number;
324
+ durationMsDelta: number;
325
+ }
326
+ /** Diff between two full eval-runs. Includes both baseline-vs-baseline and
327
+ * winner-vs-winner generation diffs when both sides expose them, plus
328
+ * run-level metadata. */
329
+ interface EvalRunDiff {
330
+ beforeRunId: string;
331
+ afterRunId: string;
332
+ beforeTimestamp: string;
333
+ afterTimestamp: string;
334
+ beforeGateDecision: GateDecision | null;
335
+ afterGateDecision: GateDecision | null;
336
+ beforeHoldoutLift: number | null;
337
+ afterHoldoutLift: number | null;
338
+ holdoutLiftDelta: number | null;
339
+ beforeTotalCostUsd: number;
340
+ afterTotalCostUsd: number;
341
+ totalCostUsdDelta: number;
342
+ beforeTotalDurationMs: number;
343
+ afterTotalDurationMs: number;
344
+ totalDurationMsDelta: number;
345
+ /** Baseline-vs-baseline diff. Null when either run has no baseline. */
346
+ baselineDiff: EvalGenerationDiff | null;
347
+ /** Highest-index-generation comparison. Null when either run has no
348
+ * recorded generations (e.g. baseline-only or errored before any
349
+ * generation completed). */
350
+ winnersDiff: EvalGenerationDiff | null;
351
+ }
352
+ /**
353
+ * Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
354
+ * unmatched cells surface in `added` / `removed`. Aggregate fields are
355
+ * recomputed from the snapshot's stored fields, not re-derived from cells —
356
+ * this keeps the diff consistent with whatever aggregation the substrate
357
+ * actually reported.
358
+ */
359
+ declare function diffGenerations(before: EvalRunGenerationSnapshot, after: EvalRunGenerationSnapshot): EvalGenerationDiff;
360
+ /**
361
+ * Diff two full eval-runs. Produces baseline-vs-baseline and
362
+ * winner-vs-winner generation diffs when both sides expose them, plus
363
+ * run-level cost / lift / gate-decision deltas.
364
+ */
365
+ declare function diffRuns(before: EvalRunEvent, after: EvalRunEvent): EvalRunDiff;
366
+ /**
367
+ * Within-run baseline → winning-generation diff. The natural "what did the
368
+ * improvement loop produce" view for a single run. Returns null when the
369
+ * run never reached a generation past baseline (errored early, or the gate
370
+ * shipped the baseline as-is).
371
+ */
372
+ declare function diffRunBaselineToWinner(run: EvalRunEvent): EvalGenerationDiff | null;
373
+ //#endregion
374
+ //#region src/contract/intake/agent-trace.d.ts
375
+ type AgentTraceContributorType = 'human' | 'ai' | 'mixed' | 'unknown';
376
+ interface AgentTraceContributor {
377
+ type: AgentTraceContributorType;
378
+ /** models.dev id, e.g. `anthropic/claude-opus-4-5-20251101`. */
379
+ model_id?: string;
380
+ }
381
+ interface AgentTraceRange {
382
+ start_line: number;
383
+ end_line: number;
384
+ content_hash?: string;
385
+ /** Per-range contributor override (agent handoffs). Wins over the
386
+ * conversation-level contributor for these lines. */
387
+ contributor?: AgentTraceContributor;
388
+ }
389
+ interface AgentTraceConversation {
390
+ url?: string;
391
+ contributor?: AgentTraceContributor;
392
+ ranges: AgentTraceRange[];
393
+ }
394
+ interface AgentTraceFile {
395
+ path: string;
396
+ conversations: AgentTraceConversation[];
397
+ }
398
+ interface AgentTraceRecord {
399
+ version: string;
400
+ id: string;
401
+ timestamp: string;
402
+ vcs?: {
403
+ type: string;
404
+ revision: string;
405
+ };
406
+ tool?: {
407
+ name?: string;
408
+ version?: string;
409
+ };
410
+ files: AgentTraceFile[];
411
+ }
412
+ /** Authorship provenance for one VCS revision, aggregated across the record's
413
+ * files/conversations/ranges. */
414
+ interface AuthoringProvenance {
415
+ commitSha: string;
416
+ /** Unique AI model ids that authored code in this commit (type ai|mixed). */
417
+ aiModels: string[];
418
+ /** Tools that produced the records (e.g. `cursor`). */
419
+ tools: string[];
420
+ conversationCount: number;
421
+ fileCount: number;
422
+ /** Total attributed lines (sum of range spans). */
423
+ lineCount: number;
424
+ /** True if any range was authored (in whole or part) by a human. */
425
+ humanInvolved: boolean;
426
+ }
427
+ type AgentTraceIndex = Map<string, AuthoringProvenance>;
428
+ /**
429
+ * Build a commit → provenance index from Agent Trace records. Multiple records
430
+ * for the same revision are merged. Records without `vcs.revision` are skipped
431
+ * (the SHA is the join key — without it there is nothing to correlate against).
432
+ */
433
+ declare function parseAgentTrace(records: AgentTraceRecord[]): AgentTraceIndex;
434
+ interface PartitionByAuthoringModelResult {
435
+ /** Runs grouped by each AI model that authored code in the run's commit. A
436
+ * run whose commit had multiple authoring models appears under EACH — the
437
+ * cohorts overlap by construction at commit granularity. */
438
+ byModel: Map<string, RunRecord[]>;
439
+ /** Runs whose `commitSha` had no Agent Trace provenance (no record, or no
440
+ * AI authorship). Kept separate — never silently folded into a cohort. */
441
+ unattributed: RunRecord[];
442
+ }
443
+ /**
444
+ * Partition runs by the AI model(s) that authored the code at each run's
445
+ * `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
446
+ * model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
447
+ * on "model A's code vs model B's code".
448
+ */
449
+ declare function partitionRunsByAuthoringModel(runs: RunRecord[], index: AgentTraceIndex): PartitionByAuthoringModelResult;
450
+ //#endregion
451
+ //#region src/contract/intake/code-agent-observation.d.ts
452
+ type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
453
+ type CodeAgentSessionTerminalStatus = 'completed' | 'failed' | 'unknown';
454
+ type CodeAgentSessionActionKind = 'tool' | 'patch' | 'terminal' | 'graph-completion';
455
+ type CodeAgentSessionActionSurface = 'tool' | 'mcp' | 'subagent' | 'skill' | 'hook' | 'web' | 'code';
456
+ type CodeAgentSessionActionStatus = 'started' | 'completed' | 'failed' | 'unknown';
457
+ interface CodeAgentSessionExecutionReceipt {
458
+ exitCode: number;
459
+ startedAtMs?: number;
460
+ completedAtMs?: number;
461
+ }
462
+ interface CodeAgentSessionAction {
463
+ id: string;
464
+ stepIndex: number;
465
+ kind: CodeAgentSessionActionKind;
466
+ surface: CodeAgentSessionActionSurface;
467
+ name: string;
468
+ status: CodeAgentSessionActionStatus;
469
+ timestampMs?: number;
470
+ costUsd?: number;
471
+ metadata: Record<string, unknown>;
472
+ }
473
+ interface CodeAgentSessionObservation {
474
+ source: CodeAgentSessionSource;
475
+ sessionId: string;
476
+ finalText: string | null;
477
+ terminal: {
478
+ status: CodeAgentSessionTerminalStatus;
479
+ explicit: boolean;
480
+ };
481
+ actions: CodeAgentSessionAction[];
482
+ }
483
+ interface ObserveCodeAgentSessionOptions {
484
+ source: CodeAgentSessionSource;
485
+ entries: unknown[];
486
+ sourcePath?: string;
487
+ execution?: CodeAgentSessionExecutionReceipt;
488
+ }
489
+ /**
490
+ * Project one provider session into the exact user-visible answer and a
491
+ * provider-neutral action stream. Raw prompts, tool inputs, and tool outputs
492
+ * stay out of this projection; callers retain the source JSONL as evidence.
493
+ */
494
+ declare function observeCodeAgentSession(options: ObserveCodeAgentSessionOptions): CodeAgentSessionObservation;
495
+ //#endregion
496
+ //#region src/contract/intake/code-agent-session.d.ts
497
+ interface ParsedCodeAgentJsonl {
498
+ entries: unknown[];
499
+ malformedLines: number;
500
+ }
501
+ interface CodeAgentSessionMetrics {
502
+ entries: number;
503
+ userMessages: number;
504
+ assistantMessages: number;
505
+ reasoningItems: number;
506
+ toolCalls: number;
507
+ toolOutputs: number;
508
+ toolErrors: number;
509
+ unclassifiedErrors: number;
510
+ patchAttempts: number;
511
+ patchSuccesses: number;
512
+ patchFailures: number;
513
+ turnsStarted: number;
514
+ turnsCompleted: number;
515
+ turnsAborted: number;
516
+ contextCompactions: number;
517
+ mcpCalls: number;
518
+ subagentCalls: number;
519
+ skillCalls: number;
520
+ hookCalls: number;
521
+ webCalls: number;
522
+ codeActions: number;
523
+ prLinks: number;
524
+ fileSnapshots: number;
525
+ graphNodes: number;
526
+ graphEdges: number;
527
+ actionCandidates: number;
528
+ verificationReports: number;
529
+ completionDecisions: number;
530
+ reliabilityRows: number;
531
+ reliabilityLift: number;
532
+ inputTokens: number;
533
+ outputTokens: number;
534
+ reasoningTokens: number;
535
+ cachedTokens: number;
536
+ cacheWriteTokens: number;
537
+ observedCostUsd: number;
538
+ observedCostCaptured?: boolean;
539
+ wallMs: number;
540
+ processScore: number;
541
+ }
542
+ interface CodeAgentSessionDiagnostic {
543
+ source: CodeAgentSessionSource;
544
+ sessionId: string;
545
+ sourcePath?: string;
546
+ entries: number;
547
+ malformedLines: number;
548
+ hasExplicitTerminalSignal: boolean;
549
+ hasFinalOutput: boolean;
550
+ hasQualityLabel: boolean;
551
+ hasTokenUsage: boolean;
552
+ hasCost: boolean;
553
+ costKind?: RunCostProvenance['kind'];
554
+ warnings: string[];
555
+ }
556
+ interface CodeAgentSessionIntakeResult {
557
+ runs: RunRecord[];
558
+ diagnostics: CodeAgentSessionDiagnostic[];
559
+ metrics: CodeAgentSessionMetrics[];
560
+ observations: CodeAgentSessionObservation[];
561
+ }
562
+ interface CodeAgentSessionIntakeOptions {
563
+ entries: unknown[];
564
+ malformedLines?: number;
565
+ sourcePath?: string;
566
+ experimentId?: string;
567
+ candidateId?: string;
568
+ seed?: number;
569
+ splitTag?: RunSplitTag;
570
+ scenarioId?: string;
571
+ model?: string;
572
+ promptHash?: string;
573
+ configHash?: string;
574
+ commitSha?: string;
575
+ score?: number;
576
+ /** Explicit cost receipt. When omitted, source-reported cost wins, then a
577
+ * token-priced estimate, then uncaptured. */
578
+ costProvenance?: RunCostProvenance;
579
+ /** Exact executor-owned process result. This is required when a provider's
580
+ * JSON stream has no terminal event, as with `opencode run --format json`. */
581
+ execution?: CodeAgentSessionExecutionReceipt;
582
+ }
583
+ /** One transcript line after the intake rule ran on it. A blank line produces
584
+ * nothing, so every value here is either a parsed entry or a counted defect. */
585
+ type CodeAgentJsonlLine = {
586
+ kind: 'entry';
587
+ lineNumber: number;
588
+ entry: unknown;
589
+ } | {
590
+ kind: 'malformed';
591
+ lineNumber: number;
592
+ };
593
+ declare function parseCodeAgentJsonl(jsonl: string): ParsedCodeAgentJsonl;
594
+ /** Reads a transcript one line at a time and never holds the file as a single
595
+ * string. `parseCodeAgentJsonl` needs the whole file in one string, so a
596
+ * session above V8's ~512MB string ceiling throws `ERR_STRING_TOO_LONG` and
597
+ * cannot be ingested at all; the largest real Codex rollout on record is 695MB.
598
+ *
599
+ * Lines break on `\n` only, which is what the string path's `split('\n')` does.
600
+ * `node:readline` also breaks on a bare `\r`, so it is deliberately not used
601
+ * here: a lone carriage return inside a line must stay inside that line for the
602
+ * two paths to report the same malformed count. */
603
+ declare function streamCodeAgentJsonlFile(path: string): AsyncGenerator<CodeAgentJsonlLine>;
604
+ /** Streaming counterpart to `parseCodeAgentJsonl` for a transcript on disk.
605
+ * It returns the same shape, so a caller that holds every entry keeps working
606
+ * above the string ceiling. The entry array still grows with the transcript;
607
+ * consume `streamCodeAgentJsonlFile` directly when memory must stay flat. */
608
+ declare function parseCodeAgentJsonlFile(path: string): Promise<ParsedCodeAgentJsonl>;
609
+ declare function fromCodexSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
610
+ declare function fromClaudeCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
611
+ declare function fromOpenCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
612
+ declare function fromKimiCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
613
+ declare function fromPiSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
614
+ //#endregion
615
+ //#region src/contract/intake/feedback-table.d.ts
616
+ interface FeedbackTableRow {
617
+ /** Stable id for this run — the unit a rater scored. Drives pairing
618
+ * across analysis primitives. */
619
+ runId: string;
620
+ /** Identifier of the rater that produced this rating. */
621
+ rater: string;
622
+ /** The rating itself. Accepts boolean (approve/reject), 0..1 scalar,
623
+ * or any numeric scale — see `scale`. */
624
+ rating: number | boolean;
625
+ /** Optional metadata carried through to `RunRecord.outcome.raw` and the
626
+ * custom-shape metadata bag. */
627
+ metadata?: Record<string, unknown>;
628
+ }
629
+ interface FeedbackTableMeta {
630
+ runId: string;
631
+ /** When omitted, defaults to `'feedback-corpus'`. Used to group related
632
+ * runs in `analyzeRuns()` lift analysis. */
633
+ experimentId?: string;
634
+ /** When omitted, defaults to `runId` — each run is its own candidate. */
635
+ candidateId?: string;
636
+ /** Observed cost in USD, when available. */
637
+ costUsd?: number;
638
+ /** Stable scenario identity. Defaults to `runId`. */
639
+ scenarioId?: string;
640
+ /** Wall-clock ms, when available. Defaults to 0. */
641
+ wallMs?: number;
642
+ /** Model identifier including snapshot. Default `unknown@unknown`. */
643
+ model?: string;
644
+ /** Optional sha256 of the prompt; default `'sha256:unknown'`. */
645
+ promptHash?: string;
646
+ /** Default `'sha256:unknown'`. */
647
+ configHash?: string;
648
+ /** Default `'unknown'`. */
649
+ commitSha?: string;
650
+ /** Default `'holdout'` — feedback corpora are by nature the holdout
651
+ * signal a closed-loop improvement aims at. */
652
+ splitTag?: RunSplitTag;
653
+ /** Free-form metadata available to consumers via the cast-out path on
654
+ * the resulting RunRecord. */
655
+ extras?: Record<string, unknown>;
656
+ }
657
+ interface FromFeedbackTableOptions {
658
+ /** Per-(run, rater) ratings. */
659
+ ratings: FeedbackTableRow[];
660
+ /** Per-run metadata. When a runId appears in `ratings` but not here, the
661
+ * adapter synthesises minimal metadata with defaults documented above. */
662
+ meta?: FeedbackTableMeta[];
663
+ /** Rating scale. Provide `{ min, max }` for non-0..1 numeric scales.
664
+ * Booleans are normalised: true → 1, false → 0. Default: assumes
665
+ * ratings are already 0..1. */
666
+ scale?: {
667
+ min: number;
668
+ max: number;
669
+ };
670
+ /** When true, the rater scores are emitted into `raterScores` (a sibling
671
+ * array `analyzeRuns()` accepts) in addition to the aggregate run score.
672
+ * Default `true` preserves rater-level signal for inter-rater analysis. */
673
+ emitRaterScores?: boolean;
674
+ }
675
+ interface FromFeedbackTableResult {
676
+ runs: RunRecord[];
677
+ /** Rater-level scores ready to pass into `analyzeRuns({ raterScores })`
678
+ * for inter-rater agreement + disagreement triage. */
679
+ raterScores: Array<{
680
+ runId: string;
681
+ rater: string;
682
+ score: number;
683
+ }>;
684
+ }
685
+ declare function fromFeedbackTable(opts: FromFeedbackTableOptions): FromFeedbackTableResult;
686
+ //#endregion
687
+ //#region src/contract/intake/otel-spans.d.ts
688
+ interface FromOtelSpansOptions {
689
+ spans: TraceSpanEvent[];
690
+ /** Default split tag for synthesized records. Defaults to `'holdout'`. */
691
+ defaultSplit?: RunSplitTag;
692
+ /** Default `experimentId` when not present on any span. */
693
+ experimentId?: string;
694
+ /**
695
+ * Explicit task-quality score for a logical run. The callback receives
696
+ * spans in deterministic time/id order. Its value must agree with any
697
+ * designated score attributes present on root or `EVALUATOR` spans.
698
+ */
699
+ scoreForRun?: (runId: string, spans: readonly TraceSpanEvent[]) => number | undefined;
700
+ }
701
+ declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
702
+ //#endregion
703
+ export { FromRunRecordDirOptions as $, observeCodeAgentSession as A, verifyCandidateExperimentComparison as At, parseAgentTrace as B, CodeAgentSessionActionKind as C, evaluatePairedMeasurements as Ct, CodeAgentSessionObservation as D, sealCandidateBenchmarkTask as Dt, CodeAgentSessionExecutionReceipt as E, sealCandidateBenchmarkSuite as Et, AgentTraceIndex as F, EvalRunDiff as G, EvalCellScoreDelta as H, AgentTraceRange as I, diffRuns as J, diffGenerations as K, AgentTraceRecord as L, AgentTraceContributorType as M, AgentTraceConversation as N, CodeAgentSessionSource as O, sealCandidateExperiment as Ot, AgentTraceFile as P, evalReportingSuite as Q, AuthoringProvenance as R, CodeAgentSessionAction as S, SealCandidateBenchmarkSuiteOptions as St, CodeAgentSessionActionSurface as T, runCandidateExperiment as Tt, EvalDimensionDelta as U, partitionRunsByAuthoringModel as V, EvalGenerationDiff as W, EvalReportingSuiteOptions as X, EvalReportingSuiteInput as Y, EvalReportingSuiteResult as Z, fromOpenCodeSession as _, EvaluatePairedMeasurementsOptions as _t, FromFeedbackTableOptions as a, CompareAgentProfileImprovementExperimentOptions as at, parseCodeAgentJsonlFile as b, PairedMeasurementEvaluation as bt, CodeAgentJsonlLine as c, measuredComparisonFromAgentProfileImprovementExperiment as ct, CodeAgentSessionIntakeResult as d, sealAgentProfileImprovementSuite as dt, FromRunRecordDirResult as et, CodeAgentSessionMetrics as f, sealAgentProfileImprovementTask as ft, fromKimiCodeSession as g, CompareCandidateExperimentOptions as gt, fromCodexSession as h, CandidateExperimentRun as ht, FeedbackTableRow as i, AgentProfileImprovementExperimentRun as it, AgentTraceContributor as j, CodeAgentSessionTerminalStatus as k, verifyCandidateExperiment as kt, CodeAgentSessionDiagnostic as l, runAgentProfileImprovementExperiment as lt, fromClaudeCodeSession as m, CandidateExperimentExecutionInput as mt, fromOtelSpans as n, fromRunRecordDir as nt, FromFeedbackTableResult as o, RunAgentProfileImprovementExperimentOptions as ot, ParsedCodeAgentJsonl as p, verifyAgentProfileImprovementExperimentComparison as pt, diffRunBaselineToWinner as q, FeedbackTableMeta as r, AgentProfileImprovementExperimentExecutionInput as rt, fromFeedbackTable as s, SealAgentProfileImprovementSuiteOptions as st, FromOtelSpansOptions as t, RunRecordRejection as tt, CodeAgentSessionIntakeOptions as u, sealAgentProfileImprovementExperiment as ut, fromPiSession as v, PairedMeasurement as vt, CodeAgentSessionActionStatus as w, measuredComparisonFromCandidateExperiment as wt, streamCodeAgentJsonlFile as x, RunCandidateExperimentOptions as xt, parseCodeAgentJsonl as y, PairedMeasurementAdapter as yt, PartitionByAuthoringModelResult as z };
704
+ //# sourceMappingURL=index-fNXZMCzX.d.ts.map