@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,1458 +0,0 @@
1
- import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
2
- import { s as validateRunRecord } from "./run-record-BmSPWXJR.js";
3
- import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
4
- import { dirname } from "node:path";
5
- //#region src/action-policy.ts
6
- function evaluateActionPolicy(action, policy = {}, options = {}) {
7
- const reasons = [];
8
- let blocked = false;
9
- let requiresApproval = Boolean(action.requiresApproval);
10
- if (policy.allowedTypes?.length && !policy.allowedTypes.includes(action.type)) {
11
- blocked = true;
12
- reasons.push(`action type "${action.type}" is not allowed`);
13
- }
14
- if (policy.blockedTypes?.includes(action.type)) {
15
- blocked = true;
16
- reasons.push(`action type "${action.type}" is blocked`);
17
- }
18
- if (policy.alwaysRequireApprovalTypes?.includes(action.type)) {
19
- requiresApproval = true;
20
- reasons.push(`action type "${action.type}" requires approval`);
21
- }
22
- if (policy.requireApprovalForExternalSideEffects && action.externalSideEffect) {
23
- requiresApproval = true;
24
- reasons.push("external side effect requires approval");
25
- }
26
- if (policy.requireApprovalAboveCostUsd !== void 0 && (action.costUsd ?? 0) > policy.requireApprovalAboveCostUsd) {
27
- requiresApproval = true;
28
- reasons.push(`cost ${action.costUsd} exceeds approval threshold ${policy.requireApprovalAboveCostUsd}`);
29
- }
30
- if (policy.maxActionCostUsd !== void 0 && (action.costUsd ?? 0) > policy.maxActionCostUsd) {
31
- blocked = true;
32
- reasons.push(`cost ${action.costUsd} exceeds max action cost ${policy.maxActionCostUsd}`);
33
- }
34
- if (policy.remainingBudgetUsd !== void 0 && (action.costUsd ?? 0) > policy.remainingBudgetUsd) {
35
- blocked = true;
36
- reasons.push(`cost ${action.costUsd} exceeds remaining budget ${policy.remainingBudgetUsd}`);
37
- }
38
- if (policy.expectedOutcomeRequired && !action.metadata?.expectedOutcome) {
39
- blocked = true;
40
- reasons.push("expected outcome is required");
41
- }
42
- if (policy.killCriteriaRequired && !action.metadata?.killCriteria) {
43
- blocked = true;
44
- reasons.push("kill criteria are required");
45
- }
46
- if (policy.autoApproveTypes?.includes(action.type) && requiresApproval) reasons.push(`action type "${action.type}" is auto-approved only when no approval policy applies`);
47
- if (!reasons.length) reasons.push(requiresApproval ? "approval required" : "action allowed");
48
- const label = blocked || requiresApproval ? {
49
- source: "policy",
50
- kind: blocked ? "policy_block" : "comment",
51
- value: {
52
- actionType: action.type,
53
- blocked,
54
- requiresApproval
55
- },
56
- reason: reasons.join("; "),
57
- severity: blocked ? "critical" : "warning",
58
- createdAt: options.createdAt ?? (/* @__PURE__ */ new Date()).toISOString(),
59
- metadata: {
60
- action,
61
- policy
62
- }
63
- } : void 0;
64
- return {
65
- allowed: !blocked,
66
- blocked,
67
- requiresApproval: !blocked && requiresApproval,
68
- reasons,
69
- label
70
- };
71
- }
72
- //#endregion
73
- //#region src/detectors/index.ts
74
- /** Same action fingerprint N consecutive steps = a stuck loop (the #1 long-horizon failure mode). */
75
- function repeatedActionDetector(opts = {}) {
76
- const max = opts.maxRepeated ?? 3;
77
- const severity = opts.severity ?? "warn";
78
- const failureClass = opts.failureClass ?? "tool_recovery_failure";
79
- let last;
80
- let streak = 0;
81
- return {
82
- id: "repeated-action",
83
- get streak() {
84
- return streak;
85
- },
86
- observe(event) {
87
- const fp = event.actionFingerprint;
88
- if (fp === void 0) return null;
89
- streak = fp === last ? streak + 1 : 1;
90
- last = fp;
91
- if (max <= 0 || streak < max) return null;
92
- return {
93
- detector: "repeated-action",
94
- severity,
95
- failureClass,
96
- reason: `stuck: repeated same action for ${streak} step(s)`,
97
- streak,
98
- ...event.label ? { evidence: { action: event.label } } : {}
99
- };
100
- },
101
- reset() {
102
- last = void 0;
103
- streak = 0;
104
- }
105
- };
106
- }
107
- /** State + score unchanged across N steps = spinning wheels. Compares each step to the previous one,
108
- * so prime with the initial state (observe it once before the first real step) to detect on step 1. */
109
- function noProgressDetector(opts = {}) {
110
- const max = opts.maxNoProgress ?? 3;
111
- const minScoreDelta = opts.minScoreDelta ?? .001;
112
- const severity = opts.severity ?? "warn";
113
- const failureClass = opts.failureClass ?? "tool_recovery_failure";
114
- let lastState;
115
- let lastScore;
116
- let streak = 0;
117
- return {
118
- id: "no-progress",
119
- get streak() {
120
- return streak;
121
- },
122
- observe(event) {
123
- const stateUnchanged = lastState !== void 0 && lastState === event.stateFingerprint;
124
- const scoreFlat = Math.abs((event.score ?? 0) - (lastScore ?? 0)) < minScoreDelta;
125
- streak = stateUnchanged && scoreFlat ? streak + 1 : 0;
126
- lastState = event.stateFingerprint;
127
- lastScore = event.score;
128
- if (max <= 0 || streak < max) return null;
129
- return {
130
- detector: "no-progress",
131
- severity,
132
- failureClass,
133
- reason: `stuck: no state/score progress for ${streak} step(s)`,
134
- streak,
135
- ...event.label ? { evidence: { state: event.label } } : {}
136
- };
137
- },
138
- reset() {
139
- lastState = void 0;
140
- lastScore = void 0;
141
- streak = 0;
142
- }
143
- };
144
- }
145
- /** N consecutive tool errors = the worker is hammering a broken approach. */
146
- function errorStreakDetector(opts = {}) {
147
- const max = opts.maxErrors ?? 3;
148
- const severity = opts.severity ?? "warn";
149
- const failureClass = opts.failureClass ?? "tool_recovery_failure";
150
- let streak = 0;
151
- return {
152
- id: "error-streak",
153
- get streak() {
154
- return streak;
155
- },
156
- observe(event) {
157
- if (event.status === void 0) return null;
158
- streak = event.status === "error" ? streak + 1 : 0;
159
- if (max <= 0 || streak < max) return null;
160
- return {
161
- detector: "error-streak",
162
- severity,
163
- failureClass,
164
- reason: `stuck: ${streak} consecutive errored step(s)`,
165
- streak,
166
- ...event.label ? { evidence: { lastError: event.label } } : {}
167
- };
168
- },
169
- reset() {
170
- streak = 0;
171
- }
172
- };
173
- }
174
- /** Fold one event through many detectors at once; returns every signal that fired this step. The
175
- * natural shape for an online pipe watching with a whole panel of detectors. */
176
- function observeAll(detectors, event) {
177
- const signals = [];
178
- for (const d of detectors) {
179
- const s = d.observe(event);
180
- if (s) signals.push(s);
181
- }
182
- return signals;
183
- }
184
- //#endregion
185
- //#region src/control-runtime.ts
186
- /**
187
- * Policy-based agent control runtime.
188
- *
189
- * This is the minimal reusable loop behind driver-agent patterns:
190
- *
191
- * observe state -> validate -> decide next action -> act -> observe -> ...
192
- *
193
- * It deliberately does not model named "topologies". Direct execution,
194
- * critic/revise, driver intervention, specialist calls, and human escalation
195
- * are all just actions chosen by the control policy.
196
- */
197
- const DEFAULT_BUDGET = {
198
- maxSteps: 8,
199
- maxWallMs: 300 * 1e3
200
- };
201
- async function runAgentControlLoop(config) {
202
- const budget = normalizeBudget(config.budget);
203
- const actionFailure = config.actionFailure ?? "continue";
204
- const controller = new AbortController();
205
- const upstreamAbort = () => controller.abort(config.signal?.reason);
206
- if (config.signal) if (config.signal.aborted) controller.abort(config.signal.reason);
207
- else config.signal.addEventListener("abort", upstreamAbort, { once: true });
208
- const started = Date.now();
209
- const wallTimer = budget.maxWallMs ? setTimeout(() => controller.abort(/* @__PURE__ */ new Error("control runtime wall timeout")), budget.maxWallMs) : void 0;
210
- const history = [];
211
- const emitter = config.store ? new TraceEmitter(config.store) : void 0;
212
- let spentCostUsd = 0;
213
- const runtimeErrors = [];
214
- const repeatedDetector = repeatedActionDetector({ maxRepeated: config.stopPolicies?.maxRepeatedActions ?? 0 });
215
- const progressDetector = noProgressDetector({
216
- maxNoProgress: config.stopPolicies?.maxNoProgressSteps ?? 0,
217
- minScoreDelta: config.stopPolicies?.minScoreDelta ?? .001
218
- });
219
- try {
220
- if (emitter) await runTrace(runtimeErrors, 0, () => emitter.startRun({
221
- scenarioId: config.scenarioId ?? "agent-control-loop",
222
- projectId: config.projectId,
223
- variantId: config.variantId,
224
- layer: "meta",
225
- tags: {
226
- intent: config.intent.slice(0, 120),
227
- maxSteps: String(budget.maxSteps),
228
- ...budget.maxCostUsd !== void 0 ? { maxCostUsd: String(budget.maxCostUsd) } : {}
229
- }
230
- }));
231
- let state;
232
- let evals;
233
- try {
234
- state = await config.observe({
235
- history,
236
- abortSignal: controller.signal
237
- });
238
- } catch (err) {
239
- const error = runtimeError("observe", 0, err);
240
- runtimeErrors.push(error);
241
- return finish(emitter, {
242
- intent: config.intent,
243
- pass: false,
244
- completed: false,
245
- reason: error.message,
246
- steps: history,
247
- finalState: void 0,
248
- finalEvals: [],
249
- wallMs: Date.now() - started,
250
- spentCostUsd,
251
- runId: emitter?.runId ?? null,
252
- failureClass: "unknown",
253
- runtimeErrors,
254
- stoppedBy: "runtime-error"
255
- });
256
- }
257
- try {
258
- evals = await config.validate({
259
- intent: config.intent,
260
- state,
261
- history,
262
- abortSignal: controller.signal
263
- });
264
- await recordEvalSpans(emitter, evals, "initial", runtimeErrors, 0);
265
- } catch (err) {
266
- const error = runtimeError("validate", 0, err);
267
- runtimeErrors.push(error);
268
- return finish(emitter, {
269
- intent: config.intent,
270
- pass: false,
271
- completed: false,
272
- reason: error.message,
273
- steps: history,
274
- finalState: state,
275
- finalEvals: [],
276
- wallMs: Date.now() - started,
277
- spentCostUsd,
278
- runId: emitter?.runId ?? null,
279
- failureClass: "unknown",
280
- runtimeErrors,
281
- stoppedBy: "runtime-error"
282
- });
283
- }
284
- progressDetector.observe({
285
- stateFingerprint: fingerprintState(state, config.stopPolicies),
286
- score: averageScore(evals)
287
- });
288
- for (let stepIndex = 0; stepIndex < budget.maxSteps; stepIndex++) {
289
- if (controller.signal.aborted) return finish(emitter, {
290
- intent: config.intent,
291
- pass: false,
292
- completed: false,
293
- reason: abortReason(controller.signal),
294
- score: void 0,
295
- steps: history,
296
- finalState: state,
297
- finalEvals: evals,
298
- wallMs: Date.now() - started,
299
- spentCostUsd,
300
- runId: emitter?.runId ?? null,
301
- failureClass: "timeout",
302
- runtimeErrors,
303
- stoppedBy: "abort"
304
- });
305
- const budgetStop = budgetStopDecision(budget, spentCostUsd);
306
- if (budgetStop.stop) return finish(emitter, {
307
- intent: config.intent,
308
- pass: false,
309
- completed: false,
310
- reason: budgetStop.reason,
311
- score: averageScore(evals),
312
- steps: history,
313
- finalState: state,
314
- finalEvals: evals,
315
- wallMs: Date.now() - started,
316
- spentCostUsd,
317
- runId: emitter?.runId ?? null,
318
- failureClass: "budget_exceeded",
319
- runtimeErrors,
320
- stoppedBy: "budget"
321
- });
322
- const ctx = makeContext(config.intent, state, evals, history, budget, stepIndex, started, spentCostUsd, controller.signal, emitter);
323
- let stop;
324
- try {
325
- stop = config.shouldStop ? await config.shouldStop(ctx) : defaultStopDecision(evals);
326
- } catch (err) {
327
- runtimeErrors.push(runtimeError("stop-policy", stepIndex, err));
328
- return finish(emitter, {
329
- intent: config.intent,
330
- pass: false,
331
- completed: false,
332
- reason: runtimeErrors[runtimeErrors.length - 1].message,
333
- score: averageScore(evals),
334
- steps: history,
335
- finalState: state,
336
- finalEvals: evals,
337
- wallMs: Date.now() - started,
338
- spentCostUsd,
339
- runId: emitter?.runId ?? null,
340
- failureClass: "unknown",
341
- runtimeErrors,
342
- stoppedBy: "runtime-error"
343
- });
344
- }
345
- if (stop.stop) return finish(emitter, {
346
- intent: config.intent,
347
- pass: stop.pass,
348
- completed: true,
349
- reason: stop.reason,
350
- score: stop.score,
351
- steps: history,
352
- finalState: state,
353
- finalEvals: evals,
354
- wallMs: Date.now() - started,
355
- spentCostUsd,
356
- runId: emitter?.runId ?? null,
357
- failureClass: stop.failureClass,
358
- runtimeErrors,
359
- stoppedBy: "stop-policy"
360
- });
361
- let decision;
362
- try {
363
- decision = await config.decide(ctx);
364
- } catch (err) {
365
- runtimeErrors.push(runtimeError("decide", stepIndex, err));
366
- return finish(emitter, {
367
- intent: config.intent,
368
- pass: false,
369
- completed: false,
370
- reason: runtimeErrors[runtimeErrors.length - 1].message,
371
- score: averageScore(evals),
372
- steps: history,
373
- finalState: state,
374
- finalEvals: evals,
375
- wallMs: Date.now() - started,
376
- spentCostUsd,
377
- runId: emitter?.runId ?? null,
378
- failureClass: "unknown",
379
- runtimeErrors,
380
- stoppedBy: "runtime-error"
381
- });
382
- }
383
- if (decision.type === "stop") {
384
- const pass = decision.pass ?? false;
385
- return finish(emitter, {
386
- intent: config.intent,
387
- pass,
388
- completed: true,
389
- reason: decision.reason,
390
- score: decision.score,
391
- steps: history,
392
- finalState: state,
393
- finalEvals: evals,
394
- wallMs: Date.now() - started,
395
- spentCostUsd,
396
- runId: emitter?.runId ?? null,
397
- failureClass: pass ? void 0 : decision.failureClass ?? "unknown",
398
- runtimeErrors,
399
- stoppedBy: "policy"
400
- });
401
- }
402
- const actionFingerprint = fingerprintAction(decision.action, config.stopPolicies);
403
- const repeatedActionSignal = repeatedDetector.observe({ actionFingerprint });
404
- const repeatedActionStreak = repeatedDetector.streak;
405
- if (repeatedActionSignal) return finish(emitter, {
406
- intent: config.intent,
407
- pass: false,
408
- completed: true,
409
- reason: repeatedActionSignal.reason,
410
- score: averageScore(evals),
411
- steps: history,
412
- finalState: state,
413
- finalEvals: evals,
414
- wallMs: Date.now() - started,
415
- spentCostUsd,
416
- runId: emitter?.runId ?? null,
417
- failureClass: "tool_recovery_failure",
418
- runtimeErrors,
419
- stoppedBy: "stop-policy"
420
- });
421
- const beforeState = state;
422
- const evalsBefore = evals;
423
- const scoreBefore = averageScore(evals);
424
- const actionStarted = Date.now();
425
- const stepHandle = emitter ? await runTrace(runtimeErrors, stepIndex, () => emitter.tool({
426
- name: `control-step-${stepIndex}`,
427
- toolName: "agent-control-action",
428
- args: decision.action,
429
- attributes: {
430
- decision: decision.reason ?? "continue",
431
- repeatedActionStreak
432
- }
433
- })) : void 0;
434
- let actionOutcome;
435
- try {
436
- const result = await config.act(decision.action, ctx);
437
- const rawCostUsd = config.getActionCostUsd?.({
438
- action: decision.action,
439
- result,
440
- state,
441
- evals,
442
- history
443
- });
444
- const costUsd = normalizeActionCostUsd(rawCostUsd, runtimeErrors, stepIndex);
445
- if (costUsd !== void 0 && Number.isFinite(costUsd) && costUsd > 0) {
446
- spentCostUsd += costUsd;
447
- await recordCostBudget(emitter, budget, spentCostUsd, stepHandle, runtimeErrors, stepIndex);
448
- }
449
- actionOutcome = {
450
- ok: true,
451
- result,
452
- ...costUsd !== void 0 ? { costUsd } : {},
453
- durationMs: Date.now() - actionStarted
454
- };
455
- } catch (err) {
456
- runtimeErrors.push(runtimeError("act", stepIndex, err));
457
- actionOutcome = {
458
- ok: false,
459
- error: runtimeErrors[runtimeErrors.length - 1].message,
460
- durationMs: Date.now() - actionStarted
461
- };
462
- if (actionFailure === "stop") {
463
- await runTrace(runtimeErrors, stepIndex, () => stepHandle?.fail(actionOutcome.error ?? "action failed"));
464
- const step = {
465
- index: stepIndex,
466
- decision,
467
- beforeState,
468
- afterState: state,
469
- evalsBefore,
470
- evalsAfter: evals,
471
- actionOutcome,
472
- startedAt: new Date(actionStarted).toISOString(),
473
- endedAt: (/* @__PURE__ */ new Date()).toISOString()
474
- };
475
- history.push(step);
476
- await runOnStep(config.onStep, step, runtimeErrors);
477
- return finish(emitter, {
478
- intent: config.intent,
479
- pass: false,
480
- completed: false,
481
- reason: actionOutcome.error ?? "action failed",
482
- score: averageScore(evals),
483
- steps: history,
484
- finalState: state,
485
- finalEvals: evals,
486
- wallMs: Date.now() - started,
487
- spentCostUsd,
488
- runId: emitter?.runId ?? null,
489
- failureClass: "unknown",
490
- runtimeErrors,
491
- stoppedBy: "runtime-error"
492
- });
493
- }
494
- }
495
- try {
496
- state = await config.observe({
497
- history,
498
- abortSignal: controller.signal
499
- });
500
- } catch (err) {
501
- runtimeErrors.push(runtimeError("observe", stepIndex, err));
502
- const step = {
503
- index: stepIndex,
504
- decision,
505
- beforeState,
506
- afterState: beforeState,
507
- evalsBefore,
508
- evalsAfter: evals,
509
- actionOutcome,
510
- startedAt: new Date(actionStarted).toISOString(),
511
- endedAt: (/* @__PURE__ */ new Date()).toISOString()
512
- };
513
- history.push(step);
514
- await runTrace(runtimeErrors, stepIndex, () => stepHandle?.fail(runtimeErrors[runtimeErrors.length - 1].message));
515
- await runOnStep(config.onStep, step, runtimeErrors);
516
- return finish(emitter, {
517
- intent: config.intent,
518
- pass: false,
519
- completed: false,
520
- reason: runtimeErrors[runtimeErrors.length - 1].message,
521
- score: averageScore(evals),
522
- steps: history,
523
- finalState: beforeState,
524
- finalEvals: evals,
525
- wallMs: Date.now() - started,
526
- spentCostUsd,
527
- runId: emitter?.runId ?? null,
528
- failureClass: "unknown",
529
- runtimeErrors,
530
- stoppedBy: "runtime-error"
531
- });
532
- }
533
- try {
534
- evals = await config.validate({
535
- intent: config.intent,
536
- state,
537
- history,
538
- abortSignal: controller.signal
539
- });
540
- await recordEvalSpans(emitter, evals, `step-${stepIndex}`, runtimeErrors, stepIndex, stepHandle?.span.spanId);
541
- } catch (err) {
542
- runtimeErrors.push(runtimeError("validate", stepIndex, err));
543
- const step = {
544
- index: stepIndex,
545
- decision,
546
- beforeState,
547
- afterState: state,
548
- evalsBefore,
549
- evalsAfter: evals,
550
- actionOutcome,
551
- startedAt: new Date(actionStarted).toISOString(),
552
- endedAt: (/* @__PURE__ */ new Date()).toISOString()
553
- };
554
- history.push(step);
555
- await runTrace(runtimeErrors, stepIndex, () => stepHandle?.fail(runtimeErrors[runtimeErrors.length - 1].message));
556
- await runOnStep(config.onStep, step, runtimeErrors);
557
- return finish(emitter, {
558
- intent: config.intent,
559
- pass: false,
560
- completed: false,
561
- reason: runtimeErrors[runtimeErrors.length - 1].message,
562
- score: averageScore(evals),
563
- steps: history,
564
- finalState: state,
565
- finalEvals: evals,
566
- wallMs: Date.now() - started,
567
- spentCostUsd,
568
- runId: emitter?.runId ?? null,
569
- failureClass: "unknown",
570
- runtimeErrors,
571
- stoppedBy: "runtime-error"
572
- });
573
- }
574
- const scoreAfter = averageScore(evals);
575
- const stateFingerprint = fingerprintState(state, config.stopPolicies);
576
- const noProgressSignal = progressDetector.observe({
577
- stateFingerprint,
578
- score: scoreAfter
579
- });
580
- const noProgressStreak = progressDetector.streak;
581
- const step = {
582
- index: stepIndex,
583
- decision,
584
- beforeState,
585
- afterState: state,
586
- evalsBefore,
587
- evalsAfter: evals,
588
- actionOutcome,
589
- startedAt: new Date(actionStarted).toISOString(),
590
- endedAt: (/* @__PURE__ */ new Date()).toISOString()
591
- };
592
- history.push(step);
593
- if (actionOutcome.ok) await runTrace(runtimeErrors, stepIndex, () => stepHandle?.end({ attributes: {
594
- actionCostUsd: actionOutcome.costUsd ?? null,
595
- spentCostUsd,
596
- scoreBefore: scoreBefore ?? null,
597
- scoreAfter: scoreAfter ?? null,
598
- noProgressStreak
599
- } }));
600
- else await runTrace(runtimeErrors, stepIndex, () => stepHandle?.fail(actionOutcome.error ?? "action failed", { attributes: {
601
- spentCostUsd,
602
- noProgressStreak
603
- } }));
604
- await runOnStep(config.onStep, step, runtimeErrors);
605
- if (noProgressSignal) return finish(emitter, {
606
- intent: config.intent,
607
- pass: false,
608
- completed: true,
609
- reason: noProgressSignal.reason,
610
- score: scoreAfter,
611
- steps: history,
612
- finalState: state,
613
- finalEvals: evals,
614
- wallMs: Date.now() - started,
615
- spentCostUsd,
616
- runId: emitter?.runId ?? null,
617
- failureClass: "tool_recovery_failure",
618
- runtimeErrors,
619
- stoppedBy: "stop-policy"
620
- });
621
- const postStepBudgetStop = budgetStopDecision(budget, spentCostUsd);
622
- if (postStepBudgetStop.stop) return finish(emitter, {
623
- intent: config.intent,
624
- pass: false,
625
- completed: false,
626
- reason: postStepBudgetStop.reason,
627
- score: scoreAfter,
628
- steps: history,
629
- finalState: state,
630
- finalEvals: evals,
631
- wallMs: Date.now() - started,
632
- spentCostUsd,
633
- runId: emitter?.runId ?? null,
634
- failureClass: "budget_exceeded",
635
- runtimeErrors,
636
- stoppedBy: "budget"
637
- });
638
- const postStepCtx = makeContext(config.intent, state, evals, history, budget, stepIndex + 1, started, spentCostUsd, controller.signal, emitter);
639
- let postStepStop;
640
- try {
641
- postStepStop = config.shouldStop ? await config.shouldStop(postStepCtx) : defaultStopDecision(evals);
642
- } catch (err) {
643
- runtimeErrors.push(runtimeError("stop-policy", stepIndex + 1, err));
644
- return finish(emitter, {
645
- intent: config.intent,
646
- pass: false,
647
- completed: false,
648
- reason: runtimeErrors[runtimeErrors.length - 1].message,
649
- score: averageScore(evals),
650
- steps: history,
651
- finalState: state,
652
- finalEvals: evals,
653
- wallMs: Date.now() - started,
654
- spentCostUsd,
655
- runId: emitter?.runId ?? null,
656
- failureClass: "unknown",
657
- runtimeErrors,
658
- stoppedBy: "runtime-error"
659
- });
660
- }
661
- if (postStepStop.stop) return finish(emitter, {
662
- intent: config.intent,
663
- pass: postStepStop.pass,
664
- completed: true,
665
- reason: postStepStop.reason,
666
- score: postStepStop.score,
667
- steps: history,
668
- finalState: state,
669
- finalEvals: evals,
670
- wallMs: Date.now() - started,
671
- spentCostUsd,
672
- runId: emitter?.runId ?? null,
673
- failureClass: postStepStop.failureClass,
674
- runtimeErrors,
675
- stoppedBy: "stop-policy"
676
- });
677
- }
678
- return finish(emitter, {
679
- intent: config.intent,
680
- pass: false,
681
- completed: false,
682
- reason: `budget exhausted: maxSteps=${budget.maxSteps}`,
683
- steps: history,
684
- finalState: state,
685
- finalEvals: evals,
686
- wallMs: Date.now() - started,
687
- spentCostUsd,
688
- runId: emitter?.runId ?? null,
689
- failureClass: "budget_exceeded",
690
- runtimeErrors,
691
- stoppedBy: "budget"
692
- });
693
- } catch (err) {
694
- runtimeErrors.push(runtimeError("act", history.length, err));
695
- return finish(emitter, {
696
- intent: config.intent,
697
- pass: false,
698
- completed: false,
699
- reason: runtimeErrors[runtimeErrors.length - 1].message,
700
- steps: history,
701
- finalState: void 0,
702
- finalEvals: [],
703
- wallMs: Date.now() - started,
704
- spentCostUsd,
705
- runId: emitter?.runId ?? null,
706
- failureClass: "unknown",
707
- runtimeErrors,
708
- stoppedBy: "runtime-error"
709
- });
710
- } finally {
711
- if (wallTimer) clearTimeout(wallTimer);
712
- if (config.signal) config.signal.removeEventListener("abort", upstreamAbort);
713
- }
714
- }
715
- function stopOnNoProgress(maxNoProgressSteps, options = {}) {
716
- return {
717
- ...options,
718
- maxNoProgressSteps
719
- };
720
- }
721
- function stopOnRepeatedAction(maxRepeatedActions, options = {}) {
722
- return {
723
- ...options,
724
- maxRepeatedActions
725
- };
726
- }
727
- function objectiveEval(input) {
728
- return {
729
- ...input,
730
- objective: true
731
- };
732
- }
733
- function subjectiveEval(input) {
734
- return {
735
- ...input,
736
- objective: false
737
- };
738
- }
739
- function normalizeBudget(input) {
740
- const raw = {
741
- ...DEFAULT_BUDGET,
742
- ...input
743
- };
744
- if (!Number.isInteger(raw.maxSteps) || raw.maxSteps < 1) throw new RangeError(`ControlRuntime budget.maxSteps must be an integer >= 1, got ${String(raw.maxSteps)}`);
745
- const budget = { maxSteps: raw.maxSteps };
746
- if (raw.maxWallMs !== void 0) {
747
- if (typeof raw.maxWallMs !== "number" || !Number.isFinite(raw.maxWallMs) || raw.maxWallMs <= 0) throw new RangeError(`ControlRuntime budget.maxWallMs must be a positive finite number, got ${String(raw.maxWallMs)}`);
748
- budget.maxWallMs = raw.maxWallMs;
749
- }
750
- if (raw.maxCostUsd !== void 0) {
751
- if (typeof raw.maxCostUsd !== "number" || !Number.isFinite(raw.maxCostUsd) || raw.maxCostUsd < 0) throw new RangeError(`ControlRuntime budget.maxCostUsd must be a nonnegative finite number, got ${String(raw.maxCostUsd)}`);
752
- budget.maxCostUsd = raw.maxCostUsd;
753
- }
754
- return budget;
755
- }
756
- function normalizeActionCostUsd(costUsd, runtimeErrors, stepIndex) {
757
- if (costUsd === void 0) return void 0;
758
- if (!Number.isFinite(costUsd) || costUsd < 0) {
759
- runtimeErrors.push(runtimeError("act", stepIndex, /* @__PURE__ */ new Error(`invalid action costUsd: ${String(costUsd)}`)));
760
- return;
761
- }
762
- return costUsd;
763
- }
764
- function allCriticalPassed(evals) {
765
- return evals.every((result) => result.passed || result.severity !== "critical" && result.severity !== "error");
766
- }
767
- function makeContext(intent, state, evals, history, budget, stepIndex, started, spentCostUsd, abortSignal, emitter) {
768
- return {
769
- intent,
770
- state,
771
- evals,
772
- history,
773
- budget,
774
- stepIndex,
775
- wallMs: Date.now() - started,
776
- spentCostUsd,
777
- remainingCostUsd: budget.maxCostUsd === void 0 ? void 0 : Math.max(0, budget.maxCostUsd - spentCostUsd),
778
- abortSignal,
779
- emitter
780
- };
781
- }
782
- function defaultStopDecision(evals) {
783
- if (!evals.length) return {
784
- stop: false,
785
- pass: false,
786
- reason: "no evals yet"
787
- };
788
- return allCriticalPassed(evals) ? {
789
- stop: true,
790
- pass: true,
791
- reason: "all critical evals passed",
792
- score: averageScore(evals)
793
- } : {
794
- stop: false,
795
- pass: false,
796
- reason: "critical evals still failing",
797
- score: averageScore(evals)
798
- };
799
- }
800
- function averageScore(evals) {
801
- const scored = evals.map((result) => result.score).filter((score) => typeof score === "number");
802
- if (!scored.length) return void 0;
803
- return Math.round(scored.reduce((sum, score) => sum + score, 0) / scored.length * 1e3) / 1e3;
804
- }
805
- function budgetStopDecision(budget, spentCostUsd) {
806
- if (budget.maxCostUsd !== void 0 && spentCostUsd >= budget.maxCostUsd) return {
807
- stop: true,
808
- reason: `budget exhausted: maxCostUsd=${budget.maxCostUsd}`
809
- };
810
- return {
811
- stop: false,
812
- reason: ""
813
- };
814
- }
815
- async function recordCostBudget(emitter, budget, spentCostUsd, handle, runtimeErrors, stepIndex) {
816
- if (!emitter || budget.maxCostUsd === void 0) return;
817
- const maxCostUsd = budget.maxCostUsd;
818
- await runTrace(runtimeErrors, stepIndex, () => emitter.recordBudget({
819
- dimension: "usd",
820
- limit: maxCostUsd,
821
- consumed: spentCostUsd,
822
- remaining: Math.max(0, maxCostUsd - spentCostUsd),
823
- breached: spentCostUsd >= maxCostUsd,
824
- spanId: handle?.span.spanId
825
- }));
826
- }
827
- async function recordEvalSpans(emitter, evals, phase, runtimeErrors, stepIndex, targetSpanId) {
828
- if (!emitter) return;
829
- for (const result of evals) await runTrace(runtimeErrors, stepIndex, () => emitter.recordJudge({
830
- judgeId: result.objective ? "objective-validator" : "subjective-judge",
831
- targetSpanId: targetSpanId ?? emitter.runId,
832
- name: `control-eval/${result.id}`,
833
- dimension: result.id,
834
- score: typeof result.score === "number" ? result.score : result.passed ? 1 : 0,
835
- rationale: result.detail,
836
- evidence: result.evidence,
837
- attributes: {
838
- phase,
839
- passed: result.passed,
840
- severity: result.severity,
841
- objective: result.objective
842
- }
843
- }));
844
- }
845
- async function runOnStep(onStep, step, runtimeErrors) {
846
- if (!onStep) return;
847
- try {
848
- await onStep(step);
849
- } catch (err) {
850
- runtimeErrors.push(runtimeError("on-step", step.index, err));
851
- }
852
- }
853
- async function runTrace(runtimeErrors, stepIndex, write) {
854
- try {
855
- return await write();
856
- } catch (err) {
857
- runtimeErrors.push(runtimeError("trace", stepIndex, err));
858
- return;
859
- }
860
- }
861
- function fingerprintState(state, policies) {
862
- if (policies?.stateFingerprint) return policies.stateFingerprint(state);
863
- return stableFingerprint(state);
864
- }
865
- function fingerprintAction(action, policies) {
866
- if (policies?.actionFingerprint) return policies.actionFingerprint(action);
867
- return stableFingerprint(action);
868
- }
869
- function stableFingerprint(value) {
870
- if (typeof value === "string") return value;
871
- if (typeof value === "number" || typeof value === "boolean" || value == null) return String(value);
872
- try {
873
- return JSON.stringify(sortForFingerprint(value));
874
- } catch {
875
- return String(value);
876
- }
877
- }
878
- function sortForFingerprint(value) {
879
- if (Array.isArray(value)) return value.map(sortForFingerprint);
880
- if (!value || typeof value !== "object") return value;
881
- const record = value;
882
- const sorted = {};
883
- for (const key of Object.keys(record).sort()) sorted[key] = sortForFingerprint(record[key]);
884
- return sorted;
885
- }
886
- function abortReason(signal) {
887
- const reason = signal.reason;
888
- if (reason instanceof Error) return reason.message;
889
- return reason ? String(reason) : "aborted";
890
- }
891
- function runtimeError(phase, stepIndex, err) {
892
- return {
893
- phase,
894
- stepIndex,
895
- message: err instanceof Error ? err.message : String(err)
896
- };
897
- }
898
- async function finish(emitter, result) {
899
- await runTrace(result.runtimeErrors, result.steps.length, () => emitter?.endRun({
900
- pass: result.pass,
901
- score: result.score ?? averageScore(result.finalEvals),
902
- failureClass: result.failureClass,
903
- notes: result.reason
904
- }));
905
- return result;
906
- }
907
- //#endregion
908
- //#region src/run-evidence.ts
909
- /**
910
- * Project a completed control-loop run into the strict RunRecord shape used by
911
- * release gates, optimizer tables, and research reports.
912
- *
913
- * The control loop owns live execution evidence. The caller still supplies the
914
- * experiment-cell metadata because prompt/config hashes, split assignment,
915
- * model snapshot, and commit SHA are product/harness concerns.
916
- */
917
- function controlRunToRunRecord(run, options) {
918
- const score = finiteScore(options.score) ?? finiteScore(run.score) ?? scoreFromEvals(run.finalEvals);
919
- const outcome = options.splitTag === "holdout" ? {
920
- ...score !== void 0 ? { holdoutScore: score } : {},
921
- raw: normalizeRawMetrics(options.raw, run, score)
922
- } : {
923
- ...score !== void 0 ? { searchScore: score } : {},
924
- raw: normalizeRawMetrics(options.raw, run, score)
925
- };
926
- const terminalOutcome = run.stoppedBy === "abort" ? "cancelled" : run.stoppedBy === "runtime-error" ? "failed" : run.completed ? "succeeded" : "incomplete";
927
- const costUsd = options.costProvenance.kind === "uncaptured" ? null : options.costProvenance.usd;
928
- if (costUsd !== null && costUsd !== run.spentCostUsd) throw new Error(`cost provenance amount ${costUsd} does not match control run spend ${run.spentCostUsd}`);
929
- return validateRunRecord({
930
- runId: options.runId ?? run.runId ?? `control:${options.experimentId}:${options.candidateId}:${options.seed}:${options.splitTag}`,
931
- experimentId: options.experimentId,
932
- candidateId: options.candidateId,
933
- seed: options.seed,
934
- model: options.model,
935
- promptHash: options.promptHash,
936
- configHash: options.configHash,
937
- commitSha: options.commitSha,
938
- wallMs: run.wallMs,
939
- ...options.queueMs !== void 0 ? { queueMs: options.queueMs } : {},
940
- costUsd,
941
- costProvenance: options.costProvenance,
942
- tokenUsage: options.tokenUsage,
943
- terminalOutcome,
944
- ...terminalOutcome !== "succeeded" ? { terminalFailureReason: run.reason } : {},
945
- ...options.judgeMetadata ? { judgeMetadata: options.judgeMetadata } : {},
946
- outcome,
947
- ...options.failureClass !== void 0 ? { failureClass: options.failureClass } : {},
948
- ...options.failureMode !== void 0 ? { failureMode: options.failureMode } : {},
949
- splitTag: options.splitTag,
950
- scenarioId: options.scenarioId
951
- });
952
- }
953
- function scoreFromEvals(evals) {
954
- const scores = evals.map((e) => e.score).filter((score) => typeof score === "number" && Number.isFinite(score));
955
- if (scores.length === 0) return void 0;
956
- return clampScore(scores.reduce((sum, score) => sum + score, 0) / scores.length);
957
- }
958
- function normalizeRawMetrics(raw, run, score) {
959
- const normalizedRaw = finiteOnly(raw ?? {});
960
- delete normalizedRaw.score;
961
- return {
962
- ...normalizedRaw,
963
- ...score !== void 0 ? { score } : {},
964
- pass: run.pass ? 1 : 0,
965
- completed: run.completed ? 1 : 0,
966
- steps: run.steps.length,
967
- runtimeErrors: run.runtimeErrors.length,
968
- execution_error_count: executionErrorCount(run)
969
- };
970
- }
971
- function executionErrorCount(run) {
972
- const thrownActionSteps = new Set(run.runtimeErrors.filter((error) => error.phase === "act").map((error) => error.stepIndex));
973
- const failedActionsWithoutRuntimeError = run.steps.filter((step) => step.actionOutcome?.ok === false && !thrownActionSteps.has(step.index)).length;
974
- return run.runtimeErrors.length + failedActionsWithoutRuntimeError;
975
- }
976
- function finiteOnly(values) {
977
- const out = {};
978
- for (const [key, value] of Object.entries(values)) if (Number.isFinite(value)) out[key] = value;
979
- return out;
980
- }
981
- function finiteScore(value) {
982
- return value !== void 0 && Number.isFinite(value) ? clampScore(value) : void 0;
983
- }
984
- function clampScore(value) {
985
- return Math.max(0, Math.min(1, value));
986
- }
987
- //#endregion
988
- //#region src/propose-review.ts
989
- /**
990
- * Propose / Verify / Review — the core multi-shot primitive.
991
- *
992
- * shot N: propose(state, priorReview) → new state
993
- * verify(state) → pass/fail, optional layers
994
- * review(state, verification, memory) → observations + next-shot
995
- * instruction + shouldContinue
996
- * memory.append(entry)
997
- *
998
- * Roles are strictly separated:
999
- *
1000
- * - The WORKER is whatever the caller wraps in `propose`. It is
1001
- * stateful — caller owns its resume/session mechanism.
1002
- * - The VERIFIER grades the state. It produces the ground truth.
1003
- * The reviewer cannot overturn or downgrade a verification layer.
1004
- * - The REVIEWER is stateless per call. Its continuity is the
1005
- * `ReviewMemoryStore` — durable JSONL by default, or any store
1006
- * implementing the interface. It reads memory + trace summary +
1007
- * verification and directs the NEXT proposer shot.
1008
- *
1009
- * This shape is load-bearing. The reviewer never grades; the verifier
1010
- * never directs. Two processes, two prompts, two concerns — which is
1011
- * what keeps the loop from confirmation-biasing itself into "all
1012
- * passed" when it didn't.
1013
- *
1014
- * Short-circuits and soft-fails are both first-class:
1015
- * - verify.pass === true → reviewer LLM call is skipped, memory
1016
- * records a success entry, loop exits.
1017
- * - review throws → the shot still counts; the loop uses the
1018
- * last-known instruction (or `fallbackInstruction`) for the next
1019
- * propose call. A transient reviewer failure must NEVER abort a
1020
- * valid arc.
1021
- *
1022
- * Composable: `propose` itself can be another `runProposeReview` call.
1023
- * That's the dogfooding path — a harness built on this primitive is in
1024
- * turn evaluable by it.
1025
- */
1026
- function inMemoryReviewStore(initial = []) {
1027
- const entries = [...initial];
1028
- return {
1029
- async load() {
1030
- return [...entries];
1031
- },
1032
- async append(entry) {
1033
- entries.push(entry);
1034
- }
1035
- };
1036
- }
1037
- function jsonlReviewStore(path) {
1038
- return {
1039
- async load() {
1040
- if (!existsSync(path)) return [];
1041
- const raw = readFileSync(path, "utf8");
1042
- const out = [];
1043
- for (const line of raw.split("\n")) {
1044
- const trimmed = line.trim();
1045
- if (!trimmed) continue;
1046
- try {
1047
- out.push(JSON.parse(trimmed));
1048
- } catch {}
1049
- }
1050
- return out;
1051
- },
1052
- async append(entry) {
1053
- mkdirSync(dirname(path), { recursive: true });
1054
- appendFileSync(path, `${JSON.stringify(entry)}\n`);
1055
- }
1056
- };
1057
- }
1058
- const DEFAULT_FALLBACK_INSTRUCTION$1 = "Inspect the verification failures above. Fix the critical issues first, then the major ones. Do not restate the failures — act on them.";
1059
- async function runProposeReview(config) {
1060
- const maxShots = config.maxShots ?? 10;
1061
- const maxWallMs = config.maxWallMs ?? 600 * 1e3;
1062
- const confidenceFloor = config.confidenceFloor ?? .3;
1063
- const confidenceFloorWindow = config.confidenceFloorWindow ?? 2;
1064
- const memory = config.memory ?? inMemoryReviewStore();
1065
- const fallbackInstruction = config.fallbackInstruction ?? DEFAULT_FALLBACK_INSTRUCTION$1;
1066
- const emitter = config.store ? new TraceEmitter(config.store) : null;
1067
- if (emitter) await emitter.startRun({
1068
- scenarioId: config.scenarioId ?? "propose-review",
1069
- projectId: config.projectId,
1070
- variantId: config.variantId,
1071
- layer: "meta",
1072
- tags: {
1073
- goal: config.goal.slice(0, 120),
1074
- maxShots: String(maxShots)
1075
- }
1076
- });
1077
- const abort = new AbortController();
1078
- const wallStart = Date.now();
1079
- const wallTimer = setTimeout(() => abort.abort(/* @__PURE__ */ new Error("propose-review wall timeout")), maxWallMs);
1080
- const shots = [];
1081
- let state = config.initialState;
1082
- let priorReview = null;
1083
- let lastVerification = { pass: false };
1084
- let failureClass;
1085
- let completed = false;
1086
- let lowConfidenceStreak = 0;
1087
- try {
1088
- for (let shot = 1; shot <= maxShots; shot++) {
1089
- if (abort.signal.aborted) {
1090
- failureClass = "timeout";
1091
- break;
1092
- }
1093
- const shotStart = Date.now();
1094
- const shotHandle = emitter ? await emitter.span({
1095
- kind: "tool",
1096
- name: `shot-${shot}`
1097
- }) : null;
1098
- let proposeOut;
1099
- try {
1100
- proposeOut = await config.propose({
1101
- shot,
1102
- goal: config.goal,
1103
- state,
1104
- priorReview,
1105
- abortSignal: abort.signal,
1106
- emitter: emitter ?? void 0
1107
- });
1108
- } catch (err) {
1109
- await shotHandle?.fail(err instanceof Error ? err : String(err));
1110
- failureClass = "unknown";
1111
- throw err;
1112
- }
1113
- state = proposeOut.state;
1114
- const traceSummary = proposeOut.traceSummary;
1115
- let verification;
1116
- try {
1117
- verification = await config.verify(state);
1118
- } catch (err) {
1119
- await shotHandle?.fail(err instanceof Error ? err : String(err));
1120
- failureClass = "unknown";
1121
- throw err;
1122
- }
1123
- lastVerification = verification;
1124
- const memorySnapshot = await memory.load();
1125
- const verificationDigest = {
1126
- pass: verification.pass,
1127
- score: verification.score,
1128
- failingLayers: verification.failingLayers ?? []
1129
- };
1130
- let review;
1131
- let reviewAvailable = true;
1132
- let reviewError;
1133
- if (verification.pass) review = {
1134
- observations: "verification passed — skipping reviewer LLM call",
1135
- diagnosis: "no failures to diagnose",
1136
- nextShotInstruction: "(done)",
1137
- shouldContinue: false,
1138
- confidence: 1
1139
- };
1140
- else try {
1141
- review = await config.review({
1142
- shot,
1143
- goal: config.goal,
1144
- state,
1145
- verification,
1146
- traceSummary,
1147
- memory: memorySnapshot
1148
- });
1149
- review = coerceReview(review);
1150
- } catch (err) {
1151
- reviewAvailable = false;
1152
- reviewError = err instanceof Error ? err.message : String(err);
1153
- const lastInstruction = memorySnapshot.length > 0 ? memorySnapshot[memorySnapshot.length - 1].nextShotInstruction : fallbackInstruction;
1154
- review = {
1155
- observations: "(reviewer unavailable — using last-known instruction)",
1156
- diagnosis: reviewError,
1157
- nextShotInstruction: lastInstruction,
1158
- shouldContinue: true,
1159
- confidence: .3
1160
- };
1161
- }
1162
- const entry = {
1163
- shot,
1164
- timestamp: Date.now(),
1165
- ...review,
1166
- verification: verificationDigest
1167
- };
1168
- await memory.append(entry);
1169
- const shotRecord = {
1170
- shot,
1171
- state,
1172
- verification,
1173
- traceSummary,
1174
- review,
1175
- reviewAvailable,
1176
- reviewError,
1177
- durationMs: Date.now() - shotStart
1178
- };
1179
- shots.push(shotRecord);
1180
- await shotHandle?.end({ attributes: {
1181
- verificationPass: verification.pass,
1182
- verificationScore: verification.score ?? null,
1183
- reviewShouldContinue: review.shouldContinue,
1184
- reviewConfidence: review.confidence,
1185
- reviewAvailable
1186
- } });
1187
- if (verification.pass) {
1188
- completed = true;
1189
- break;
1190
- }
1191
- if (!review.shouldContinue) break;
1192
- if (confidenceFloorWindow > 0 && review.confidence <= confidenceFloor) {
1193
- lowConfidenceStreak += 1;
1194
- if (lowConfidenceStreak >= confidenceFloorWindow) break;
1195
- } else lowConfidenceStreak = 0;
1196
- priorReview = review;
1197
- }
1198
- if (!completed && !failureClass) failureClass = shots.length >= maxShots ? "budget_exceeded" : "unknown";
1199
- } finally {
1200
- clearTimeout(wallTimer);
1201
- }
1202
- const score = lastVerification.pass ? 1 : typeof lastVerification.score === "number" ? lastVerification.score : 0;
1203
- if (emitter) await emitter.endRun({
1204
- pass: completed,
1205
- score,
1206
- failureClass,
1207
- notes: `${shots.length} shot(s); final pass=${lastVerification.pass}`
1208
- });
1209
- return {
1210
- runId: emitter?.runId ?? null,
1211
- completed,
1212
- shots,
1213
- finalState: state,
1214
- finalVerification: lastVerification,
1215
- failureClass,
1216
- wallMs: Date.now() - wallStart,
1217
- score
1218
- };
1219
- }
1220
- const REVIEWER_SYSTEM_PROMPT = `You are a senior reviewer directing a multi-shot build loop.
1221
- You do NOT grade — the verifier already did. Your job is to direct the worker's next shot.
1222
- You are blind to the worker's inner monologue. You see what it DID, not what it thought.
1223
- Return STRICT JSON matching the schema. No prose outside the JSON.`;
1224
- function createLlmReviewer(cfg) {
1225
- const renderState = cfg.renderState ?? ((s) => safeJson(s));
1226
- const renderTraceSummary = cfg.renderTraceSummary ?? ((s) => s === void 0 ? "(none)" : safeJson(s));
1227
- const system = cfg.systemPromptAddendum ? `${REVIEWER_SYSTEM_PROMPT}\n\n${cfg.systemPromptAddendum}` : REVIEWER_SYSTEM_PROMPT;
1228
- return async (input) => {
1229
- const memoryBlock = input.memory.length === 0 ? "(no prior shots — this is shot 1)" : input.memory.map((m) => [
1230
- `shot ${m.shot} — verification.pass=${m.verification.pass}` + (typeof m.verification.score === "number" ? ` score=${m.verification.score.toFixed(2)}` : "") + ` confidence=${m.confidence.toFixed(2)} failing=[${(m.verification.failingLayers ?? []).join(",")}]`,
1231
- ` observations: ${m.observations.slice(0, 400)}`,
1232
- ` diagnosis: ${m.diagnosis.slice(0, 400)}`,
1233
- ` instruction given: ${m.nextShotInstruction.slice(0, 400)}`
1234
- ].join("\n")).join("\n\n");
1235
- const user = [
1236
- `=== GOAL ===`,
1237
- input.goal,
1238
- ``,
1239
- `=== SHOT NUMBER ===`,
1240
- String(input.shot),
1241
- ``,
1242
- `=== CURRENT STATE ===`,
1243
- renderState(input.state),
1244
- ``,
1245
- `=== TRACE SUMMARY ===`,
1246
- renderTraceSummary(input.traceSummary),
1247
- ``,
1248
- `=== VERIFICATION ===`,
1249
- summarizeVerification(input.verification),
1250
- ``,
1251
- `=== REVIEWER MEMORY (prior shots) ===`,
1252
- memoryBlock,
1253
- ``,
1254
- `=== YOUR TASK ===`,
1255
- `Return STRICT JSON:`,
1256
- `{`,
1257
- ` "observations": string (20..2000 chars, first-person worker behavior — quote counts, errors, loops)`,
1258
- ` "diagnosis": string (20..1500 chars, root cause, NOT a restatement of verification)`,
1259
- ` "nextShotInstruction": string (40..3000 chars, concrete directive to the worker)`,
1260
- ` "shouldContinue": boolean (false if verification.pass, or if thrashing, or unachievable)`,
1261
- ` "confidence": number in [0,1]`,
1262
- `}`
1263
- ].join("\n");
1264
- return coerceReview(await cfg.callJson({
1265
- system,
1266
- user
1267
- }));
1268
- };
1269
- }
1270
- function coerceReview(raw) {
1271
- if (!raw || typeof raw !== "object") throw new Error("reviewer returned non-object");
1272
- const observations = typeof raw.observations === "string" ? raw.observations : "";
1273
- const diagnosis = typeof raw.diagnosis === "string" ? raw.diagnosis : "";
1274
- const nextShotInstruction = typeof raw.nextShotInstruction === "string" ? raw.nextShotInstruction : "";
1275
- if (!observations || !diagnosis || !nextShotInstruction) throw new Error("reviewer missing required string fields");
1276
- if (typeof raw.shouldContinue !== "boolean") throw new Error("reviewer missing shouldContinue boolean");
1277
- const confidenceRaw = Number(raw.confidence);
1278
- if (!Number.isFinite(confidenceRaw)) throw new Error("reviewer confidence not finite");
1279
- return {
1280
- observations,
1281
- diagnosis,
1282
- nextShotInstruction,
1283
- shouldContinue: raw.shouldContinue,
1284
- confidence: Math.max(0, Math.min(1, confidenceRaw))
1285
- };
1286
- }
1287
- function summarizeVerification(v) {
1288
- return `pass=${v.pass}` + (typeof v.score === "number" ? ` score=${v.score.toFixed(3)}` : "") + (v.failingLayers && v.failingLayers.length > 0 ? ` failing=[${v.failingLayers.join(", ")}]` : "") + (v.details === void 0 ? "" : `\n${safeJson(v.details).slice(0, 1500)}`);
1289
- }
1290
- function safeJson(x) {
1291
- try {
1292
- return JSON.stringify(x, null, 2);
1293
- } catch {
1294
- return String(x);
1295
- }
1296
- }
1297
- //#endregion
1298
- //#region src/propose-review-control.ts
1299
- const DEFAULT_FALLBACK_INSTRUCTION = "Inspect the verification failures above. Fix the critical issues first, then the major ones. Do not restate the failures — act on them.";
1300
- async function runProposeReviewAsControlLoop(config) {
1301
- const maxShots = config.maxShots ?? 10;
1302
- const confidenceFloor = config.confidenceFloor ?? .3;
1303
- const confidenceFloorWindow = config.confidenceFloorWindow ?? 2;
1304
- const memory = config.memory ?? inMemoryReviewStore();
1305
- const fallbackInstruction = config.fallbackInstruction ?? DEFAULT_FALLBACK_INSTRUCTION;
1306
- const failureClassFromVerification = config.failureClassFromVerification ?? controlFailureClassFromVerification;
1307
- let lowConfidenceStreak = 0;
1308
- let current = {
1309
- shot: 0,
1310
- state: config.initialState,
1311
- priorReview: null,
1312
- verification: { pass: false },
1313
- memory: await memory.load(),
1314
- completed: false,
1315
- reviewAvailable: false
1316
- };
1317
- return runAgentControlLoop({
1318
- intent: config.goal,
1319
- budget: {
1320
- maxSteps: maxShots,
1321
- maxWallMs: config.maxWallMs
1322
- },
1323
- store: config.store,
1324
- scenarioId: config.scenarioId ?? "propose-review-control",
1325
- projectId: config.projectId,
1326
- variantId: config.variantId,
1327
- actionFailure: config.actionFailure ?? "stop",
1328
- observe: () => current,
1329
- validate: ({ state }) => [objectiveEval({
1330
- id: "verification",
1331
- passed: state.verification.pass,
1332
- score: state.verification.score,
1333
- severity: "critical",
1334
- detail: state.verification.pass ? "verification passed" : `verification failed${state.verification.failingLayers?.length ? `: ${state.verification.failingLayers.join(", ")}` : ""}`
1335
- })],
1336
- shouldStop: ({ state }) => {
1337
- if (state.verification.pass) return {
1338
- stop: true,
1339
- pass: true,
1340
- reason: "verification passed",
1341
- score: state.verification.score
1342
- };
1343
- if (state.completed) return {
1344
- stop: true,
1345
- pass: false,
1346
- reason: "reviewer stopped continuation",
1347
- score: state.verification.score,
1348
- failureClass: failureClassFromVerification(state.verification)
1349
- };
1350
- return {
1351
- stop: false,
1352
- pass: false,
1353
- reason: "verification still failing",
1354
- score: state.verification.score
1355
- };
1356
- },
1357
- decide: ({ state }) => ({
1358
- type: "continue",
1359
- action: {
1360
- type: "propose-review-shot",
1361
- shot: state.shot + 1
1362
- },
1363
- reason: state.priorReview?.nextShotInstruction ?? fallbackInstruction
1364
- }),
1365
- act: async (action, ctx) => {
1366
- const shot = action.shot;
1367
- const proposeOut = await config.propose({
1368
- shot,
1369
- goal: config.goal,
1370
- state: current.state,
1371
- priorReview: current.priorReview,
1372
- abortSignal: ctx.abortSignal,
1373
- emitter: ctx.emitter
1374
- });
1375
- const nextState = proposeOut.state;
1376
- const verification = await config.verify(nextState);
1377
- let review = null;
1378
- let reviewAvailable = false;
1379
- let reviewError;
1380
- let shouldContinue = !verification.pass;
1381
- if (!verification.pass) try {
1382
- review = await config.review({
1383
- shot,
1384
- goal: config.goal,
1385
- state: nextState,
1386
- verification,
1387
- traceSummary: proposeOut.traceSummary,
1388
- memory: await memory.load()
1389
- });
1390
- reviewAvailable = true;
1391
- shouldContinue = review.shouldContinue;
1392
- lowConfidenceStreak = review.confidence <= confidenceFloor ? lowConfidenceStreak + 1 : 0;
1393
- if (confidenceFloorWindow > 0 && lowConfidenceStreak >= confidenceFloorWindow) shouldContinue = false;
1394
- } catch (err) {
1395
- reviewError = err instanceof Error ? err.message : String(err);
1396
- review = current.priorReview ?? {
1397
- observations: "Reviewer unavailable.",
1398
- diagnosis: reviewError,
1399
- nextShotInstruction: fallbackInstruction,
1400
- shouldContinue: true,
1401
- confidence: 0
1402
- };
1403
- shouldContinue = true;
1404
- }
1405
- else review = {
1406
- observations: "Verification passed.",
1407
- diagnosis: "No further revision needed.",
1408
- nextShotInstruction: "",
1409
- shouldContinue: false,
1410
- confidence: 1
1411
- };
1412
- const entry = {
1413
- ...review ?? {
1414
- observations: "No review.",
1415
- diagnosis: "",
1416
- nextShotInstruction: fallbackInstruction,
1417
- shouldContinue,
1418
- confidence: 0
1419
- },
1420
- shot,
1421
- timestamp: Date.now(),
1422
- verification: {
1423
- pass: verification.pass,
1424
- score: verification.score,
1425
- failingLayers: verification.failingLayers
1426
- }
1427
- };
1428
- await memory.append(entry);
1429
- current = {
1430
- shot,
1431
- state: nextState,
1432
- priorReview: review,
1433
- verification,
1434
- traceSummary: proposeOut.traceSummary,
1435
- memory: await memory.load(),
1436
- completed: verification.pass || !shouldContinue,
1437
- reviewAvailable,
1438
- reviewError
1439
- };
1440
- return {
1441
- state: nextState,
1442
- verification,
1443
- traceSummary: proposeOut.traceSummary,
1444
- review,
1445
- reviewAvailable,
1446
- reviewError
1447
- };
1448
- }
1449
- });
1450
- }
1451
- function controlFailureClassFromVerification(verification) {
1452
- if (verification.pass) return void 0;
1453
- return verification.failingLayers?.length ? "instruction_following" : "unknown";
1454
- }
1455
- //#endregion
1456
- export { observeAll as _, jsonlReviewStore as a, scoreFromEvals as c, runAgentControlLoop as d, stopOnNoProgress as f, noProgressDetector as g, errorStreakDetector as h, inMemoryReviewStore as i, allCriticalPassed as l, subjectiveEval as m, runProposeReviewAsControlLoop as n, runProposeReview as o, stopOnRepeatedAction as p, createLlmReviewer as r, controlRunToRunRecord as s, controlFailureClassFromVerification as t, objectiveEval as u, repeatedActionDetector as v, evaluateActionPolicy as y };
1457
-
1458
- //# sourceMappingURL=propose-review-control-dSNPjFUH.js.map