@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,639 +1,26 @@
1
- import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
- import { t as canonicalize } from "./pre-registration-DakwTRXk.js";
1
+ import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
2
+ import { c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, s as agentProfileHash, t as extractProducedState } from "./produced-state-DU79a81m.js";
3
3
  import { buildAgentProfileCell } from "./profile-cell.js";
4
- import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
5
- import { f as maximumChargeForLlmRequest, h as ModelSubstitutionError, l as costReceiptFromLlm, u as costReceiptFromLlmError, y as assertServedModel } from "./llm-client-DzvMUsS_.js";
6
- import { r as canonicalString } from "./canonical-D011XM8r.js";
7
- import { c as SearchLedgerError, l as SearchLedgerIntegrityError, o as SEARCH_LEDGER_FILE_CONTEXT, s as SearchLedgerConflictError } from "./single-run-lock-DFWHEB09.js";
8
- import { r as contentHash } from "./verdict-cache-BCcOh0kF.js";
9
- import { p as mapConcurrent, t as FileLedgerJournal } from "./ledger-core-DXZIqu17.js";
10
- import { Ct as JudgeParseError, H as planCampaignRun, P as surfaceContentHash, St as summarizeBackendIntegrity, U as runCampaign, _t as recoverTruncatedJson, bt as assertRealBackend, h as labelTrustRank, k as assertCodeSurfaceIdentity } from "./skillopt-optimization-method-CQdVeM8k.js";
11
- import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
- import { t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
13
- import { o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BmSPWXJR.js";
14
- import { l as pairHoldout } from "./promotion-policy-CrLrmys8.js";
15
- import { a as scoreAnalystFindings } from "./benchmark-CWeqGl7x.js";
16
- import { l as campaignCellToRunRecord } from "./reward-hacking-BDToousL.js";
17
- import { z } from "zod";
4
+ import { t as comparePairedArms } from "./paired-arms-D-XRF_fy.js";
5
+ import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
6
+ import { o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BvHPVS-i.js";
7
+ import { n as contentHash } from "./verdict-cache-mZf5FEiY.js";
8
+ import { K as runCampaign, Q as summarizeBackendIntegrity, Z as assertRealBackend, h as labelTrustRank, k as surfaceContentHash, q as planCampaignRun, w as assertCodeSurfaceIdentity } from "./llm-judge-dZ8P6nGI.js";
9
+ import { l as campaignCellToRunRecord } from "./reward-hacking-DNgjilrV.js";
10
+ import { t as CostAccountingIncompleteError } from "./cost-ledger-BSe92yAV.js";
11
+ import { t as FileLedgerJournal, u as mapConcurrent } from "./ledger-core-BmZt19oQ.js";
12
+ import { r as canonicalString } from "./canonical-D-XsTQ6_.js";
13
+ import { C as SEARCH_LEDGER_FILE_CONTEXT, E as SearchLedgerIntegrityError, T as SearchLedgerError, w as SearchLedgerConflictError } from "./external-optimizer-subprocess-DrJ9hR8u.js";
14
+ import { o as pairHoldout } from "./power-preflight-DEw-uC7q.js";
15
+ import { a as scoreAnalystFindings } from "./benchmark-BhT16ep9.js";
16
+ import "./external-optimizer-process-BTiNB-RH.js";
17
+ import "./skillopt-optimization-method-jjdnc3YK.js";
18
+ import { createHash } from "node:crypto";
18
19
  import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, readdirSync, readlinkSync, realpathSync, rmSync, statSync, writeFileSync } from "node:fs";
20
+ import { z } from "zod";
19
21
  import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
20
- import { createHash, randomUUID } from "node:crypto";
21
- import { harnessSupportsModel } from "@tangle-network/agent-interface";
22
- import { execFileSync } from "node:child_process";
23
22
  import { devNull, tmpdir } from "node:os";
24
- //#region src/completion-verifier.ts
25
- /**
26
- * Completion verifier — the task-completion oracle.
27
- *
28
- * Answers the only eval question that is not a proxy: did the agent actually
29
- * COMPLETE the task — produce every required deliverable, persisted and
30
- * correct — rather than describe what should be done. A fluent transcript
31
- * that never produces the artifact scores zero here.
32
- *
33
- * Per requirement, a two-stage check:
34
- * 1. Structural — a produced item (vault artifact / approved proposal /
35
- * tool call) of the right kind is matched against the requirement and
36
- * carries non-empty content. Deterministic; no LLM.
37
- * 2. Correctness — only if structurally present AND the matched item
38
- * carries content, one targeted check decides whether that item
39
- * actually fulfils the requirement. A hallucinated artifact fails here;
40
- * an absent one already failed stage 1.
41
- *
42
- * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
43
- * checker failures — are excluded from the denominator, never scored as
44
- * zeros). Quality dimensions are meaningless on an incomplete task — callers
45
- * gate on `fullyComplete` / `completionRate` before scoring quality.
46
- */
47
- /**
48
- * Construct a `CompletionVerdict` from the per-requirement checks, deriving
49
- * `completionRate` / `fullyComplete` and the spine fields (`valid` =
50
- * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
51
- * requirements — a verdict over nothing is a misconfiguration, mirroring
52
- * `verifyCompletion`'s gold-spec guard.
53
- */
54
- function completionVerdict(input) {
55
- if (input.requirements.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`);
56
- const measurable = input.requirements.filter((r) => !r.unmeasured);
57
- const unmeasuredCount = input.requirements.length - measurable.length;
58
- if (measurable.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? "unknown reason"})`);
59
- const satisfiedCount = measurable.filter((r) => r.satisfied).length;
60
- const completionRate = satisfiedCount / measurable.length;
61
- const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length;
62
- return {
63
- taskId: input.taskId,
64
- requirements: input.requirements,
65
- completionRate,
66
- fullyComplete,
67
- unmeasuredCount,
68
- valid: fullyComplete,
69
- score: completionRate
70
- };
71
- }
72
- const STOPWORDS = /* @__PURE__ */ new Set([
73
- "the",
74
- "a",
75
- "an",
76
- "of",
77
- "for",
78
- "and",
79
- "or",
80
- "to",
81
- "in",
82
- "on",
83
- "with",
84
- "by"
85
- ]);
86
- const REQUIREMENT_FORM_STOPWORDS = /* @__PURE__ */ new Set([
87
- "generated",
88
- "generate",
89
- "view",
90
- "render",
91
- "rendered",
92
- "persisted",
93
- "persist",
94
- "artifact",
95
- "file",
96
- "document",
97
- "note",
98
- "proposal",
99
- "deliverable",
100
- "output",
101
- "created",
102
- "create",
103
- "produce",
104
- "produced",
105
- "flag"
106
- ]);
107
- const MATCH_THRESHOLD = .5;
108
- const MIN_CONTENT_CHARS = 50;
109
- function tokens(s, extraStop) {
110
- return new Set(s.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)));
111
- }
112
- /**
113
- * Recall of the requirement's tokens within a candidate's identifying text.
114
- * Recall, not Jaccard — a candidate's path/id legitimately carries extra
115
- * tokens the requirement does not name. The requirement side drops
116
- * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.
117
- */
118
- function tokenRecall(requirementText, candidateText) {
119
- const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS);
120
- if (req.size === 0) return 0;
121
- const cand = tokens(candidateText);
122
- let hit = 0;
123
- for (const t of req) if (cand.has(t)) hit++;
124
- return hit / req.size;
125
- }
126
- function artifactCandidates(req, reqIndex, artifacts) {
127
- const reqText = `${req.title} ${req.category ?? ""}`;
128
- const out = [];
129
- artifacts.forEach((a, i) => {
130
- if ((a.content ?? "").trim().length < MIN_CONTENT_CHARS) return;
131
- let score = tokenRecall(reqText, `${a.path ?? ""} ${a.kind} ${(a.content ?? "").slice(0, 4e3)}`);
132
- if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) score = Math.max(score, 1);
133
- if (score < MATCH_THRESHOLD) return;
134
- out.push({
135
- reqIndex,
136
- itemKey: `artifact:${i}`,
137
- score,
138
- evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,
139
- content: a.content ?? null
140
- });
141
- });
142
- return out;
143
- }
144
- function proposalCandidates(req, reqIndex, proposals) {
145
- const reqText = `${req.title} ${req.category ?? ""}`;
146
- const out = [];
147
- for (const p of proposals) {
148
- if (p.status !== "approved") continue;
149
- const body = (p.content ?? "").trim();
150
- if (body.length < MIN_CONTENT_CHARS) continue;
151
- const score = tokenRecall(reqText, `${p.title} ${body}`);
152
- if (score < MATCH_THRESHOLD) continue;
153
- out.push({
154
- reqIndex,
155
- itemKey: `proposal:${p.id}`,
156
- score,
157
- evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,
158
- content: body
159
- });
160
- }
161
- return out;
162
- }
163
- function toolCallCandidates(req, reqIndex, toolCalls) {
164
- const out = [];
165
- toolCalls.forEach((name, i) => {
166
- const score = tokenRecall(req.title, name);
167
- if (score < MATCH_THRESHOLD) return;
168
- out.push({
169
- reqIndex,
170
- itemKey: `tool:${i}`,
171
- score,
172
- evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,
173
- content: null
174
- });
175
- });
176
- return out;
177
- }
178
- /**
179
- * Verify whether a run completed the task. `checkCorrectness` is injected —
180
- * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
181
- *
182
- * Throws on a gold spec with no requirements: an eval task that requires
183
- * nothing is a misconfiguration, not a vacuously-complete task.
184
- */
185
- async function verifyCompletion(gold, state, checkCorrectness) {
186
- if (gold.requirements.length === 0) throw new Error(`verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`);
187
- const candidates = [];
188
- gold.requirements.forEach((req, i) => {
189
- const by = req.satisfiedBy ?? "any";
190
- if (by === "artifact" || by === "any") candidates.push(...artifactCandidates(req, i, state.artifacts));
191
- if (by === "proposal" || by === "any") candidates.push(...proposalCandidates(req, i, state.proposals));
192
- if (by === "tool-call" || by === "any") candidates.push(...toolCallCandidates(req, i, state.toolCalls));
193
- });
194
- candidates.sort((a, b) => b.score - a.score);
195
- const assigned = /* @__PURE__ */ new Map();
196
- const itemTaken = /* @__PURE__ */ new Set();
197
- for (const c of candidates) {
198
- if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue;
199
- assigned.set(c.reqIndex, c);
200
- itemTaken.add(c.itemKey);
201
- }
202
- const requirements = [];
203
- for (let i = 0; i < gold.requirements.length; i++) {
204
- const req = gold.requirements[i];
205
- const match = assigned.get(i);
206
- const evidence = [];
207
- let correct = null;
208
- let unmeasuredReason;
209
- if (match) {
210
- evidence.push(match.evidence);
211
- if (match.content !== null) try {
212
- const r = await checkCorrectness(req, match.content);
213
- correct = r.correct;
214
- evidence.push(`correctness: ${r.correct ? "pass" : "fail"} — ${r.reason}`);
215
- } catch (err) {
216
- unmeasuredReason = err instanceof JudgeParseError ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}` : `checker call failed: ${err instanceof Error ? err.message : String(err)}`;
217
- evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`);
218
- }
219
- else evidence.push("correctness: not assessed — matched item carries no content");
220
- } else {
221
- const by = req.satisfiedBy ?? "any";
222
- const kind = by === "any" ? "artifact/proposal/tool-call" : by;
223
- evidence.push(`no produced ${kind} matched this requirement`);
224
- }
225
- const structurallyPresent = match !== void 0;
226
- const unmeasured = unmeasuredReason !== void 0;
227
- const satisfied = structurallyPresent && !unmeasured && correct !== false;
228
- requirements.push({
229
- reqId: req.reqId,
230
- title: req.title,
231
- structurallyPresent,
232
- correct,
233
- satisfied,
234
- ...unmeasured ? {
235
- unmeasured: true,
236
- unmeasuredReason
237
- } : {},
238
- evidence
239
- });
240
- }
241
- return completionVerdict({
242
- taskId: gold.taskId,
243
- requirements
244
- });
245
- }
246
- /**
247
- * Parse the correctness checker's model response. Tolerates a response
248
- * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
249
- * verdict boolean usually lands in the first few tokens, so a recovered
250
- * prefix with a boolean `correct` is a real measurement, not a guess.
251
- * Fails loud (JudgeParseError) when no boolean verdict is recoverable.
252
- */
253
- function parseCorrectnessResponse(raw) {
254
- const readVerdict = (candidate) => {
255
- if (candidate === null || typeof candidate !== "object") return null;
256
- const { correct, reason } = candidate;
257
- if (typeof correct !== "boolean") return null;
258
- return {
259
- correct,
260
- reason: typeof reason === "string" ? reason : ""
261
- };
262
- };
263
- const match = raw.match(/\{[\s\S]*\}/);
264
- if (match) try {
265
- const strict = readVerdict(JSON.parse(match[0]));
266
- if (strict) return strict;
267
- } catch {}
268
- const start = raw.indexOf("{");
269
- if (start !== -1) {
270
- const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)));
271
- if (recovered) return recovered;
272
- }
273
- throw new JudgeParseError("correctness-checker", raw);
274
- }
275
- /**
276
- * Production `CorrectnessChecker` — one LLM call per matched artifact,
277
- * deterministic (temperature 0), structured JSON out. Judges fulfilment
278
- * only: a plan, a gesture, or a description of what should be done does not
279
- * fulfil a requirement — the artifact must BE the deliverable.
280
- */
281
- function createLlmCorrectnessChecker(chat, opts = {}) {
282
- const model = opts.model ?? "claude-sonnet-4-6";
283
- const maxContentChars = opts.maxContentChars ?? 8e3;
284
- const maxAttempts = opts.maxAttempts ?? 2;
285
- const costLedger = opts.costLedger ?? new CostLedger();
286
- const sink = opts.rawSink;
287
- const record = async (event) => {
288
- try {
289
- await sink?.record(event);
290
- } catch {}
291
- };
292
- return async (requirement, content) => {
293
- const request = {
294
- model,
295
- messages: [{
296
- role: "system",
297
- content: "You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}."
298
- }, {
299
- role: "user",
300
- content: `Requirement: ${requirement.title}\n${requirement.category ? `Category: ${requirement.category}\n` : ""}\nProduced artifact:\n${content.slice(0, maxContentChars)}`
301
- }],
302
- temperature: 0,
303
- maxTokens: 200
304
- };
305
- let lastErr;
306
- for (let attempt = 0; attempt < maxAttempts; attempt++) {
307
- const started = Date.now();
308
- await record({
309
- eventId: randomUUID(),
310
- provider: chat.transport,
311
- model,
312
- endpoint: "/chat",
313
- baseUrl: "",
314
- attemptIndex: attempt,
315
- direction: "request",
316
- timestamp: started,
317
- requestBody: request,
318
- redactedFields: []
319
- });
320
- try {
321
- const paid = await costLedger.runPaidCall({
322
- channel: "verifier",
323
- phase: opts.costPhase ?? "completion.correctness",
324
- actor: "correctness-checker",
325
- model,
326
- maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
327
- tags: {
328
- ...opts.costTags,
329
- requirementId: requirement.reqId,
330
- attempt: String(attempt)
331
- },
332
- signal: opts.signal,
333
- execute: (signal, callId) => chat.chat(request, {
334
- signal,
335
- idempotencyKey: callId
336
- }),
337
- receipt: costReceiptFromLlm,
338
- receiptFromError: costReceiptFromLlmError
339
- });
340
- if (!paid.succeeded) throw paid.error;
341
- const resp = paid.value;
342
- assertServedModel(model, resp.servedModel, {
343
- allowUnreported: true,
344
- context: `correctness checker for requirement ${requirement.reqId}`
345
- });
346
- const raw = resp.content;
347
- await record({
348
- eventId: randomUUID(),
349
- provider: chat.transport,
350
- model,
351
- endpoint: "/chat",
352
- baseUrl: "",
353
- attemptIndex: attempt,
354
- direction: "response",
355
- timestamp: Date.now(),
356
- durationMs: Date.now() - started,
357
- responseBody: resp,
358
- redactedFields: []
359
- });
360
- return parseCorrectnessResponse(raw);
361
- } catch (err) {
362
- lastErr = err;
363
- await record({
364
- eventId: randomUUID(),
365
- provider: chat.transport,
366
- model,
367
- endpoint: "/chat",
368
- baseUrl: "",
369
- attemptIndex: attempt,
370
- direction: "error",
371
- timestamp: Date.now(),
372
- durationMs: Date.now() - started,
373
- errorMessage: err instanceof Error ? err.message : String(err),
374
- redactedFields: []
375
- });
376
- if (err instanceof ModelSubstitutionError) throw err;
377
- }
378
- }
379
- throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
380
- };
381
- }
382
- /** Stopwords for requirement-title tokenization — drops the imperative verbs
383
- * ('review', 'update', …) common to deliverable titles so recall keys on the
384
- * substantive nouns, not the boilerplate ask. */
385
- const TITLE_STOPWORDS = /* @__PURE__ */ new Set([
386
- "the",
387
- "a",
388
- "an",
389
- "and",
390
- "or",
391
- "for",
392
- "to",
393
- "of",
394
- "in",
395
- "on",
396
- "with",
397
- "review",
398
- "update",
399
- "new",
400
- "proposed"
401
- ]);
402
- /**
403
- * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
404
- * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
405
- * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
406
- * of the requirement title's significant tokens. No network.
407
- *
408
- * Polarity-blind: token recall credits a negation that contains the
409
- * requirement's tokens ("I will NOT produce the comparison" recalls every token
410
- * of "produce the comparison"). The structural match stage is ALSO lexical, so
411
- * pairing the two collapses to a single gameable gate. Use this only as an
412
- * opt-in structural pre-filter or for tasks whose requirements have no polarity
413
- * to invert; for produced-state grading the correctness checker MUST be semantic
414
- * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
415
- */
416
- function createTokenRecallChecker(opts = {}) {
417
- const minRecall = opts.minRecall ?? .5;
418
- const minLen = opts.minContentLength ?? 120;
419
- return async (requirement, content) => {
420
- const body = content.trim();
421
- if (body.length < minLen) return {
422
- correct: false,
423
- reason: `content too thin (${body.length} chars) to be the deliverable`
424
- };
425
- const titleTokens = requirement.title.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t));
426
- if (titleTokens.length === 0) return {
427
- correct: true,
428
- reason: "requirement title has no significant tokens — structural match accepted"
429
- };
430
- const lower = body.toLowerCase();
431
- const hits = titleTokens.filter((t) => lower.includes(t)).length;
432
- return hits / titleTokens.length >= minRecall ? {
433
- correct: true,
434
- reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`
435
- } : {
436
- correct: false,
437
- reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`
438
- };
439
- };
440
- }
441
- //#endregion
442
- //#region src/produced-state.ts
443
- function artifactKind(mimeType) {
444
- if (!mimeType) return "file";
445
- if (mimeType.includes("json")) return "json";
446
- if (mimeType.startsWith("text/")) return "text";
447
- return "file";
448
- }
449
- /**
450
- * Normalize a run's runtime event stream into `ProducedState`.
451
- *
452
- * Pure and total — unrecognized event types are skipped. `toolCalls` is
453
- * deduplicated by name in first-seen order (completion cares about a tool's
454
- * presence, not its call count). An artifact with neither a name nor a uri
455
- * still yields an entry keyed by its `artifactId` so it is never silently
456
- * dropped; an artifact with no `content` yields empty content, which the
457
- * completion oracle's structural check then rejects on its own.
458
- */
459
- function extractProducedState(events) {
460
- const artifacts = [];
461
- const proposals = [];
462
- const toolCalls = [];
463
- const seenTools = /* @__PURE__ */ new Set();
464
- for (const ev of events) if (ev.type === "tool_call") {
465
- const name = ev.toolName;
466
- if (name && !seenTools.has(name)) {
467
- seenTools.add(name);
468
- toolCalls.push(name);
469
- }
470
- } else if (ev.type === "artifact") {
471
- const a = ev;
472
- artifacts.push({
473
- kind: artifactKind(a.mimeType),
474
- path: a.name ?? a.uri ?? a.artifactId,
475
- content: a.content ?? ""
476
- });
477
- } else if (ev.type === "proposal_created") {
478
- const p = ev;
479
- proposals.push({
480
- id: p.proposalId,
481
- title: p.title,
482
- status: p.status ?? "pending",
483
- ...p.content !== void 0 ? { content: p.content } : {}
484
- });
485
- }
486
- return {
487
- artifacts,
488
- proposals,
489
- toolCalls
490
- };
491
- }
492
- //#endregion
493
- //#region src/agent-profile.ts
494
- /**
495
- * The agentic coding harnesses an eval sweeps by default — the ones we care about
496
- * ranking. This is the SINGLE source of that list; consumers import it instead of
497
- * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
498
- * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
499
- * harness) to widen beyond these.
500
- */
501
- const CODING_HARNESSES = [
502
- "opencode",
503
- "claude-code",
504
- "codex",
505
- "kimi-code"
506
- ];
507
- /** Model sentinel for a vendor-locked harness that supports none of the swept models:
508
- * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
509
- * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
510
- * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
511
- * table that would rot as router catalogs change. */
512
- const HARNESS_NATIVE_MODEL = "default";
513
- /**
514
- * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
515
- * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
516
- * which models do we evaluate" lives, so no product hand-rolls its own harness list
517
- * or column→profile mapping (the pattern that let those copies drift and silently
518
- * break the harness pivot).
519
- *
520
- * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
521
- * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
522
- * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
523
- * and results join back by harness/model via {@link harnessAxisOf} with no
524
- * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
525
- * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
526
- * requested harness runs; `keepIncompatible` forces every pair verbatim.
527
- *
528
- * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
529
- * everything we care about" switch, identical in shape whether one harness or all.
530
- */
531
- function expandProfileAxes(spec) {
532
- const harnesses = spec.harnesses ?? CODING_HARNESSES;
533
- if (harnesses.length === 0) throw new ValidationError("expandProfileAxes: no harnesses to sweep");
534
- const baseModel = spec.base.model?.default;
535
- const models = spec.models ?? (baseModel ? [baseModel] : []);
536
- if (models.length === 0) throw new ValidationError("expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied");
537
- const out = [];
538
- const seen = /* @__PURE__ */ new Set();
539
- for (const harness of harnesses) {
540
- const supported = spec.keepIncompatible ? models : models.filter((model) => harnessSupportsModel(harness, model));
541
- const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL];
542
- for (const model of effective) {
543
- const profile = {
544
- ...spec.base,
545
- harness,
546
- name: `${spec.base.name ?? "agent"}/${harness}/${model}`,
547
- model: {
548
- ...spec.base.model,
549
- default: model
550
- },
551
- metadata: {
552
- ...spec.base.metadata ?? {},
553
- harness,
554
- harnessModel: model
555
- }
556
- };
557
- const id = agentProfileId(profile);
558
- if (seen.has(id)) continue;
559
- seen.add(id);
560
- out.push(profile);
561
- }
562
- }
563
- if (out.length === 0) throw new ValidationError(`expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(", ")}], models=[${models.join(", ")}]).`);
564
- return out;
565
- }
566
- /**
567
- * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
568
- * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
569
- * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
570
- * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
571
- * in the hand-rolled copies).
572
- */
573
- function harnessAxisOf(profile) {
574
- const m = profile.metadata;
575
- const harness = m?.harness;
576
- const model = m?.harnessModel;
577
- if (typeof harness === "string" && typeof model === "string") return {
578
- harness,
579
- model
580
- };
581
- }
582
- /**
583
- * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
584
- * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
585
- * keys, and directory names where two profiles must not collapse onto one row.
586
- * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
587
- * eval matrices while keeping filenames readable.
588
- */
589
- function agentProfileId(profile) {
590
- return `${pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? "profile"}-${agentProfileHash(profile).slice(0, 16)}`;
591
- }
592
- /**
593
- * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
594
- * model id because run records reject bare/missing model aliases.
595
- */
596
- function agentProfileModelId(profile) {
597
- const model = profile.model?.default?.trim();
598
- if (!model) throw new ValidationError(`AgentProfile "${agentProfileDisplayLabel(profile) ?? "unnamed profile"}" has no model.default — cannot record eval run`);
599
- return model;
600
- }
601
- function agentProfileDisplayLabel(profile) {
602
- return profile.name?.trim() || profile.version?.trim() || void 0;
603
- }
604
- function pathSafeProfileLabel(label) {
605
- return label?.trim().replace(/[^A-Za-z0-9._-]+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "") || void 0;
606
- }
607
- function compact(input) {
608
- const out = {};
609
- for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
610
- return out;
611
- }
612
- /**
613
- * Deterministic behaviour identity for the canonical
614
- * `@tangle-network/agent-interface` AgentProfile.
615
- *
616
- * `name` and `description` are labels and do not affect the hash. Profile
617
- * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
618
- * and extensions do affect the hash. Resource array order is hash-bearing
619
- * because mount order can change agent behaviour. Undefined fields are treated
620
- * as absent; explicit `null` fields remain hash-bearing.
621
- */
622
- function agentProfileHash(profile) {
623
- const model = agentProfileModelId(profile);
624
- const behaviour = {
625
- ...profile,
626
- name: void 0,
627
- description: void 0,
628
- tags: profile.tags ? [...profile.tags].sort() : void 0,
629
- model: compact({
630
- ...profile.model,
631
- default: model
632
- })
633
- };
634
- return createHash("sha256").update(JSON.stringify(canonicalize(behaviour))).digest("hex");
635
- }
636
- //#endregion
23
+ import { execFileSync } from "node:child_process";
637
24
  //#region src/campaign/analyst-surface.ts
638
25
  function buildTraceAnalystSurfaceDispatch(options) {
639
26
  return async (surface, scenario, context) => {
@@ -3790,6 +3177,6 @@ function resolveWorktreePath(surface, worktreeDir) {
3790
3177
  return verifyCodeSurface(surface, worktreeDir).path;
3791
3178
  }
3792
3179
  //#endregion
3793
- export { planEvalFixtureRun as A, harnessAxisOf as B, LabeledScenarioStoreError as C, discoverEvalFixtures as D, neutralizationGate as E, HARNESS_NATIVE_MODEL as F, parseCorrectnessResponse as G, completionVerdict as H, agentProfileHash as I, verifyCompletion as K, agentProfileId as L, buildTraceAnalystSurfaceDispatch as M, traceAnalystQualityJudge as N, loadEvalFixture as O, CODING_HARNESSES as P, agentProfileModelId as R, FsLabeledScenarioStore as S, rolloutArgumentDiff as T, createLlmCorrectnessChecker as U, extractProducedState as V, createTokenRecallChecker as W, renderScoreboardMarkdown as _, autoevalsScorerJudge as a, userStoryScoreboard as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, makePlaybackDispatch as g, runProfileMatrix as h, verifyCodeSurface as i, analyzeCrossSurfaceInteractions as j, loadEvalFixtureScenarios as k, SEARCH_LEDGER_SCHEMA as l, ProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, scoreUserStory as v, classifyUngroundedLiterals as w, neutralizeText as x, scoreboardSummary as y, expandProfileAxes as z };
3180
+ export { planEvalFixtureRun as A, LabeledScenarioStoreError as C, discoverEvalFixtures as D, neutralizationGate as E, buildTraceAnalystSurfaceDispatch as M, traceAnalystQualityJudge as N, loadEvalFixture as O, FsLabeledScenarioStore as S, rolloutArgumentDiff as T, renderScoreboardMarkdown as _, autoevalsScorerJudge as a, userStoryScoreboard as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, makePlaybackDispatch as g, runProfileMatrix as h, verifyCodeSurface as i, analyzeCrossSurfaceInteractions as j, loadEvalFixtureScenarios as k, SEARCH_LEDGER_SCHEMA as l, ProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, scoreUserStory as v, classifyUngroundedLiterals as w, neutralizeText as x, scoreboardSummary as y };
3794
3181
 
3795
- //# sourceMappingURL=campaign-C2TTzQII.js.map
3182
+ //# sourceMappingURL=campaign-BYjBAypg.js.map