@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -0,0 +1,586 @@
1
+ import { s as ValidationError } from "./errors-Dngq5h35.js";
2
+ import { t as canonicalize } from "./pre-registration-DakwTRXk.js";
3
+ import { I as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-dZ8P6nGI.js";
4
+ import { i as CostLedger } from "./cost-ledger-BSe92yAV.js";
5
+ import { t as certificationEvidenceDigest } from "./verdict-BndeTAh_.js";
6
+ import { f as maximumChargeForLlmRequest, g as assertServedModel, h as ModelSubstitutionError, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-d0-2TT1g.js";
7
+ import { createHash, randomUUID } from "node:crypto";
8
+ import { harnessSupportsModel } from "@tangle-network/agent-interface";
9
+ //#region src/agent-profile.ts
10
+ /**
11
+ * The agentic coding harnesses an eval sweeps by default — the ones we care about
12
+ * ranking. This is the SINGLE source of that list; consumers import it instead of
13
+ * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
14
+ * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
15
+ * harness) to widen beyond these.
16
+ */
17
+ const CODING_HARNESSES = [
18
+ "opencode",
19
+ "claude-code",
20
+ "codex",
21
+ "kimi-code"
22
+ ];
23
+ /** Model sentinel for a vendor-locked harness that supports none of the swept models:
24
+ * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
25
+ * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
26
+ * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
27
+ * table that would rot as router catalogs change. */
28
+ const HARNESS_NATIVE_MODEL = "default";
29
+ /**
30
+ * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
31
+ * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
32
+ * which models do we evaluate" lives, so no product hand-rolls its own harness list
33
+ * or column→profile mapping (the pattern that let those copies drift and silently
34
+ * break the harness pivot).
35
+ *
36
+ * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
37
+ * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
38
+ * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
39
+ * and results join back by harness/model via {@link harnessAxisOf} with no
40
+ * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
41
+ * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
42
+ * requested harness runs; `keepIncompatible` forces every pair verbatim.
43
+ *
44
+ * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
45
+ * everything we care about" switch, identical in shape whether one harness or all.
46
+ */
47
+ function expandProfileAxes(spec) {
48
+ const harnesses = spec.harnesses ?? CODING_HARNESSES;
49
+ if (harnesses.length === 0) throw new ValidationError("expandProfileAxes: no harnesses to sweep");
50
+ const baseModel = spec.base.model?.default;
51
+ const models = spec.models ?? (baseModel ? [baseModel] : []);
52
+ if (models.length === 0) throw new ValidationError("expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied");
53
+ const out = [];
54
+ const seen = /* @__PURE__ */ new Set();
55
+ for (const harness of harnesses) {
56
+ const supported = spec.keepIncompatible ? models : models.filter((model) => harnessSupportsModel(harness, model));
57
+ const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL];
58
+ for (const model of effective) {
59
+ const profile = {
60
+ ...spec.base,
61
+ harness,
62
+ name: `${spec.base.name ?? "agent"}/${harness}/${model}`,
63
+ model: {
64
+ ...spec.base.model,
65
+ default: model
66
+ },
67
+ metadata: {
68
+ ...spec.base.metadata ?? {},
69
+ harness,
70
+ harnessModel: model
71
+ }
72
+ };
73
+ const id = agentProfileId(profile);
74
+ if (seen.has(id)) continue;
75
+ seen.add(id);
76
+ out.push(profile);
77
+ }
78
+ }
79
+ if (out.length === 0) throw new ValidationError(`expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(", ")}], models=[${models.join(", ")}]).`);
80
+ return out;
81
+ }
82
+ /**
83
+ * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
84
+ * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
85
+ * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
86
+ * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
87
+ * in the hand-rolled copies).
88
+ */
89
+ function harnessAxisOf(profile) {
90
+ const m = profile.metadata;
91
+ const harness = m?.harness;
92
+ const model = m?.harnessModel;
93
+ if (typeof harness === "string" && typeof model === "string") return {
94
+ harness,
95
+ model
96
+ };
97
+ }
98
+ /**
99
+ * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
100
+ * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
101
+ * keys, and directory names where two profiles must not collapse onto one row.
102
+ * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
103
+ * eval matrices while keeping filenames readable.
104
+ */
105
+ function agentProfileId(profile) {
106
+ return `${pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? "profile"}-${agentProfileHash(profile).slice(0, 16)}`;
107
+ }
108
+ /**
109
+ * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
110
+ * model id because run records reject bare/missing model aliases.
111
+ */
112
+ function agentProfileModelId(profile) {
113
+ const model = profile.model?.default?.trim();
114
+ if (!model) throw new ValidationError(`AgentProfile "${agentProfileDisplayLabel(profile) ?? "unnamed profile"}" has no model.default — cannot record eval run`);
115
+ return model;
116
+ }
117
+ function agentProfileDisplayLabel(profile) {
118
+ return profile.name?.trim() || profile.version?.trim() || void 0;
119
+ }
120
+ function pathSafeProfileLabel(label) {
121
+ return label?.trim().replace(/[^A-Za-z0-9._-]+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "") || void 0;
122
+ }
123
+ function compact(input) {
124
+ const out = {};
125
+ for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
126
+ return out;
127
+ }
128
+ /**
129
+ * Deterministic behaviour identity for the canonical
130
+ * `@tangle-network/agent-interface` AgentProfile.
131
+ *
132
+ * `name` and `description` are labels and do not affect the hash. Profile
133
+ * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
134
+ * and extensions do affect the hash. Resource array order is hash-bearing
135
+ * because mount order can change agent behaviour. Undefined fields are treated
136
+ * as absent; explicit `null` fields remain hash-bearing.
137
+ */
138
+ function agentProfileHash(profile) {
139
+ const model = agentProfileModelId(profile);
140
+ const behaviour = {
141
+ ...profile,
142
+ name: void 0,
143
+ description: void 0,
144
+ tags: profile.tags ? [...profile.tags].sort() : void 0,
145
+ model: compact({
146
+ ...profile.model,
147
+ default: model
148
+ })
149
+ };
150
+ return createHash("sha256").update(JSON.stringify(canonicalize(behaviour))).digest("hex");
151
+ }
152
+ //#endregion
153
+ //#region src/completion-verifier.ts
154
+ /**
155
+ * Completion verifier — the task-completion oracle.
156
+ *
157
+ * Answers the only eval question that is not a proxy: did the agent actually
158
+ * COMPLETE the task — produce every required deliverable, persisted and
159
+ * correct — rather than describe what should be done. A fluent transcript
160
+ * that never produces the artifact scores zero here.
161
+ *
162
+ * Per requirement, a two-stage check:
163
+ * 1. Structural — a produced item (vault artifact / approved proposal /
164
+ * tool call) of the right kind is matched against the requirement and
165
+ * carries non-empty content. Deterministic; no LLM.
166
+ * 2. Correctness — only if structurally present AND the matched item
167
+ * carries content, one targeted check decides whether that item
168
+ * actually fulfils the requirement. A hallucinated artifact fails here;
169
+ * an absent one already failed stage 1.
170
+ *
171
+ * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
172
+ * checker failures — are excluded from the denominator, never scored as
173
+ * zeros). Quality dimensions are meaningless on an incomplete task — callers
174
+ * gate on `fullyComplete` / `completionRate` before scoring quality.
175
+ */
176
+ /**
177
+ * Construct a `CompletionVerdict` from the per-requirement checks, deriving
178
+ * `completionRate` / `fullyComplete` and the spine fields (`valid` =
179
+ * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
180
+ * requirements — a verdict over nothing is a misconfiguration, mirroring
181
+ * `verifyCompletion`'s gold-spec guard.
182
+ */
183
+ function completionVerdict(input) {
184
+ if (input.requirements.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`);
185
+ const measurable = input.requirements.filter((r) => !r.unmeasured);
186
+ const unmeasuredCount = input.requirements.length - measurable.length;
187
+ if (measurable.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? "unknown reason"})`);
188
+ const satisfiedCount = measurable.filter((r) => r.satisfied).length;
189
+ const completionRate = satisfiedCount / measurable.length;
190
+ const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length;
191
+ return {
192
+ taskId: input.taskId,
193
+ requirements: input.requirements,
194
+ completionRate,
195
+ fullyComplete,
196
+ unmeasuredCount,
197
+ valid: fullyComplete,
198
+ score: completionRate,
199
+ ...input.certification === void 0 ? {} : { certification: input.certification }
200
+ };
201
+ }
202
+ const STOPWORDS = /* @__PURE__ */ new Set([
203
+ "the",
204
+ "a",
205
+ "an",
206
+ "of",
207
+ "for",
208
+ "and",
209
+ "or",
210
+ "to",
211
+ "in",
212
+ "on",
213
+ "with",
214
+ "by"
215
+ ]);
216
+ const REQUIREMENT_FORM_STOPWORDS = /* @__PURE__ */ new Set([
217
+ "generated",
218
+ "generate",
219
+ "view",
220
+ "render",
221
+ "rendered",
222
+ "persisted",
223
+ "persist",
224
+ "artifact",
225
+ "file",
226
+ "document",
227
+ "note",
228
+ "proposal",
229
+ "deliverable",
230
+ "output",
231
+ "created",
232
+ "create",
233
+ "produce",
234
+ "produced",
235
+ "flag"
236
+ ]);
237
+ const MATCH_THRESHOLD = .5;
238
+ const MIN_CONTENT_CHARS = 50;
239
+ function tokens(s, extraStop) {
240
+ return new Set(s.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)));
241
+ }
242
+ /**
243
+ * Recall of the requirement's tokens within a candidate's identifying text.
244
+ * Recall, not Jaccard — a candidate's path/id legitimately carries extra
245
+ * tokens the requirement does not name. The requirement side drops
246
+ * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.
247
+ */
248
+ function tokenRecall(requirementText, candidateText) {
249
+ const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS);
250
+ if (req.size === 0) return 0;
251
+ const cand = tokens(candidateText);
252
+ let hit = 0;
253
+ for (const t of req) if (cand.has(t)) hit++;
254
+ return hit / req.size;
255
+ }
256
+ function artifactCandidates(req, reqIndex, artifacts) {
257
+ const reqText = `${req.title} ${req.category ?? ""}`;
258
+ const out = [];
259
+ artifacts.forEach((a, i) => {
260
+ if ((a.content ?? "").trim().length < MIN_CONTENT_CHARS) return;
261
+ let score = tokenRecall(reqText, `${a.path ?? ""} ${a.kind} ${(a.content ?? "").slice(0, 4e3)}`);
262
+ if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) score = Math.max(score, 1);
263
+ if (score < MATCH_THRESHOLD) return;
264
+ out.push({
265
+ reqIndex,
266
+ itemKey: `artifact:${i}`,
267
+ score,
268
+ evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,
269
+ content: a.content ?? null
270
+ });
271
+ });
272
+ return out;
273
+ }
274
+ function proposalCandidates(req, reqIndex, proposals) {
275
+ const reqText = `${req.title} ${req.category ?? ""}`;
276
+ const out = [];
277
+ for (const p of proposals) {
278
+ if (p.status !== "approved") continue;
279
+ const body = (p.content ?? "").trim();
280
+ if (body.length < MIN_CONTENT_CHARS) continue;
281
+ const score = tokenRecall(reqText, `${p.title} ${body}`);
282
+ if (score < MATCH_THRESHOLD) continue;
283
+ out.push({
284
+ reqIndex,
285
+ itemKey: `proposal:${p.id}`,
286
+ score,
287
+ evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,
288
+ content: body
289
+ });
290
+ }
291
+ return out;
292
+ }
293
+ function toolCallCandidates(req, reqIndex, toolCalls) {
294
+ const out = [];
295
+ toolCalls.forEach((name, i) => {
296
+ const score = tokenRecall(req.title, name);
297
+ if (score < MATCH_THRESHOLD) return;
298
+ out.push({
299
+ reqIndex,
300
+ itemKey: `tool:${i}`,
301
+ score,
302
+ evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,
303
+ content: null
304
+ });
305
+ });
306
+ return out;
307
+ }
308
+ /**
309
+ * Verify whether a run completed the task. `checkCorrectness` is injected —
310
+ * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
311
+ *
312
+ * Throws on a gold spec with no requirements: an eval task that requires
313
+ * nothing is a misconfiguration, not a vacuously-complete task.
314
+ */
315
+ async function verifyCompletion(gold, state, checkCorrectness) {
316
+ if (gold.requirements.length === 0) throw new Error(`verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`);
317
+ const candidates = [];
318
+ gold.requirements.forEach((req, i) => {
319
+ const by = req.satisfiedBy ?? "any";
320
+ if (by === "artifact" || by === "any") candidates.push(...artifactCandidates(req, i, state.artifacts));
321
+ if (by === "proposal" || by === "any") candidates.push(...proposalCandidates(req, i, state.proposals));
322
+ if (by === "tool-call" || by === "any") candidates.push(...toolCallCandidates(req, i, state.toolCalls));
323
+ });
324
+ candidates.sort((a, b) => b.score - a.score);
325
+ const assigned = /* @__PURE__ */ new Map();
326
+ const itemTaken = /* @__PURE__ */ new Set();
327
+ for (const c of candidates) {
328
+ if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue;
329
+ assigned.set(c.reqIndex, c);
330
+ itemTaken.add(c.itemKey);
331
+ }
332
+ const requirements = [];
333
+ for (let i = 0; i < gold.requirements.length; i++) {
334
+ const req = gold.requirements[i];
335
+ const match = assigned.get(i);
336
+ const evidence = [];
337
+ let correct = null;
338
+ let unmeasuredReason;
339
+ if (match) {
340
+ evidence.push(match.evidence);
341
+ if (match.content !== null) try {
342
+ const r = await checkCorrectness(req, match.content);
343
+ correct = r.correct;
344
+ evidence.push(`correctness: ${r.correct ? "pass" : "fail"} — ${r.reason}`);
345
+ } catch (err) {
346
+ unmeasuredReason = err instanceof JudgeParseError ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}` : `checker call failed: ${err instanceof Error ? err.message : String(err)}`;
347
+ evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`);
348
+ }
349
+ else evidence.push("correctness: not assessed — matched item carries no content");
350
+ } else {
351
+ const by = req.satisfiedBy ?? "any";
352
+ const kind = by === "any" ? "artifact/proposal/tool-call" : by;
353
+ evidence.push(`no produced ${kind} matched this requirement`);
354
+ }
355
+ const structurallyPresent = match !== void 0;
356
+ const unmeasured = unmeasuredReason !== void 0;
357
+ const satisfied = structurallyPresent && !unmeasured && correct !== false;
358
+ requirements.push({
359
+ reqId: req.reqId,
360
+ title: req.title,
361
+ structurallyPresent,
362
+ correct,
363
+ satisfied,
364
+ ...unmeasured ? {
365
+ unmeasured: true,
366
+ unmeasuredReason
367
+ } : {},
368
+ evidence
369
+ });
370
+ }
371
+ const attestation = checkCorrectness.attestation;
372
+ const certification = attestation ? {
373
+ strategy: attestation.strategy,
374
+ checker: attestation.checker,
375
+ assumptions: ["structural matching is lexical token recall over produced items — the correctness stage only sees items it matched", ...attestation.assumptions],
376
+ evidenceDigest: certificationEvidenceDigest({
377
+ taskId: gold.taskId,
378
+ requirements
379
+ })
380
+ } : void 0;
381
+ return completionVerdict({
382
+ taskId: gold.taskId,
383
+ requirements,
384
+ ...certification === void 0 ? {} : { certification }
385
+ });
386
+ }
387
+ /**
388
+ * Parse the correctness checker's model response. Tolerates a response
389
+ * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
390
+ * verdict boolean usually lands in the first few tokens, so a recovered
391
+ * prefix with a boolean `correct` is a real measurement, not a guess.
392
+ * Fails loud (JudgeParseError) when no boolean verdict is recoverable.
393
+ */
394
+ function parseCorrectnessResponse(raw) {
395
+ const readVerdict = (candidate) => {
396
+ if (candidate === null || typeof candidate !== "object") return null;
397
+ const { correct, reason } = candidate;
398
+ if (typeof correct !== "boolean") return null;
399
+ return {
400
+ correct,
401
+ reason: typeof reason === "string" ? reason : ""
402
+ };
403
+ };
404
+ const match = raw.match(/\{[\s\S]*\}/);
405
+ if (match) try {
406
+ const strict = readVerdict(JSON.parse(match[0]));
407
+ if (strict) return strict;
408
+ } catch {}
409
+ const start = raw.indexOf("{");
410
+ if (start !== -1) {
411
+ const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)));
412
+ if (recovered) return recovered;
413
+ }
414
+ throw new JudgeParseError("correctness-checker", raw);
415
+ }
416
+ /**
417
+ * Production `CorrectnessChecker` — one LLM call per matched artifact,
418
+ * deterministic (temperature 0), structured JSON out. Judges fulfilment
419
+ * only: a plan, a gesture, or a description of what should be done does not
420
+ * fulfil a requirement — the artifact must BE the deliverable.
421
+ */
422
+ function createLlmCorrectnessChecker(chat, opts = {}) {
423
+ const model = opts.model ?? "claude-sonnet-4-6";
424
+ const maxContentChars = opts.maxContentChars ?? 8e3;
425
+ const maxAttempts = opts.maxAttempts ?? 2;
426
+ const costLedger = opts.costLedger ?? new CostLedger();
427
+ const sink = opts.rawSink;
428
+ const record = async (event) => {
429
+ try {
430
+ await sink?.record(event);
431
+ } catch {}
432
+ };
433
+ const checker = async (requirement, content) => {
434
+ const request = {
435
+ model,
436
+ messages: [{
437
+ role: "system",
438
+ content: "You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}."
439
+ }, {
440
+ role: "user",
441
+ content: `Requirement: ${requirement.title}\n${requirement.category ? `Category: ${requirement.category}\n` : ""}\nProduced artifact:\n${content.slice(0, maxContentChars)}`
442
+ }],
443
+ temperature: 0,
444
+ maxTokens: 200
445
+ };
446
+ let lastErr;
447
+ for (let attempt = 0; attempt < maxAttempts; attempt++) {
448
+ const started = Date.now();
449
+ await record({
450
+ eventId: randomUUID(),
451
+ provider: chat.transport,
452
+ model,
453
+ endpoint: "/chat",
454
+ baseUrl: "",
455
+ attemptIndex: attempt,
456
+ direction: "request",
457
+ timestamp: started,
458
+ requestBody: request,
459
+ redactedFields: []
460
+ });
461
+ try {
462
+ const paid = await costLedger.runPaidCall({
463
+ channel: "verifier",
464
+ phase: opts.costPhase ?? "completion.correctness",
465
+ actor: "correctness-checker",
466
+ model,
467
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
468
+ tags: {
469
+ ...opts.costTags,
470
+ requirementId: requirement.reqId,
471
+ attempt: String(attempt)
472
+ },
473
+ signal: opts.signal,
474
+ execute: (signal, callId) => chat.chat(request, {
475
+ signal,
476
+ idempotencyKey: callId
477
+ }),
478
+ receipt: costReceiptFromLlm,
479
+ receiptFromError: costReceiptFromLlmError
480
+ });
481
+ if (!paid.succeeded) throw paid.error;
482
+ const resp = paid.value;
483
+ assertServedModel(model, resp.servedModel, {
484
+ allowUnreported: true,
485
+ context: `correctness checker for requirement ${requirement.reqId}`
486
+ });
487
+ const raw = resp.content;
488
+ await record({
489
+ eventId: randomUUID(),
490
+ provider: chat.transport,
491
+ model,
492
+ endpoint: "/chat",
493
+ baseUrl: "",
494
+ attemptIndex: attempt,
495
+ direction: "response",
496
+ timestamp: Date.now(),
497
+ durationMs: Date.now() - started,
498
+ responseBody: resp,
499
+ redactedFields: []
500
+ });
501
+ return parseCorrectnessResponse(raw);
502
+ } catch (err) {
503
+ lastErr = err;
504
+ await record({
505
+ eventId: randomUUID(),
506
+ provider: chat.transport,
507
+ model,
508
+ endpoint: "/chat",
509
+ baseUrl: "",
510
+ attemptIndex: attempt,
511
+ direction: "error",
512
+ timestamp: Date.now(),
513
+ durationMs: Date.now() - started,
514
+ errorMessage: err instanceof Error ? err.message : String(err),
515
+ redactedFields: []
516
+ });
517
+ if (err instanceof ModelSubstitutionError) throw err;
518
+ }
519
+ }
520
+ throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
521
+ };
522
+ checker.attestation = {
523
+ strategy: "judge",
524
+ checker: {
525
+ name: "llm-correctness-checker",
526
+ version: model
527
+ },
528
+ assumptions: ["served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)"]
529
+ };
530
+ return checker;
531
+ }
532
+ //#endregion
533
+ //#region src/produced-state.ts
534
+ function artifactKind(mimeType) {
535
+ if (!mimeType) return "file";
536
+ if (mimeType.includes("json")) return "json";
537
+ if (mimeType.startsWith("text/")) return "text";
538
+ return "file";
539
+ }
540
+ /**
541
+ * Normalize a run's runtime event stream into `ProducedState`.
542
+ *
543
+ * Pure and total — unrecognized event types are skipped. `toolCalls` is
544
+ * deduplicated by name in first-seen order (completion cares about a tool's
545
+ * presence, not its call count). An artifact with neither a name nor a uri
546
+ * still yields an entry keyed by its `artifactId` so it is never silently
547
+ * dropped; an artifact with no `content` yields empty content, which the
548
+ * completion oracle's structural check then rejects on its own.
549
+ */
550
+ function extractProducedState(events) {
551
+ const artifacts = [];
552
+ const proposals = [];
553
+ const toolCalls = [];
554
+ const seenTools = /* @__PURE__ */ new Set();
555
+ for (const ev of events) if (ev.type === "tool_call") {
556
+ const name = ev.toolName;
557
+ if (name && !seenTools.has(name)) {
558
+ seenTools.add(name);
559
+ toolCalls.push(name);
560
+ }
561
+ } else if (ev.type === "artifact") {
562
+ const a = ev;
563
+ artifacts.push({
564
+ kind: artifactKind(a.mimeType),
565
+ path: a.name ?? a.uri ?? a.artifactId,
566
+ content: a.content ?? ""
567
+ });
568
+ } else if (ev.type === "proposal_created") {
569
+ const p = ev;
570
+ proposals.push({
571
+ id: p.proposalId,
572
+ title: p.title,
573
+ status: p.status ?? "pending",
574
+ ...p.content !== void 0 ? { content: p.content } : {}
575
+ });
576
+ }
577
+ return {
578
+ artifacts,
579
+ proposals,
580
+ toolCalls
581
+ };
582
+ }
583
+ //#endregion
584
+ export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
585
+
586
+ //# sourceMappingURL=produced-state-DU79a81m.js.map