@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -1,16 +1,17 @@
1
1
  import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
2
  import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
3
  import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
4
- import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-D3EoChAU.js";
5
- import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
4
+ import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-DzvMUsS_.js";
5
+ import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
6
6
  import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
7
7
  import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
- import { M as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D5iN0Xzb.js";
9
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
8
+ import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BMQEv1wG.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
10
10
  import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
11
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
12
  import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
13
13
  import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
14
+ import { c as primeProtocolSha256, d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-BfSalTfR.js";
14
15
  import { z } from "zod";
15
16
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
16
17
  import * as nodePath from "node:path";
@@ -1207,7 +1208,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1207
1208
  "package.json",
1208
1209
  "pnpm-lock.yaml"
1209
1210
  ]);
1210
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "4785a1e0d785431f287362808dae777cf73125c551c18260117a711a5f2a6263";
1211
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "96a42e1ad7cc0c9b00b2430ae09092c14bd5de385c696e849cb50507b9d77f02";
1211
1212
  /** The published benchmark evidence was produced at this package version, by
1212
1213
  * the retired one-shot direct runner, before trace analysts moved to the
1213
1214
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -1252,8 +1253,10 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1252
1253
  "src/analyst/benchmark-verification-artifacts.ts",
1253
1254
  "src/analyst/benchmark-verification-outcome.ts",
1254
1255
  "src/analyst/benchmark.ts",
1256
+ "src/analyst/definition.ts",
1255
1257
  "src/analyst/dspy-rlm-engine.ts",
1256
1258
  "src/analyst/engine.ts",
1259
+ "src/analyst/equal-terms.ts",
1257
1260
  "src/analyst/exact-types.ts",
1258
1261
  "src/analyst/finding-signature.ts",
1259
1262
  "src/analyst/finding-subject.ts",
@@ -1261,6 +1264,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1261
1264
  "src/analyst/parse-tolerant.ts",
1262
1265
  "src/analyst/prime-bridge-transport.ts",
1263
1266
  "src/analyst/prime-protocol.ts",
1267
+ "src/analyst/reply-contract.ts",
1264
1268
  "src/analyst/tool-groups.ts",
1265
1269
  "src/analyst/trace-tool-callback.ts",
1266
1270
  "src/analyst/types.ts",
@@ -1279,7 +1283,9 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1279
1283
  "src/concurrency.ts",
1280
1284
  "src/cost-ledger.ts",
1281
1285
  "src/errors.ts",
1286
+ "src/integrity/served-model.ts",
1282
1287
  "src/judge-calibration.ts",
1288
+ "src/judge-families.ts",
1283
1289
  "src/ledger-core/atomic-file-lock.ts",
1284
1290
  "src/ledger-core/canonical.ts",
1285
1291
  "src/ledger-core/deep-freeze.ts",
@@ -1309,7 +1315,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1309
1315
  "src/trace/raw-provider-sink.ts",
1310
1316
  "src/verdict-cache.ts"
1311
1317
  ]);
1312
- const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "46af7a98d6df73d845394827283390bac7a2fb9712f418b14b1093f19a21faea";
1318
+ const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "cc354effd79c8dfc8669c230706c67c63a06e3aff013f540806361263bae8860";
1313
1319
  function analystBenchmarkImplementationDigest() {
1314
1320
  return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
1315
1321
  }
@@ -2236,19 +2242,24 @@ Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/
2236
2242
  Give the rationale as the concrete downstream evidence visible at the consequence step.
2237
2243
  Submit as soon as every candidate failure block has a supported verdict.
2238
2244
  Return no finding for a clean trajectory.`;
2239
- /** One-shot JSON transport prompt for the direct runner. */
2240
- function publicBenchmarkSystemPrompt(dataset) {
2241
- const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
2242
- return `${publicBenchmarkTaskPrompt(dataset)}
2243
-
2244
- ${fieldContract}
2245
-
2246
- Return exactly one JSON object with:
2245
+ /** Reply-envelope contract shared by both one-shot datasets. */
2246
+ const PUBLIC_BENCHMARK_ENVELOPE_CONTRACT = `Return exactly one JSON object with:
2247
2247
  - "report": a concise evidence-based explanation, at most 4000 characters
2248
2248
  - "findings": the strict finding array
2249
2249
  Use an empty findings array when the trace does not support a finding.
2250
2250
  Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
2251
2251
  The runner constructs exact trace URIs and action previews from each selected step.`;
2252
+ /** Per-dataset field grammar for the one-shot JSON reply. */
2253
+ function publicBenchmarkFieldContract(dataset) {
2254
+ return dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
2255
+ }
2256
+ /** One-shot JSON transport prompt for the direct runner. */
2257
+ function publicBenchmarkSystemPrompt(dataset) {
2258
+ return [
2259
+ publicBenchmarkTaskPrompt(dataset),
2260
+ publicBenchmarkFieldContract(dataset),
2261
+ PUBLIC_BENCHMARK_ENVELOPE_CONTRACT
2262
+ ].join("\n\n");
2252
2263
  }
2253
2264
  /** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
2254
2265
  function publicBenchmarkRlmInstructions(dataset) {
@@ -2256,6 +2267,7 @@ function publicBenchmarkRlmInstructions(dataset) {
2256
2267
  return `${publicBenchmarkTaskPrompt(dataset)}
2257
2268
  ${outputContract}`;
2258
2269
  }
2270
+ /** Task text shared by every runner shape on one dataset. */
2259
2271
  function publicBenchmarkTaskPrompt(dataset) {
2260
2272
  return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
2261
2273
  }
@@ -3592,61 +3604,300 @@ function fileContext() {
3592
3604
  };
3593
3605
  }
3594
3606
  //#endregion
3607
+ //#region src/analyst/definition.ts
3608
+ /**
3609
+ * AnalystDefinition — the declarative unit behind an analyst arm.
3610
+ *
3611
+ * An arm is one way of EXECUTING an analysis question: a one-shot JSON call, a
3612
+ * bridge-reached RLM, a recursive engine with trace tools. What the arm SAYS —
3613
+ * the question, the task text, the reply grammar, how evidence reaches the
3614
+ * model, the repair-turn and budget terms — is protocol, not execution, so it
3615
+ * lives here as one inspectable value. `bindAnalyst` (./bind) compiles a
3616
+ * definition plus a transport binding into a runnable arm, and the parity
3617
+ * suite holds the compiled arm to the byte against the arm's entry point, so a
3618
+ * definition cannot drift from what its arm actually sends.
3619
+ *
3620
+ * Three rules carried over from the repair-arm comparison contract
3621
+ * (trace-repair's `repairArmAsymmetries`), made structural here:
3622
+ *
3623
+ * one contract the reply grammar is a `ReplyContract` value on the
3624
+ * definition, never prose inside a runner body.
3625
+ * one repair turn `analystDefinitionAsymmetries` refuses a set whose
3626
+ * definitions declare unequal repair turns, because a second
3627
+ * attempt is a second sample the other arms never got.
3628
+ * declared difference what arms MAY differ in — the evidence projection, the
3629
+ * reasoning effort, the budget — is declared per definition
3630
+ * and rendered beside the comparison instead of being
3631
+ * inferred from two runners' source.
3632
+ */
3633
+ /**
3634
+ * Thrown at bind time when a definition asks for something no strategy can
3635
+ * compile — an unknown projection × transport pair, a repair-turn count the
3636
+ * exchange machinery cannot grant, a reasoning effort the arm cannot map. The
3637
+ * message names the construct so an expressiveness gap is a loud, attributable
3638
+ * failure instead of a silently narrowed protocol.
3639
+ */
3640
+ var AnalystExpressivenessError = class extends Error {};
3641
+ /**
3642
+ * Digest of everything a definition can send to its model. An inline
3643
+ * definition hashes under the historical prime-protocol domain, so its digest
3644
+ * equals the digest its bespoke arm always recorded; other projections hash
3645
+ * under the definition domain.
3646
+ */
3647
+ function analystDefinitionProtocolSha256(definition) {
3648
+ const { projection, replyContract } = definition;
3649
+ if (projection.mode === "inline") return primeProtocolSha256({
3650
+ question: definition.question,
3651
+ ...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
3652
+ contractLines: replyContract.contractLines,
3653
+ repairContractLines: replyContract.repairContractLines,
3654
+ limits: {
3655
+ ...definition.contractLimits,
3656
+ maxInlineTrajectoryChars: projection.maxInlineChars,
3657
+ chunkedProjectionAttributeByteCap: projection.cappedAttributeBytes
3658
+ }
3659
+ });
3660
+ return createHash("sha256").update(JSON.stringify({
3661
+ kind: "analyst-definition-protocol",
3662
+ mode: projection.mode,
3663
+ question: definition.question,
3664
+ taskDefinition: definition.taskDefinition ?? null,
3665
+ contractLines: replyContract.contractLines,
3666
+ repairContractLines: replyContract.repairContractLines,
3667
+ limits: definition.contractLimits,
3668
+ projection: projection.mode === "chunked" ? { attributeByteCaps: projection.attributeByteCaps } : { toolGroup: projection.toolGroup }
3669
+ })).digest("hex");
3670
+ }
3671
+ /**
3672
+ * Refuse a set of definitions that cannot be compared on equal terms, and
3673
+ * render what still differs between the ones that can. The hard rule is the
3674
+ * repair turn: a malformed reply must earn the same number of retries in every
3675
+ * arm, because a retry is a second sample. Projection, reasoning effort, and
3676
+ * budget differences are declared and reported, never hidden.
3677
+ */
3678
+ function analystDefinitionAsymmetries(definitions) {
3679
+ const { ids, repairTurns } = assertEqualDeclarativeTerms("analyst definition", definitions.map((definition) => ({
3680
+ id: definition.id,
3681
+ repairTurns: definition.repair.turns
3682
+ })));
3683
+ const firstMode = definitions[0].projection.mode;
3684
+ return {
3685
+ ids,
3686
+ repairTurns,
3687
+ sharedProjectionMode: definitions.every((definition) => definition.projection.mode === firstMode) ? firstMode : null,
3688
+ asymmetries: definitions.map((definition) => ({
3689
+ id: definition.id,
3690
+ projectionMode: definition.projection.mode,
3691
+ reasoningEffort: definition.profile.model?.reasoningEffort ?? null,
3692
+ timeoutMs: definition.budget.timeoutMs,
3693
+ maxCostUsd: definition.budget.maxCostUsd ?? null,
3694
+ maxOutputTokens: definition.budget.maxOutputTokens ?? null,
3695
+ protocolSha256: definition.protocolSha256,
3696
+ definitionSha256: analystDefinitionProtocolSha256(definition)
3697
+ }))
3698
+ };
3699
+ }
3700
+ //#endregion
3701
+ //#region src/analyst/reply-contract.ts
3702
+ /**
3703
+ * The reply grammar an analyst arm holds a model to, independent of transport.
3704
+ *
3705
+ * One contract serves every arm shape: the inline bridge protocol
3706
+ * (`runPrimeExchange`) reads the base fields, and one-shot JSON arms
3707
+ * additionally use the strict-envelope and all-rejected knobs. `PrimeReplyContract`
3708
+ * in ./prime-protocol is a type alias of this contract, so a consumer written
3709
+ * against the prime protocol names the same grammar object.
3710
+ */
3711
+ /**
3712
+ * Decode a parsed reply value under a contract: strict envelope when declared,
3713
+ * then per-row decoding, then the all-rejected policy. Shape before count: a
3714
+ * malformed row never consumes an accepted slot.
3715
+ */
3716
+ function decodeReplyRows(contract, value) {
3717
+ let rawRows;
3718
+ let extras = {};
3719
+ if (contract.parseEnvelope) {
3720
+ const envelope = contract.parseEnvelope(value);
3721
+ rawRows = envelope.rows;
3722
+ extras = envelope.extras;
3723
+ } else {
3724
+ const field = (typeof value === "object" && value !== null && !Array.isArray(value) ? value : void 0)?.[contract.rowsField];
3725
+ if (!Array.isArray(field)) throw new ValidationError(`reply has no "${contract.rowsField}" array`);
3726
+ rawRows = field;
3727
+ }
3728
+ const rows = [];
3729
+ const rejected = [];
3730
+ let overflow = 0;
3731
+ rawRows.forEach((row, index) => {
3732
+ const decoded = contract.decodeRow(row, index);
3733
+ if (!decoded.ok) {
3734
+ rejected.push({
3735
+ index,
3736
+ reason: decoded.reason
3737
+ });
3738
+ return;
3739
+ }
3740
+ if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
3741
+ overflow += 1;
3742
+ return;
3743
+ }
3744
+ rows.push(decoded.row);
3745
+ });
3746
+ if (rows.length === 0 && rawRows.length > 0 && contract.whenAllRowsRejected === "fail") throw new ValidationError(`${contract.allRejectedMessage ?? "every reported row was malformed"}: ${rejected.map((entry) => entry.reason).join(" | ")}`);
3747
+ return {
3748
+ rows,
3749
+ extras,
3750
+ rejected,
3751
+ reportedRows: rawRows.length,
3752
+ overflow
3753
+ };
3754
+ }
3755
+ //#endregion
3595
3756
  //#region src/analyst/benchmark-public-model.ts
3596
- /** One-shot JSON baseline. This is not a recursive trace analyst. */
3757
+ /** The direct arm as a declarative unit for one public dataset. */
3758
+ function publicDirectAnalystDefinition(dataset, args) {
3759
+ const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
3760
+ const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
3761
+ return {
3762
+ id: "direct",
3763
+ description: "One-shot JSON baseline over the caller-owned model path.",
3764
+ version: "1.0.0",
3765
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
3766
+ profile: { model: { reasoningEffort: "none" } },
3767
+ question: "",
3768
+ taskDefinition: publicBenchmarkTaskPrompt(dataset),
3769
+ projection: {
3770
+ mode: "chunked",
3771
+ attributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS
3772
+ },
3773
+ replyContract: directReplyContract(dataset),
3774
+ contractLimits: dataset === "agentrx" ? { maxFindings: 1 } : {
3775
+ maxBlocks: 16,
3776
+ maxBlockSteps: 12
3777
+ },
3778
+ budget: {
3779
+ timeoutMs: args.timeoutMs,
3780
+ maxCostUsd: args.maxCostUsd,
3781
+ maxOutputTokens: args.maxOutputTokens
3782
+ },
3783
+ repair: { turns: 0 },
3784
+ protocolSha256: publicBenchmarkProtocolSha256(dataset),
3785
+ binding: {
3786
+ kind: "chunked",
3787
+ subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$2(dataset, caseId),
3788
+ baseMetadata: {
3789
+ analysisMode: "direct-baseline",
3790
+ outputAdapter
3791
+ },
3792
+ costActor: actor,
3793
+ costPhase: "analyst.public-benchmark",
3794
+ userMessage: (rendered) => `TRACE DATA:\n${rendered}\n\nReturn the analysis JSON object.`,
3795
+ async expandRows({ subject, rows, store, analystId, producedAt, providerModel, signal }) {
3796
+ const converted = await publicBenchmarkPredictionsToFindings({
3797
+ dataset,
3798
+ trajectoryId: subject,
3799
+ predictions: rows,
3800
+ store,
3801
+ analystId,
3802
+ providerModel: requiredString(providerModel ?? "", "finding providerModel"),
3803
+ producedAt: requiredString(producedAt ?? "", "finding producedAt"),
3804
+ ...signal ? { signal } : {}
3805
+ });
3806
+ return {
3807
+ findings: converted.findings,
3808
+ diagnostics: converted.diagnostics
3809
+ };
3810
+ },
3811
+ ...dataset === "codetracebench" ? { verifyFindings: async (args) => {
3812
+ await validateCodeTraceFindingEvidence({
3813
+ trajectoryId: args.subject,
3814
+ findings: [...args.findings],
3815
+ store: args.store,
3816
+ ...args.signal ? { signal: args.signal } : {}
3817
+ });
3818
+ } } : {}
3819
+ }
3820
+ };
3821
+ }
3822
+ /** Thin shell: validate config, declare the definition, run the chunked strategy. */
3597
3823
  function createPublicBenchmarkDirectRunner(dataset, config) {
3598
3824
  if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
3825
+ const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
3826
+ return runChunkedAnalystDefinition(publicDirectAnalystDefinition(dataset, {
3827
+ timeoutMs: positiveSafeInteger(config.timeoutMs, "timeoutMs"),
3828
+ maxOutputTokens,
3829
+ maxCostUsd: config.maxCostUsdPerAnalysis ?? 1
3830
+ }), config);
3831
+ }
3832
+ /**
3833
+ * Compile a chunked-projection definition into a runnable one-shot JSON arm
3834
+ * over the caller-owned model path. Prompt content, the projection ladder,
3835
+ * the reply grammar, and the budget declaration come from the definition;
3836
+ * caching, cost settlement, and the model proxy are transport machinery.
3837
+ */
3838
+ function runChunkedAnalystDefinition(definition, config) {
3839
+ const { projection, binding, replyContract } = definition;
3840
+ if (projection.mode !== "chunked" || binding.kind !== "chunked") throw new AnalystExpressivenessError(`the chunked one-shot strategy compiles only chunked projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
3841
+ if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
3842
+ if (definition.repair.turns !== 0) throw new AnalystExpressivenessError(`the one-shot JSON exchange grants no repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
3843
+ const reasoningEffort = definition.profile.model?.reasoningEffort;
3844
+ if (reasoningEffort !== "none") throw new AnalystExpressivenessError(`the one-shot JSON strategy runs with thinking disabled and can express only reasoning effort 'none'; definition '${definition.id}' declares '${reasoningEffort}'`);
3845
+ if (!replyContract.parseEnvelope) throw new AnalystExpressivenessError(`the one-shot JSON strategy needs a strict reply envelope; definition '${definition.id}' declares no parseEnvelope`);
3846
+ if (definition.taskDefinition === void 0) throw new AnalystExpressivenessError(`the one-shot JSON strategy composes its system prompt from the task definition; definition '${definition.id}' declares none`);
3599
3847
  const model = requiredString(config.model, "model");
3600
3848
  const callRef = requiredString(config.callRef, "callRef");
3601
3849
  if (typeof config.call !== "function") throw new TypeError("call must be a function");
3602
3850
  if (typeof config.recordExecution !== "function") throw new TypeError("recordExecution must be a function");
3603
3851
  const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
3604
3852
  const timeoutMs = positiveSafeInteger(config.timeoutMs, "timeoutMs");
3853
+ const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
3854
+ assertDeclaredBudget(definition, {
3855
+ timeoutMs,
3856
+ maxOutputTokens,
3857
+ maxCostUsd
3858
+ });
3605
3859
  const maxReasoningTokens = config.maxReasoningTokens ?? maxOutputTokens * 4;
3606
3860
  const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
3607
3861
  const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
3608
3862
  const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
3609
3863
  const pricing = config.pricing ?? pricingForModel$2(model);
3610
- const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
3611
3864
  const costLedger = config.costLedger ?? new CostLedger();
3612
3865
  const durability = config.durability ? {
3613
3866
  runIdentitySha256: requiredString(config.durability.runIdentitySha256, "durability.runIdentitySha256"),
3614
3867
  responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
3615
3868
  } : void 0;
3616
- const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
3617
- const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
3869
+ const systemPrompt = [definition.taskDefinition, ...replyContract.contractLines].join("\n\n");
3618
3870
  return {
3619
- id: "direct",
3871
+ id: definition.id,
3620
3872
  async analyze(input, context) {
3621
- const trajectoryId = trajectoryIdFromCaseId$2(dataset, context.caseId);
3873
+ const trajectoryId = binding.subjectFromCaseId(context.caseId);
3622
3874
  const costTags = {
3623
- analystId: actor,
3875
+ analystId: binding.costActor,
3624
3876
  benchmarkCaseId: context.caseId,
3625
3877
  benchmarkRepetition: String(context.repetition)
3626
3878
  };
3627
3879
  let rawPredictions = [];
3628
- let rejectedBlocks = [];
3880
+ let rejectedRows = [];
3629
3881
  let modelFindings = [];
3630
3882
  let providerModel = model;
3631
3883
  let producedAt;
3632
3884
  let modelMetadata = {
3633
- analysisMode: "direct-baseline",
3634
- outputAdapter,
3635
- protocolSha256: publicBenchmarkProtocolSha256(dataset),
3885
+ ...binding.baseMetadata,
3886
+ protocolSha256: definition.protocolSha256,
3636
3887
  callRef
3637
3888
  };
3638
3889
  try {
3639
- if (!input.traceStore) throw new Error(`${dataset} model runner requires a trace store`);
3640
- const preparedContext = await prepareSingleTraceContext(input.traceStore, context);
3641
- if (preparedContext === void 0) throw new Error(`${dataset} trace '${trajectoryId}' has no readable spans`);
3890
+ if (!input.traceStore) throw new Error(`chunked analyst '${definition.id}' requires a trace store`);
3891
+ const preparedContext = await prepareSingleTraceContext(input.traceStore, context, projection.attributeByteCaps);
3892
+ if (preparedContext === void 0) throw new Error(`trace '${trajectoryId}' has no readable spans`);
3642
3893
  const request = {
3643
3894
  model,
3644
3895
  messages: [{
3645
3896
  role: "system",
3646
- content: publicBenchmarkSystemPrompt(dataset)
3897
+ content: systemPrompt
3647
3898
  }, {
3648
3899
  role: "user",
3649
- content: `TRACE DATA:\n${preparedContext}\n\nReturn the analysis JSON object.`
3900
+ content: binding.userMessage(preparedContext)
3650
3901
  }],
3651
3902
  jsonMode: true,
3652
3903
  thinking: "disabled",
@@ -3676,14 +3927,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3676
3927
  error: cached.error,
3677
3928
  metadata: modelMetadata
3678
3929
  };
3679
- const response = parsePublicBenchmarkModelResponse(dataset, cached.response);
3680
- rawPredictions = response.findings;
3681
- rejectedBlocks = response.rejectedBlocks;
3930
+ const response = decodeReplyRows(replyContract, cached.response);
3931
+ rawPredictions = response.rows;
3932
+ rejectedRows = response.rejected.map((entry) => entry.reason);
3682
3933
  providerModel = cached.metadata.providerModel;
3683
3934
  producedAt = cached.metadata.producedAt;
3684
3935
  modelMetadata = {
3685
3936
  ...modelMetadata,
3686
- report: response.report,
3937
+ ...response.extras,
3687
3938
  providerModel: cached.metadata.providerModel,
3688
3939
  providerDurationMs: cached.metadata.providerDurationMs,
3689
3940
  finishReason: cached.metadata.finishReason
@@ -3712,8 +3963,8 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3712
3963
  },
3713
3964
  costLedger,
3714
3965
  channel: "analyst",
3715
- phase: "analyst.public-benchmark",
3716
- actor,
3966
+ phase: binding.costPhase,
3967
+ actor: binding.costActor,
3717
3968
  tags: costTags,
3718
3969
  callId: providerCallId,
3719
3970
  ...context.signal ? { signal: context.signal } : {}
@@ -3732,7 +3983,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3732
3983
  ...context.signal ? { signal: context.signal } : {},
3733
3984
  idempotencyKey: providerCallId
3734
3985
  });
3735
- const response = parsePublicBenchmarkModelResponse(dataset, completed.value);
3986
+ const response = decodeReplyRows(replyContract, completed.value);
3736
3987
  const responseProducedAt = (/* @__PURE__ */ new Date()).toISOString();
3737
3988
  const receipt = requiredSettledReceipt(costLedger, providerCallId);
3738
3989
  if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
@@ -3778,26 +4029,25 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3778
4029
  }
3779
4030
  });
3780
4031
  const response = completed.response;
3781
- rawPredictions = response.findings;
3782
- rejectedBlocks = response.rejectedBlocks;
4032
+ rawPredictions = response.rows;
4033
+ rejectedRows = response.rejected.map((entry) => entry.reason);
3783
4034
  providerModel = completed.result.model;
3784
4035
  producedAt = completed.producedAt;
3785
4036
  modelMetadata = {
3786
4037
  ...modelMetadata,
3787
4038
  responseSource: "provider",
3788
- report: response.report,
4039
+ ...response.extras,
3789
4040
  providerModel: completed.result.model,
3790
4041
  providerDurationMs: completed.result.durationMs,
3791
4042
  finishReason: completed.result.finishReason ?? null,
3792
4043
  cost: costReceiptMetadata(completed.receipt)
3793
4044
  };
3794
4045
  }
3795
- const converted = await publicBenchmarkPredictionsToFindings({
3796
- dataset,
3797
- trajectoryId,
3798
- predictions: rawPredictions,
4046
+ const converted = await binding.expandRows({
4047
+ subject: trajectoryId,
4048
+ rows: rawPredictions,
3799
4049
  store: input.traceStore,
3800
- analystId: "direct",
4050
+ analystId: definition.id,
3801
4051
  providerModel,
3802
4052
  producedAt: requiredString(producedAt ?? "", "finding producedAt"),
3803
4053
  ...context.signal ? { signal: context.signal } : {}
@@ -3807,11 +4057,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3807
4057
  ...modelMetadata,
3808
4058
  blockDiagnostics: {
3809
4059
  ...converted.diagnostics,
3810
- rejectedBlocks
4060
+ rejectedBlocks: rejectedRows
3811
4061
  }
3812
4062
  };
3813
- if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
3814
- trajectoryId,
4063
+ if (binding.verifyFindings) await binding.verifyFindings({
4064
+ subject: trajectoryId,
3815
4065
  findings: modelFindings,
3816
4066
  store: input.traceStore,
3817
4067
  ...context.signal ? { signal: context.signal } : {}
@@ -3844,6 +4094,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3844
4094
  }
3845
4095
  };
3846
4096
  }
4097
+ /** A definition that declares one budget while the transport runs another is refused. */
4098
+ function assertDeclaredBudget(definition, effective) {
4099
+ const declared = definition.budget;
4100
+ if (declared.timeoutMs !== effective.timeoutMs || declared.maxOutputTokens !== effective.maxOutputTokens || declared.maxCostUsd !== effective.maxCostUsd) throw new AnalystExpressivenessError(`definition '${definition.id}' declares budget ${JSON.stringify(declared)} but the bound transport runs ${JSON.stringify(effective)}; the declaration must state what executes`);
4101
+ }
3847
4102
  function settleCachedResponse(costLedger, cached) {
3848
4103
  const settled = costLedger.list().find((receipt) => receipt.callId === cached.callId);
3849
4104
  const pending = costLedger.listPending?.().find((record) => record.callId === cached.callId);
@@ -4001,27 +4256,56 @@ const AgentRxModelResponseSchema = z.object({
4001
4256
  report: z.string().min(1).max(4e3),
4002
4257
  findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
4003
4258
  }).strict();
4004
- function parsePublicBenchmarkModelResponse(dataset, value) {
4259
+ /**
4260
+ * The one-shot reply grammar per dataset. The envelope is the contract and
4261
+ * stays strict. Individual CodeTraceBench blocks are model output: one
4262
+ * malformed block must not void a case whose remaining blocks are usable and
4263
+ * whose provider call is already paid for, so rows decode individually and
4264
+ * every rejection is reported.
4265
+ */
4266
+ function directReplyContract(dataset) {
4005
4267
  if (dataset === "agentrx") return {
4006
- ...AgentRxModelResponseSchema.parse(value),
4007
- rejectedBlocks: []
4008
- };
4009
- const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
4010
- const findings = [];
4011
- const rejectedBlocks = [];
4012
- for (const [index, block] of envelope.findings.entries()) {
4013
- const parsed = CodeTraceBlockPredictionSchema.safeParse(block);
4014
- if (parsed.success) {
4015
- findings.push(parsed.data);
4016
- continue;
4268
+ rowsField: "findings",
4269
+ contractLines: [publicBenchmarkFieldContract("agentrx"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
4270
+ repairContractLines: [],
4271
+ parseEnvelope(value) {
4272
+ const parsed = AgentRxModelResponseSchema.parse(value);
4273
+ return {
4274
+ rows: parsed.findings,
4275
+ extras: { report: parsed.report }
4276
+ };
4277
+ },
4278
+ decodeRow(row) {
4279
+ return {
4280
+ ok: true,
4281
+ row
4282
+ };
4017
4283
  }
4018
- rejectedBlocks.push(`block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`);
4019
- }
4020
- if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
4284
+ };
4021
4285
  return {
4022
- report: envelope.report,
4023
- findings,
4024
- rejectedBlocks
4286
+ rowsField: "findings",
4287
+ contractLines: [publicBenchmarkFieldContract("codetracebench"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
4288
+ repairContractLines: [],
4289
+ parseEnvelope(value) {
4290
+ const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
4291
+ return {
4292
+ rows: envelope.findings,
4293
+ extras: { report: envelope.report }
4294
+ };
4295
+ },
4296
+ decodeRow(row, index) {
4297
+ const parsed = CodeTraceBlockPredictionSchema.safeParse(row);
4298
+ if (parsed.success) return {
4299
+ ok: true,
4300
+ row: parsed.data
4301
+ };
4302
+ return {
4303
+ ok: false,
4304
+ reason: `block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`
4305
+ };
4306
+ },
4307
+ whenAllRowsRejected: "fail",
4308
+ allRejectedMessage: "every reported failure block was malformed"
4025
4309
  };
4026
4310
  }
4027
4311
  async function publicBenchmarkPredictionsToFindings(options) {
@@ -4088,12 +4372,12 @@ async function publicBenchmarkPredictionsToFindings(options) {
4088
4372
  ...options.signal ? { signal: options.signal } : {}
4089
4373
  });
4090
4374
  }
4091
- async function prepareSingleTraceContext(store, context) {
4375
+ async function prepareSingleTraceContext(store, context, attributeByteCaps) {
4092
4376
  const storeContext = context.signal ? { signal: context.signal } : void 0;
4093
4377
  const overview = await store.getOverview(void 0, storeContext);
4094
4378
  if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public model benchmark requires exactly one trace, received ${overview.total_traces}`);
4095
4379
  const traceId = overview.sample_trace_ids[0];
4096
- for (const perAttributeByteCap of TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS) {
4380
+ for (const perAttributeByteCap of attributeByteCaps) {
4097
4381
  const viewed = await store.viewTrace({
4098
4382
  trace_id: traceId,
4099
4383
  per_attribute_byte_cap: perAttributeByteCap
@@ -4243,18 +4527,156 @@ function contiguousSegments(sortedSteps) {
4243
4527
  }
4244
4528
  //#endregion
4245
4529
  //#region src/analyst/benchmark-public-rlm.ts
4246
- /** Public benchmark candidate that runs the actual recursive trace analyst. */
4530
+ /** The dspy-rlm arm as a declarative unit for one public dataset. */
4531
+ function publicRlmAnalystDefinition(dataset, args) {
4532
+ return {
4533
+ id: "dspy-rlm",
4534
+ description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
4535
+ version: "1.0.0",
4536
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
4537
+ profile: {},
4538
+ question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
4539
+ taskDefinition: args.instructions,
4540
+ projection: {
4541
+ mode: "repl-variable",
4542
+ toolGroup: "singleTrace"
4543
+ },
4544
+ replyContract: {
4545
+ rowsField: "findings",
4546
+ contractLines: [RAW_FINDING_SCHEMA_PROMPT],
4547
+ repairContractLines: [],
4548
+ decodeRow(row) {
4549
+ const parsed = RawAnalystFindingSchema.safeParse(row);
4550
+ if (parsed.success) return {
4551
+ ok: true,
4552
+ row: parsed.data
4553
+ };
4554
+ return {
4555
+ ok: false,
4556
+ reason: parsed.error.issues.map((issue) => `${issue.path.join(".")}: ${issue.message}`).join("; ")
4557
+ };
4558
+ }
4559
+ },
4560
+ contractLimits: {
4561
+ maxIterations: args.engineLimits.maxIterations,
4562
+ maxLlmCalls: args.engineLimits.maxLlmCalls,
4563
+ maxToolCalls: args.engineLimits.maxToolCalls,
4564
+ maxOutputChars: args.engineLimits.maxOutputChars
4565
+ },
4566
+ budget: {
4567
+ timeoutMs: args.timeoutMs,
4568
+ maxCostUsd: args.maxCostUsd,
4569
+ maxOutputTokens: args.maxOutputTokens,
4570
+ engineLimits: args.engineLimits
4571
+ },
4572
+ repair: { turns: 1 },
4573
+ protocolSha256: args.protocolSha256,
4574
+ binding: {
4575
+ kind: "repl-variable",
4576
+ traceAnalystId: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
4577
+ subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$1(dataset, caseId),
4578
+ baseMetadata: {
4579
+ analysisMode: "recursive",
4580
+ engine: "dspy-rlm"
4581
+ },
4582
+ findingBaseMetadata: {
4583
+ analysis_mode: "recursive",
4584
+ engine: "dspy-rlm"
4585
+ },
4586
+ costPhase: "analyst.public-benchmark.dspy-rlm",
4587
+ ...dataset === "codetracebench" ? { metadataFromSubject: codeTraceBlockMetadataFromSubject } : {},
4588
+ async adapt({ subject, findings, analystId, store, signal }) {
4589
+ return adaptPublicBenchmarkFindings({
4590
+ dataset,
4591
+ trajectoryId: subject,
4592
+ findings: [...findings],
4593
+ analystId,
4594
+ store,
4595
+ ...signal ? { signal } : {}
4596
+ });
4597
+ },
4598
+ ...dataset === "codetracebench" ? { consensus: codeTraceConsensusPort() } : {},
4599
+ abstentionFallback: (fallbackConfig) => createPublicBenchmarkDirectRunner(dataset, fallbackConfig)
4600
+ }
4601
+ };
4602
+ }
4603
+ /** Step-level majority consensus on the CodeTraceBench block grammar. */
4604
+ function codeTraceConsensusPort() {
4605
+ return {
4606
+ vote(samples) {
4607
+ const consensus = consensusCodeTraceBlocks(samples.map((sample) => [...sample]));
4608
+ return {
4609
+ blocks: consensus.blocks,
4610
+ decision: consensus.decision
4611
+ };
4612
+ },
4613
+ async expand({ subject, blocks, store, analystId, producedAt, signal }) {
4614
+ const expanded = await expandCodeTraceFailureBlocks({
4615
+ trajectoryId: subject,
4616
+ blocks,
4617
+ store,
4618
+ analystId,
4619
+ producedAt,
4620
+ ...signal ? { signal } : {}
4621
+ });
4622
+ return {
4623
+ findings: expanded.findings,
4624
+ diagnostics: expanded.diagnostics
4625
+ };
4626
+ },
4627
+ sampleRecord(assignments) {
4628
+ return {
4629
+ blocks: sampleBlockRecords(assignments),
4630
+ steps: assignments.map((assignment) => assignment.step)
4631
+ };
4632
+ }
4633
+ };
4634
+ }
4635
+ /** Thin shell: validate config, declare the definition, run the repl-variable strategy. */
4247
4636
  function createPublicBenchmarkRlmRunner(dataset, config) {
4248
- const costLedger = config.costLedger ?? new CostLedger();
4249
4637
  const samples = config.dspyRlm?.samples ?? 1;
4250
4638
  if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
4251
4639
  if (samples > 1 && dataset !== "codetracebench") throw new Error("dspyRlm.samples > 1 requires the codetracebench dataset; step-level consensus is defined on its block grammar");
4252
- const limits = {
4640
+ return runReplVariableAnalystDefinition(publicRlmAnalystDefinition(dataset, {
4641
+ instructions: config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset),
4642
+ protocolSha256: effectiveAnalystProtocolSha256(dataset, config.instructionsOverride),
4643
+ timeoutMs: config.timeoutMs,
4644
+ maxOutputTokens: config.maxOutputTokens,
4645
+ maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
4646
+ engineLimits: rlmEngineLimits(config)
4647
+ }), config);
4648
+ }
4649
+ /** Engine iteration limits with this arm's defaults applied. */
4650
+ function rlmEngineLimits(config) {
4651
+ return {
4253
4652
  maxIterations: config.dspyRlm?.maxIterations ?? 14,
4254
4653
  maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
4255
4654
  maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
4256
4655
  maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
4257
4656
  };
4657
+ }
4658
+ /**
4659
+ * Compile a repl-variable definition into a runnable recursive-engine arm over
4660
+ * the caller-owned model path. The question, instructions, tool group, and
4661
+ * iteration limits come from the definition; the engine, model proxy, sampling
4662
+ * loop, and abstention floor are transport machinery.
4663
+ */
4664
+ function runReplVariableAnalystDefinition(definition, config) {
4665
+ const { projection, binding } = definition;
4666
+ if (projection.mode !== "repl-variable" || binding.kind !== "repl-variable") throw new AnalystExpressivenessError(`the repl-variable strategy compiles only repl-variable projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
4667
+ const instructions = definition.taskDefinition;
4668
+ if (instructions === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy runs the definition's task text as engine instructions; definition '${definition.id}' declares none`);
4669
+ if (config.instructionsOverride !== void 0 && config.instructionsOverride.text !== instructions) throw new AnalystExpressivenessError(`definition '${definition.id}' declares instructions that differ from the transport's instructionsOverride; one text must execute`);
4670
+ const area = definition.area;
4671
+ if (area === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy stamps the definition's area on every finding; definition '${definition.id}' declares none`);
4672
+ const limits = definition.budget.engineLimits;
4673
+ if (limits === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy needs declared engine limits; definition '${definition.id}' declares none`);
4674
+ const effectiveLimits = rlmEngineLimits(config);
4675
+ if (limits.maxIterations !== effectiveLimits.maxIterations || limits.maxLlmCalls !== effectiveLimits.maxLlmCalls || limits.maxToolCalls !== effectiveLimits.maxToolCalls || limits.maxOutputChars !== effectiveLimits.maxOutputChars) throw new AnalystExpressivenessError(`definition '${definition.id}' declares engine limits ${JSON.stringify(limits)} but the bound transport runs ${JSON.stringify(effectiveLimits)}; the declaration must state what executes`);
4676
+ const costLedger = config.costLedger ?? new CostLedger();
4677
+ const samples = config.dspyRlm?.samples ?? 1;
4678
+ if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
4679
+ if (samples > 1 && binding.consensus === void 0) throw new AnalystExpressivenessError(`samples > 1 needs a consensus port; definition '${definition.id}' declares none`);
4258
4680
  const pricing = config.pricing ?? pricingForModel$1(config.model);
4259
4681
  const engine = createDspyRlmTraceEngine({
4260
4682
  call: config.call,
@@ -4277,18 +4699,26 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4277
4699
  ...config.dspyRlm?.traceToolTimeoutMs === void 0 ? {} : { traceToolTimeoutMs: config.dspyRlm.traceToolTimeoutMs },
4278
4700
  ...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
4279
4701
  });
4280
- const instructions = config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset);
4281
- const protocolSha256 = effectiveAnalystProtocolSha256(dataset, config.instructionsOverride);
4282
- const definition = publicBenchmarkDefinition(dataset, limits, instructions);
4702
+ const protocolSha256 = definition.protocolSha256;
4703
+ const traceDefinition = {
4704
+ id: binding.traceAnalystId,
4705
+ description: definition.description,
4706
+ area,
4707
+ version: definition.version,
4708
+ question: definition.question,
4709
+ instructions,
4710
+ toolGroup: projection.toolGroup,
4711
+ limits
4712
+ };
4283
4713
  const { instructionsOverride: _rlmOnlyOverride, ...directConfig } = config;
4284
- const abstentionFallbackRunner = createPublicBenchmarkDirectRunner(dataset, {
4714
+ const abstentionFallbackRunner = binding.abstentionFallback({
4285
4715
  ...directConfig,
4286
4716
  costLedger
4287
4717
  });
4288
4718
  return {
4289
- id: "dspy-rlm",
4719
+ id: definition.id,
4290
4720
  async analyze(input, context) {
4291
- const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
4721
+ const trajectoryId = binding.subjectFromCaseId(context.caseId);
4292
4722
  const tags = {
4293
4723
  benchmarkCaseId: context.caseId,
4294
4724
  benchmarkRepetition: String(context.repetition)
@@ -4296,7 +4726,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4296
4726
  let usage;
4297
4727
  let rawFindings = [];
4298
4728
  try {
4299
- if (!input.traceStore) throw new Error(`${dataset} DSPy RLM runner requires a trace store`);
4729
+ if (!input.traceStore) throw new Error(`repl-variable analyst '${definition.id}' requires a trace store`);
4300
4730
  if (samples > 1) {
4301
4731
  const store = input.traceStore;
4302
4732
  const caseUsageFilter = {
@@ -4310,14 +4740,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4310
4740
  for (let sample = 0; sample < samples; sample += 1) {
4311
4741
  let sampleUsage;
4312
4742
  const completed = await runTraceAnalyst({
4313
- definition,
4743
+ definition: traceDefinition,
4314
4744
  engine,
4315
4745
  store,
4316
4746
  context: {
4317
4747
  runId: context.caseId,
4318
4748
  correlationId: `${context.caseId}:${context.repetition}:sample-${sample}`,
4319
4749
  costLedger,
4320
- costPhase: "analyst.public-benchmark.dspy-rlm",
4750
+ costPhase: binding.costPhase,
4321
4751
  tags,
4322
4752
  recordUsage: (receipt) => {
4323
4753
  sampleUsage = receipt;
@@ -4328,8 +4758,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4328
4758
  });
4329
4759
  const producedAt = (/* @__PURE__ */ new Date()).toISOString();
4330
4760
  const sampleFindings = completed.findings.map((finding) => makeFinding({
4331
- analyst_id: "dspy-rlm",
4332
- area: "incorrect",
4761
+ analyst_id: definition.id,
4762
+ area,
4333
4763
  subject: finding.subject,
4334
4764
  claim: finding.claim,
4335
4765
  rationale: finding.rationale,
@@ -4338,20 +4768,18 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4338
4768
  evidence_refs: evidenceRefsFromRawFinding(finding),
4339
4769
  recommended_action: finding.recommended_action,
4340
4770
  metadata: {
4341
- analysis_mode: "recursive",
4342
- engine: "dspy-rlm",
4771
+ ...binding.findingBaseMetadata,
4343
4772
  model: config.model,
4344
4773
  sample,
4345
- ...codeTraceBlockMetadataFromSubject(finding.subject) ?? {}
4774
+ ...binding.metadataFromSubject?.(finding.subject) ?? {}
4346
4775
  },
4347
4776
  produced_at: producedAt
4348
4777
  }));
4349
4778
  rawFindings = [...rawFindings, ...sampleFindings];
4350
- const adapted = await adaptPublicBenchmarkFindings({
4351
- dataset,
4352
- trajectoryId,
4779
+ const adapted = await binding.adapt({
4780
+ subject: trajectoryId,
4353
4781
  findings: sampleFindings,
4354
- analystId: "dspy-rlm",
4782
+ analystId: definition.id,
4355
4783
  store,
4356
4784
  ...context.signal ? { signal: context.signal } : {}
4357
4785
  });
@@ -4366,18 +4794,17 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4366
4794
  modelCalls: completed.modelCalls,
4367
4795
  toolCalls: completed.toolCalls,
4368
4796
  runtime: completed.runtime,
4369
- blocks: sampleBlockRecords(assignments),
4370
- steps: assignments.map((assignment) => assignment.step),
4797
+ ...binding.consensus.sampleRecord(assignments),
4371
4798
  ...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
4372
4799
  ...sampleUsage ? { usage: sampleUsage } : {}
4373
4800
  });
4374
4801
  }
4375
- const consensus = consensusCodeTraceBlocks(sampleAssignments);
4376
- const expanded = await expandCodeTraceFailureBlocks({
4377
- trajectoryId,
4802
+ const consensus = binding.consensus.vote(sampleAssignments);
4803
+ const expanded = await binding.consensus.expand({
4804
+ subject: trajectoryId,
4378
4805
  blocks: consensus.blocks,
4379
4806
  store,
4380
- analystId: "dspy-rlm",
4807
+ analystId: definition.id,
4381
4808
  producedAt: (/* @__PURE__ */ new Date()).toISOString(),
4382
4809
  ...context.signal ? { signal: context.signal } : {}
4383
4810
  });
@@ -4388,8 +4815,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4388
4815
  findings: fallback && !fallback.error ? fallback.findings : expanded.findings,
4389
4816
  usage,
4390
4817
  metadata: {
4391
- analysisMode: "recursive",
4392
- engine: "dspy-rlm",
4818
+ ...binding.baseMetadata,
4393
4819
  protocolSha256,
4394
4820
  samples,
4395
4821
  sampleRuns,
@@ -4406,14 +4832,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4406
4832
  };
4407
4833
  }
4408
4834
  const completed = await runTraceAnalyst({
4409
- definition,
4835
+ definition: traceDefinition,
4410
4836
  engine,
4411
4837
  store: input.traceStore,
4412
4838
  context: {
4413
4839
  runId: context.caseId,
4414
4840
  correlationId: `${context.caseId}:${context.repetition}`,
4415
4841
  costLedger,
4416
- costPhase: "analyst.public-benchmark.dspy-rlm",
4842
+ costPhase: binding.costPhase,
4417
4843
  tags,
4418
4844
  recordUsage: (receipt) => {
4419
4845
  usage = receipt;
@@ -4423,8 +4849,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4423
4849
  });
4424
4850
  const producedAt = (/* @__PURE__ */ new Date()).toISOString();
4425
4851
  rawFindings = completed.findings.map((finding) => makeFinding({
4426
- analyst_id: "dspy-rlm",
4427
- area: dataset === "agentrx" ? "root-cause" : "incorrect",
4852
+ analyst_id: definition.id,
4853
+ area,
4428
4854
  subject: finding.subject,
4429
4855
  claim: finding.claim,
4430
4856
  rationale: finding.rationale,
@@ -4433,18 +4859,16 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4433
4859
  evidence_refs: evidenceRefsFromRawFinding(finding),
4434
4860
  recommended_action: finding.recommended_action,
4435
4861
  metadata: {
4436
- analysis_mode: "recursive",
4437
- engine: "dspy-rlm",
4862
+ ...binding.findingBaseMetadata,
4438
4863
  model: config.model,
4439
- ...dataset === "codetracebench" ? codeTraceBlockMetadataFromSubject(finding.subject) : {}
4864
+ ...binding.metadataFromSubject?.(finding.subject) ?? {}
4440
4865
  },
4441
4866
  produced_at: producedAt
4442
4867
  }));
4443
- const adapted = await adaptPublicBenchmarkFindings({
4444
- dataset,
4445
- trajectoryId,
4868
+ const adapted = await binding.adapt({
4869
+ subject: trajectoryId,
4446
4870
  findings: rawFindings,
4447
- analystId: "dspy-rlm",
4871
+ analystId: definition.id,
4448
4872
  store: input.traceStore,
4449
4873
  ...context.signal ? { signal: context.signal } : {}
4450
4874
  });
@@ -4463,8 +4887,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4463
4887
  findings: fallback && !fallback.error ? fallback.findings : adapted.findings,
4464
4888
  usage,
4465
4889
  metadata: {
4466
- analysisMode: "recursive",
4467
- engine: "dspy-rlm",
4890
+ ...binding.baseMetadata,
4468
4891
  protocolSha256,
4469
4892
  ...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
4470
4893
  answer: completed.answer,
@@ -4487,8 +4910,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4487
4910
  usage,
4488
4911
  error: publicBenchmarkError(error, []),
4489
4912
  metadata: {
4490
- analysisMode: "recursive",
4491
- engine: "dspy-rlm",
4913
+ ...binding.baseMetadata,
4492
4914
  ...samples > 1 ? { samples } : {},
4493
4915
  rawFindings
4494
4916
  }
@@ -4516,18 +4938,6 @@ function sampleBlockRecords(assignments) {
4516
4938
  acceptedSteps
4517
4939
  }));
4518
4940
  }
4519
- function publicBenchmarkDefinition(dataset, limits, instructions) {
4520
- return {
4521
- id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
4522
- description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
4523
- area: dataset === "agentrx" ? "root-cause" : "incorrect",
4524
- version: "1.0.0",
4525
- question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
4526
- instructions,
4527
- toolGroup: "singleTrace",
4528
- limits
4529
- };
4530
- }
4531
4941
  function pricingForModel$1(model) {
4532
4942
  const pricing = resolveModelPricing(model);
4533
4943
  if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
@@ -4891,455 +5301,31 @@ function nodeHttpPrimeBridgeTransport() {
4891
5301
  };
4892
5302
  }
4893
5303
  //#endregion
4894
- //#region src/analyst/prime-protocol.ts
4895
- function buildPrimePrompt(spec) {
4896
- return [
4897
- `QUESTION: ${spec.question}`,
4898
- "",
4899
- ...spec.taskDefinition === void 0 ? [] : [
4900
- "TASK DEFINITION:",
4901
- spec.taskDefinition,
4902
- ""
4903
- ],
4904
- ...spec.contractLines,
4905
- "",
4906
- spec.trajectoryHeader,
4907
- spec.renderedTrajectory,
4908
- ...spec.trailer === void 0 ? [] : ["", spec.trailer]
4909
- ].join("\n");
4910
- }
4911
- /** Carries the malformed reply and the contract — never the trajectory. */
4912
- function buildPrimeRepairPrompt(spec) {
4913
- return [
4914
- "Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
4915
- `(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
4916
- "fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
4917
- ...spec.repairContractLines,
4918
- "",
4919
- "PREVIOUS REPLY:",
4920
- spec.previousReply
4921
- ].join("\n");
4922
- }
4923
- /**
4924
- * Recover the reply's JSON object.
4925
- *
4926
- * Distinct from `extractJsonPayload` in ../llm-client, which serves a response
4927
- * that DECLARES a JSON root and therefore must not scan onward. A prime reply
4928
- * is prose plus a fenced block, and when the model emits several fences the
4929
- * last one is its answer — so fences are scanned in reverse, and only then is a
4930
- * brace-to-brace slice tried.
4931
- */
4932
- function extractPrimeJsonObject(text) {
4933
- const direct = parsePrimeJsonObject(text);
4934
- if (direct) return direct;
4935
- const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
4936
- for (let index = fenced.length - 1; index >= 0; index -= 1) {
4937
- const candidate = parsePrimeJsonObject(fenced[index][1]);
4938
- if (candidate) return candidate;
4939
- }
4940
- const start = text.indexOf("{");
4941
- const end = text.lastIndexOf("}");
4942
- if (start >= 0 && end > start) {
4943
- const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
4944
- if (candidate) return candidate;
4945
- }
4946
- return null;
4947
- }
4948
- /** Why the reply cannot be read as a prime answer, or null when it can. */
4949
- function primeReplyDefect(parsed, rowsField) {
4950
- if (parsed === null) return "no parseable JSON object";
4951
- if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
4952
- return null;
4953
- }
4954
- function parsePrimeJsonObject(text) {
4955
- try {
4956
- const value = JSON.parse(text.trim());
4957
- return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
4958
- } catch {
4959
- return null;
4960
- }
4961
- }
4962
- function emptyPrimeRawUsage() {
4963
- return {
4964
- calls: null,
4965
- inputTokens: null,
4966
- outputTokens: null,
4967
- bridgeEstimated: false
4968
- };
4969
- }
4970
- /** Read the bridge's OpenAI-shaped `usage` object. */
4971
- function normalizePrimeUsage(raw) {
4972
- if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
4973
- const record = raw;
4974
- return {
4975
- calls: tokenCountOrNull(record.model_requests),
4976
- inputTokens: tokenCountOrNull(record.prompt_tokens),
4977
- outputTokens: tokenCountOrNull(record.completion_tokens),
4978
- bridgeEstimated: record.estimated === true
4979
- };
4980
- }
4981
- /**
4982
- * Sum two turns. Each side poisons independently: two turns that both report
4983
- * input and neither report output yield a real input total beside a null
4984
- * output, because discarding a measured count is as wrong as inventing one.
4985
- */
4986
- function mergePrimeRawUsage(a, b) {
4987
- return {
4988
- calls: sumOrNull(a.calls, b.calls),
4989
- inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
4990
- outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
4991
- bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
4992
- };
4993
- }
4994
- function sumOrNull(a, b) {
4995
- return a !== null && b !== null ? a + b : null;
4996
- }
4997
- function tokenCountOrNull(value) {
4998
- return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
4999
- }
5000
- /**
5001
- * Bind raw prime usage to agent-eval's typed receipt.
5002
- *
5003
- * `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
5004
- * cannot round-trip through `tokens` without writing a zero nobody measured.
5005
- * The complete-accounting field therefore stays null, the reported side is
5006
- * carried verbatim in `partialTokens`, and its price becomes the receipt's
5007
- * `knownCostUsd` lower bound.
5008
- *
5009
- * Only agent-eval calls this; consumers with no pricing table read
5010
- * `PrimeRawUsage` directly.
5011
- */
5012
- function analystUsageReceiptFromPrimeUsage(usage, pricing) {
5013
- const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
5014
- const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
5015
- if (inputTokens !== null && outputTokens !== null) return {
5016
- calls,
5017
- tokens: {
5018
- input: inputTokens,
5019
- output: outputTokens
5020
- },
5021
- cost: {
5022
- kind: "estimated",
5023
- usd: priceTokens(inputTokens, outputTokens, pricing)
5024
- },
5025
- ...estimatedTokens
5026
- };
5027
- if (inputTokens === null && outputTokens === null) return {
5028
- calls,
5029
- tokens: null,
5030
- cost: {
5031
- kind: "uncaptured",
5032
- usd: null
5033
- },
5034
- ...estimatedTokens
5035
- };
5036
- return {
5037
- calls,
5038
- tokens: null,
5039
- partialTokens: {
5040
- input: inputTokens,
5041
- output: outputTokens
5042
- },
5043
- cost: {
5044
- kind: "uncaptured",
5045
- usd: null
5046
- },
5047
- knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
5048
- ...estimatedTokens
5049
- };
5050
- }
5051
- function priceTokens(input, output, pricing) {
5052
- return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
5053
- }
5054
- /**
5055
- * Run the protocol: one call, one bounded repair turn on a structurally
5056
- * malformed reply, then decode. Zero valid rows from a well-formed reply is an
5057
- * honest null, not a failure.
5058
- */
5059
- async function runPrimeExchange(options) {
5060
- const { contract } = options;
5061
- const turns = [];
5062
- const repair = {
5063
- attempted: false,
5064
- succeeded: null
5065
- };
5066
- const first = await callPrimeTurn(options, options.prompt);
5067
- if (!first.ok) return {
5068
- ok: false,
5069
- failure: first.failure,
5070
- usage: mergeTurns(turns),
5071
- turns,
5072
- repair
5073
- };
5074
- turns.push({
5075
- turn: "first",
5076
- usage: first.usage,
5077
- rawUsage: first.rawUsage
5078
- });
5079
- let reply = first.content;
5080
- let parsed = extractPrimeJsonObject(reply);
5081
- let defect = primeReplyDefect(parsed, contract.rowsField);
5082
- if (defect !== null && options.repair) {
5083
- repair.attempted = true;
5084
- const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
5085
- defect,
5086
- previousReply: reply,
5087
- repairContractLines: contract.repairContractLines
5088
- }));
5089
- if (!second.ok) return {
5090
- ok: false,
5091
- failure: second.failure,
5092
- usage: mergeTurns(turns),
5093
- turns,
5094
- repair,
5095
- reply
5096
- };
5097
- turns.push({
5098
- turn: "repair",
5099
- usage: second.usage,
5100
- rawUsage: second.rawUsage
5101
- });
5102
- reply = second.content;
5103
- parsed = extractPrimeJsonObject(reply);
5104
- defect = primeReplyDefect(parsed, contract.rowsField);
5105
- repair.succeeded = defect === null;
5106
- }
5107
- const usage = mergeTurns(turns);
5108
- if (defect !== null) return {
5109
- ok: false,
5110
- failure: {
5111
- kind: "malformed-reply",
5112
- message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
5113
- },
5114
- usage,
5115
- turns,
5116
- repair,
5117
- reply
5118
- };
5119
- const rawRows = parsed[contract.rowsField];
5120
- const rows = [];
5121
- const rejected = [];
5122
- let overflow = 0;
5123
- rawRows.forEach((row, index) => {
5124
- const decoded = contract.decodeRow(row, index);
5125
- if (!decoded.ok) {
5126
- rejected.push({
5127
- index,
5128
- reason: decoded.reason
5129
- });
5130
- return;
5131
- }
5132
- if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
5133
- overflow += 1;
5134
- return;
5135
- }
5136
- rows.push(decoded.row);
5137
- });
5138
- const answer = parsed.answer;
5139
- return {
5140
- ok: true,
5141
- answer: typeof answer === "string" ? answer : null,
5142
- rows,
5143
- rejected,
5144
- reportedRows: rawRows.length,
5145
- overflow,
5146
- usage,
5147
- turns,
5148
- repair,
5149
- reply
5150
- };
5151
- }
5152
- async function callPrimeTurn(options, content) {
5153
- const { transport, url, model, timeoutMs, signal } = options;
5154
- const controller = new AbortController();
5155
- const forwardAbort = () => controller.abort(signal?.reason);
5156
- if (signal?.aborted) controller.abort(signal.reason);
5157
- else signal?.addEventListener("abort", forwardAbort, { once: true });
5158
- const deadline = setTimeout(() => controller.abort(), timeoutMs);
5159
- let result;
5160
- try {
5161
- result = await transport({
5162
- url,
5163
- body: {
5164
- model,
5165
- messages: [{
5166
- role: "user",
5167
- content
5168
- }]
5169
- },
5170
- signal: controller.signal
5171
- });
5172
- } catch (error) {
5173
- if (signal?.aborted) return {
5174
- ok: false,
5175
- failure: {
5176
- kind: "aborted",
5177
- message: "prime exchange cancelled by the caller",
5178
- cause: error
5179
- }
5180
- };
5181
- if (controller.signal.aborted) return {
5182
- ok: false,
5183
- failure: {
5184
- kind: "deadline",
5185
- message: `bridge call exceeded ${timeoutMs}ms`
5186
- }
5187
- };
5188
- return {
5189
- ok: false,
5190
- failure: {
5191
- kind: "transport",
5192
- message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
5193
- }
5194
- };
5195
- } finally {
5196
- clearTimeout(deadline);
5197
- signal?.removeEventListener("abort", forwardAbort);
5198
- }
5199
- if (result.status !== 200) {
5200
- const bodySnippet = result.text.slice(0, 500);
5201
- return {
5202
- ok: false,
5203
- failure: {
5204
- kind: "http-status",
5205
- message: `bridge HTTP ${result.status}: ${bodySnippet}`,
5206
- status: result.status,
5207
- bodySnippet
5208
- }
5209
- };
5210
- }
5211
- let response;
5212
- try {
5213
- response = JSON.parse(result.text);
5214
- } catch {
5215
- return {
5216
- ok: false,
5217
- failure: {
5218
- kind: "unparseable-json",
5219
- message: `bridge returned unparseable JSON (${result.text.length} bytes)`
5220
- }
5221
- };
5222
- }
5223
- const replyContent = primeReplyContent(response);
5224
- if (replyContent === null) return {
5225
- ok: false,
5226
- failure: {
5227
- kind: "no-content",
5228
- message: "bridge reply carries no message content"
5229
- }
5230
- };
5231
- const rawUsage = primeReplyUsage(response);
5232
- return {
5233
- ok: true,
5234
- content: replyContent,
5235
- usage: normalizePrimeUsage(rawUsage),
5236
- rawUsage
5237
- };
5238
- }
5239
- /**
5240
- * Fold from the FIRST turn, never from an empty receipt: an all-null identity
5241
- * would poison every side it merged with and erase counts the bridge reported.
5242
- */
5243
- function mergeTurns(turns) {
5244
- if (turns.length === 0) return emptyPrimeRawUsage();
5245
- return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
5246
- }
5247
- function primeReplyContent(response) {
5248
- if (typeof response !== "object" || response === null) return null;
5249
- const choices = response.choices;
5250
- if (!Array.isArray(choices) || choices.length === 0) return null;
5251
- const message = choices[0]?.message;
5252
- if (typeof message !== "object" || message === null) return null;
5253
- const content = message.content;
5254
- return typeof content === "string" && content.length > 0 ? content : null;
5255
- }
5256
- function primeReplyUsage(response) {
5257
- if (typeof response !== "object" || response === null) return null;
5258
- return response.usage ?? null;
5259
- }
5260
- /**
5261
- * Render, measure, fall back to the capped projection, re-measure, fail loud.
5262
- *
5263
- * Inline is the only delivery prime has, so an oversized trajectory is a
5264
- * refusal rather than a silent truncation: dropping spans would understate the
5265
- * trajectory and the analyst would answer a question about a different run.
5266
- */
5267
- async function projectPrimeTrajectory(source, limits) {
5268
- let fetch = "full";
5269
- let items = await source.full();
5270
- if (items === null) {
5271
- fetch = "capped";
5272
- items = await source.capped();
5273
- }
5274
- let rendered = JSON.stringify(items);
5275
- if (rendered.length > limits.maxInlineChars && fetch === "full") {
5276
- fetch = "capped";
5277
- items = await source.capped();
5278
- rendered = JSON.stringify(items);
5279
- }
5280
- if (rendered.length > limits.maxInlineChars) return {
5281
- ok: false,
5282
- reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
5283
- renderedChars: rendered.length
5284
- };
5285
- return {
5286
- ok: true,
5287
- items,
5288
- rendered,
5289
- delivery: {
5290
- mode: "inline-json",
5291
- fetch,
5292
- renderedChars: rendered.length
5293
- }
5294
- };
5295
- }
5296
- /**
5297
- * Digest of everything a consumer can send to the bridge under the prime
5298
- * protocol, recorded per observation so a prime result names the exact contract
5299
- * that produced it.
5300
- *
5301
- * Computed over the ACTUALLY composed contract, so two consumers that both
5302
- * stamp `analyst_id: 'prime'` while asking materially different questions get
5303
- * different digests by construction. That is what makes 'prime' a reproducible
5304
- * claim rather than a label.
5305
- */
5306
- function primeProtocolSha256(identity) {
5307
- return createHash("sha256").update(JSON.stringify({
5308
- kind: "prime-analyst-protocol",
5309
- question: identity.question,
5310
- taskPrompt: identity.taskDefinition ?? null,
5311
- outputContract: identity.contractLines,
5312
- repairContract: buildPrimeRepairPrompt({
5313
- defect: "<defect>",
5314
- previousReply: "<previous-reply>",
5315
- repairContractLines: identity.repairContractLines
5316
- }),
5317
- limits: identity.limits
5318
- })).digest("hex");
5319
- }
5320
- //#endregion
5321
5304
  //#region src/analyst/benchmark-runner-prime.ts
5322
5305
  /**
5323
5306
  * Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
5324
5307
  * cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
5325
5308
  * analyst.
5326
5309
  *
5327
- * The runner consumes the same prepared benchmark cases every other runner
5328
- * receives the trace store already carries the appended final-verification
5329
- * spans and produces findings through the same published block expansion, so
5330
- * a prime observation and a dspy-rlm observation differ only in which analyst
5331
- * produced the blocks.
5310
+ * The arm is expressed as an `AnalystDefinition`
5311
+ * (`primeCodeTraceAnalystDefinition`): the question, task text, output
5312
+ * contract, inline projection budget, and repair-turn declaration are all
5313
+ * definition content, and `createPrimeBenchmarkRunner` is a thin shell that
5314
+ * builds the definition and runs it through the inline strategy below. The
5315
+ * same strategy is what `bindAnalyst` (./bind) dispatches to, so a compiled
5316
+ * definition and this entry point send byte-identical requests — the parity
5317
+ * suite asserts exactly that.
5332
5318
  *
5333
- * The protocol itself — prompt composition, the bounded repair turn, reply
5319
+ * The protocol machinery — prompt composition, the bounded repair turn, reply
5334
5320
  * extraction, the projection ladder, usage normalization — lives in
5335
- * `./prime-protocol`, which knows nothing about CodeTraceBench. This file is
5336
- * the benchmark's binding to it: the block row grammar, the store-backed
5337
- * projection source, and the benchmark observation shape.
5321
+ * `./prime-protocol`, which knows nothing about CodeTraceBench. This file adds
5322
+ * the benchmark's binding to it (block row grammar, store-backed projection,
5323
+ * observation shape) plus the projection-generic inline execution strategy.
5338
5324
  *
5339
5325
  * Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
5340
5326
  * the viewTrace span projection as a REPL variable; prime has no REPL, so the
5341
5327
  * same projection is serialized into the prompt. When the full projection is
5342
- * oversized the runner falls back to chunked viewSpans over the same
5328
+ * oversized the strategy falls back to chunked viewSpans over the same
5343
5329
  * projection surface with a per-attribute byte cap, and fails loud if the
5344
5330
  * result still exceeds the inline budget.
5345
5331
  *
@@ -5441,56 +5427,142 @@ const PRIME_BLOCK_CONTRACT = {
5441
5427
  }
5442
5428
  };
5443
5429
  /**
5444
- * Digest of everything this runner can send to the bridge, recorded per
5430
+ * Digest of everything this arm can send to the bridge, recorded per
5445
5431
  * observation so a prime result names the exact contract that produced it.
5446
5432
  */
5447
5433
  function primeAnalystProtocolSha256() {
5448
5434
  return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
5449
5435
  }
5450
- /** CodeTraceBench-only: the prompt and output contract speak its block grammar. */
5436
+ /**
5437
+ * The prime arm as a declarative unit. CodeTraceBench-only: the question,
5438
+ * task text, and block grammar speak its incorrect-step definition.
5439
+ */
5440
+ function primeCodeTraceAnalystDefinition(args) {
5441
+ return {
5442
+ id: PRIME_ANALYST_ID,
5443
+ description: "One-shot RLM over an OpenAI-compatible bridge answering the CodeTraceBench incorrect-step task.",
5444
+ version: "1.0.0",
5445
+ area: "incorrect",
5446
+ profile: {},
5447
+ question: PRIME_QUESTION,
5448
+ taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5449
+ projection: {
5450
+ mode: "inline",
5451
+ maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS,
5452
+ cappedAttributeBytes: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5453
+ },
5454
+ replyContract: PRIME_BLOCK_CONTRACT,
5455
+ contractLimits: {
5456
+ maxBlocks: 16,
5457
+ maxBlockSteps: 12
5458
+ },
5459
+ budget: { timeoutMs: args.timeoutMs },
5460
+ repair: { turns: args.repairTurns },
5461
+ protocolSha256: primeAnalystProtocolSha256(),
5462
+ binding: {
5463
+ kind: "inline",
5464
+ subjectFromCaseId: trajectoryIdFromCaseId,
5465
+ baseMetadata: {
5466
+ analysisMode: "prime-rlm",
5467
+ engine: "prime"
5468
+ },
5469
+ header(subject, spans) {
5470
+ const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
5471
+ if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${subject}'`);
5472
+ return `TRAJECTORY (trace_id ${subject}; ${stepSpans.length} assistant step spans; full span projection as JSON):`;
5473
+ },
5474
+ trailer(_subject, spans) {
5475
+ const finalVerification = spans.filter(isFinalVerificationSpan);
5476
+ return finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself.";
5477
+ },
5478
+ async expandRows({ subject, rows, store, analystId, signal }) {
5479
+ const expanded = await expandCodeTraceFailureBlocks({
5480
+ trajectoryId: subject,
5481
+ blocks: rows,
5482
+ store,
5483
+ analystId,
5484
+ ...signal ? { signal } : {}
5485
+ });
5486
+ return {
5487
+ findings: expanded.findings,
5488
+ diagnostics: expanded.diagnostics
5489
+ };
5490
+ }
5491
+ }
5492
+ };
5493
+ }
5494
+ /** Thin shell: validate options, declare the definition, run the inline strategy. */
5451
5495
  function createPrimeBenchmarkRunner(options) {
5452
- const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
5453
- const model = requiredString(options.model, "model");
5454
5496
  const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
5455
5497
  const repair = options.repair;
5456
5498
  if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
5457
- const pricing = options.pricing ?? pricingForModel(model);
5458
- const transport = options.transport ?? nodeHttpPrimeBridgeTransport();
5499
+ return runInlineAnalystDefinition(primeCodeTraceAnalystDefinition({
5500
+ timeoutMs,
5501
+ repairTurns: repair ? 1 : 0
5502
+ }), {
5503
+ baseUrl: options.baseUrl,
5504
+ model: options.model,
5505
+ ...options.transport ? { transport: options.transport } : {},
5506
+ ...options.pricing ? { pricing: options.pricing } : {}
5507
+ });
5508
+ }
5509
+ /**
5510
+ * Compile an inline-projection definition into a runnable arm. Projection,
5511
+ * prompt composition, the bounded repair turn, and usage accounting are all
5512
+ * driven by the definition; nothing in this strategy names a benchmark.
5513
+ */
5514
+ function runInlineAnalystDefinition(definition, transports) {
5515
+ const { projection, binding } = definition;
5516
+ if (projection.mode !== "inline" || binding.kind !== "inline") throw new AnalystExpressivenessError(`the inline strategy compiles only inline projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
5517
+ if (definition.repair.turns > 1) throw new AnalystExpressivenessError(`the inline exchange grants at most one bounded repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
5518
+ const baseUrl = requiredString(transports.baseUrl, "baseUrl").replace(/\/+$/, "");
5519
+ const model = requiredString(transports.model, "model");
5520
+ const timeoutMs = positiveSafeInteger(definition.budget.timeoutMs, "timeoutMs");
5521
+ const repair = definition.repair.turns === 1;
5522
+ const pricing = transports.pricing ?? pricingForModel(model);
5523
+ const transport = transports.transport ?? nodeHttpPrimeBridgeTransport();
5459
5524
  const url = `${baseUrl}/v1/chat/completions`;
5460
5525
  return {
5461
- id: PRIME_ANALYST_ID,
5526
+ id: definition.id,
5462
5527
  async analyze(input, context) {
5463
- const trajectoryId = trajectoryIdFromCaseId(context.caseId);
5528
+ const subject = binding.subjectFromCaseId(context.caseId);
5464
5529
  let usage;
5465
5530
  let metadata = {
5466
- analysisMode: "prime-rlm",
5467
- engine: "prime",
5531
+ ...binding.baseMetadata,
5468
5532
  bridgeUrl: baseUrl,
5469
5533
  model,
5470
- protocolSha256: primeAnalystProtocolSha256()
5534
+ protocolSha256: definition.protocolSha256
5471
5535
  };
5472
5536
  try {
5473
5537
  const store = input.traceStore;
5474
- if (!store) throw new Error("codetracebench prime runner requires a trace store");
5475
- const projection = await projectPrimeTrajectory(codeTraceProjectionSource(store, trajectoryId, context.signal ? { signal: context.signal } : void 0), { maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS });
5476
- if (!projection.ok) throw new PrimeTraceProjectionError(projection.reason);
5538
+ if (!store) throw new Error(`inline analyst '${definition.id}' requires a trace store`);
5539
+ const storeContext = context.signal ? { signal: context.signal } : void 0;
5540
+ const projected = await projectPrimeTrajectory(inlineProjectionSource(store, subject, projection.cappedAttributeBytes, storeContext), { maxInlineChars: projection.maxInlineChars });
5541
+ if (!projected.ok) throw new PrimeTraceProjectionError(projected.reason);
5477
5542
  const delivery = {
5478
- mode: projection.delivery.mode,
5479
- fetch: projection.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
5480
- perAttributeByteCap: projection.delivery.fetch === "full" ? null : CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP,
5481
- renderedChars: projection.delivery.renderedChars
5543
+ mode: projected.delivery.mode,
5544
+ fetch: projected.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
5545
+ perAttributeByteCap: projected.delivery.fetch === "full" ? null : projection.cappedAttributeBytes,
5546
+ renderedChars: projected.delivery.renderedChars
5482
5547
  };
5483
5548
  metadata = {
5484
5549
  ...metadata,
5485
5550
  delivery
5486
5551
  };
5487
- const prompt = buildCodeTracePrompt(trajectoryId, projection.items, projection.rendered);
5552
+ const prompt = buildPrimePrompt({
5553
+ question: definition.question,
5554
+ ...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
5555
+ contractLines: definition.replyContract.contractLines,
5556
+ trajectoryHeader: binding.header(subject, projected.items),
5557
+ renderedTrajectory: projected.rendered,
5558
+ trailer: binding.trailer(subject, projected.items)
5559
+ });
5488
5560
  metadata = {
5489
5561
  ...metadata,
5490
5562
  promptChars: prompt.length
5491
5563
  };
5492
5564
  const outcome = await runPrimeExchange({
5493
- contract: PRIME_BLOCK_CONTRACT,
5565
+ contract: definition.replyContract,
5494
5566
  prompt,
5495
5567
  transport,
5496
5568
  url,
@@ -5518,11 +5590,11 @@ function createPrimeBenchmarkRunner(options) {
5518
5590
  };
5519
5591
  throw primeFailureError(outcome.failure);
5520
5592
  }
5521
- const expanded = await expandCodeTraceFailureBlocks({
5522
- trajectoryId,
5523
- blocks: outcome.rows,
5593
+ const expanded = await binding.expandRows({
5594
+ subject,
5595
+ rows: outcome.rows,
5524
5596
  store,
5525
- analystId: PRIME_ANALYST_ID,
5597
+ analystId: definition.id,
5526
5598
  ...context.signal ? { signal: context.signal } : {}
5527
5599
  });
5528
5600
  return {
@@ -5553,13 +5625,13 @@ function createPrimeBenchmarkRunner(options) {
5553
5625
  * the full viewTrace projection, or the chunked viewSpans projection at a
5554
5626
  * per-attribute byte cap.
5555
5627
  */
5556
- function codeTraceProjectionSource(store, trajectoryId, context) {
5628
+ function inlineProjectionSource(store, trajectoryId, cappedAttributeBytes, context) {
5557
5629
  return {
5558
5630
  async full() {
5559
5631
  return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
5560
5632
  },
5561
- capped: () => projectSpansChunked(store, trajectoryId, context),
5562
- cappedDescription: `per-attribute cap ${CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP}`
5633
+ capped: () => projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context),
5634
+ cappedDescription: `per-attribute cap ${cappedAttributeBytes}`
5563
5635
  };
5564
5636
  }
5565
5637
  /**
@@ -5568,7 +5640,7 @@ function codeTraceProjectionSource(store, trajectoryId, context) {
5568
5640
  * id must project or the case fails loud — a silently dropped span would
5569
5641
  * understate the trajectory.
5570
5642
  */
5571
- async function projectSpansChunked(store, trajectoryId, context) {
5643
+ async function projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context) {
5572
5644
  const enumeration = await store.viewTrace({
5573
5645
  trace_id: trajectoryId,
5574
5646
  per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
@@ -5587,26 +5659,13 @@ async function projectSpansChunked(store, trajectoryId, context) {
5587
5659
  const result = await store.viewSpans({
5588
5660
  trace_id: trajectoryId,
5589
5661
  span_ids: chunk,
5590
- per_attribute_byte_cap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5662
+ per_attribute_byte_cap: cappedAttributeBytes
5591
5663
  }, context);
5592
5664
  if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
5593
5665
  projected.push(...result.spans);
5594
5666
  }
5595
5667
  return projected;
5596
5668
  }
5597
- function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
5598
- const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
5599
- if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
5600
- const finalVerification = spans.filter(isFinalVerificationSpan);
5601
- return buildPrimePrompt({
5602
- question: PRIME_QUESTION,
5603
- taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5604
- contractLines: PRIME_OUTPUT_CONTRACT_LINES,
5605
- trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
5606
- renderedTrajectory: renderedSpans,
5607
- trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
5608
- });
5609
- }
5610
5669
  /** Map the protocol's terminal reason onto this benchmark's typed error classes. */
5611
5670
  function primeFailureError(failure) {
5612
5671
  switch (failure.kind) {
@@ -6351,6 +6410,6 @@ function shellQuote(value) {
6351
6410
  return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
6352
6411
  }
6353
6412
  //#endregion
6354
- export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, createPublicBenchmarkRlmRunner as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, selectPublicBenchmarkRows as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, nodeHttpPrimeBridgeTransport as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, publicBenchmarkDistributions as b, buildPrimeRepairPrompt as c, summarizeAgentRxCalibration as ct, mergePrimeRawUsage as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, normalizePrimeUsage as f, agentRxPredictionsToFindings as ft, runPrimeExchange as g, projectPrimeTrajectory as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, emptyPrimeRawUsage as l, codeTraceBenchCase as lt, primeReplyDefect as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, analystUsageReceiptFromPrimeUsage as o, AGENT_RX_UPSTREAM_REVISION as ot, primeProtocolSha256 as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, buildPrimePrompt as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, extractPrimeJsonObject as u, codeTracerPredictionsToFindings as ut, loadPublicBenchmarkRows as v, createPublicBenchmarkDirectRunner as w, publicBenchmarkSelectionReport as x, preparePublicAnalystBenchmark as y, publicBenchmarkRlmInstructions as z };
6413
+ export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, analystDefinitionAsymmetries as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, AnalystExpressivenessError as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, summarizeAgentRxCalibration as ct, publicBenchmarkDistributions as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, publicBenchmarkSelectionReport as f, agentRxPredictionsToFindings as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, loadPublicBenchmarkRows as l, codeTraceBenchCase as lt, createPublicBenchmarkRlmRunner as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, primeCodeTraceAnalystDefinition as o, AGENT_RX_UPSTREAM_REVISION as ot, selectPublicBenchmarkRows as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, runInlineAnalystDefinition as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, preparePublicAnalystBenchmark as u, codeTracerPredictionsToFindings as ut, createPublicBenchmarkDirectRunner as v, analystDefinitionProtocolSha256 as w, decodeReplyRows as x, publicDirectAnalystDefinition as y, publicBenchmarkRlmInstructions as z };
6355
6414
 
6356
- //# sourceMappingURL=benchmark-command-CA_NFOmy.js.map
6415
+ //# sourceMappingURL=benchmark-command-BCafwNrf.js.map