@tangle-network/agent-eval 0.144.6 → 0.144.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
  14. package/dist/benchmark-command-BKENp2s5.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
  22. package/dist/campaign--HVSuvV0.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
  160. package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/campaign-proposers.md +5 -0
  207. package/docs/charter.md +112 -0
  208. package/docs/experiment.md +104 -0
  209. package/docs/prime-analyst.md +1 -0
  210. package/docs/trace-analysis.md +26 -0
  211. package/docs/trace-repair-admission.md +194 -0
  212. package/docs/trace-repair-analyst-arms.md +121 -0
  213. package/docs/trace-repair-continuation.md +107 -0
  214. package/docs/trace-repair-grader.md +163 -0
  215. package/docs/trajectory-replay.md +110 -0
  216. package/docs/verification-strategies.md +103 -0
  217. package/package.json +19 -2
  218. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  219. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  220. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  221. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  222. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  223. package/dist/campaign-CXsdyym7.js.map +0 -1
  224. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  225. package/dist/index-4XwggC10.d.ts.map +0 -1
  226. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  227. package/dist/integrity-fdt8XPAv.js.map +0 -1
  228. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  229. package/dist/replay-Krvb114g.d.ts.map +0 -1
  230. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  231. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  232. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  233. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  234. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  235. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  236. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  237. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  238. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  239. package/dist/verdict-Dps8_okt.d.ts +0 -37
  240. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -1,16 +1,17 @@
1
1
  import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
2
  import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
3
  import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
4
- import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-Dv5BiKLE.js";
5
- import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
4
+ import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-DzvMUsS_.js";
5
+ import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
6
6
  import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
7
7
  import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
- import { M as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D5iN0Xzb.js";
9
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
8
+ import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BMQEv1wG.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
10
10
  import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
11
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
12
  import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
13
13
  import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
14
+ import { c as primeProtocolSha256, d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-BfSalTfR.js";
14
15
  import { z } from "zod";
15
16
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
16
17
  import * as nodePath from "node:path";
@@ -1207,7 +1208,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1207
1208
  "package.json",
1208
1209
  "pnpm-lock.yaml"
1209
1210
  ]);
1210
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "c0525dfe7f6931aeefe0e4c98d366cf6558c176e56477c5271d35274f383cf61";
1211
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "788e0d4e228836e84b4cf31492c9a0f7c884efcce088623988f9c378c768ef7c";
1211
1212
  /** The published benchmark evidence was produced at this package version, by
1212
1213
  * the retired one-shot direct runner, before trace analysts moved to the
1213
1214
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -1252,8 +1253,10 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1252
1253
  "src/analyst/benchmark-verification-artifacts.ts",
1253
1254
  "src/analyst/benchmark-verification-outcome.ts",
1254
1255
  "src/analyst/benchmark.ts",
1256
+ "src/analyst/definition.ts",
1255
1257
  "src/analyst/dspy-rlm-engine.ts",
1256
1258
  "src/analyst/engine.ts",
1259
+ "src/analyst/equal-terms.ts",
1257
1260
  "src/analyst/exact-types.ts",
1258
1261
  "src/analyst/finding-signature.ts",
1259
1262
  "src/analyst/finding-subject.ts",
@@ -1261,6 +1264,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1261
1264
  "src/analyst/parse-tolerant.ts",
1262
1265
  "src/analyst/prime-bridge-transport.ts",
1263
1266
  "src/analyst/prime-protocol.ts",
1267
+ "src/analyst/reply-contract.ts",
1264
1268
  "src/analyst/tool-groups.ts",
1265
1269
  "src/analyst/trace-tool-callback.ts",
1266
1270
  "src/analyst/types.ts",
@@ -1311,7 +1315,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1311
1315
  "src/trace/raw-provider-sink.ts",
1312
1316
  "src/verdict-cache.ts"
1313
1317
  ]);
1314
- const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "47592eaceab13f9cf40530e02a4af3d9ff5bcd6a7eb7d92b110508cc6cd323e5";
1318
+ const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "cc354effd79c8dfc8669c230706c67c63a06e3aff013f540806361263bae8860";
1315
1319
  function analystBenchmarkImplementationDigest() {
1316
1320
  return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
1317
1321
  }
@@ -2238,19 +2242,24 @@ Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/
2238
2242
  Give the rationale as the concrete downstream evidence visible at the consequence step.
2239
2243
  Submit as soon as every candidate failure block has a supported verdict.
2240
2244
  Return no finding for a clean trajectory.`;
2241
- /** One-shot JSON transport prompt for the direct runner. */
2242
- function publicBenchmarkSystemPrompt(dataset) {
2243
- const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
2244
- return `${publicBenchmarkTaskPrompt(dataset)}
2245
-
2246
- ${fieldContract}
2247
-
2248
- Return exactly one JSON object with:
2245
+ /** Reply-envelope contract shared by both one-shot datasets. */
2246
+ const PUBLIC_BENCHMARK_ENVELOPE_CONTRACT = `Return exactly one JSON object with:
2249
2247
  - "report": a concise evidence-based explanation, at most 4000 characters
2250
2248
  - "findings": the strict finding array
2251
2249
  Use an empty findings array when the trace does not support a finding.
2252
2250
  Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
2253
2251
  The runner constructs exact trace URIs and action previews from each selected step.`;
2252
+ /** Per-dataset field grammar for the one-shot JSON reply. */
2253
+ function publicBenchmarkFieldContract(dataset) {
2254
+ return dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
2255
+ }
2256
+ /** One-shot JSON transport prompt for the direct runner. */
2257
+ function publicBenchmarkSystemPrompt(dataset) {
2258
+ return [
2259
+ publicBenchmarkTaskPrompt(dataset),
2260
+ publicBenchmarkFieldContract(dataset),
2261
+ PUBLIC_BENCHMARK_ENVELOPE_CONTRACT
2262
+ ].join("\n\n");
2254
2263
  }
2255
2264
  /** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
2256
2265
  function publicBenchmarkRlmInstructions(dataset) {
@@ -2258,6 +2267,7 @@ function publicBenchmarkRlmInstructions(dataset) {
2258
2267
  return `${publicBenchmarkTaskPrompt(dataset)}
2259
2268
  ${outputContract}`;
2260
2269
  }
2270
+ /** Task text shared by every runner shape on one dataset. */
2261
2271
  function publicBenchmarkTaskPrompt(dataset) {
2262
2272
  return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
2263
2273
  }
@@ -3594,61 +3604,300 @@ function fileContext() {
3594
3604
  };
3595
3605
  }
3596
3606
  //#endregion
3607
+ //#region src/analyst/definition.ts
3608
+ /**
3609
+ * AnalystDefinition — the declarative unit behind an analyst arm.
3610
+ *
3611
+ * An arm is one way of EXECUTING an analysis question: a one-shot JSON call, a
3612
+ * bridge-reached RLM, a recursive engine with trace tools. What the arm SAYS —
3613
+ * the question, the task text, the reply grammar, how evidence reaches the
3614
+ * model, the repair-turn and budget terms — is protocol, not execution, so it
3615
+ * lives here as one inspectable value. `bindAnalyst` (./bind) compiles a
3616
+ * definition plus a transport binding into a runnable arm, and the parity
3617
+ * suite holds the compiled arm to the byte against the arm's entry point, so a
3618
+ * definition cannot drift from what its arm actually sends.
3619
+ *
3620
+ * Three rules carried over from the repair-arm comparison contract
3621
+ * (trace-repair's `repairArmAsymmetries`), made structural here:
3622
+ *
3623
+ * one contract the reply grammar is a `ReplyContract` value on the
3624
+ * definition, never prose inside a runner body.
3625
+ * one repair turn `analystDefinitionAsymmetries` refuses a set whose
3626
+ * definitions declare unequal repair turns, because a second
3627
+ * attempt is a second sample the other arms never got.
3628
+ * declared difference what arms MAY differ in — the evidence projection, the
3629
+ * reasoning effort, the budget — is declared per definition
3630
+ * and rendered beside the comparison instead of being
3631
+ * inferred from two runners' source.
3632
+ */
3633
+ /**
3634
+ * Thrown at bind time when a definition asks for something no strategy can
3635
+ * compile — an unknown projection × transport pair, a repair-turn count the
3636
+ * exchange machinery cannot grant, a reasoning effort the arm cannot map. The
3637
+ * message names the construct so an expressiveness gap is a loud, attributable
3638
+ * failure instead of a silently narrowed protocol.
3639
+ */
3640
+ var AnalystExpressivenessError = class extends Error {};
3641
+ /**
3642
+ * Digest of everything a definition can send to its model. An inline
3643
+ * definition hashes under the historical prime-protocol domain, so its digest
3644
+ * equals the digest its bespoke arm always recorded; other projections hash
3645
+ * under the definition domain.
3646
+ */
3647
+ function analystDefinitionProtocolSha256(definition) {
3648
+ const { projection, replyContract } = definition;
3649
+ if (projection.mode === "inline") return primeProtocolSha256({
3650
+ question: definition.question,
3651
+ ...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
3652
+ contractLines: replyContract.contractLines,
3653
+ repairContractLines: replyContract.repairContractLines,
3654
+ limits: {
3655
+ ...definition.contractLimits,
3656
+ maxInlineTrajectoryChars: projection.maxInlineChars,
3657
+ chunkedProjectionAttributeByteCap: projection.cappedAttributeBytes
3658
+ }
3659
+ });
3660
+ return createHash("sha256").update(JSON.stringify({
3661
+ kind: "analyst-definition-protocol",
3662
+ mode: projection.mode,
3663
+ question: definition.question,
3664
+ taskDefinition: definition.taskDefinition ?? null,
3665
+ contractLines: replyContract.contractLines,
3666
+ repairContractLines: replyContract.repairContractLines,
3667
+ limits: definition.contractLimits,
3668
+ projection: projection.mode === "chunked" ? { attributeByteCaps: projection.attributeByteCaps } : { toolGroup: projection.toolGroup }
3669
+ })).digest("hex");
3670
+ }
3671
+ /**
3672
+ * Refuse a set of definitions that cannot be compared on equal terms, and
3673
+ * render what still differs between the ones that can. The hard rule is the
3674
+ * repair turn: a malformed reply must earn the same number of retries in every
3675
+ * arm, because a retry is a second sample. Projection, reasoning effort, and
3676
+ * budget differences are declared and reported, never hidden.
3677
+ */
3678
+ function analystDefinitionAsymmetries(definitions) {
3679
+ const { ids, repairTurns } = assertEqualDeclarativeTerms("analyst definition", definitions.map((definition) => ({
3680
+ id: definition.id,
3681
+ repairTurns: definition.repair.turns
3682
+ })));
3683
+ const firstMode = definitions[0].projection.mode;
3684
+ return {
3685
+ ids,
3686
+ repairTurns,
3687
+ sharedProjectionMode: definitions.every((definition) => definition.projection.mode === firstMode) ? firstMode : null,
3688
+ asymmetries: definitions.map((definition) => ({
3689
+ id: definition.id,
3690
+ projectionMode: definition.projection.mode,
3691
+ reasoningEffort: definition.profile.model?.reasoningEffort ?? null,
3692
+ timeoutMs: definition.budget.timeoutMs,
3693
+ maxCostUsd: definition.budget.maxCostUsd ?? null,
3694
+ maxOutputTokens: definition.budget.maxOutputTokens ?? null,
3695
+ protocolSha256: definition.protocolSha256,
3696
+ definitionSha256: analystDefinitionProtocolSha256(definition)
3697
+ }))
3698
+ };
3699
+ }
3700
+ //#endregion
3701
+ //#region src/analyst/reply-contract.ts
3702
+ /**
3703
+ * The reply grammar an analyst arm holds a model to, independent of transport.
3704
+ *
3705
+ * One contract serves every arm shape: the inline bridge protocol
3706
+ * (`runPrimeExchange`) reads the base fields, and one-shot JSON arms
3707
+ * additionally use the strict-envelope and all-rejected knobs. `PrimeReplyContract`
3708
+ * in ./prime-protocol is a type alias of this contract, so a consumer written
3709
+ * against the prime protocol names the same grammar object.
3710
+ */
3711
+ /**
3712
+ * Decode a parsed reply value under a contract: strict envelope when declared,
3713
+ * then per-row decoding, then the all-rejected policy. Shape before count: a
3714
+ * malformed row never consumes an accepted slot.
3715
+ */
3716
+ function decodeReplyRows(contract, value) {
3717
+ let rawRows;
3718
+ let extras = {};
3719
+ if (contract.parseEnvelope) {
3720
+ const envelope = contract.parseEnvelope(value);
3721
+ rawRows = envelope.rows;
3722
+ extras = envelope.extras;
3723
+ } else {
3724
+ const field = (typeof value === "object" && value !== null && !Array.isArray(value) ? value : void 0)?.[contract.rowsField];
3725
+ if (!Array.isArray(field)) throw new ValidationError(`reply has no "${contract.rowsField}" array`);
3726
+ rawRows = field;
3727
+ }
3728
+ const rows = [];
3729
+ const rejected = [];
3730
+ let overflow = 0;
3731
+ rawRows.forEach((row, index) => {
3732
+ const decoded = contract.decodeRow(row, index);
3733
+ if (!decoded.ok) {
3734
+ rejected.push({
3735
+ index,
3736
+ reason: decoded.reason
3737
+ });
3738
+ return;
3739
+ }
3740
+ if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
3741
+ overflow += 1;
3742
+ return;
3743
+ }
3744
+ rows.push(decoded.row);
3745
+ });
3746
+ if (rows.length === 0 && rawRows.length > 0 && contract.whenAllRowsRejected === "fail") throw new ValidationError(`${contract.allRejectedMessage ?? "every reported row was malformed"}: ${rejected.map((entry) => entry.reason).join(" | ")}`);
3747
+ return {
3748
+ rows,
3749
+ extras,
3750
+ rejected,
3751
+ reportedRows: rawRows.length,
3752
+ overflow
3753
+ };
3754
+ }
3755
+ //#endregion
3597
3756
  //#region src/analyst/benchmark-public-model.ts
3598
- /** One-shot JSON baseline. This is not a recursive trace analyst. */
3757
+ /** The direct arm as a declarative unit for one public dataset. */
3758
+ function publicDirectAnalystDefinition(dataset, args) {
3759
+ const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
3760
+ const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
3761
+ return {
3762
+ id: "direct",
3763
+ description: "One-shot JSON baseline over the caller-owned model path.",
3764
+ version: "1.0.0",
3765
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
3766
+ profile: { model: { reasoningEffort: "none" } },
3767
+ question: "",
3768
+ taskDefinition: publicBenchmarkTaskPrompt(dataset),
3769
+ projection: {
3770
+ mode: "chunked",
3771
+ attributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS
3772
+ },
3773
+ replyContract: directReplyContract(dataset),
3774
+ contractLimits: dataset === "agentrx" ? { maxFindings: 1 } : {
3775
+ maxBlocks: 16,
3776
+ maxBlockSteps: 12
3777
+ },
3778
+ budget: {
3779
+ timeoutMs: args.timeoutMs,
3780
+ maxCostUsd: args.maxCostUsd,
3781
+ maxOutputTokens: args.maxOutputTokens
3782
+ },
3783
+ repair: { turns: 0 },
3784
+ protocolSha256: publicBenchmarkProtocolSha256(dataset),
3785
+ binding: {
3786
+ kind: "chunked",
3787
+ subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$2(dataset, caseId),
3788
+ baseMetadata: {
3789
+ analysisMode: "direct-baseline",
3790
+ outputAdapter
3791
+ },
3792
+ costActor: actor,
3793
+ costPhase: "analyst.public-benchmark",
3794
+ userMessage: (rendered) => `TRACE DATA:\n${rendered}\n\nReturn the analysis JSON object.`,
3795
+ async expandRows({ subject, rows, store, analystId, producedAt, providerModel, signal }) {
3796
+ const converted = await publicBenchmarkPredictionsToFindings({
3797
+ dataset,
3798
+ trajectoryId: subject,
3799
+ predictions: rows,
3800
+ store,
3801
+ analystId,
3802
+ providerModel: requiredString(providerModel ?? "", "finding providerModel"),
3803
+ producedAt: requiredString(producedAt ?? "", "finding producedAt"),
3804
+ ...signal ? { signal } : {}
3805
+ });
3806
+ return {
3807
+ findings: converted.findings,
3808
+ diagnostics: converted.diagnostics
3809
+ };
3810
+ },
3811
+ ...dataset === "codetracebench" ? { verifyFindings: async (args) => {
3812
+ await validateCodeTraceFindingEvidence({
3813
+ trajectoryId: args.subject,
3814
+ findings: [...args.findings],
3815
+ store: args.store,
3816
+ ...args.signal ? { signal: args.signal } : {}
3817
+ });
3818
+ } } : {}
3819
+ }
3820
+ };
3821
+ }
3822
+ /** Thin shell: validate config, declare the definition, run the chunked strategy. */
3599
3823
  function createPublicBenchmarkDirectRunner(dataset, config) {
3600
3824
  if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
3825
+ const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
3826
+ return runChunkedAnalystDefinition(publicDirectAnalystDefinition(dataset, {
3827
+ timeoutMs: positiveSafeInteger(config.timeoutMs, "timeoutMs"),
3828
+ maxOutputTokens,
3829
+ maxCostUsd: config.maxCostUsdPerAnalysis ?? 1
3830
+ }), config);
3831
+ }
3832
+ /**
3833
+ * Compile a chunked-projection definition into a runnable one-shot JSON arm
3834
+ * over the caller-owned model path. Prompt content, the projection ladder,
3835
+ * the reply grammar, and the budget declaration come from the definition;
3836
+ * caching, cost settlement, and the model proxy are transport machinery.
3837
+ */
3838
+ function runChunkedAnalystDefinition(definition, config) {
3839
+ const { projection, binding, replyContract } = definition;
3840
+ if (projection.mode !== "chunked" || binding.kind !== "chunked") throw new AnalystExpressivenessError(`the chunked one-shot strategy compiles only chunked projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
3841
+ if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
3842
+ if (definition.repair.turns !== 0) throw new AnalystExpressivenessError(`the one-shot JSON exchange grants no repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
3843
+ const reasoningEffort = definition.profile.model?.reasoningEffort;
3844
+ if (reasoningEffort !== "none") throw new AnalystExpressivenessError(`the one-shot JSON strategy runs with thinking disabled and can express only reasoning effort 'none'; definition '${definition.id}' declares '${reasoningEffort}'`);
3845
+ if (!replyContract.parseEnvelope) throw new AnalystExpressivenessError(`the one-shot JSON strategy needs a strict reply envelope; definition '${definition.id}' declares no parseEnvelope`);
3846
+ if (definition.taskDefinition === void 0) throw new AnalystExpressivenessError(`the one-shot JSON strategy composes its system prompt from the task definition; definition '${definition.id}' declares none`);
3601
3847
  const model = requiredString(config.model, "model");
3602
3848
  const callRef = requiredString(config.callRef, "callRef");
3603
3849
  if (typeof config.call !== "function") throw new TypeError("call must be a function");
3604
3850
  if (typeof config.recordExecution !== "function") throw new TypeError("recordExecution must be a function");
3605
3851
  const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
3606
3852
  const timeoutMs = positiveSafeInteger(config.timeoutMs, "timeoutMs");
3853
+ const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
3854
+ assertDeclaredBudget(definition, {
3855
+ timeoutMs,
3856
+ maxOutputTokens,
3857
+ maxCostUsd
3858
+ });
3607
3859
  const maxReasoningTokens = config.maxReasoningTokens ?? maxOutputTokens * 4;
3608
3860
  const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
3609
3861
  const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
3610
3862
  const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
3611
3863
  const pricing = config.pricing ?? pricingForModel$2(model);
3612
- const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
3613
3864
  const costLedger = config.costLedger ?? new CostLedger();
3614
3865
  const durability = config.durability ? {
3615
3866
  runIdentitySha256: requiredString(config.durability.runIdentitySha256, "durability.runIdentitySha256"),
3616
3867
  responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
3617
3868
  } : void 0;
3618
- const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
3619
- const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
3869
+ const systemPrompt = [definition.taskDefinition, ...replyContract.contractLines].join("\n\n");
3620
3870
  return {
3621
- id: "direct",
3871
+ id: definition.id,
3622
3872
  async analyze(input, context) {
3623
- const trajectoryId = trajectoryIdFromCaseId$2(dataset, context.caseId);
3873
+ const trajectoryId = binding.subjectFromCaseId(context.caseId);
3624
3874
  const costTags = {
3625
- analystId: actor,
3875
+ analystId: binding.costActor,
3626
3876
  benchmarkCaseId: context.caseId,
3627
3877
  benchmarkRepetition: String(context.repetition)
3628
3878
  };
3629
3879
  let rawPredictions = [];
3630
- let rejectedBlocks = [];
3880
+ let rejectedRows = [];
3631
3881
  let modelFindings = [];
3632
3882
  let providerModel = model;
3633
3883
  let producedAt;
3634
3884
  let modelMetadata = {
3635
- analysisMode: "direct-baseline",
3636
- outputAdapter,
3637
- protocolSha256: publicBenchmarkProtocolSha256(dataset),
3885
+ ...binding.baseMetadata,
3886
+ protocolSha256: definition.protocolSha256,
3638
3887
  callRef
3639
3888
  };
3640
3889
  try {
3641
- if (!input.traceStore) throw new Error(`${dataset} model runner requires a trace store`);
3642
- const preparedContext = await prepareSingleTraceContext(input.traceStore, context);
3643
- if (preparedContext === void 0) throw new Error(`${dataset} trace '${trajectoryId}' has no readable spans`);
3890
+ if (!input.traceStore) throw new Error(`chunked analyst '${definition.id}' requires a trace store`);
3891
+ const preparedContext = await prepareSingleTraceContext(input.traceStore, context, projection.attributeByteCaps);
3892
+ if (preparedContext === void 0) throw new Error(`trace '${trajectoryId}' has no readable spans`);
3644
3893
  const request = {
3645
3894
  model,
3646
3895
  messages: [{
3647
3896
  role: "system",
3648
- content: publicBenchmarkSystemPrompt(dataset)
3897
+ content: systemPrompt
3649
3898
  }, {
3650
3899
  role: "user",
3651
- content: `TRACE DATA:\n${preparedContext}\n\nReturn the analysis JSON object.`
3900
+ content: binding.userMessage(preparedContext)
3652
3901
  }],
3653
3902
  jsonMode: true,
3654
3903
  thinking: "disabled",
@@ -3678,14 +3927,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3678
3927
  error: cached.error,
3679
3928
  metadata: modelMetadata
3680
3929
  };
3681
- const response = parsePublicBenchmarkModelResponse(dataset, cached.response);
3682
- rawPredictions = response.findings;
3683
- rejectedBlocks = response.rejectedBlocks;
3930
+ const response = decodeReplyRows(replyContract, cached.response);
3931
+ rawPredictions = response.rows;
3932
+ rejectedRows = response.rejected.map((entry) => entry.reason);
3684
3933
  providerModel = cached.metadata.providerModel;
3685
3934
  producedAt = cached.metadata.producedAt;
3686
3935
  modelMetadata = {
3687
3936
  ...modelMetadata,
3688
- report: response.report,
3937
+ ...response.extras,
3689
3938
  providerModel: cached.metadata.providerModel,
3690
3939
  providerDurationMs: cached.metadata.providerDurationMs,
3691
3940
  finishReason: cached.metadata.finishReason
@@ -3714,8 +3963,8 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3714
3963
  },
3715
3964
  costLedger,
3716
3965
  channel: "analyst",
3717
- phase: "analyst.public-benchmark",
3718
- actor,
3966
+ phase: binding.costPhase,
3967
+ actor: binding.costActor,
3719
3968
  tags: costTags,
3720
3969
  callId: providerCallId,
3721
3970
  ...context.signal ? { signal: context.signal } : {}
@@ -3734,7 +3983,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3734
3983
  ...context.signal ? { signal: context.signal } : {},
3735
3984
  idempotencyKey: providerCallId
3736
3985
  });
3737
- const response = parsePublicBenchmarkModelResponse(dataset, completed.value);
3986
+ const response = decodeReplyRows(replyContract, completed.value);
3738
3987
  const responseProducedAt = (/* @__PURE__ */ new Date()).toISOString();
3739
3988
  const receipt = requiredSettledReceipt(costLedger, providerCallId);
3740
3989
  if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
@@ -3780,26 +4029,25 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3780
4029
  }
3781
4030
  });
3782
4031
  const response = completed.response;
3783
- rawPredictions = response.findings;
3784
- rejectedBlocks = response.rejectedBlocks;
4032
+ rawPredictions = response.rows;
4033
+ rejectedRows = response.rejected.map((entry) => entry.reason);
3785
4034
  providerModel = completed.result.model;
3786
4035
  producedAt = completed.producedAt;
3787
4036
  modelMetadata = {
3788
4037
  ...modelMetadata,
3789
4038
  responseSource: "provider",
3790
- report: response.report,
4039
+ ...response.extras,
3791
4040
  providerModel: completed.result.model,
3792
4041
  providerDurationMs: completed.result.durationMs,
3793
4042
  finishReason: completed.result.finishReason ?? null,
3794
4043
  cost: costReceiptMetadata(completed.receipt)
3795
4044
  };
3796
4045
  }
3797
- const converted = await publicBenchmarkPredictionsToFindings({
3798
- dataset,
3799
- trajectoryId,
3800
- predictions: rawPredictions,
4046
+ const converted = await binding.expandRows({
4047
+ subject: trajectoryId,
4048
+ rows: rawPredictions,
3801
4049
  store: input.traceStore,
3802
- analystId: "direct",
4050
+ analystId: definition.id,
3803
4051
  providerModel,
3804
4052
  producedAt: requiredString(producedAt ?? "", "finding producedAt"),
3805
4053
  ...context.signal ? { signal: context.signal } : {}
@@ -3809,11 +4057,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3809
4057
  ...modelMetadata,
3810
4058
  blockDiagnostics: {
3811
4059
  ...converted.diagnostics,
3812
- rejectedBlocks
4060
+ rejectedBlocks: rejectedRows
3813
4061
  }
3814
4062
  };
3815
- if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
3816
- trajectoryId,
4063
+ if (binding.verifyFindings) await binding.verifyFindings({
4064
+ subject: trajectoryId,
3817
4065
  findings: modelFindings,
3818
4066
  store: input.traceStore,
3819
4067
  ...context.signal ? { signal: context.signal } : {}
@@ -3846,6 +4094,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3846
4094
  }
3847
4095
  };
3848
4096
  }
4097
+ /** A definition that declares one budget while the transport runs another is refused. */
4098
+ function assertDeclaredBudget(definition, effective) {
4099
+ const declared = definition.budget;
4100
+ if (declared.timeoutMs !== effective.timeoutMs || declared.maxOutputTokens !== effective.maxOutputTokens || declared.maxCostUsd !== effective.maxCostUsd) throw new AnalystExpressivenessError(`definition '${definition.id}' declares budget ${JSON.stringify(declared)} but the bound transport runs ${JSON.stringify(effective)}; the declaration must state what executes`);
4101
+ }
3849
4102
  function settleCachedResponse(costLedger, cached) {
3850
4103
  const settled = costLedger.list().find((receipt) => receipt.callId === cached.callId);
3851
4104
  const pending = costLedger.listPending?.().find((record) => record.callId === cached.callId);
@@ -4003,27 +4256,56 @@ const AgentRxModelResponseSchema = z.object({
4003
4256
  report: z.string().min(1).max(4e3),
4004
4257
  findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
4005
4258
  }).strict();
4006
- function parsePublicBenchmarkModelResponse(dataset, value) {
4259
+ /**
4260
+ * The one-shot reply grammar per dataset. The envelope is the contract and
4261
+ * stays strict. Individual CodeTraceBench blocks are model output: one
4262
+ * malformed block must not void a case whose remaining blocks are usable and
4263
+ * whose provider call is already paid for, so rows decode individually and
4264
+ * every rejection is reported.
4265
+ */
4266
+ function directReplyContract(dataset) {
4007
4267
  if (dataset === "agentrx") return {
4008
- ...AgentRxModelResponseSchema.parse(value),
4009
- rejectedBlocks: []
4010
- };
4011
- const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
4012
- const findings = [];
4013
- const rejectedBlocks = [];
4014
- for (const [index, block] of envelope.findings.entries()) {
4015
- const parsed = CodeTraceBlockPredictionSchema.safeParse(block);
4016
- if (parsed.success) {
4017
- findings.push(parsed.data);
4018
- continue;
4268
+ rowsField: "findings",
4269
+ contractLines: [publicBenchmarkFieldContract("agentrx"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
4270
+ repairContractLines: [],
4271
+ parseEnvelope(value) {
4272
+ const parsed = AgentRxModelResponseSchema.parse(value);
4273
+ return {
4274
+ rows: parsed.findings,
4275
+ extras: { report: parsed.report }
4276
+ };
4277
+ },
4278
+ decodeRow(row) {
4279
+ return {
4280
+ ok: true,
4281
+ row
4282
+ };
4019
4283
  }
4020
- rejectedBlocks.push(`block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`);
4021
- }
4022
- if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
4284
+ };
4023
4285
  return {
4024
- report: envelope.report,
4025
- findings,
4026
- rejectedBlocks
4286
+ rowsField: "findings",
4287
+ contractLines: [publicBenchmarkFieldContract("codetracebench"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
4288
+ repairContractLines: [],
4289
+ parseEnvelope(value) {
4290
+ const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
4291
+ return {
4292
+ rows: envelope.findings,
4293
+ extras: { report: envelope.report }
4294
+ };
4295
+ },
4296
+ decodeRow(row, index) {
4297
+ const parsed = CodeTraceBlockPredictionSchema.safeParse(row);
4298
+ if (parsed.success) return {
4299
+ ok: true,
4300
+ row: parsed.data
4301
+ };
4302
+ return {
4303
+ ok: false,
4304
+ reason: `block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`
4305
+ };
4306
+ },
4307
+ whenAllRowsRejected: "fail",
4308
+ allRejectedMessage: "every reported failure block was malformed"
4027
4309
  };
4028
4310
  }
4029
4311
  async function publicBenchmarkPredictionsToFindings(options) {
@@ -4090,12 +4372,12 @@ async function publicBenchmarkPredictionsToFindings(options) {
4090
4372
  ...options.signal ? { signal: options.signal } : {}
4091
4373
  });
4092
4374
  }
4093
- async function prepareSingleTraceContext(store, context) {
4375
+ async function prepareSingleTraceContext(store, context, attributeByteCaps) {
4094
4376
  const storeContext = context.signal ? { signal: context.signal } : void 0;
4095
4377
  const overview = await store.getOverview(void 0, storeContext);
4096
4378
  if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public model benchmark requires exactly one trace, received ${overview.total_traces}`);
4097
4379
  const traceId = overview.sample_trace_ids[0];
4098
- for (const perAttributeByteCap of TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS) {
4380
+ for (const perAttributeByteCap of attributeByteCaps) {
4099
4381
  const viewed = await store.viewTrace({
4100
4382
  trace_id: traceId,
4101
4383
  per_attribute_byte_cap: perAttributeByteCap
@@ -4245,18 +4527,156 @@ function contiguousSegments(sortedSteps) {
4245
4527
  }
4246
4528
  //#endregion
4247
4529
  //#region src/analyst/benchmark-public-rlm.ts
4248
- /** Public benchmark candidate that runs the actual recursive trace analyst. */
4530
+ /** The dspy-rlm arm as a declarative unit for one public dataset. */
4531
+ function publicRlmAnalystDefinition(dataset, args) {
4532
+ return {
4533
+ id: "dspy-rlm",
4534
+ description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
4535
+ version: "1.0.0",
4536
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
4537
+ profile: {},
4538
+ question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
4539
+ taskDefinition: args.instructions,
4540
+ projection: {
4541
+ mode: "repl-variable",
4542
+ toolGroup: "singleTrace"
4543
+ },
4544
+ replyContract: {
4545
+ rowsField: "findings",
4546
+ contractLines: [RAW_FINDING_SCHEMA_PROMPT],
4547
+ repairContractLines: [],
4548
+ decodeRow(row) {
4549
+ const parsed = RawAnalystFindingSchema.safeParse(row);
4550
+ if (parsed.success) return {
4551
+ ok: true,
4552
+ row: parsed.data
4553
+ };
4554
+ return {
4555
+ ok: false,
4556
+ reason: parsed.error.issues.map((issue) => `${issue.path.join(".")}: ${issue.message}`).join("; ")
4557
+ };
4558
+ }
4559
+ },
4560
+ contractLimits: {
4561
+ maxIterations: args.engineLimits.maxIterations,
4562
+ maxLlmCalls: args.engineLimits.maxLlmCalls,
4563
+ maxToolCalls: args.engineLimits.maxToolCalls,
4564
+ maxOutputChars: args.engineLimits.maxOutputChars
4565
+ },
4566
+ budget: {
4567
+ timeoutMs: args.timeoutMs,
4568
+ maxCostUsd: args.maxCostUsd,
4569
+ maxOutputTokens: args.maxOutputTokens,
4570
+ engineLimits: args.engineLimits
4571
+ },
4572
+ repair: { turns: 1 },
4573
+ protocolSha256: args.protocolSha256,
4574
+ binding: {
4575
+ kind: "repl-variable",
4576
+ traceAnalystId: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
4577
+ subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$1(dataset, caseId),
4578
+ baseMetadata: {
4579
+ analysisMode: "recursive",
4580
+ engine: "dspy-rlm"
4581
+ },
4582
+ findingBaseMetadata: {
4583
+ analysis_mode: "recursive",
4584
+ engine: "dspy-rlm"
4585
+ },
4586
+ costPhase: "analyst.public-benchmark.dspy-rlm",
4587
+ ...dataset === "codetracebench" ? { metadataFromSubject: codeTraceBlockMetadataFromSubject } : {},
4588
+ async adapt({ subject, findings, analystId, store, signal }) {
4589
+ return adaptPublicBenchmarkFindings({
4590
+ dataset,
4591
+ trajectoryId: subject,
4592
+ findings: [...findings],
4593
+ analystId,
4594
+ store,
4595
+ ...signal ? { signal } : {}
4596
+ });
4597
+ },
4598
+ ...dataset === "codetracebench" ? { consensus: codeTraceConsensusPort() } : {},
4599
+ abstentionFallback: (fallbackConfig) => createPublicBenchmarkDirectRunner(dataset, fallbackConfig)
4600
+ }
4601
+ };
4602
+ }
4603
+ /** Step-level majority consensus on the CodeTraceBench block grammar. */
4604
+ function codeTraceConsensusPort() {
4605
+ return {
4606
+ vote(samples) {
4607
+ const consensus = consensusCodeTraceBlocks(samples.map((sample) => [...sample]));
4608
+ return {
4609
+ blocks: consensus.blocks,
4610
+ decision: consensus.decision
4611
+ };
4612
+ },
4613
+ async expand({ subject, blocks, store, analystId, producedAt, signal }) {
4614
+ const expanded = await expandCodeTraceFailureBlocks({
4615
+ trajectoryId: subject,
4616
+ blocks,
4617
+ store,
4618
+ analystId,
4619
+ producedAt,
4620
+ ...signal ? { signal } : {}
4621
+ });
4622
+ return {
4623
+ findings: expanded.findings,
4624
+ diagnostics: expanded.diagnostics
4625
+ };
4626
+ },
4627
+ sampleRecord(assignments) {
4628
+ return {
4629
+ blocks: sampleBlockRecords(assignments),
4630
+ steps: assignments.map((assignment) => assignment.step)
4631
+ };
4632
+ }
4633
+ };
4634
+ }
4635
+ /** Thin shell: validate config, declare the definition, run the repl-variable strategy. */
4249
4636
  function createPublicBenchmarkRlmRunner(dataset, config) {
4250
- const costLedger = config.costLedger ?? new CostLedger();
4251
4637
  const samples = config.dspyRlm?.samples ?? 1;
4252
4638
  if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
4253
4639
  if (samples > 1 && dataset !== "codetracebench") throw new Error("dspyRlm.samples > 1 requires the codetracebench dataset; step-level consensus is defined on its block grammar");
4254
- const limits = {
4640
+ return runReplVariableAnalystDefinition(publicRlmAnalystDefinition(dataset, {
4641
+ instructions: config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset),
4642
+ protocolSha256: effectiveAnalystProtocolSha256(dataset, config.instructionsOverride),
4643
+ timeoutMs: config.timeoutMs,
4644
+ maxOutputTokens: config.maxOutputTokens,
4645
+ maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
4646
+ engineLimits: rlmEngineLimits(config)
4647
+ }), config);
4648
+ }
4649
+ /** Engine iteration limits with this arm's defaults applied. */
4650
+ function rlmEngineLimits(config) {
4651
+ return {
4255
4652
  maxIterations: config.dspyRlm?.maxIterations ?? 14,
4256
4653
  maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
4257
4654
  maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
4258
4655
  maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
4259
4656
  };
4657
+ }
4658
+ /**
4659
+ * Compile a repl-variable definition into a runnable recursive-engine arm over
4660
+ * the caller-owned model path. The question, instructions, tool group, and
4661
+ * iteration limits come from the definition; the engine, model proxy, sampling
4662
+ * loop, and abstention floor are transport machinery.
4663
+ */
4664
+ function runReplVariableAnalystDefinition(definition, config) {
4665
+ const { projection, binding } = definition;
4666
+ if (projection.mode !== "repl-variable" || binding.kind !== "repl-variable") throw new AnalystExpressivenessError(`the repl-variable strategy compiles only repl-variable projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
4667
+ const instructions = definition.taskDefinition;
4668
+ if (instructions === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy runs the definition's task text as engine instructions; definition '${definition.id}' declares none`);
4669
+ if (config.instructionsOverride !== void 0 && config.instructionsOverride.text !== instructions) throw new AnalystExpressivenessError(`definition '${definition.id}' declares instructions that differ from the transport's instructionsOverride; one text must execute`);
4670
+ const area = definition.area;
4671
+ if (area === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy stamps the definition's area on every finding; definition '${definition.id}' declares none`);
4672
+ const limits = definition.budget.engineLimits;
4673
+ if (limits === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy needs declared engine limits; definition '${definition.id}' declares none`);
4674
+ const effectiveLimits = rlmEngineLimits(config);
4675
+ if (limits.maxIterations !== effectiveLimits.maxIterations || limits.maxLlmCalls !== effectiveLimits.maxLlmCalls || limits.maxToolCalls !== effectiveLimits.maxToolCalls || limits.maxOutputChars !== effectiveLimits.maxOutputChars) throw new AnalystExpressivenessError(`definition '${definition.id}' declares engine limits ${JSON.stringify(limits)} but the bound transport runs ${JSON.stringify(effectiveLimits)}; the declaration must state what executes`);
4676
+ const costLedger = config.costLedger ?? new CostLedger();
4677
+ const samples = config.dspyRlm?.samples ?? 1;
4678
+ if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
4679
+ if (samples > 1 && binding.consensus === void 0) throw new AnalystExpressivenessError(`samples > 1 needs a consensus port; definition '${definition.id}' declares none`);
4260
4680
  const pricing = config.pricing ?? pricingForModel$1(config.model);
4261
4681
  const engine = createDspyRlmTraceEngine({
4262
4682
  call: config.call,
@@ -4279,18 +4699,26 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4279
4699
  ...config.dspyRlm?.traceToolTimeoutMs === void 0 ? {} : { traceToolTimeoutMs: config.dspyRlm.traceToolTimeoutMs },
4280
4700
  ...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
4281
4701
  });
4282
- const instructions = config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset);
4283
- const protocolSha256 = effectiveAnalystProtocolSha256(dataset, config.instructionsOverride);
4284
- const definition = publicBenchmarkDefinition(dataset, limits, instructions);
4702
+ const protocolSha256 = definition.protocolSha256;
4703
+ const traceDefinition = {
4704
+ id: binding.traceAnalystId,
4705
+ description: definition.description,
4706
+ area,
4707
+ version: definition.version,
4708
+ question: definition.question,
4709
+ instructions,
4710
+ toolGroup: projection.toolGroup,
4711
+ limits
4712
+ };
4285
4713
  const { instructionsOverride: _rlmOnlyOverride, ...directConfig } = config;
4286
- const abstentionFallbackRunner = createPublicBenchmarkDirectRunner(dataset, {
4714
+ const abstentionFallbackRunner = binding.abstentionFallback({
4287
4715
  ...directConfig,
4288
4716
  costLedger
4289
4717
  });
4290
4718
  return {
4291
- id: "dspy-rlm",
4719
+ id: definition.id,
4292
4720
  async analyze(input, context) {
4293
- const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
4721
+ const trajectoryId = binding.subjectFromCaseId(context.caseId);
4294
4722
  const tags = {
4295
4723
  benchmarkCaseId: context.caseId,
4296
4724
  benchmarkRepetition: String(context.repetition)
@@ -4298,7 +4726,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4298
4726
  let usage;
4299
4727
  let rawFindings = [];
4300
4728
  try {
4301
- if (!input.traceStore) throw new Error(`${dataset} DSPy RLM runner requires a trace store`);
4729
+ if (!input.traceStore) throw new Error(`repl-variable analyst '${definition.id}' requires a trace store`);
4302
4730
  if (samples > 1) {
4303
4731
  const store = input.traceStore;
4304
4732
  const caseUsageFilter = {
@@ -4312,14 +4740,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4312
4740
  for (let sample = 0; sample < samples; sample += 1) {
4313
4741
  let sampleUsage;
4314
4742
  const completed = await runTraceAnalyst({
4315
- definition,
4743
+ definition: traceDefinition,
4316
4744
  engine,
4317
4745
  store,
4318
4746
  context: {
4319
4747
  runId: context.caseId,
4320
4748
  correlationId: `${context.caseId}:${context.repetition}:sample-${sample}`,
4321
4749
  costLedger,
4322
- costPhase: "analyst.public-benchmark.dspy-rlm",
4750
+ costPhase: binding.costPhase,
4323
4751
  tags,
4324
4752
  recordUsage: (receipt) => {
4325
4753
  sampleUsage = receipt;
@@ -4330,8 +4758,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4330
4758
  });
4331
4759
  const producedAt = (/* @__PURE__ */ new Date()).toISOString();
4332
4760
  const sampleFindings = completed.findings.map((finding) => makeFinding({
4333
- analyst_id: "dspy-rlm",
4334
- area: "incorrect",
4761
+ analyst_id: definition.id,
4762
+ area,
4335
4763
  subject: finding.subject,
4336
4764
  claim: finding.claim,
4337
4765
  rationale: finding.rationale,
@@ -4340,20 +4768,18 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4340
4768
  evidence_refs: evidenceRefsFromRawFinding(finding),
4341
4769
  recommended_action: finding.recommended_action,
4342
4770
  metadata: {
4343
- analysis_mode: "recursive",
4344
- engine: "dspy-rlm",
4771
+ ...binding.findingBaseMetadata,
4345
4772
  model: config.model,
4346
4773
  sample,
4347
- ...codeTraceBlockMetadataFromSubject(finding.subject) ?? {}
4774
+ ...binding.metadataFromSubject?.(finding.subject) ?? {}
4348
4775
  },
4349
4776
  produced_at: producedAt
4350
4777
  }));
4351
4778
  rawFindings = [...rawFindings, ...sampleFindings];
4352
- const adapted = await adaptPublicBenchmarkFindings({
4353
- dataset,
4354
- trajectoryId,
4779
+ const adapted = await binding.adapt({
4780
+ subject: trajectoryId,
4355
4781
  findings: sampleFindings,
4356
- analystId: "dspy-rlm",
4782
+ analystId: definition.id,
4357
4783
  store,
4358
4784
  ...context.signal ? { signal: context.signal } : {}
4359
4785
  });
@@ -4368,18 +4794,17 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4368
4794
  modelCalls: completed.modelCalls,
4369
4795
  toolCalls: completed.toolCalls,
4370
4796
  runtime: completed.runtime,
4371
- blocks: sampleBlockRecords(assignments),
4372
- steps: assignments.map((assignment) => assignment.step),
4797
+ ...binding.consensus.sampleRecord(assignments),
4373
4798
  ...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
4374
4799
  ...sampleUsage ? { usage: sampleUsage } : {}
4375
4800
  });
4376
4801
  }
4377
- const consensus = consensusCodeTraceBlocks(sampleAssignments);
4378
- const expanded = await expandCodeTraceFailureBlocks({
4379
- trajectoryId,
4802
+ const consensus = binding.consensus.vote(sampleAssignments);
4803
+ const expanded = await binding.consensus.expand({
4804
+ subject: trajectoryId,
4380
4805
  blocks: consensus.blocks,
4381
4806
  store,
4382
- analystId: "dspy-rlm",
4807
+ analystId: definition.id,
4383
4808
  producedAt: (/* @__PURE__ */ new Date()).toISOString(),
4384
4809
  ...context.signal ? { signal: context.signal } : {}
4385
4810
  });
@@ -4390,8 +4815,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4390
4815
  findings: fallback && !fallback.error ? fallback.findings : expanded.findings,
4391
4816
  usage,
4392
4817
  metadata: {
4393
- analysisMode: "recursive",
4394
- engine: "dspy-rlm",
4818
+ ...binding.baseMetadata,
4395
4819
  protocolSha256,
4396
4820
  samples,
4397
4821
  sampleRuns,
@@ -4408,14 +4832,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4408
4832
  };
4409
4833
  }
4410
4834
  const completed = await runTraceAnalyst({
4411
- definition,
4835
+ definition: traceDefinition,
4412
4836
  engine,
4413
4837
  store: input.traceStore,
4414
4838
  context: {
4415
4839
  runId: context.caseId,
4416
4840
  correlationId: `${context.caseId}:${context.repetition}`,
4417
4841
  costLedger,
4418
- costPhase: "analyst.public-benchmark.dspy-rlm",
4842
+ costPhase: binding.costPhase,
4419
4843
  tags,
4420
4844
  recordUsage: (receipt) => {
4421
4845
  usage = receipt;
@@ -4425,8 +4849,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4425
4849
  });
4426
4850
  const producedAt = (/* @__PURE__ */ new Date()).toISOString();
4427
4851
  rawFindings = completed.findings.map((finding) => makeFinding({
4428
- analyst_id: "dspy-rlm",
4429
- area: dataset === "agentrx" ? "root-cause" : "incorrect",
4852
+ analyst_id: definition.id,
4853
+ area,
4430
4854
  subject: finding.subject,
4431
4855
  claim: finding.claim,
4432
4856
  rationale: finding.rationale,
@@ -4435,18 +4859,16 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4435
4859
  evidence_refs: evidenceRefsFromRawFinding(finding),
4436
4860
  recommended_action: finding.recommended_action,
4437
4861
  metadata: {
4438
- analysis_mode: "recursive",
4439
- engine: "dspy-rlm",
4862
+ ...binding.findingBaseMetadata,
4440
4863
  model: config.model,
4441
- ...dataset === "codetracebench" ? codeTraceBlockMetadataFromSubject(finding.subject) : {}
4864
+ ...binding.metadataFromSubject?.(finding.subject) ?? {}
4442
4865
  },
4443
4866
  produced_at: producedAt
4444
4867
  }));
4445
- const adapted = await adaptPublicBenchmarkFindings({
4446
- dataset,
4447
- trajectoryId,
4868
+ const adapted = await binding.adapt({
4869
+ subject: trajectoryId,
4448
4870
  findings: rawFindings,
4449
- analystId: "dspy-rlm",
4871
+ analystId: definition.id,
4450
4872
  store: input.traceStore,
4451
4873
  ...context.signal ? { signal: context.signal } : {}
4452
4874
  });
@@ -4465,8 +4887,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4465
4887
  findings: fallback && !fallback.error ? fallback.findings : adapted.findings,
4466
4888
  usage,
4467
4889
  metadata: {
4468
- analysisMode: "recursive",
4469
- engine: "dspy-rlm",
4890
+ ...binding.baseMetadata,
4470
4891
  protocolSha256,
4471
4892
  ...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
4472
4893
  answer: completed.answer,
@@ -4489,8 +4910,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4489
4910
  usage,
4490
4911
  error: publicBenchmarkError(error, []),
4491
4912
  metadata: {
4492
- analysisMode: "recursive",
4493
- engine: "dspy-rlm",
4913
+ ...binding.baseMetadata,
4494
4914
  ...samples > 1 ? { samples } : {},
4495
4915
  rawFindings
4496
4916
  }
@@ -4518,18 +4938,6 @@ function sampleBlockRecords(assignments) {
4518
4938
  acceptedSteps
4519
4939
  }));
4520
4940
  }
4521
- function publicBenchmarkDefinition(dataset, limits, instructions) {
4522
- return {
4523
- id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
4524
- description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
4525
- area: dataset === "agentrx" ? "root-cause" : "incorrect",
4526
- version: "1.0.0",
4527
- question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
4528
- instructions,
4529
- toolGroup: "singleTrace",
4530
- limits
4531
- };
4532
- }
4533
4941
  function pricingForModel$1(model) {
4534
4942
  const pricing = resolveModelPricing(model);
4535
4943
  if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
@@ -4893,455 +5301,31 @@ function nodeHttpPrimeBridgeTransport() {
4893
5301
  };
4894
5302
  }
4895
5303
  //#endregion
4896
- //#region src/analyst/prime-protocol.ts
4897
- function buildPrimePrompt(spec) {
4898
- return [
4899
- `QUESTION: ${spec.question}`,
4900
- "",
4901
- ...spec.taskDefinition === void 0 ? [] : [
4902
- "TASK DEFINITION:",
4903
- spec.taskDefinition,
4904
- ""
4905
- ],
4906
- ...spec.contractLines,
4907
- "",
4908
- spec.trajectoryHeader,
4909
- spec.renderedTrajectory,
4910
- ...spec.trailer === void 0 ? [] : ["", spec.trailer]
4911
- ].join("\n");
4912
- }
4913
- /** Carries the malformed reply and the contract — never the trajectory. */
4914
- function buildPrimeRepairPrompt(spec) {
4915
- return [
4916
- "Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
4917
- `(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
4918
- "fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
4919
- ...spec.repairContractLines,
4920
- "",
4921
- "PREVIOUS REPLY:",
4922
- spec.previousReply
4923
- ].join("\n");
4924
- }
4925
- /**
4926
- * Recover the reply's JSON object.
4927
- *
4928
- * Distinct from `extractJsonPayload` in ../llm-client, which serves a response
4929
- * that DECLARES a JSON root and therefore must not scan onward. A prime reply
4930
- * is prose plus a fenced block, and when the model emits several fences the
4931
- * last one is its answer — so fences are scanned in reverse, and only then is a
4932
- * brace-to-brace slice tried.
4933
- */
4934
- function extractPrimeJsonObject(text) {
4935
- const direct = parsePrimeJsonObject(text);
4936
- if (direct) return direct;
4937
- const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
4938
- for (let index = fenced.length - 1; index >= 0; index -= 1) {
4939
- const candidate = parsePrimeJsonObject(fenced[index][1]);
4940
- if (candidate) return candidate;
4941
- }
4942
- const start = text.indexOf("{");
4943
- const end = text.lastIndexOf("}");
4944
- if (start >= 0 && end > start) {
4945
- const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
4946
- if (candidate) return candidate;
4947
- }
4948
- return null;
4949
- }
4950
- /** Why the reply cannot be read as a prime answer, or null when it can. */
4951
- function primeReplyDefect(parsed, rowsField) {
4952
- if (parsed === null) return "no parseable JSON object";
4953
- if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
4954
- return null;
4955
- }
4956
- function parsePrimeJsonObject(text) {
4957
- try {
4958
- const value = JSON.parse(text.trim());
4959
- return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
4960
- } catch {
4961
- return null;
4962
- }
4963
- }
4964
- function emptyPrimeRawUsage() {
4965
- return {
4966
- calls: null,
4967
- inputTokens: null,
4968
- outputTokens: null,
4969
- bridgeEstimated: false
4970
- };
4971
- }
4972
- /** Read the bridge's OpenAI-shaped `usage` object. */
4973
- function normalizePrimeUsage(raw) {
4974
- if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
4975
- const record = raw;
4976
- return {
4977
- calls: tokenCountOrNull(record.model_requests),
4978
- inputTokens: tokenCountOrNull(record.prompt_tokens),
4979
- outputTokens: tokenCountOrNull(record.completion_tokens),
4980
- bridgeEstimated: record.estimated === true
4981
- };
4982
- }
4983
- /**
4984
- * Sum two turns. Each side poisons independently: two turns that both report
4985
- * input and neither report output yield a real input total beside a null
4986
- * output, because discarding a measured count is as wrong as inventing one.
4987
- */
4988
- function mergePrimeRawUsage(a, b) {
4989
- return {
4990
- calls: sumOrNull(a.calls, b.calls),
4991
- inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
4992
- outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
4993
- bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
4994
- };
4995
- }
4996
- function sumOrNull(a, b) {
4997
- return a !== null && b !== null ? a + b : null;
4998
- }
4999
- function tokenCountOrNull(value) {
5000
- return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
5001
- }
5002
- /**
5003
- * Bind raw prime usage to agent-eval's typed receipt.
5004
- *
5005
- * `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
5006
- * cannot round-trip through `tokens` without writing a zero nobody measured.
5007
- * The complete-accounting field therefore stays null, the reported side is
5008
- * carried verbatim in `partialTokens`, and its price becomes the receipt's
5009
- * `knownCostUsd` lower bound.
5010
- *
5011
- * Only agent-eval calls this; consumers with no pricing table read
5012
- * `PrimeRawUsage` directly.
5013
- */
5014
- function analystUsageReceiptFromPrimeUsage(usage, pricing) {
5015
- const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
5016
- const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
5017
- if (inputTokens !== null && outputTokens !== null) return {
5018
- calls,
5019
- tokens: {
5020
- input: inputTokens,
5021
- output: outputTokens
5022
- },
5023
- cost: {
5024
- kind: "estimated",
5025
- usd: priceTokens(inputTokens, outputTokens, pricing)
5026
- },
5027
- ...estimatedTokens
5028
- };
5029
- if (inputTokens === null && outputTokens === null) return {
5030
- calls,
5031
- tokens: null,
5032
- cost: {
5033
- kind: "uncaptured",
5034
- usd: null
5035
- },
5036
- ...estimatedTokens
5037
- };
5038
- return {
5039
- calls,
5040
- tokens: null,
5041
- partialTokens: {
5042
- input: inputTokens,
5043
- output: outputTokens
5044
- },
5045
- cost: {
5046
- kind: "uncaptured",
5047
- usd: null
5048
- },
5049
- knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
5050
- ...estimatedTokens
5051
- };
5052
- }
5053
- function priceTokens(input, output, pricing) {
5054
- return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
5055
- }
5056
- /**
5057
- * Run the protocol: one call, one bounded repair turn on a structurally
5058
- * malformed reply, then decode. Zero valid rows from a well-formed reply is an
5059
- * honest null, not a failure.
5060
- */
5061
- async function runPrimeExchange(options) {
5062
- const { contract } = options;
5063
- const turns = [];
5064
- const repair = {
5065
- attempted: false,
5066
- succeeded: null
5067
- };
5068
- const first = await callPrimeTurn(options, options.prompt);
5069
- if (!first.ok) return {
5070
- ok: false,
5071
- failure: first.failure,
5072
- usage: mergeTurns(turns),
5073
- turns,
5074
- repair
5075
- };
5076
- turns.push({
5077
- turn: "first",
5078
- usage: first.usage,
5079
- rawUsage: first.rawUsage
5080
- });
5081
- let reply = first.content;
5082
- let parsed = extractPrimeJsonObject(reply);
5083
- let defect = primeReplyDefect(parsed, contract.rowsField);
5084
- if (defect !== null && options.repair) {
5085
- repair.attempted = true;
5086
- const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
5087
- defect,
5088
- previousReply: reply,
5089
- repairContractLines: contract.repairContractLines
5090
- }));
5091
- if (!second.ok) return {
5092
- ok: false,
5093
- failure: second.failure,
5094
- usage: mergeTurns(turns),
5095
- turns,
5096
- repair,
5097
- reply
5098
- };
5099
- turns.push({
5100
- turn: "repair",
5101
- usage: second.usage,
5102
- rawUsage: second.rawUsage
5103
- });
5104
- reply = second.content;
5105
- parsed = extractPrimeJsonObject(reply);
5106
- defect = primeReplyDefect(parsed, contract.rowsField);
5107
- repair.succeeded = defect === null;
5108
- }
5109
- const usage = mergeTurns(turns);
5110
- if (defect !== null) return {
5111
- ok: false,
5112
- failure: {
5113
- kind: "malformed-reply",
5114
- message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
5115
- },
5116
- usage,
5117
- turns,
5118
- repair,
5119
- reply
5120
- };
5121
- const rawRows = parsed[contract.rowsField];
5122
- const rows = [];
5123
- const rejected = [];
5124
- let overflow = 0;
5125
- rawRows.forEach((row, index) => {
5126
- const decoded = contract.decodeRow(row, index);
5127
- if (!decoded.ok) {
5128
- rejected.push({
5129
- index,
5130
- reason: decoded.reason
5131
- });
5132
- return;
5133
- }
5134
- if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
5135
- overflow += 1;
5136
- return;
5137
- }
5138
- rows.push(decoded.row);
5139
- });
5140
- const answer = parsed.answer;
5141
- return {
5142
- ok: true,
5143
- answer: typeof answer === "string" ? answer : null,
5144
- rows,
5145
- rejected,
5146
- reportedRows: rawRows.length,
5147
- overflow,
5148
- usage,
5149
- turns,
5150
- repair,
5151
- reply
5152
- };
5153
- }
5154
- async function callPrimeTurn(options, content) {
5155
- const { transport, url, model, timeoutMs, signal } = options;
5156
- const controller = new AbortController();
5157
- const forwardAbort = () => controller.abort(signal?.reason);
5158
- if (signal?.aborted) controller.abort(signal.reason);
5159
- else signal?.addEventListener("abort", forwardAbort, { once: true });
5160
- const deadline = setTimeout(() => controller.abort(), timeoutMs);
5161
- let result;
5162
- try {
5163
- result = await transport({
5164
- url,
5165
- body: {
5166
- model,
5167
- messages: [{
5168
- role: "user",
5169
- content
5170
- }]
5171
- },
5172
- signal: controller.signal
5173
- });
5174
- } catch (error) {
5175
- if (signal?.aborted) return {
5176
- ok: false,
5177
- failure: {
5178
- kind: "aborted",
5179
- message: "prime exchange cancelled by the caller",
5180
- cause: error
5181
- }
5182
- };
5183
- if (controller.signal.aborted) return {
5184
- ok: false,
5185
- failure: {
5186
- kind: "deadline",
5187
- message: `bridge call exceeded ${timeoutMs}ms`
5188
- }
5189
- };
5190
- return {
5191
- ok: false,
5192
- failure: {
5193
- kind: "transport",
5194
- message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
5195
- }
5196
- };
5197
- } finally {
5198
- clearTimeout(deadline);
5199
- signal?.removeEventListener("abort", forwardAbort);
5200
- }
5201
- if (result.status !== 200) {
5202
- const bodySnippet = result.text.slice(0, 500);
5203
- return {
5204
- ok: false,
5205
- failure: {
5206
- kind: "http-status",
5207
- message: `bridge HTTP ${result.status}: ${bodySnippet}`,
5208
- status: result.status,
5209
- bodySnippet
5210
- }
5211
- };
5212
- }
5213
- let response;
5214
- try {
5215
- response = JSON.parse(result.text);
5216
- } catch {
5217
- return {
5218
- ok: false,
5219
- failure: {
5220
- kind: "unparseable-json",
5221
- message: `bridge returned unparseable JSON (${result.text.length} bytes)`
5222
- }
5223
- };
5224
- }
5225
- const replyContent = primeReplyContent(response);
5226
- if (replyContent === null) return {
5227
- ok: false,
5228
- failure: {
5229
- kind: "no-content",
5230
- message: "bridge reply carries no message content"
5231
- }
5232
- };
5233
- const rawUsage = primeReplyUsage(response);
5234
- return {
5235
- ok: true,
5236
- content: replyContent,
5237
- usage: normalizePrimeUsage(rawUsage),
5238
- rawUsage
5239
- };
5240
- }
5241
- /**
5242
- * Fold from the FIRST turn, never from an empty receipt: an all-null identity
5243
- * would poison every side it merged with and erase counts the bridge reported.
5244
- */
5245
- function mergeTurns(turns) {
5246
- if (turns.length === 0) return emptyPrimeRawUsage();
5247
- return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
5248
- }
5249
- function primeReplyContent(response) {
5250
- if (typeof response !== "object" || response === null) return null;
5251
- const choices = response.choices;
5252
- if (!Array.isArray(choices) || choices.length === 0) return null;
5253
- const message = choices[0]?.message;
5254
- if (typeof message !== "object" || message === null) return null;
5255
- const content = message.content;
5256
- return typeof content === "string" && content.length > 0 ? content : null;
5257
- }
5258
- function primeReplyUsage(response) {
5259
- if (typeof response !== "object" || response === null) return null;
5260
- return response.usage ?? null;
5261
- }
5262
- /**
5263
- * Render, measure, fall back to the capped projection, re-measure, fail loud.
5264
- *
5265
- * Inline is the only delivery prime has, so an oversized trajectory is a
5266
- * refusal rather than a silent truncation: dropping spans would understate the
5267
- * trajectory and the analyst would answer a question about a different run.
5268
- */
5269
- async function projectPrimeTrajectory(source, limits) {
5270
- let fetch = "full";
5271
- let items = await source.full();
5272
- if (items === null) {
5273
- fetch = "capped";
5274
- items = await source.capped();
5275
- }
5276
- let rendered = JSON.stringify(items);
5277
- if (rendered.length > limits.maxInlineChars && fetch === "full") {
5278
- fetch = "capped";
5279
- items = await source.capped();
5280
- rendered = JSON.stringify(items);
5281
- }
5282
- if (rendered.length > limits.maxInlineChars) return {
5283
- ok: false,
5284
- reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
5285
- renderedChars: rendered.length
5286
- };
5287
- return {
5288
- ok: true,
5289
- items,
5290
- rendered,
5291
- delivery: {
5292
- mode: "inline-json",
5293
- fetch,
5294
- renderedChars: rendered.length
5295
- }
5296
- };
5297
- }
5298
- /**
5299
- * Digest of everything a consumer can send to the bridge under the prime
5300
- * protocol, recorded per observation so a prime result names the exact contract
5301
- * that produced it.
5302
- *
5303
- * Computed over the ACTUALLY composed contract, so two consumers that both
5304
- * stamp `analyst_id: 'prime'` while asking materially different questions get
5305
- * different digests by construction. That is what makes 'prime' a reproducible
5306
- * claim rather than a label.
5307
- */
5308
- function primeProtocolSha256(identity) {
5309
- return createHash("sha256").update(JSON.stringify({
5310
- kind: "prime-analyst-protocol",
5311
- question: identity.question,
5312
- taskPrompt: identity.taskDefinition ?? null,
5313
- outputContract: identity.contractLines,
5314
- repairContract: buildPrimeRepairPrompt({
5315
- defect: "<defect>",
5316
- previousReply: "<previous-reply>",
5317
- repairContractLines: identity.repairContractLines
5318
- }),
5319
- limits: identity.limits
5320
- })).digest("hex");
5321
- }
5322
- //#endregion
5323
5304
  //#region src/analyst/benchmark-runner-prime.ts
5324
5305
  /**
5325
5306
  * Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
5326
5307
  * cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
5327
5308
  * analyst.
5328
5309
  *
5329
- * The runner consumes the same prepared benchmark cases every other runner
5330
- * receives the trace store already carries the appended final-verification
5331
- * spans and produces findings through the same published block expansion, so
5332
- * a prime observation and a dspy-rlm observation differ only in which analyst
5333
- * produced the blocks.
5310
+ * The arm is expressed as an `AnalystDefinition`
5311
+ * (`primeCodeTraceAnalystDefinition`): the question, task text, output
5312
+ * contract, inline projection budget, and repair-turn declaration are all
5313
+ * definition content, and `createPrimeBenchmarkRunner` is a thin shell that
5314
+ * builds the definition and runs it through the inline strategy below. The
5315
+ * same strategy is what `bindAnalyst` (./bind) dispatches to, so a compiled
5316
+ * definition and this entry point send byte-identical requests — the parity
5317
+ * suite asserts exactly that.
5334
5318
  *
5335
- * The protocol itself — prompt composition, the bounded repair turn, reply
5319
+ * The protocol machinery — prompt composition, the bounded repair turn, reply
5336
5320
  * extraction, the projection ladder, usage normalization — lives in
5337
- * `./prime-protocol`, which knows nothing about CodeTraceBench. This file is
5338
- * the benchmark's binding to it: the block row grammar, the store-backed
5339
- * projection source, and the benchmark observation shape.
5321
+ * `./prime-protocol`, which knows nothing about CodeTraceBench. This file adds
5322
+ * the benchmark's binding to it (block row grammar, store-backed projection,
5323
+ * observation shape) plus the projection-generic inline execution strategy.
5340
5324
  *
5341
5325
  * Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
5342
5326
  * the viewTrace span projection as a REPL variable; prime has no REPL, so the
5343
5327
  * same projection is serialized into the prompt. When the full projection is
5344
- * oversized the runner falls back to chunked viewSpans over the same
5328
+ * oversized the strategy falls back to chunked viewSpans over the same
5345
5329
  * projection surface with a per-attribute byte cap, and fails loud if the
5346
5330
  * result still exceeds the inline budget.
5347
5331
  *
@@ -5443,56 +5427,142 @@ const PRIME_BLOCK_CONTRACT = {
5443
5427
  }
5444
5428
  };
5445
5429
  /**
5446
- * Digest of everything this runner can send to the bridge, recorded per
5430
+ * Digest of everything this arm can send to the bridge, recorded per
5447
5431
  * observation so a prime result names the exact contract that produced it.
5448
5432
  */
5449
5433
  function primeAnalystProtocolSha256() {
5450
5434
  return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
5451
5435
  }
5452
- /** CodeTraceBench-only: the prompt and output contract speak its block grammar. */
5436
+ /**
5437
+ * The prime arm as a declarative unit. CodeTraceBench-only: the question,
5438
+ * task text, and block grammar speak its incorrect-step definition.
5439
+ */
5440
+ function primeCodeTraceAnalystDefinition(args) {
5441
+ return {
5442
+ id: PRIME_ANALYST_ID,
5443
+ description: "One-shot RLM over an OpenAI-compatible bridge answering the CodeTraceBench incorrect-step task.",
5444
+ version: "1.0.0",
5445
+ area: "incorrect",
5446
+ profile: {},
5447
+ question: PRIME_QUESTION,
5448
+ taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5449
+ projection: {
5450
+ mode: "inline",
5451
+ maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS,
5452
+ cappedAttributeBytes: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5453
+ },
5454
+ replyContract: PRIME_BLOCK_CONTRACT,
5455
+ contractLimits: {
5456
+ maxBlocks: 16,
5457
+ maxBlockSteps: 12
5458
+ },
5459
+ budget: { timeoutMs: args.timeoutMs },
5460
+ repair: { turns: args.repairTurns },
5461
+ protocolSha256: primeAnalystProtocolSha256(),
5462
+ binding: {
5463
+ kind: "inline",
5464
+ subjectFromCaseId: trajectoryIdFromCaseId,
5465
+ baseMetadata: {
5466
+ analysisMode: "prime-rlm",
5467
+ engine: "prime"
5468
+ },
5469
+ header(subject, spans) {
5470
+ const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
5471
+ if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${subject}'`);
5472
+ return `TRAJECTORY (trace_id ${subject}; ${stepSpans.length} assistant step spans; full span projection as JSON):`;
5473
+ },
5474
+ trailer(_subject, spans) {
5475
+ const finalVerification = spans.filter(isFinalVerificationSpan);
5476
+ return finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself.";
5477
+ },
5478
+ async expandRows({ subject, rows, store, analystId, signal }) {
5479
+ const expanded = await expandCodeTraceFailureBlocks({
5480
+ trajectoryId: subject,
5481
+ blocks: rows,
5482
+ store,
5483
+ analystId,
5484
+ ...signal ? { signal } : {}
5485
+ });
5486
+ return {
5487
+ findings: expanded.findings,
5488
+ diagnostics: expanded.diagnostics
5489
+ };
5490
+ }
5491
+ }
5492
+ };
5493
+ }
5494
+ /** Thin shell: validate options, declare the definition, run the inline strategy. */
5453
5495
  function createPrimeBenchmarkRunner(options) {
5454
- const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
5455
- const model = requiredString(options.model, "model");
5456
5496
  const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
5457
5497
  const repair = options.repair;
5458
5498
  if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
5459
- const pricing = options.pricing ?? pricingForModel(model);
5460
- const transport = options.transport ?? nodeHttpPrimeBridgeTransport();
5499
+ return runInlineAnalystDefinition(primeCodeTraceAnalystDefinition({
5500
+ timeoutMs,
5501
+ repairTurns: repair ? 1 : 0
5502
+ }), {
5503
+ baseUrl: options.baseUrl,
5504
+ model: options.model,
5505
+ ...options.transport ? { transport: options.transport } : {},
5506
+ ...options.pricing ? { pricing: options.pricing } : {}
5507
+ });
5508
+ }
5509
+ /**
5510
+ * Compile an inline-projection definition into a runnable arm. Projection,
5511
+ * prompt composition, the bounded repair turn, and usage accounting are all
5512
+ * driven by the definition; nothing in this strategy names a benchmark.
5513
+ */
5514
+ function runInlineAnalystDefinition(definition, transports) {
5515
+ const { projection, binding } = definition;
5516
+ if (projection.mode !== "inline" || binding.kind !== "inline") throw new AnalystExpressivenessError(`the inline strategy compiles only inline projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
5517
+ if (definition.repair.turns > 1) throw new AnalystExpressivenessError(`the inline exchange grants at most one bounded repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
5518
+ const baseUrl = requiredString(transports.baseUrl, "baseUrl").replace(/\/+$/, "");
5519
+ const model = requiredString(transports.model, "model");
5520
+ const timeoutMs = positiveSafeInteger(definition.budget.timeoutMs, "timeoutMs");
5521
+ const repair = definition.repair.turns === 1;
5522
+ const pricing = transports.pricing ?? pricingForModel(model);
5523
+ const transport = transports.transport ?? nodeHttpPrimeBridgeTransport();
5461
5524
  const url = `${baseUrl}/v1/chat/completions`;
5462
5525
  return {
5463
- id: PRIME_ANALYST_ID,
5526
+ id: definition.id,
5464
5527
  async analyze(input, context) {
5465
- const trajectoryId = trajectoryIdFromCaseId(context.caseId);
5528
+ const subject = binding.subjectFromCaseId(context.caseId);
5466
5529
  let usage;
5467
5530
  let metadata = {
5468
- analysisMode: "prime-rlm",
5469
- engine: "prime",
5531
+ ...binding.baseMetadata,
5470
5532
  bridgeUrl: baseUrl,
5471
5533
  model,
5472
- protocolSha256: primeAnalystProtocolSha256()
5534
+ protocolSha256: definition.protocolSha256
5473
5535
  };
5474
5536
  try {
5475
5537
  const store = input.traceStore;
5476
- if (!store) throw new Error("codetracebench prime runner requires a trace store");
5477
- const projection = await projectPrimeTrajectory(codeTraceProjectionSource(store, trajectoryId, context.signal ? { signal: context.signal } : void 0), { maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS });
5478
- if (!projection.ok) throw new PrimeTraceProjectionError(projection.reason);
5538
+ if (!store) throw new Error(`inline analyst '${definition.id}' requires a trace store`);
5539
+ const storeContext = context.signal ? { signal: context.signal } : void 0;
5540
+ const projected = await projectPrimeTrajectory(inlineProjectionSource(store, subject, projection.cappedAttributeBytes, storeContext), { maxInlineChars: projection.maxInlineChars });
5541
+ if (!projected.ok) throw new PrimeTraceProjectionError(projected.reason);
5479
5542
  const delivery = {
5480
- mode: projection.delivery.mode,
5481
- fetch: projection.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
5482
- perAttributeByteCap: projection.delivery.fetch === "full" ? null : CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP,
5483
- renderedChars: projection.delivery.renderedChars
5543
+ mode: projected.delivery.mode,
5544
+ fetch: projected.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
5545
+ perAttributeByteCap: projected.delivery.fetch === "full" ? null : projection.cappedAttributeBytes,
5546
+ renderedChars: projected.delivery.renderedChars
5484
5547
  };
5485
5548
  metadata = {
5486
5549
  ...metadata,
5487
5550
  delivery
5488
5551
  };
5489
- const prompt = buildCodeTracePrompt(trajectoryId, projection.items, projection.rendered);
5552
+ const prompt = buildPrimePrompt({
5553
+ question: definition.question,
5554
+ ...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
5555
+ contractLines: definition.replyContract.contractLines,
5556
+ trajectoryHeader: binding.header(subject, projected.items),
5557
+ renderedTrajectory: projected.rendered,
5558
+ trailer: binding.trailer(subject, projected.items)
5559
+ });
5490
5560
  metadata = {
5491
5561
  ...metadata,
5492
5562
  promptChars: prompt.length
5493
5563
  };
5494
5564
  const outcome = await runPrimeExchange({
5495
- contract: PRIME_BLOCK_CONTRACT,
5565
+ contract: definition.replyContract,
5496
5566
  prompt,
5497
5567
  transport,
5498
5568
  url,
@@ -5520,11 +5590,11 @@ function createPrimeBenchmarkRunner(options) {
5520
5590
  };
5521
5591
  throw primeFailureError(outcome.failure);
5522
5592
  }
5523
- const expanded = await expandCodeTraceFailureBlocks({
5524
- trajectoryId,
5525
- blocks: outcome.rows,
5593
+ const expanded = await binding.expandRows({
5594
+ subject,
5595
+ rows: outcome.rows,
5526
5596
  store,
5527
- analystId: PRIME_ANALYST_ID,
5597
+ analystId: definition.id,
5528
5598
  ...context.signal ? { signal: context.signal } : {}
5529
5599
  });
5530
5600
  return {
@@ -5555,13 +5625,13 @@ function createPrimeBenchmarkRunner(options) {
5555
5625
  * the full viewTrace projection, or the chunked viewSpans projection at a
5556
5626
  * per-attribute byte cap.
5557
5627
  */
5558
- function codeTraceProjectionSource(store, trajectoryId, context) {
5628
+ function inlineProjectionSource(store, trajectoryId, cappedAttributeBytes, context) {
5559
5629
  return {
5560
5630
  async full() {
5561
5631
  return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
5562
5632
  },
5563
- capped: () => projectSpansChunked(store, trajectoryId, context),
5564
- cappedDescription: `per-attribute cap ${CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP}`
5633
+ capped: () => projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context),
5634
+ cappedDescription: `per-attribute cap ${cappedAttributeBytes}`
5565
5635
  };
5566
5636
  }
5567
5637
  /**
@@ -5570,7 +5640,7 @@ function codeTraceProjectionSource(store, trajectoryId, context) {
5570
5640
  * id must project or the case fails loud — a silently dropped span would
5571
5641
  * understate the trajectory.
5572
5642
  */
5573
- async function projectSpansChunked(store, trajectoryId, context) {
5643
+ async function projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context) {
5574
5644
  const enumeration = await store.viewTrace({
5575
5645
  trace_id: trajectoryId,
5576
5646
  per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
@@ -5589,26 +5659,13 @@ async function projectSpansChunked(store, trajectoryId, context) {
5589
5659
  const result = await store.viewSpans({
5590
5660
  trace_id: trajectoryId,
5591
5661
  span_ids: chunk,
5592
- per_attribute_byte_cap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5662
+ per_attribute_byte_cap: cappedAttributeBytes
5593
5663
  }, context);
5594
5664
  if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
5595
5665
  projected.push(...result.spans);
5596
5666
  }
5597
5667
  return projected;
5598
5668
  }
5599
- function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
5600
- const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
5601
- if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
5602
- const finalVerification = spans.filter(isFinalVerificationSpan);
5603
- return buildPrimePrompt({
5604
- question: PRIME_QUESTION,
5605
- taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5606
- contractLines: PRIME_OUTPUT_CONTRACT_LINES,
5607
- trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
5608
- renderedTrajectory: renderedSpans,
5609
- trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
5610
- });
5611
- }
5612
5669
  /** Map the protocol's terminal reason onto this benchmark's typed error classes. */
5613
5670
  function primeFailureError(failure) {
5614
5671
  switch (failure.kind) {
@@ -6353,6 +6410,6 @@ function shellQuote(value) {
6353
6410
  return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
6354
6411
  }
6355
6412
  //#endregion
6356
- export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, createPublicBenchmarkRlmRunner as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, selectPublicBenchmarkRows as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, nodeHttpPrimeBridgeTransport as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, publicBenchmarkDistributions as b, buildPrimeRepairPrompt as c, summarizeAgentRxCalibration as ct, mergePrimeRawUsage as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, normalizePrimeUsage as f, agentRxPredictionsToFindings as ft, runPrimeExchange as g, projectPrimeTrajectory as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, emptyPrimeRawUsage as l, codeTraceBenchCase as lt, primeReplyDefect as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, analystUsageReceiptFromPrimeUsage as o, AGENT_RX_UPSTREAM_REVISION as ot, primeProtocolSha256 as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, buildPrimePrompt as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, extractPrimeJsonObject as u, codeTracerPredictionsToFindings as ut, loadPublicBenchmarkRows as v, createPublicBenchmarkDirectRunner as w, publicBenchmarkSelectionReport as x, preparePublicAnalystBenchmark as y, publicBenchmarkRlmInstructions as z };
6413
+ export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, analystDefinitionAsymmetries as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, AnalystExpressivenessError as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, summarizeAgentRxCalibration as ct, publicBenchmarkDistributions as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, publicBenchmarkSelectionReport as f, agentRxPredictionsToFindings as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, loadPublicBenchmarkRows as l, codeTraceBenchCase as lt, createPublicBenchmarkRlmRunner as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, primeCodeTraceAnalystDefinition as o, AGENT_RX_UPSTREAM_REVISION as ot, selectPublicBenchmarkRows as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, runInlineAnalystDefinition as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, preparePublicAnalystBenchmark as u, codeTracerPredictionsToFindings as ut, createPublicBenchmarkDirectRunner as v, analystDefinitionProtocolSha256 as w, decodeReplyRows as x, publicDirectAnalystDefinition as y, publicBenchmarkRlmInstructions as z };
6357
6414
 
6358
- //# sourceMappingURL=benchmark-command-CQd78YHt.js.map
6415
+ //# sourceMappingURL=benchmark-command-BKENp2s5.js.map