@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,781 @@
1
+ import { s as TraceStore } from "../store-CT9YIIve.js";
2
+ import { i as TraceEmitter } from "../emitter-DGQGoLyj.js";
3
+ import { i as CounterfactualRunner, t as CounterfactualContext } from "../counterfactual-CxmxAONP.js";
4
+ import { a as parseObservationOutput, i as isSubmitAction, n as SUBMIT_ACTION_SIGNATURE, o as parseRecordedReturncode, r as deriveFailureSignature, t as RecordedTrajectoryStep } from "../steps-BArUxhna.js";
5
+ //#region src/trajectory-replay/corpus.d.ts
6
+ interface CorpusSpec {
7
+ readonly name: string;
8
+ readonly labelsPath: string;
9
+ readonly preparedDir: string;
10
+ }
11
+ /** Parses `name=<labelsPath>::<preparedDir>` (paths may contain `=`, not `::`). */
12
+ declare function parseCorpusFlag(value: string): CorpusSpec;
13
+ type ReplayExclusionReason = 'no-swe-raw-trajectory' | 'ambiguous-swe-raw-trajectory' | 'unreadable-raw-trajectory' | 'no-docker-image' | 'no-gold-incorrect-step' | 'gold-only-submit-step' | 'missing-steps-json' | 'gold-step-outside-steps' | 'cwd-underivable';
14
+ interface ExcludedCase {
15
+ readonly corpus: string;
16
+ readonly trajId: string;
17
+ readonly reason: ReplayExclusionReason;
18
+ readonly detail?: string;
19
+ }
20
+ type CwdSource = 'run-config' | 'docker-config' | 'pwd-observation';
21
+ /** Everything needed to replay one trajectory. */
22
+ interface CaseResources {
23
+ readonly corpus: string;
24
+ readonly trajId: string;
25
+ readonly stepsPath: string;
26
+ readonly steps: readonly RecordedTrajectoryStep[];
27
+ readonly taskStatement: string | null;
28
+ readonly image: string;
29
+ readonly cwd: string;
30
+ readonly cwdSource: CwdSource;
31
+ /** Per-step timeout the recorded run used, when the raw config carries one. */
32
+ readonly recordedStepTimeoutMs: number | null;
33
+ }
34
+ interface ReplayableCase extends CaseResources {
35
+ /** 1-based gold incorrect step ids, ascending. */
36
+ readonly goldIncorrectSteps: readonly number[];
37
+ /** k — the first gold incorrect step that is a real mid-trajectory action;
38
+ * submit-command golds are skipped (see SUBMIT_ACTION_SIGNATURE). */
39
+ readonly k: number;
40
+ /** Gold steps before k skipped because their action is the submit command. */
41
+ readonly submitGoldsSkipped: number;
42
+ /** Recorded returncode at k; null when the observation carries none. */
43
+ readonly recordedReturncodeAtK: number | null;
44
+ }
45
+ interface LabelEntry {
46
+ readonly traj_id: string;
47
+ readonly incorrect_stages?: readonly {
48
+ readonly incorrect_step_ids?: readonly number[];
49
+ }[];
50
+ }
51
+ declare function readLabelEntries(labelsPath: string): LabelEntry[];
52
+ declare function goldIncorrectSteps(entry: LabelEntry): number[];
53
+ type ResourceResolution = {
54
+ readonly resolved: true;
55
+ readonly resources: CaseResources;
56
+ } | {
57
+ readonly resolved: false;
58
+ readonly reason: ReplayExclusionReason;
59
+ readonly detail?: string;
60
+ };
61
+ /**
62
+ * Resolves the replay resources for one trajectory, independent of gold
63
+ * labels — the finding wire uses this with a finding-supplied step instead.
64
+ */
65
+ declare function resolveCaseResources(corpus: CorpusSpec, trajId: string): ResourceResolution;
66
+ interface EnumerationResult {
67
+ readonly replayable: ReplayableCase[];
68
+ readonly excluded: ExcludedCase[];
69
+ readonly labelEntryCount: number;
70
+ }
71
+ /**
72
+ * Replayable = a raw trajectory with a recorded image AND at least one gold
73
+ * incorrect step that is a real mid-trajectory action. Gold steps whose action
74
+ * is the submit command are skipped when choosing k (counted per case); a case
75
+ * whose golds are ALL submit steps is excluded as gold-only-submit-step.
76
+ * Exclusion reasons are reported in resolution order:
77
+ * raw trajectory → image → gold labels → steps.json → k range → cwd.
78
+ */
79
+ declare function enumerateReplayableCases(corpora: readonly CorpusSpec[]): EnumerationResult;
80
+ //#endregion
81
+ //#region src/trajectory-replay/exec.d.ts
82
+ /**
83
+ * The execution boundary replay runs across.
84
+ *
85
+ * A replay needs one thing from its environment: a session that runs a shell
86
+ * command inside the trajectory's own image and reports the exit code and
87
+ * output. That is the whole contract. Concrete backends — a sandbox platform
88
+ * client, a docker exec, an SSH shell — live with the consumer that owns the
89
+ * infrastructure, so this package depends on no sandbox client.
90
+ */
91
+ interface ReplayExecResult {
92
+ exitCode: number;
93
+ stdout: string;
94
+ stderr: string;
95
+ }
96
+ interface ReplayExecSession {
97
+ exec(command: string, timeoutMs: number): Promise<ReplayExecResult>;
98
+ close(): Promise<void>;
99
+ }
100
+ interface ReplayExecBackend {
101
+ /** One fresh execution environment per call; the caller closes it. */
102
+ open(): Promise<ReplayExecSession>;
103
+ }
104
+ /** Builds a backend pinned to one image. Callers that resolve images
105
+ * internally (batch, corpus wire, finding verification) take this instead of
106
+ * a backend, so every case runs against its own image. */
107
+ type ReplayExecBackendFactory = (image: string) => ReplayExecBackend;
108
+ /**
109
+ * mini-SWE runs every action as a fresh /bin/sh subshell from a fixed
110
+ * workdir. Reproduce that exactly — and stay quote-proof for arbitrary
111
+ * recorded actions — by piping the base64 of the action into `sh` after
112
+ * cd-ing to the workdir. Exit code is sh's, i.e. the action's.
113
+ */
114
+ declare function wrapActionForExec(action: string, cwd: string): string;
115
+ //#endregion
116
+ //#region src/trajectory-replay/fix.d.ts
117
+ interface ChatUsage {
118
+ readonly promptTokens: number;
119
+ readonly completionTokens: number;
120
+ }
121
+ type ChatOutcome = {
122
+ readonly succeeded: true;
123
+ readonly value: {
124
+ content: string;
125
+ usage: ChatUsage | null;
126
+ };
127
+ } | {
128
+ readonly succeeded: false;
129
+ readonly error: string;
130
+ };
131
+ interface ChatCompletionCaller {
132
+ complete(system: string, user: string): Promise<ChatOutcome>;
133
+ }
134
+ interface FixPromptInput {
135
+ readonly taskStatement: string | null;
136
+ readonly steps: readonly RecordedTrajectoryStep[];
137
+ /** 1-based step_id of the incorrect step. */
138
+ readonly k: number;
139
+ /** Steps of context on each side of k (default 3). */
140
+ readonly contextRadius?: number;
141
+ }
142
+ /** Head+tail excerpt with an elision marker; identity below the limit. */
143
+ declare function clipText(text: string, limit: number): string;
144
+ declare function buildFixPrompt(input: FixPromptInput): {
145
+ system: string;
146
+ user: string;
147
+ };
148
+ /** One prior attempt of the fix loop, rendered into the retry prompt. */
149
+ interface FailedFixAttempt {
150
+ readonly attempt: number;
151
+ /** Null when the model call itself failed before producing a command. */
152
+ readonly command: string | null;
153
+ readonly exitCode: number | null;
154
+ readonly stdoutTail: string | null;
155
+ readonly stderrTail: string | null;
156
+ readonly llmError: string | null;
157
+ }
158
+ /**
159
+ * Retry prompt for fix-loop attempts ≥2: the original context plus every prior
160
+ * attempt with its REAL executed output, and permission to answer with a short
161
+ * script (the block still executes as one /bin/sh unit).
162
+ */
163
+ declare function buildRetryFixPrompt(input: FixPromptInput, priorAttempts: readonly FailedFixAttempt[], maxScriptCommands?: number): {
164
+ system: string;
165
+ user: string;
166
+ };
167
+ /** Non-empty, non-comment lines of a fix script — the loop's script-size cap. */
168
+ declare function countScriptCommands(script: string): number;
169
+ /** Last fenced code block, else the whole trimmed content; null when empty. */
170
+ declare function extractFixCommand(content: string): string | null;
171
+ type FixGenerationOutcome = {
172
+ readonly succeeded: true;
173
+ readonly value: {
174
+ command: string;
175
+ usage: ChatUsage | null;
176
+ };
177
+ } | {
178
+ readonly succeeded: false;
179
+ readonly error: string;
180
+ };
181
+ declare function generateFixCommand(caller: ChatCompletionCaller, input: FixPromptInput): Promise<FixGenerationOutcome>;
182
+ //#endregion
183
+ //#region src/trajectory-replay/fix-loop.d.ts
184
+ /** Result of executing one corrected command as a full arm. */
185
+ interface FixArmExecution {
186
+ readonly exitCode: number;
187
+ readonly prefixExecuted: number;
188
+ readonly prefixDivergences: number;
189
+ /** Divergent prefix steps over executed ones, in percent. */
190
+ readonly prefixDivergencePct: number;
191
+ readonly failureVanished: boolean;
192
+ readonly stdout: string;
193
+ readonly stderr: string;
194
+ }
195
+ /** Runs one corrected command as a full arm: fresh sandbox + prefix replay. */
196
+ type FixArmExecutor = (command: string, attempt: number) => Promise<FixArmExecution>;
197
+ interface FixLoopOptions {
198
+ /** Total LLM attempts per case (≥1); 1 degenerates to the one-shot path. */
199
+ readonly maxAttempts: number;
200
+ /** Command-line cap on retry scripts (default 5). */
201
+ readonly maxScriptCommands?: number;
202
+ /** Chars kept per stdout/stderr tail in records and retry prompts. */
203
+ readonly outputTailChars?: number;
204
+ readonly onProgress?: (message: string) => void;
205
+ }
206
+ interface FixLoopAttemptRecord {
207
+ readonly attempt: number;
208
+ readonly command: string | null;
209
+ readonly llmError: string | null;
210
+ readonly usage: {
211
+ readonly promptTokens: number;
212
+ readonly completionTokens: number;
213
+ } | null;
214
+ readonly executed: boolean;
215
+ readonly exitCode: number | null;
216
+ readonly prefixExecuted: number | null;
217
+ readonly prefixDivergences: number | null;
218
+ readonly prefixDivergencePct: number | null;
219
+ readonly failureVanished: boolean | null;
220
+ readonly stdoutTail: string | null;
221
+ readonly stderrTail: string | null;
222
+ readonly armBError: string | null;
223
+ readonly wallMs: number;
224
+ }
225
+ interface FixLoopResult {
226
+ readonly flipped: boolean;
227
+ readonly flippedAtAttempt: number | null;
228
+ readonly attempts: readonly FixLoopAttemptRecord[];
229
+ /** True when a sandbox error ended the loop before the attempt budget. */
230
+ readonly aborted: boolean;
231
+ readonly llmCalls: number;
232
+ /** Calls that produced no runnable fix: transport errors, empty replies,
233
+ * and retry scripts over the command cap. */
234
+ readonly llmFailures: number;
235
+ readonly promptTokens: number;
236
+ readonly completionTokens: number;
237
+ /** Successful calls whose provider reported no usage. Their tokens are absent
238
+ * from the two totals above, so a nonzero count makes those totals a lower
239
+ * bound rather than a measurement. */
240
+ readonly callsWithoutUsage: number;
241
+ }
242
+ declare function runFixLoop(caller: ChatCompletionCaller, input: FixPromptInput, executor: FixArmExecutor, options: FixLoopOptions): Promise<FixLoopResult>;
243
+ //#endregion
244
+ //#region src/trajectory-replay/image-preparer.d.ts
245
+ /**
246
+ * Replay-ready image derivation.
247
+ *
248
+ * A recorded trajectory names the image it ran in, but the image is not always
249
+ * runnable as-is: sandbox platforms pin customer commands to a non-root
250
+ * identity, so a root-owned working tree must be chowned before the replay can
251
+ * write to it. `ImagePreparer` is that step, injectable so a consumer whose
252
+ * images are already replay-ready supplies its own no-op or none at all.
253
+ */
254
+ type ImagePreparation = {
255
+ readonly succeeded: true;
256
+ readonly value: {
257
+ derivedImage: string;
258
+ pulled: boolean;
259
+ built: boolean;
260
+ };
261
+ } | {
262
+ readonly succeeded: false;
263
+ readonly error: string;
264
+ };
265
+ interface ImagePreparer {
266
+ ensure(image: string, cwd: string): Promise<ImagePreparation>;
267
+ }
268
+ interface DockerImagePreparerOptions {
269
+ readonly pullTimeoutMs?: number;
270
+ readonly buildTimeoutMs?: number;
271
+ }
272
+ declare function derivedImageTag(image: string, cwd: string): string;
273
+ /**
274
+ * Pulls the base image when absent and builds `FROM <base>; RUN chown -R
275
+ * 1000:1000 <cwd>` tagged by content hash, so repeated batches reuse both
276
+ * the pull and the build. cwd `/` skips the chown (never chown -R /) and
277
+ * replays on the base image directly.
278
+ */
279
+ declare function dockerImagePreparer(options?: DockerImagePreparerOptions): ImagePreparer;
280
+ //#endregion
281
+ //#region src/trajectory-replay/batch.d.ts
282
+ interface ReplayBatchOptions {
283
+ readonly corpora: readonly CorpusSpec[];
284
+ readonly out: string;
285
+ /** 'generate' = one LLM call per arm-A-reproduced case, then arm B.
286
+ * 'loop' = iterative: failed arms feed their real output into up to
287
+ * `fixAttempts` prompts, each executed in its own fresh session. */
288
+ readonly fix: 'none' | 'generate' | 'loop';
289
+ /** Attempt budget per case in loop mode (default 3). */
290
+ readonly fixAttempts?: number;
291
+ readonly fixCaller?: ChatCompletionCaller;
292
+ readonly fixModelLabel?: string;
293
+ /** Cap on LLM fix calls; eligible cases beyond it are seeded-sampled out. */
294
+ readonly maxFixCases?: number;
295
+ readonly seed?: number;
296
+ /** Overrides the per-case recorded step timeout. */
297
+ readonly stepTimeoutMs?: number;
298
+ readonly prefixLimit?: number;
299
+ /** Run only cases whose trajId contains this substring (smoke knob). */
300
+ readonly caseFilter?: string;
301
+ /** Run only the first N replayable cases (smoke knob). */
302
+ readonly caseLimit?: number;
303
+ /** Derives the replay-ready image per case. Defaults to the docker preparer. */
304
+ readonly preparer?: ImagePreparer;
305
+ /** Builds the exec backend for a case's derived image. */
306
+ readonly backendFactory: ReplayExecBackendFactory;
307
+ readonly onProgress?: (message: string) => void;
308
+ }
309
+ interface ReplayBatchFixResult {
310
+ /** false when the case was eligible but seeded-sampled out of the cap. */
311
+ readonly attempted: boolean;
312
+ readonly sampledOut: boolean;
313
+ readonly command: string | null;
314
+ readonly llmError: string | null;
315
+ /** Loop mode: token totals summed across every attempt (null when the
316
+ * provider reported no usage). */
317
+ readonly usage: ChatUsage | null;
318
+ readonly armBExit: number | null;
319
+ readonly armBPrefixExecuted: number | null;
320
+ readonly armBPrefixDivergences: number | null;
321
+ readonly armBPrefixDivergencePct: number | null;
322
+ readonly failureVanished: boolean | null;
323
+ readonly armBError: string | null;
324
+ /** Loop mode only: the full per-attempt trail. Null in generate mode. */
325
+ readonly attempts: readonly FixLoopAttemptRecord[] | null;
326
+ /** 1-based attempt that flipped the failure; null when none did.
327
+ * Generate mode: 1 when the single attempt flipped. */
328
+ readonly flippedAtAttempt: number | null;
329
+ }
330
+ interface ReplayBatchCaseRow {
331
+ readonly corpus: string;
332
+ readonly trajId: string;
333
+ readonly image: string;
334
+ readonly derivedImage: string | null;
335
+ readonly cwd: string;
336
+ readonly cwdSource: string;
337
+ readonly k: number;
338
+ readonly stepCount: number;
339
+ readonly goldIncorrectSteps: readonly number[];
340
+ /** Gold steps before k skipped because their action is the submit command. */
341
+ readonly submitGoldsSkipped: number;
342
+ readonly recordedReturncodeAtK: number | null;
343
+ readonly signature: string | null;
344
+ readonly status: 'ok' | 'image-unavailable' | 'replay-error';
345
+ readonly error: string | null;
346
+ readonly imagePulled: boolean;
347
+ readonly imageBuilt: boolean;
348
+ readonly prefixExecuted: number | null;
349
+ readonly prefixDivergences: number | null;
350
+ readonly prefixDivergencePct: number | null;
351
+ /** Prefix steps whose recorded returncode equalled the replayed exit. */
352
+ readonly prefixConfirmed: number | null;
353
+ readonly prefixReturncodeMismatches: number | null;
354
+ /** Prefix steps the recording carries no returncode for: unverifiable, and
355
+ * counted as divergences because agreement was never established. */
356
+ readonly prefixUnknownExpectations: number | null;
357
+ readonly armAExit: number | null;
358
+ readonly armAReturncodeMatch: boolean;
359
+ readonly armASignatureMatch: boolean;
360
+ /** Headline predicate: prefix divergence within tolerance AND arm A
361
+ * reproduced the recorded returncode at k. */
362
+ readonly replayed: boolean;
363
+ readonly fix: ReplayBatchFixResult | null;
364
+ readonly wallMs: number;
365
+ }
366
+ interface ReplayBatchReport {
367
+ readonly generatedAt: string;
368
+ readonly corpora: readonly {
369
+ name: string;
370
+ labelsPath: string;
371
+ preparedDir: string;
372
+ }[];
373
+ readonly totals: {
374
+ readonly labelEntries: number;
375
+ readonly replayable: number;
376
+ readonly executed: number;
377
+ readonly excludedByReason: Record<string, number>;
378
+ /** Per-corpus submit-gold accounting: cases dropped because every gold is
379
+ * the submit command, and golds skipped inside still-replayable cases. */
380
+ readonly submitGoldsByCorpus: Record<string, {
381
+ submitOnlyCases: number;
382
+ goldsSkippedWithinReplayable: number;
383
+ }>;
384
+ };
385
+ readonly headline: {
386
+ readonly replayabilityRate: {
387
+ numerator: number;
388
+ denominator: number;
389
+ value: number | null;
390
+ };
391
+ readonly signatureStrictRate: {
392
+ numerator: number;
393
+ denominator: number;
394
+ value: number | null;
395
+ };
396
+ /** Corpus-level replay fidelity over every executed prefix step. A corpus
397
+ * whose recordings cannot adjudicate the replay lands here as
398
+ * `unknownExpectations`, not as a clean replay. */
399
+ readonly prefixFidelity: {
400
+ readonly executedSteps: number;
401
+ readonly divergentSteps: number;
402
+ readonly returncodeMismatches: number;
403
+ readonly unknownExpectations: number;
404
+ /** Divergent over executed steps; null when no prefix step ran. */
405
+ readonly divergencePct: number | null;
406
+ readonly tolerancePct: number;
407
+ readonly casesWithinTolerance: number;
408
+ readonly casesExecuted: number;
409
+ };
410
+ readonly fixFlipRate: {
411
+ numerator: number;
412
+ denominator: number;
413
+ value: number | null;
414
+ } | null;
415
+ /** Fix-flip restricted to cases whose recorded returncode at k is nonzero —
416
+ * real recorded failures, where "the failure vanished" is not vacuous. */
417
+ readonly fixFlipRateNonzeroRc: {
418
+ numerator: number;
419
+ denominator: number;
420
+ value: number | null;
421
+ } | null;
422
+ /** Loop mode only: flips at attempt 1 over cases whose attempt 1 executed —
423
+ * the number directly comparable to the one-shot fixFlipRate. */
424
+ readonly fixFlipAttempt1: {
425
+ numerator: number;
426
+ denominator: number;
427
+ value: number | null;
428
+ } | null;
429
+ /** Loop mode only: flip count keyed by the attempt number that flipped. */
430
+ readonly flipsByAttempt: Record<string, number> | null;
431
+ };
432
+ readonly llm: {
433
+ readonly model: string;
434
+ readonly calls: number;
435
+ readonly failures: number;
436
+ readonly promptTokens: number;
437
+ readonly completionTokens: number;
438
+ /** Successful calls whose provider reported no usage. Their tokens are
439
+ * absent from the two totals above, so a nonzero count here means the
440
+ * totals are a lower bound, not a measurement. */
441
+ readonly callsWithoutUsage: number;
442
+ } | null;
443
+ readonly excluded: readonly {
444
+ corpus: string;
445
+ trajId: string;
446
+ reason: string;
447
+ detail?: string;
448
+ }[];
449
+ readonly pullFailures: readonly {
450
+ corpus: string;
451
+ trajId: string;
452
+ image: string;
453
+ error: string;
454
+ }[];
455
+ readonly cases: readonly ReplayBatchCaseRow[];
456
+ }
457
+ declare function seededSample<T>(items: readonly T[], size: number, seed: number): Set<T>;
458
+ declare function runReplayBatch(options: ReplayBatchOptions): Promise<ReplayBatchReport>;
459
+ declare function renderBatchReport(report: ReplayBatchReport): string;
460
+ //#endregion
461
+ //#region src/trajectory-replay/verify.d.ts
462
+ interface IngestedTrajectory {
463
+ runId: string;
464
+ stepCount: number;
465
+ }
466
+ /**
467
+ * Emit one tool span per trajectory step, in order, with a monotonic
468
+ * injected clock so `buildTrajectory` ordering is deterministic even when
469
+ * two spans would share a Date.now() millisecond.
470
+ */
471
+ declare function ingestRecordedTrajectory(store: TraceStore, steps: readonly RecordedTrajectoryStep[], caseId: string): Promise<IngestedTrajectory>;
472
+ /** Why a replayed prefix step failed to confirm the recording. */
473
+ type PrefixDivergenceKind =
474
+ /** The recording carries a returncode and the replayed exit differs from it. */
475
+ 'returncode-mismatch' |
476
+ /** The recording carries no returncode, so agreement cannot be established. */
477
+ 'unknown-expectation';
478
+ interface PrefixDivergence {
479
+ step: number;
480
+ kind: PrefixDivergenceKind;
481
+ /** null exactly when `kind` is `unknown-expectation`. */
482
+ expectedReturncode: number | null;
483
+ actualExit: number;
484
+ }
485
+ interface PrefixReplayResult {
486
+ prefixExecuted: number;
487
+ /** Every step that did not confirm the recording, of both kinds. */
488
+ prefixDivergences: PrefixDivergence[];
489
+ /** Steps whose recorded returncode equalled the replayed exit. */
490
+ prefixConfirmed: number;
491
+ prefixReturncodeMismatches: number;
492
+ prefixUnknownExpectations: number;
493
+ /** Divergent steps over executed steps, in percent, one decimal.
494
+ * 0 for an empty prefix: no step ran, so none diverged. */
495
+ prefixDivergencePct: number;
496
+ /** `prefixDivergencePct` within `PREFIX_DIVERGENCE_TOLERANCE_PCT`. */
497
+ prefixWithinTolerance: boolean;
498
+ wallMs: number;
499
+ }
500
+ /** Admission tolerance: a prefix replay is faithful enough to build a verdict
501
+ * on when at most this percentage of its executed steps diverged. */
502
+ declare const PREFIX_DIVERGENCE_TOLERANCE_PCT = 10;
503
+ /**
504
+ * Compare one replayed prefix step against its recording. Returns null only
505
+ * when the recording positively confirms the replay.
506
+ */
507
+ declare function classifyPrefixStep(step: number, expectedReturncode: number | null, actualExit: number): PrefixDivergence | null;
508
+ /** Roll per-step classifications up into the rate the admission pre-pass gates on. */
509
+ declare function summarizePrefixReplay(prefixExecuted: number, prefixDivergences: PrefixDivergence[], wallMs: number): PrefixReplayResult;
510
+ interface ArmExecutionResult {
511
+ command: string;
512
+ exitCode: number;
513
+ stdout: string;
514
+ stderr: string;
515
+ wallMs: number;
516
+ }
517
+ interface SandboxCounterfactualRunnerOptions {
518
+ cwd: string;
519
+ stepTimeoutMs: number;
520
+ /** Execute at most this many prefix steps (from the start). Fast-iteration
521
+ * knob; a truncated prefix weakens state fidelity and the verdict says so. */
522
+ prefixLimit?: number;
523
+ onProgress?: (message: string) => void;
524
+ }
525
+ /**
526
+ * Replays `ctx.prefix` in a fresh session, then executes the mutated step.
527
+ * Divergences are recorded and never abort the replay. Results land on
528
+ * `lastPrefix` / `lastArm` for the caller; spans for every exec land in the
529
+ * counterfactual meta-run.
530
+ */
531
+ declare class SandboxCounterfactualRunner implements CounterfactualRunner {
532
+ private readonly backend;
533
+ private readonly options;
534
+ lastPrefix: PrefixReplayResult | null;
535
+ lastArm: ArmExecutionResult | null;
536
+ constructor(backend: ReplayExecBackend, options: SandboxCounterfactualRunnerOptions);
537
+ executeFrom(ctx: CounterfactualContext, emitter: TraceEmitter): Promise<void>;
538
+ }
539
+ interface ReplayVerifyOptions {
540
+ stepsPath: string;
541
+ image: string;
542
+ /** 1-based step_id of the error-critical step. */
543
+ at: number;
544
+ /** Corrected command for arm B. Omit to run arm A only. */
545
+ fixCommand?: string;
546
+ cwd: string;
547
+ out: string;
548
+ caseId?: string;
549
+ /** Override the auto-derived failure signature substring. */
550
+ signature?: string;
551
+ stepTimeoutMs?: number;
552
+ prefixLimit?: number;
553
+ /** Label of the driver backing the exec backend (reported, not probed). */
554
+ driverLabel?: string;
555
+ /** Execution environment for both arms; each arm opens its own session. */
556
+ backend: ReplayExecBackend;
557
+ onProgress?: (message: string) => void;
558
+ }
559
+ interface ReplayArmVerdict {
560
+ command: string;
561
+ exitCode: number;
562
+ wallMs: number;
563
+ prefix: PrefixReplayResult;
564
+ }
565
+ interface ReplayVerdict {
566
+ case: string;
567
+ image: string;
568
+ driver: string;
569
+ k: number;
570
+ cwd: string;
571
+ recordedReturncode: number | null;
572
+ signature: string | null;
573
+ signatureBasis: 'returncode+output-substring' | 'returncode-only';
574
+ prefixExecuted: number;
575
+ prefixDivergences: PrefixDivergence[];
576
+ /** Arm A's prefix divergence rate — the number admission gates on. */
577
+ prefixDivergencePct: number;
578
+ prefixConfirmed: number;
579
+ prefixReturncodeMismatches: number;
580
+ prefixUnknownExpectations: number;
581
+ prefixWithinTolerance: boolean;
582
+ armA: ReplayArmVerdict & {
583
+ failureSignatureMatch: boolean;
584
+ };
585
+ armB: (ReplayArmVerdict & {
586
+ failureVanished: boolean;
587
+ }) | null;
588
+ timings: {
589
+ armAMs: number;
590
+ armBMs: number | null;
591
+ totalMs: number;
592
+ };
593
+ runIds: {
594
+ original: string;
595
+ armA: string;
596
+ armB: string | null;
597
+ };
598
+ }
599
+ declare function replayVerify(options: ReplayVerifyOptions): Promise<ReplayVerdict>;
600
+ //#endregion
601
+ //#region src/trajectory-replay/findings.d.ts
602
+ interface VerifiableFinding {
603
+ readonly finding_id?: string;
604
+ readonly analyst_id?: string;
605
+ readonly subject?: string;
606
+ readonly area?: string;
607
+ readonly claim?: string;
608
+ readonly evidence_refs?: readonly {
609
+ readonly kind?: string;
610
+ readonly uri?: string;
611
+ readonly excerpt?: string;
612
+ }[];
613
+ readonly metadata?: Readonly<Record<string, unknown>>;
614
+ }
615
+ /**
616
+ * 1-based step the finding accuses, or null when the finding names none.
617
+ * `metadata.block_first_step` wins over the subject: the analyst records the
618
+ * block's first incorrect step there even when the subject names a later
619
+ * step of the same block.
620
+ */
621
+ declare function findingReplayStep(finding: VerifiableFinding): number | null;
622
+ /** Trajectory id from the finding's `trace://<traj>/…` evidence refs, or null. */
623
+ declare function findingTrajectoryId(finding: VerifiableFinding): string | null;
624
+ /**
625
+ * Where the executable trajectory lives.
626
+ * `direct` — one trajectory's steps.json plus a replay-ready image; the
627
+ * caller owns image preparation.
628
+ * `corpus` — labeled trajectory corpora; each finding's trajectory is
629
+ * resolved by its `trace://` evidence and the image is derived through
630
+ * `preparer` (the docker uid-1000 preparer unless overridden; `null` when
631
+ * the images are already replay-ready).
632
+ */
633
+ type FindingReplaySource = {
634
+ readonly kind: 'direct';
635
+ readonly stepsPath: string;
636
+ readonly image: string;
637
+ readonly cwd: string;
638
+ readonly caseId?: string;
639
+ } | {
640
+ readonly kind: 'corpus';
641
+ readonly corpora: readonly CorpusSpec[];
642
+ readonly preparer?: ImagePreparer | null;
643
+ };
644
+ interface ResolvedFindingReplay {
645
+ readonly caseId: string;
646
+ readonly stepsPath: string;
647
+ /** Raw image; corpus-mode execution derives the replay image from it. */
648
+ readonly image: string;
649
+ readonly cwd: string;
650
+ /** 1-based step_id arm A executes. */
651
+ readonly at: number;
652
+ readonly recordedReturncode: number;
653
+ readonly recordedStepTimeoutMs: number | null;
654
+ }
655
+ type FindingReplayability = {
656
+ readonly replayable: true;
657
+ readonly resolved: ResolvedFindingReplay;
658
+ } | {
659
+ readonly replayable: false;
660
+ readonly reason: string;
661
+ };
662
+ /**
663
+ * Decides whether one finding can be executed against the source, and with
664
+ * what invocation. Never throws for a finding-shaped problem — every dead end
665
+ * becomes a `not-replayable` reason the receipt can carry verbatim.
666
+ */
667
+ declare function resolveFindingReplayability(finding: VerifiableFinding, source: FindingReplaySource): FindingReplayability;
668
+ type FindingVerificationStatus = 'reproduced' | 'fix-flipped' | 'not-replayable' | 'divergent';
669
+ interface FindingVerification {
670
+ readonly finding_id: string | null;
671
+ readonly subject: string | null;
672
+ readonly trajectory_id: string | null;
673
+ /** 1-based step the proof executed at; null when not replayable. */
674
+ readonly step: number | null;
675
+ readonly verified: FindingVerificationStatus;
676
+ /** Present exactly when `verified` is `not-replayable`. */
677
+ readonly reason: string | null;
678
+ /** Receipt directory: receipt.json plus, when executed, replay-verdict.json + report.md. */
679
+ readonly receipt: string;
680
+ readonly verdict_path: string | null;
681
+ /** Receipt dir of the executed proof this finding shares (same case, step, and fix). */
682
+ readonly deduplicated_with: string | null;
683
+ }
684
+ interface VerifyFindingsRun {
685
+ readonly out: string;
686
+ readonly verifications: readonly FindingVerification[];
687
+ readonly counts: Readonly<Record<FindingVerificationStatus, number>>;
688
+ /** Executions actually performed (deduplicated proofs count once). */
689
+ readonly executions: number;
690
+ }
691
+ interface VerifyFindingsOptions {
692
+ readonly source: FindingReplaySource;
693
+ /** Receipt root; one subdirectory per finding plus verifications.json. */
694
+ readonly out: string;
695
+ /** Corrected command for arm B on every executed finding; omit for arm A only. */
696
+ readonly fixCommand?: string;
697
+ readonly stepTimeoutMs?: number;
698
+ readonly prefixLimit?: number;
699
+ /** Builds the exec backend for a finding's replay image. */
700
+ readonly backendFactory: ReplayExecBackendFactory;
701
+ /** Runs once before the first proof, only when some finding is replayable.
702
+ * Throw to refuse execution against absent or degraded infrastructure. */
703
+ readonly preflight?: () => Promise<void>;
704
+ readonly onProgress?: (message: string) => void;
705
+ }
706
+ /**
707
+ * Arm A reproduced on a prefix the recording confirmed → the fix flipping it
708
+ * beats plain reproduction; anything else diverged. A proof standing on a
709
+ * prefix outside the divergence tolerance is divergent no matter what arm A
710
+ * did: the state it ran against is not the recorded state.
711
+ */
712
+ declare function classifyVerdict(verdict: ReplayVerdict): FindingVerificationStatus;
713
+ /**
714
+ * Verifies every finding against the source: resolves replayability, executes
715
+ * one proof per distinct (case, step, fix) — findings accusing the same step
716
+ * share the executed proof — and writes a receipt directory per finding plus a
717
+ * run-level verifications.json.
718
+ */
719
+ declare function verifyFindings(findings: readonly VerifiableFinding[], options: VerifyFindingsOptions): Promise<VerifyFindingsRun>;
720
+ /** Markdown section an analysis report appends when finding verification ran. */
721
+ declare function renderVerifiedFindingsSection(run: VerifyFindingsRun): string;
722
+ /**
723
+ * Accepts the two shapes findings travel in: a bare JSON array of analyst
724
+ * findings, or an object with a `findings` array (e.g. an extracted
725
+ * `observations[n]` from a result.json).
726
+ */
727
+ declare function readFindingsFile(path: string): VerifiableFinding[];
728
+ //#endregion
729
+ //#region src/trajectory-replay/wire.d.ts
730
+ interface IncorrectStepsSubject {
731
+ readonly firstStep: number;
732
+ readonly lastStep: number;
733
+ readonly escapeStatus: 'escaped' | 'unescaped';
734
+ readonly consequenceStep: number;
735
+ }
736
+ /** Null when the subject is not an incorrect-steps finding subject. */
737
+ declare function parseIncorrectStepsSubject(subject: string): IncorrectStepsSubject | null;
738
+ interface AnalystReplayFinding {
739
+ readonly trajId: string;
740
+ readonly subject: string;
741
+ }
742
+ interface ResolvedReplayInvocation {
743
+ readonly resources: CaseResources;
744
+ readonly subject: IncorrectStepsSubject;
745
+ /** 1-based step_id arm A executes: the finding's first incorrect step. */
746
+ readonly at: number;
747
+ }
748
+ /**
749
+ * Maps a finding onto replay resources, searching the given corpora for the
750
+ * trajectory. Throws with the precise reason when the finding cannot be
751
+ * replayed (malformed subject, unknown trajectory, non-replayable case, step
752
+ * out of range) — the caller surfaces that reason instead of a proof.
753
+ */
754
+ declare function resolveFindingInvocation(finding: AnalystReplayFinding, corpora: readonly CorpusSpec[]): ResolvedReplayInvocation;
755
+ interface ReplayFindingOptions {
756
+ readonly corpora: readonly CorpusSpec[];
757
+ readonly out: string;
758
+ /** Generates the arm-B corrected command; omit to run arm A only. */
759
+ readonly fixCaller?: ChatCompletionCaller;
760
+ /** Pre-supplied arm-B command; mutually exclusive with fixCaller. */
761
+ readonly fixCommand?: string;
762
+ readonly stepTimeoutMs?: number;
763
+ readonly prefixLimit?: number;
764
+ /** Builds the exec backend for the resolved trajectory image. */
765
+ readonly backendFactory: ReplayExecBackendFactory;
766
+ readonly onProgress?: (message: string) => void;
767
+ }
768
+ interface ReplayFindingResult {
769
+ readonly invocation: ResolvedReplayInvocation;
770
+ readonly fixCommand: string | null;
771
+ readonly verdict: ReplayVerdict;
772
+ }
773
+ /**
774
+ * Finding in, executed proof out. The image comes from the corpus resources;
775
+ * the backend factory receives it as-is, so a factory backed by infrastructure
776
+ * that needs a derived image must run an `ImagePreparer` first.
777
+ */
778
+ declare function replayVerifyFinding(finding: AnalystReplayFinding, options: ReplayFindingOptions): Promise<ReplayFindingResult>;
779
+ //#endregion
780
+ export { type AnalystReplayFinding, type ArmExecutionResult, type CaseResources, type ChatCompletionCaller, type ChatOutcome, type ChatUsage, type CorpusSpec, type CwdSource, type DockerImagePreparerOptions, type EnumerationResult, type ExcludedCase, type FailedFixAttempt, type FindingReplaySource, type FindingReplayability, type FindingVerification, type FindingVerificationStatus, type FixArmExecution, type FixArmExecutor, type FixGenerationOutcome, type FixLoopAttemptRecord, type FixLoopOptions, type FixLoopResult, type FixPromptInput, type ImagePreparation, type ImagePreparer, type IncorrectStepsSubject, type IngestedTrajectory, PREFIX_DIVERGENCE_TOLERANCE_PCT, type PrefixDivergence, type PrefixDivergenceKind, type PrefixReplayResult, type RecordedTrajectoryStep, type ReplayArmVerdict, type ReplayBatchCaseRow, type ReplayBatchFixResult, type ReplayBatchOptions, type ReplayBatchReport, type ReplayExclusionReason, type ReplayExecBackend, type ReplayExecBackendFactory, type ReplayExecResult, type ReplayExecSession, type ReplayFindingOptions, type ReplayFindingResult, type ReplayVerdict, type ReplayVerifyOptions, type ReplayableCase, type ResolvedFindingReplay, type ResolvedReplayInvocation, type ResourceResolution, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, type SandboxCounterfactualRunnerOptions, type VerifiableFinding, type VerifyFindingsOptions, type VerifyFindingsRun, buildFixPrompt, buildRetryFixPrompt, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isSubmitAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, verifyFindings, wrapActionForExec };
781
+ //# sourceMappingURL=index.d.ts.map