@tangle-network/agent-eval 0.144.6 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
  14. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
  22. package/dist/campaign-Tdy3h62h.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  160. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/charter.md +112 -0
  207. package/docs/experiment.md +104 -0
  208. package/docs/prime-analyst.md +1 -0
  209. package/docs/trace-analysis.md +26 -0
  210. package/docs/trace-repair-admission.md +194 -0
  211. package/docs/trace-repair-analyst-arms.md +121 -0
  212. package/docs/trace-repair-continuation.md +107 -0
  213. package/docs/trace-repair-grader.md +163 -0
  214. package/docs/trajectory-replay.md +110 -0
  215. package/docs/verification-strategies.md +103 -0
  216. package/package.json +19 -2
  217. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  218. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  219. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  220. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  221. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  222. package/dist/campaign-CXsdyym7.js.map +0 -1
  223. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  224. package/dist/index-4XwggC10.d.ts.map +0 -1
  225. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  226. package/dist/integrity-fdt8XPAv.js.map +0 -1
  227. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  228. package/dist/replay-Krvb114g.d.ts.map +0 -1
  229. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  230. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  231. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  232. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  233. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  234. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  235. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  236. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  237. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  238. package/dist/verdict-Dps8_okt.d.ts +0 -37
  239. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,2103 @@
1
+ import { t as TraceEmitter } from "../emitter-CPBAhxum.js";
2
+ import { n as InMemoryTraceStore } from "../store-DNe_Uv1Q.js";
3
+ import { n as runCounterfactual } from "../counterfactual-CWPTrMH7.js";
4
+ import { a as parseObservationOutput, i as isSubmitAction, n as SUBMIT_ACTION_SIGNATURE, o as parseRecordedReturncode, r as deriveFailureSignature, t as wrapActionForExec } from "../exec-BLtYZdWo.js";
5
+ import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
6
+ import { join } from "node:path";
7
+ import { createHash } from "node:crypto";
8
+ import { execFile } from "node:child_process";
9
+ import { tmpdir } from "node:os";
10
+ import { promisify } from "node:util";
11
+ //#region src/trajectory-replay/corpus.ts
12
+ /**
13
+ * Corpus enumeration for batch replay verification.
14
+ *
15
+ * A labeled trajectory corpus is a gold label file (array of trajectory
16
+ * entries with `incorrect_stages[].incorrect_step_ids`) plus a prepared
17
+ * directory:
18
+ * <prepared>/normalized/<traj_id>/steps.json normalized steps
19
+ * <prepared>/normalized/<traj_id>/task.md optional task statement
20
+ * <prepared>/extracted/<traj_id>/swe_raw/** raw mini-SWE trajectory
21
+ *
22
+ * Only trajectories that record their environment are replayable: the raw
23
+ * trajectory must carry `info.docker_config.base_image`. Every excluded case
24
+ * is returned with a machine-readable reason — the batch report surfaces the
25
+ * full exclusion table, never a silently shrunk denominator.
26
+ */
27
+ /** Parses `name=<labelsPath>::<preparedDir>` (paths may contain `=`, not `::`). */
28
+ function parseCorpusFlag(value) {
29
+ const eq = value.indexOf("=");
30
+ if (eq <= 0) throw new Error(`--corpus must be name=<labels>::<prepared>, got: ${value}`);
31
+ const name = value.slice(0, eq);
32
+ const rest = value.slice(eq + 1);
33
+ const sep = rest.indexOf("::");
34
+ if (sep <= 0 || sep === rest.length - 2) throw new Error(`--corpus must be name=<labels>::<prepared>, got: ${value}`);
35
+ return {
36
+ name,
37
+ labelsPath: rest.slice(0, sep),
38
+ preparedDir: rest.slice(sep + 2)
39
+ };
40
+ }
41
+ function readLabelEntries(labelsPath) {
42
+ const parsed = JSON.parse(readFileSync(labelsPath, "utf8"));
43
+ if (!Array.isArray(parsed)) throw new Error(`${labelsPath} is not a JSON array of label entries`);
44
+ for (const entry of parsed) if (typeof entry.traj_id !== "string") throw new Error(`${labelsPath} has an entry without a string traj_id`);
45
+ return parsed;
46
+ }
47
+ function goldIncorrectSteps(entry) {
48
+ const ids = /* @__PURE__ */ new Set();
49
+ for (const stage of entry.incorrect_stages ?? []) for (const id of stage.incorrect_step_ids ?? []) ids.add(id);
50
+ return [...ids].sort((a, b) => a - b);
51
+ }
52
+ function findRawTrajFiles(sweRawDir) {
53
+ if (!existsSync(sweRawDir)) return [];
54
+ return readdirSync(sweRawDir, {
55
+ recursive: true,
56
+ encoding: "utf8"
57
+ }).filter((relative) => relative.endsWith(".traj.json")).map((relative) => join(sweRawDir, relative)).sort();
58
+ }
59
+ /** First non-empty output line of the trajectory's first bare `pwd` step. */
60
+ function cwdFromPwdObservation(steps) {
61
+ for (const step of steps) {
62
+ if (step.action.trim() !== "pwd") continue;
63
+ const line = parseObservationOutput(step.observation).split("\n").map((l) => l.trim()).find((l) => l.startsWith("/"));
64
+ if (line) return line;
65
+ }
66
+ return null;
67
+ }
68
+ function taskStatementOf(preparedDir, trajId, raw) {
69
+ const taskPath = join(preparedDir, "normalized", trajId, "task.md");
70
+ if (existsSync(taskPath)) {
71
+ const text = readFileSync(taskPath, "utf8").trim();
72
+ if (text.length > 0) return text;
73
+ }
74
+ const userMessage = raw.messages?.find((m) => m.role === "user");
75
+ if (typeof userMessage?.content === "string" && userMessage.content.trim().length > 0) return userMessage.content.trim();
76
+ return null;
77
+ }
78
+ /**
79
+ * Resolves the replay resources for one trajectory, independent of gold
80
+ * labels — the finding wire uses this with a finding-supplied step instead.
81
+ */
82
+ function resolveCaseResources(corpus, trajId) {
83
+ const sweRawDir = join(corpus.preparedDir, "extracted", trajId, "swe_raw");
84
+ const rawFiles = findRawTrajFiles(sweRawDir);
85
+ if (rawFiles.length === 0) return {
86
+ resolved: false,
87
+ reason: "no-swe-raw-trajectory",
88
+ detail: sweRawDir
89
+ };
90
+ if (rawFiles.length > 1) return {
91
+ resolved: false,
92
+ reason: "ambiguous-swe-raw-trajectory",
93
+ detail: rawFiles.join(", ")
94
+ };
95
+ let raw;
96
+ try {
97
+ raw = JSON.parse(readFileSync(rawFiles[0], "utf8"));
98
+ } catch (err) {
99
+ return {
100
+ resolved: false,
101
+ reason: "unreadable-raw-trajectory",
102
+ detail: `${rawFiles[0]}: ${err instanceof Error ? err.message : String(err)}`
103
+ };
104
+ }
105
+ const dockerConfig = raw.info?.docker_config;
106
+ const image = dockerConfig?.base_image;
107
+ if (typeof image !== "string" || image.length === 0) return {
108
+ resolved: false,
109
+ reason: "no-docker-image",
110
+ detail: rawFiles[0]
111
+ };
112
+ const stepsPath = join(corpus.preparedDir, "normalized", trajId, "steps.json");
113
+ if (!existsSync(stepsPath)) return {
114
+ resolved: false,
115
+ reason: "missing-steps-json",
116
+ detail: stepsPath
117
+ };
118
+ const steps = JSON.parse(readFileSync(stepsPath, "utf8"));
119
+ const runConfigCwd = raw.info?.config?.environment?.cwd;
120
+ const dockerCwd = dockerConfig?.cwd;
121
+ let cwd;
122
+ let cwdSource;
123
+ if (typeof runConfigCwd === "string" && runConfigCwd.length > 0) {
124
+ cwd = runConfigCwd;
125
+ cwdSource = "run-config";
126
+ } else if (typeof dockerCwd === "string" && dockerCwd.length > 0) {
127
+ cwd = dockerCwd;
128
+ cwdSource = "docker-config";
129
+ } else {
130
+ const observed = cwdFromPwdObservation(steps);
131
+ if (!observed) return {
132
+ resolved: false,
133
+ reason: "cwd-underivable"
134
+ };
135
+ cwd = observed;
136
+ cwdSource = "pwd-observation";
137
+ }
138
+ const recordedTimeout = raw.info?.config?.environment?.timeout;
139
+ return {
140
+ resolved: true,
141
+ resources: {
142
+ corpus: corpus.name,
143
+ trajId,
144
+ stepsPath,
145
+ steps,
146
+ taskStatement: taskStatementOf(corpus.preparedDir, trajId, raw),
147
+ image,
148
+ cwd,
149
+ cwdSource,
150
+ recordedStepTimeoutMs: typeof recordedTimeout === "number" && recordedTimeout > 0 ? recordedTimeout * 1e3 : null
151
+ }
152
+ };
153
+ }
154
+ /**
155
+ * Replayable = a raw trajectory with a recorded image AND at least one gold
156
+ * incorrect step that is a real mid-trajectory action. Gold steps whose action
157
+ * is the submit command are skipped when choosing k (counted per case); a case
158
+ * whose golds are ALL submit steps is excluded as gold-only-submit-step.
159
+ * Exclusion reasons are reported in resolution order:
160
+ * raw trajectory → image → gold labels → steps.json → k range → cwd.
161
+ */
162
+ function enumerateReplayableCases(corpora) {
163
+ const replayable = [];
164
+ const excluded = [];
165
+ let labelEntryCount = 0;
166
+ for (const corpus of corpora) for (const entry of readLabelEntries(corpus.labelsPath)) {
167
+ labelEntryCount += 1;
168
+ const trajId = entry.traj_id;
169
+ const resolution = resolveCaseResources(corpus, trajId);
170
+ if (!resolution.resolved) {
171
+ excluded.push({
172
+ corpus: corpus.name,
173
+ trajId,
174
+ reason: resolution.reason,
175
+ detail: resolution.detail
176
+ });
177
+ continue;
178
+ }
179
+ const gold = goldIncorrectSteps(entry);
180
+ if (gold.length === 0) {
181
+ excluded.push({
182
+ corpus: corpus.name,
183
+ trajId,
184
+ reason: "no-gold-incorrect-step"
185
+ });
186
+ continue;
187
+ }
188
+ const resources = resolution.resources;
189
+ let submitGoldsSkipped = 0;
190
+ let target = null;
191
+ let missingGoldId = null;
192
+ for (const goldId of gold) {
193
+ const step = resources.steps.find((s) => s.step_id === goldId);
194
+ if (!step) {
195
+ missingGoldId = goldId;
196
+ break;
197
+ }
198
+ if (isSubmitAction(step.action)) {
199
+ submitGoldsSkipped += 1;
200
+ continue;
201
+ }
202
+ target = step;
203
+ break;
204
+ }
205
+ if (missingGoldId !== null) {
206
+ excluded.push({
207
+ corpus: corpus.name,
208
+ trajId,
209
+ reason: "gold-step-outside-steps",
210
+ detail: `k=${missingGoldId}, steps=${resources.steps.length}`
211
+ });
212
+ continue;
213
+ }
214
+ if (!target) {
215
+ excluded.push({
216
+ corpus: corpus.name,
217
+ trajId,
218
+ reason: "gold-only-submit-step",
219
+ detail: `${submitGoldsSkipped} gold step(s), all submit commands`
220
+ });
221
+ continue;
222
+ }
223
+ replayable.push({
224
+ ...resources,
225
+ goldIncorrectSteps: gold,
226
+ k: target.step_id,
227
+ submitGoldsSkipped,
228
+ recordedReturncodeAtK: parseRecordedReturncode(target.observation)
229
+ });
230
+ }
231
+ return {
232
+ replayable,
233
+ excluded,
234
+ labelEntryCount
235
+ };
236
+ }
237
+ //#endregion
238
+ //#region src/trajectory-replay/fix.ts
239
+ /**
240
+ * Counterfactual patch synthesis: turn a recorded incorrect step into a
241
+ * corrected shell command for replay arm B.
242
+ *
243
+ * The caller is a typed outcome boundary ({ succeeded, value, error }) so a
244
+ * provider failure is a per-case report row, never a thrown batch abort and
245
+ * never a silent empty fix. Concrete chat transports live with the consumer;
246
+ * this module only builds prompts and reads replies.
247
+ */
248
+ /** Head+tail excerpt with an elision marker; identity below the limit. */
249
+ function clipText(text, limit) {
250
+ if (text.length <= limit) return text;
251
+ const half = Math.floor(limit / 2);
252
+ return `${text.slice(0, half)}\n… [${text.length - limit} chars elided] …\n${text.slice(-half)}`;
253
+ }
254
+ function renderStep(step, marker) {
255
+ const rc = parseRecordedReturncode(step.observation);
256
+ const output = clipText(parseObservationOutput(step.observation).trim(), 1600);
257
+ return [
258
+ `### step ${step.step_id}${marker}`,
259
+ "```sh",
260
+ clipText(step.action, 2e3),
261
+ "```",
262
+ `returncode: ${rc ?? "none recorded"}`,
263
+ output.length > 0 ? `output:\n\`\`\`\n${output}\n\`\`\`` : "output: (empty)"
264
+ ].join("\n");
265
+ }
266
+ /** Shared user-prompt sections: task statement, ±radius context, failing step. */
267
+ function promptBody(input) {
268
+ const radius = input.contextRadius ?? 3;
269
+ const target = input.steps.find((s) => s.step_id === input.k);
270
+ if (!target) throw new Error(`trajectory-replay: no step with step_id ${input.k}`);
271
+ const context = input.steps.filter((s) => s.step_id !== input.k && Math.abs(s.step_id - input.k) <= radius);
272
+ return [
273
+ "## Task the agent was solving",
274
+ input.taskStatement ? clipText(input.taskStatement, 4e3) : "(no task statement recorded)",
275
+ "",
276
+ "## Surrounding steps",
277
+ ...context.map((s) => renderStep(s, "")),
278
+ "",
279
+ "## Failing step to correct",
280
+ renderStep(target, " (INCORRECT — correct this one)")
281
+ ];
282
+ }
283
+ function buildFixPrompt(input) {
284
+ return {
285
+ system: [
286
+ "You repair one failed shell command from a recorded coding-agent trajectory.",
287
+ "The trajectory replays inside the original docker image; every command runs as a fresh /bin/sh subshell from a fixed working directory.",
288
+ "You are given the failing step, its recorded output, surrounding steps, and the task statement.",
289
+ "Reply with exactly ONE corrected shell command (compound commands with && or pipes are fine) inside a single ```sh fenced block.",
290
+ "The corrected command must accomplish the failing step's intent and exit 0. No prose outside the fenced block."
291
+ ].join("\n"),
292
+ user: [
293
+ ...promptBody(input),
294
+ "",
295
+ "Output the single corrected replacement for the failing step now."
296
+ ].join("\n")
297
+ };
298
+ }
299
+ function renderFailedAttempt(prior) {
300
+ if (prior.command === null) return [`### attempt ${prior.attempt}`, `model call failed before producing a command: ${prior.llmError ?? "unknown error"}`].join("\n");
301
+ const stdout = (prior.stdoutTail ?? "").trim();
302
+ const stderr = (prior.stderrTail ?? "").trim();
303
+ return [
304
+ `### attempt ${prior.attempt}`,
305
+ "```sh",
306
+ clipText(prior.command, 2e3),
307
+ "```",
308
+ `exit code: ${prior.exitCode ?? "not executed"}`,
309
+ stdout.length > 0 ? `stdout:\n\`\`\`\n${stdout}\n\`\`\`` : "stdout: (empty)",
310
+ stderr.length > 0 ? `stderr:\n\`\`\`\n${stderr}\n\`\`\`` : "stderr: (empty)"
311
+ ].join("\n");
312
+ }
313
+ /**
314
+ * Retry prompt for fix-loop attempts ≥2: the original context plus every prior
315
+ * attempt with its REAL executed output, and permission to answer with a short
316
+ * script (the block still executes as one /bin/sh unit).
317
+ */
318
+ function buildRetryFixPrompt(input, priorAttempts, maxScriptCommands = 5) {
319
+ if (priorAttempts.length === 0) throw new Error("trajectory-replay: buildRetryFixPrompt requires at least one prior attempt");
320
+ return {
321
+ system: [
322
+ "You repair one failed shell command from a recorded coding-agent trajectory.",
323
+ "The trajectory replays inside the original docker image; every command runs as a fresh /bin/sh subshell from a fixed working directory.",
324
+ "Earlier corrected commands were executed for real and failed; their actual output is included below.",
325
+ `Reply with a corrected fix inside a single \`\`\`sh fenced block: either one command, or a short script of at most ${maxScriptCommands} commands (one per line).`,
326
+ "The whole block executes as ONE /bin/sh unit from the fixed working directory and must exit 0.",
327
+ "Do not repeat a command that already failed. Keep reasoning brief. No prose outside the fenced block."
328
+ ].join("\n"),
329
+ user: [
330
+ ...promptBody(input),
331
+ "",
332
+ "## Previous fix attempts (executed for real — all failed)",
333
+ ...priorAttempts.map((prior) => renderFailedAttempt(prior)),
334
+ "",
335
+ `Output a corrected fix now — one \`\`\`sh block, at most ${maxScriptCommands} commands.`
336
+ ].join("\n")
337
+ };
338
+ }
339
+ /** Non-empty, non-comment lines of a fix script — the loop's script-size cap. */
340
+ function countScriptCommands(script) {
341
+ return script.split("\n").filter((line) => line.trim().length > 0 && !line.trim().startsWith("#")).length;
342
+ }
343
+ /** Last fenced code block, else the whole trimmed content; null when empty. */
344
+ function extractFixCommand(content) {
345
+ const blocks = [...content.matchAll(/```(?:sh|bash|shell)?\n([\s\S]*?)```/g)];
346
+ const command = (blocks.length > 0 ? blocks[blocks.length - 1][1] : content).trim();
347
+ if (command.length === 0 || command.includes("```")) return null;
348
+ return command;
349
+ }
350
+ async function generateFixCommand(caller, input) {
351
+ const { system, user } = buildFixPrompt(input);
352
+ const outcome = await caller.complete(system, user);
353
+ if (!outcome.succeeded) return outcome;
354
+ const command = extractFixCommand(outcome.value.content);
355
+ if (command === null) return {
356
+ succeeded: false,
357
+ error: `completion carried no usable command: ${clipText(outcome.value.content, 300)}`
358
+ };
359
+ return {
360
+ succeeded: true,
361
+ value: {
362
+ command,
363
+ usage: outcome.value.usage
364
+ }
365
+ };
366
+ }
367
+ //#endregion
368
+ //#region src/trajectory-replay/fix-loop.ts
369
+ /**
370
+ * Iterative counterfactual fix loop.
371
+ *
372
+ * Attempt 1 is exactly the one-shot generator (same prompt, same caller), so
373
+ * flip@1 stays comparable to one-shot fix mode. When an attempt does not flip
374
+ * the failure — nonzero exit, signature still present, or the model call
375
+ * itself failed — the next attempt's prompt carries every prior command and
376
+ * its REAL executed stdout/stderr, up to a fixed attempt budget.
377
+ *
378
+ * Isolation invariant: every attempt executes through the injected executor,
379
+ * which must provide a FRESH sandbox with the same replayed prefix. A used
380
+ * sandbox is never mutated mid-arm; a flip therefore always proves the
381
+ * corrected step against the recorded prefix state, not against debris from
382
+ * an earlier attempt.
383
+ */
384
+ function toFailedAttempt(record) {
385
+ return {
386
+ attempt: record.attempt,
387
+ command: record.command,
388
+ exitCode: record.exitCode,
389
+ stdoutTail: record.stdoutTail,
390
+ stderrTail: record.stderrTail,
391
+ llmError: record.llmError ?? record.armBError
392
+ };
393
+ }
394
+ async function runFixLoop(caller, input, executor, options) {
395
+ if (!Number.isInteger(options.maxAttempts) || options.maxAttempts < 1) throw new Error(`trajectory-replay: maxAttempts must be a positive integer, got ${options.maxAttempts}`);
396
+ const scriptCap = options.maxScriptCommands ?? 5;
397
+ const tailChars = options.outputTailChars ?? 1600;
398
+ const attempts = [];
399
+ let llmCalls = 0;
400
+ let llmFailures = 0;
401
+ let promptTokens = 0;
402
+ let completionTokens = 0;
403
+ let callsWithoutUsage = 0;
404
+ const unexecuted = (attempt, llmError, usage, command, started) => ({
405
+ attempt,
406
+ command,
407
+ llmError,
408
+ usage,
409
+ executed: false,
410
+ exitCode: null,
411
+ prefixExecuted: null,
412
+ prefixDivergences: null,
413
+ prefixDivergencePct: null,
414
+ failureVanished: null,
415
+ stdoutTail: null,
416
+ stderrTail: null,
417
+ armBError: null,
418
+ wallMs: Date.now() - started
419
+ });
420
+ for (let attempt = 1; attempt <= options.maxAttempts; attempt++) {
421
+ const started = Date.now();
422
+ const prompt = attempt === 1 ? buildFixPrompt(input) : buildRetryFixPrompt(input, attempts.map(toFailedAttempt), scriptCap);
423
+ llmCalls += 1;
424
+ const outcome = await caller.complete(prompt.system, prompt.user);
425
+ if (!outcome.succeeded) {
426
+ llmFailures += 1;
427
+ attempts.push(unexecuted(attempt, outcome.error, null, null, started));
428
+ options.onProgress?.(`fix-loop attempt ${attempt}: LLM failed — ${outcome.error.slice(0, 160)}`);
429
+ continue;
430
+ }
431
+ const usage = outcome.value.usage;
432
+ if (usage === null || usage === void 0) callsWithoutUsage += 1;
433
+ promptTokens += usage?.promptTokens ?? 0;
434
+ completionTokens += usage?.completionTokens ?? 0;
435
+ const command = extractFixCommand(outcome.value.content);
436
+ if (command === null) {
437
+ llmFailures += 1;
438
+ attempts.push(unexecuted(attempt, `completion carried no usable command: ${clipText(outcome.value.content, 300)}`, usage, null, started));
439
+ continue;
440
+ }
441
+ if (attempt > 1) {
442
+ const commandLines = countScriptCommands(command);
443
+ if (commandLines > scriptCap) {
444
+ llmFailures += 1;
445
+ attempts.push(unexecuted(attempt, `script exceeds ${scriptCap} command lines (${commandLines})`, usage, null, started));
446
+ options.onProgress?.(`fix-loop attempt ${attempt}: rejected script with ${commandLines} command lines`);
447
+ continue;
448
+ }
449
+ }
450
+ let execution;
451
+ try {
452
+ execution = await executor(command, attempt);
453
+ } catch (err) {
454
+ const message = err instanceof Error ? err.message : String(err);
455
+ attempts.push({
456
+ ...unexecuted(attempt, "", usage, command, started),
457
+ llmError: null,
458
+ armBError: message.slice(0, 500)
459
+ });
460
+ options.onProgress?.(`fix-loop attempt ${attempt}: sandbox error — ${message.slice(0, 160)}`);
461
+ return {
462
+ flipped: false,
463
+ flippedAtAttempt: null,
464
+ attempts,
465
+ aborted: true,
466
+ llmCalls,
467
+ llmFailures,
468
+ promptTokens,
469
+ completionTokens,
470
+ callsWithoutUsage
471
+ };
472
+ }
473
+ attempts.push({
474
+ attempt,
475
+ command,
476
+ llmError: null,
477
+ usage,
478
+ executed: true,
479
+ exitCode: execution.exitCode,
480
+ prefixExecuted: execution.prefixExecuted,
481
+ prefixDivergences: execution.prefixDivergences,
482
+ prefixDivergencePct: execution.prefixDivergencePct,
483
+ failureVanished: execution.failureVanished,
484
+ stdoutTail: clipText(execution.stdout, tailChars),
485
+ stderrTail: clipText(execution.stderr, tailChars),
486
+ armBError: null,
487
+ wallMs: Date.now() - started
488
+ });
489
+ options.onProgress?.(`fix-loop attempt ${attempt}: exit=${execution.exitCode} failureVanished=${execution.failureVanished}`);
490
+ if (execution.failureVanished) return {
491
+ flipped: true,
492
+ flippedAtAttempt: attempt,
493
+ attempts,
494
+ aborted: false,
495
+ llmCalls,
496
+ llmFailures,
497
+ promptTokens,
498
+ completionTokens,
499
+ callsWithoutUsage
500
+ };
501
+ }
502
+ return {
503
+ flipped: false,
504
+ flippedAtAttempt: null,
505
+ attempts,
506
+ aborted: false,
507
+ llmCalls,
508
+ llmFailures,
509
+ promptTokens,
510
+ completionTokens,
511
+ callsWithoutUsage
512
+ };
513
+ }
514
+ //#endregion
515
+ //#region src/trajectory-replay/image-preparer.ts
516
+ /**
517
+ * Replay-ready image derivation.
518
+ *
519
+ * A recorded trajectory names the image it ran in, but the image is not always
520
+ * runnable as-is: sandbox platforms pin customer commands to a non-root
521
+ * identity, so a root-owned working tree must be chowned before the replay can
522
+ * write to it. `ImagePreparer` is that step, injectable so a consumer whose
523
+ * images are already replay-ready supplies its own no-op or none at all.
524
+ */
525
+ const execFileAsync = promisify(execFile);
526
+ function derivedImageTag(image, cwd) {
527
+ return `ctb-replay:${createHash("sha256").update(`${image}\n${cwd}`).digest("hex").slice(0, 12)}-uid1000`;
528
+ }
529
+ /**
530
+ * Pulls the base image when absent and builds `FROM <base>; RUN chown -R
531
+ * 1000:1000 <cwd>` tagged by content hash, so repeated batches reuse both
532
+ * the pull and the build. cwd `/` skips the chown (never chown -R /) and
533
+ * replays on the base image directly.
534
+ */
535
+ function dockerImagePreparer(options = {}) {
536
+ const pullTimeoutMs = options.pullTimeoutMs ?? 12e5;
537
+ const buildTimeoutMs = options.buildTimeoutMs ?? 9e5;
538
+ const imageExists = async (tag) => {
539
+ try {
540
+ await execFileAsync("docker", [
541
+ "image",
542
+ "inspect",
543
+ tag
544
+ ], { maxBuffer: 8 * 1024 * 1024 });
545
+ return true;
546
+ } catch {
547
+ return false;
548
+ }
549
+ };
550
+ const errorTail = (err) => {
551
+ return (err && typeof err === "object" && "stderr" in err && typeof err.stderr === "string" && err.stderr.length > 0 ? err.stderr : err instanceof Error ? err.message : String(err)).trim().split("\n").slice(-3).join(" | ").slice(0, 400);
552
+ };
553
+ return { async ensure(image, cwd) {
554
+ if (cwd === "/") {
555
+ if (await imageExists(image)) return {
556
+ succeeded: true,
557
+ value: {
558
+ derivedImage: image,
559
+ pulled: false,
560
+ built: false
561
+ }
562
+ };
563
+ try {
564
+ await execFileAsync("docker", ["pull", image], {
565
+ timeout: pullTimeoutMs,
566
+ maxBuffer: 32 * 1024 * 1024
567
+ });
568
+ } catch (err) {
569
+ return {
570
+ succeeded: false,
571
+ error: `pull ${image}: ${errorTail(err)}`
572
+ };
573
+ }
574
+ return {
575
+ succeeded: true,
576
+ value: {
577
+ derivedImage: image,
578
+ pulled: true,
579
+ built: false
580
+ }
581
+ };
582
+ }
583
+ const derived = derivedImageTag(image, cwd);
584
+ if (await imageExists(derived)) return {
585
+ succeeded: true,
586
+ value: {
587
+ derivedImage: derived,
588
+ pulled: false,
589
+ built: false
590
+ }
591
+ };
592
+ let pulled = false;
593
+ if (!await imageExists(image)) try {
594
+ await execFileAsync("docker", ["pull", image], {
595
+ timeout: pullTimeoutMs,
596
+ maxBuffer: 32 * 1024 * 1024
597
+ });
598
+ pulled = true;
599
+ } catch (err) {
600
+ return {
601
+ succeeded: false,
602
+ error: `pull ${image}: ${errorTail(err)}`
603
+ };
604
+ }
605
+ const contextDir = mkdtempSync(join(tmpdir(), "ctb-replay-image-"));
606
+ const quotedCwd = `'${cwd.replaceAll("'", `'\\''`)}'`;
607
+ writeFileSync(join(contextDir, "Dockerfile"), `FROM ${image}\nRUN chown -R 1000:1000 ${quotedCwd}\n`);
608
+ try {
609
+ await execFileAsync("docker", [
610
+ "build",
611
+ "-t",
612
+ derived,
613
+ contextDir
614
+ ], {
615
+ timeout: buildTimeoutMs,
616
+ maxBuffer: 32 * 1024 * 1024
617
+ });
618
+ } catch (err) {
619
+ return {
620
+ succeeded: false,
621
+ error: `build ${derived} from ${image}: ${errorTail(err)}`
622
+ };
623
+ }
624
+ return {
625
+ succeeded: true,
626
+ value: {
627
+ derivedImage: derived,
628
+ pulled,
629
+ built: true
630
+ }
631
+ };
632
+ } };
633
+ }
634
+ //#endregion
635
+ //#region src/trajectory-replay/verify.ts
636
+ /**
637
+ * Execution-replay verification for recorded shell trajectories.
638
+ *
639
+ * Turns a cited claim ("step k is the error-critical step") into an executed
640
+ * proof: replay steps 1..k-1 inside the trajectory's own image, then
641
+ * arm A — run the recorded step k and check the recorded failure signature
642
+ * reproduces (returncode + a stable output substring), and
643
+ * arm B — run a corrected step k and check the failure vanishes.
644
+ *
645
+ * Built on the counterfactual scaffold: the trajectory is ingested into a
646
+ * TraceStore, each arm is a `runCounterfactual` meta-run, and the
647
+ * `SandboxCounterfactualRunner` here supplies the `executeFrom` callback that
648
+ * scaffold deliberately leaves to consumers.
649
+ *
650
+ * Honest limits:
651
+ * - Only trajectories that record their image can be replayed; tasks whose
652
+ * environment needs external compose peers cannot be replayed this way.
653
+ * - Each arm replays the prefix in its own fresh session, so a backend
654
+ * without copy-on-write forks pays the prefix twice.
655
+ * - Prefix divergence is recorded per step and surfaced in the verdict, never
656
+ * hidden: a high divergence rate is a finding about replayability, not an
657
+ * error of the harness.
658
+ *
659
+ * Agreement requires positive evidence. A prefix step counts as confirmed only
660
+ * when the recording carries a returncode AND the replayed exit equals it. A
661
+ * step the recording cannot adjudicate is a divergence of its own kind
662
+ * (`unknown-expectation`), never silent agreement — otherwise a replay that
663
+ * fails on every step reports a perfect prefix and every downstream verdict
664
+ * built on it is meaningless.
665
+ */
666
+ /**
667
+ * Emit one tool span per trajectory step, in order, with a monotonic
668
+ * injected clock so `buildTrajectory` ordering is deterministic even when
669
+ * two spans would share a Date.now() millisecond.
670
+ */
671
+ async function ingestRecordedTrajectory(store, steps, caseId) {
672
+ let tick = 0;
673
+ const emitter = new TraceEmitter(store, { now: () => ++tick });
674
+ await emitter.startRun({
675
+ scenarioId: caseId,
676
+ tags: {
677
+ source: "trajectory-replay",
678
+ caseId
679
+ }
680
+ });
681
+ for (const step of steps) {
682
+ if (typeof step.action !== "string" || step.action.length === 0) throw new Error(`trajectory-replay: step ${step.step_id} has no action`);
683
+ await (await emitter.tool({
684
+ name: `step:${step.step_id}`,
685
+ toolName: "shell",
686
+ args: { command: step.action },
687
+ result: {
688
+ returncode: parseRecordedReturncode(step.observation),
689
+ observation: step.observation
690
+ }
691
+ })).end();
692
+ }
693
+ await emitter.endRun({
694
+ pass: true,
695
+ notes: "recorded trajectory (ingested for replay)"
696
+ });
697
+ return {
698
+ runId: emitter.runId,
699
+ stepCount: steps.length
700
+ };
701
+ }
702
+ /** Admission tolerance: a prefix replay is faithful enough to build a verdict
703
+ * on when at most this percentage of its executed steps diverged. */
704
+ const PREFIX_DIVERGENCE_TOLERANCE_PCT = 10;
705
+ /**
706
+ * Compare one replayed prefix step against its recording. Returns null only
707
+ * when the recording positively confirms the replay.
708
+ */
709
+ function classifyPrefixStep(step, expectedReturncode, actualExit) {
710
+ if (expectedReturncode === null) return {
711
+ step,
712
+ kind: "unknown-expectation",
713
+ expectedReturncode: null,
714
+ actualExit
715
+ };
716
+ if (expectedReturncode !== actualExit) return {
717
+ step,
718
+ kind: "returncode-mismatch",
719
+ expectedReturncode,
720
+ actualExit
721
+ };
722
+ return null;
723
+ }
724
+ /** Roll per-step classifications up into the rate the admission pre-pass gates on. */
725
+ function summarizePrefixReplay(prefixExecuted, prefixDivergences, wallMs) {
726
+ if (prefixDivergences.length > prefixExecuted) throw new Error(`trajectory-replay: ${prefixDivergences.length} divergences over ${prefixExecuted} executed prefix steps`);
727
+ const pct = prefixExecuted === 0 ? 0 : Number((prefixDivergences.length / prefixExecuted * 100).toFixed(1));
728
+ return {
729
+ prefixExecuted,
730
+ prefixDivergences,
731
+ prefixConfirmed: prefixExecuted - prefixDivergences.length,
732
+ prefixReturncodeMismatches: prefixDivergences.filter((d) => d.kind === "returncode-mismatch").length,
733
+ prefixUnknownExpectations: prefixDivergences.filter((d) => d.kind === "unknown-expectation").length,
734
+ prefixDivergencePct: pct,
735
+ prefixWithinTolerance: pct <= 10,
736
+ wallMs
737
+ };
738
+ }
739
+ function toolCommand(span) {
740
+ const args = span.args;
741
+ if (!args || typeof args.command !== "string") throw new Error(`trajectory-replay: span ${span.name} carries no command`);
742
+ return args.command;
743
+ }
744
+ function recordedReturncodeOf(span) {
745
+ const result = span.result;
746
+ return typeof result?.returncode === "number" ? result.returncode : null;
747
+ }
748
+ /**
749
+ * Replays `ctx.prefix` in a fresh session, then executes the mutated step.
750
+ * Divergences are recorded and never abort the replay. Results land on
751
+ * `lastPrefix` / `lastArm` for the caller; spans for every exec land in the
752
+ * counterfactual meta-run.
753
+ */
754
+ var SandboxCounterfactualRunner = class {
755
+ backend;
756
+ options;
757
+ lastPrefix = null;
758
+ lastArm = null;
759
+ constructor(backend, options) {
760
+ this.backend = backend;
761
+ this.options = options;
762
+ }
763
+ async executeFrom(ctx, emitter) {
764
+ const { cwd, stepTimeoutMs, prefixLimit, onProgress } = this.options;
765
+ const session = await this.backend.open();
766
+ try {
767
+ const toolSteps = ctx.prefix.filter((s) => s.span.kind === "tool");
768
+ const toExecute = prefixLimit !== void 0 ? toolSteps.slice(0, prefixLimit) : toolSteps;
769
+ const divergences = [];
770
+ const prefixStart = Date.now();
771
+ for (const step of toExecute) {
772
+ const command = toolCommand(step.span);
773
+ const expected = recordedReturncodeOf(step.span);
774
+ const started = Date.now();
775
+ const result = await session.exec(wrapActionForExec(command, cwd), stepTimeoutMs);
776
+ await (await emitter.sandbox({
777
+ name: step.span.name,
778
+ command,
779
+ exitCode: result.exitCode,
780
+ wallMs: Date.now() - started
781
+ })).end();
782
+ const divergence = classifyPrefixStep(step.index + 1, expected, result.exitCode);
783
+ if (divergence) divergences.push(divergence);
784
+ onProgress?.(`prefix ${step.span.name}: exit=${result.exitCode}` + (divergence ? ` (${divergence.kind}${expected === null ? "" : `, recorded ${expected}`})` : ""));
785
+ }
786
+ this.lastPrefix = summarizePrefixReplay(toExecute.length, divergences, Date.now() - prefixStart);
787
+ if (ctx.mutatedStep.span.kind !== "tool") throw new Error("trajectory-replay: mutation target is not a tool span");
788
+ const armCommand = toolCommand(ctx.mutatedStep.span);
789
+ const armStart = Date.now();
790
+ const armResult = await session.exec(wrapActionForExec(armCommand, cwd), stepTimeoutMs);
791
+ const armWallMs = Date.now() - armStart;
792
+ await (await emitter.sandbox({
793
+ name: `candidate:${ctx.mutatedStep.span.name}`,
794
+ command: armCommand,
795
+ exitCode: armResult.exitCode,
796
+ wallMs: armWallMs
797
+ })).end();
798
+ this.lastArm = {
799
+ command: armCommand,
800
+ exitCode: armResult.exitCode,
801
+ stdout: armResult.stdout,
802
+ stderr: armResult.stderr,
803
+ wallMs: armWallMs
804
+ };
805
+ onProgress?.(`candidate step: exit=${armResult.exitCode} in ${armWallMs}ms`);
806
+ await emitter.endRun({
807
+ pass: armResult.exitCode === 0,
808
+ notes: `prefix ${toExecute.length} steps, ${divergences.length} divergences (${this.lastPrefix.prefixDivergencePct}%)`
809
+ });
810
+ } finally {
811
+ await session.close();
812
+ }
813
+ }
814
+ };
815
+ function excerpt(text, limit = 2400) {
816
+ if (text.length <= limit) return text;
817
+ const head = text.slice(0, limit / 2);
818
+ const tail = text.slice(-limit / 2);
819
+ return `${head}\n… [${text.length - limit} chars elided] …\n${tail}`;
820
+ }
821
+ async function replayVerify(options) {
822
+ const totalStart = Date.now();
823
+ const steps = JSON.parse(readFileSync(options.stepsPath, "utf8"));
824
+ if (!Array.isArray(steps) || steps.length === 0) throw new Error(`trajectory-replay: ${options.stepsPath} is not a non-empty steps array`);
825
+ const index = options.at - 1;
826
+ if (index < 0 || index >= steps.length) throw new Error(`trajectory-replay: at ${options.at} out of range [1, ${steps.length}]`);
827
+ const target = steps[index];
828
+ if (target.step_id !== options.at) throw new Error(`trajectory-replay: steps[${index}].step_id=${target.step_id} != at ${options.at}; steps.json must be ordered with 1-based contiguous step_ids`);
829
+ const caseId = options.caseId ?? options.stepsPath;
830
+ const recordedReturncode = parseRecordedReturncode(target.observation);
831
+ const signature = options.signature ?? deriveFailureSignature(target.observation);
832
+ const signatureBasis = signature ? "returncode+output-substring" : "returncode-only";
833
+ const runnerOptions = {
834
+ cwd: options.cwd,
835
+ stepTimeoutMs: options.stepTimeoutMs ?? 3e5,
836
+ prefixLimit: options.prefixLimit,
837
+ onProgress: options.onProgress
838
+ };
839
+ const store = new InMemoryTraceStore();
840
+ const { runId } = await ingestRecordedTrajectory(store, steps, caseId);
841
+ options.onProgress?.(`arm A: replaying ${index} prefix steps then recorded step ${options.at}`);
842
+ const armARunner = new SandboxCounterfactualRunner(options.backend, runnerOptions);
843
+ const armAStart = Date.now();
844
+ const armAResult = await runCounterfactual(store, runId, {
845
+ kind: "custom",
846
+ at: index,
847
+ describe: "arm-A identity replay",
848
+ apply: (s) => s
849
+ }, armARunner);
850
+ const armAMs = Date.now() - armAStart;
851
+ const armAExec = requireExec(armARunner, "arm A");
852
+ const armAOutput = `${armAExec.stdout}\n${armAExec.stderr}`;
853
+ const failureSignatureMatch = recordedReturncode !== null && armAExec.exitCode === recordedReturncode && (signature ? armAOutput.includes(signature) : true);
854
+ let armBRunner = null;
855
+ let armBExec = null;
856
+ let armBRunId = null;
857
+ let armBMs = null;
858
+ if (options.fixCommand) {
859
+ const fixCommand = options.fixCommand;
860
+ options.onProgress?.("arm B: replaying prefix then corrected step");
861
+ armBRunner = new SandboxCounterfactualRunner(options.backend, runnerOptions);
862
+ const armBStart = Date.now();
863
+ const armBResult = await runCounterfactual(store, runId, {
864
+ kind: "custom",
865
+ at: index,
866
+ describe: "arm-B corrected step",
867
+ apply: (step) => ({
868
+ ...step,
869
+ span: {
870
+ ...step.span,
871
+ args: { command: fixCommand }
872
+ }
873
+ })
874
+ }, armBRunner);
875
+ armBMs = Date.now() - armBStart;
876
+ armBRunId = armBResult.counterfactualRunId;
877
+ armBExec = requireExec(armBRunner, "arm B");
878
+ }
879
+ const armBOutput = armBExec ? `${armBExec.stdout}\n${armBExec.stderr}` : null;
880
+ const armAPrefix = requirePrefix(armARunner, "arm A");
881
+ const verdict = {
882
+ case: caseId,
883
+ image: options.image,
884
+ driver: options.driverLabel ?? "docker",
885
+ k: options.at,
886
+ cwd: options.cwd,
887
+ recordedReturncode,
888
+ signature,
889
+ signatureBasis,
890
+ prefixExecuted: armAPrefix.prefixExecuted,
891
+ prefixDivergences: armAPrefix.prefixDivergences,
892
+ prefixDivergencePct: armAPrefix.prefixDivergencePct,
893
+ prefixConfirmed: armAPrefix.prefixConfirmed,
894
+ prefixReturncodeMismatches: armAPrefix.prefixReturncodeMismatches,
895
+ prefixUnknownExpectations: armAPrefix.prefixUnknownExpectations,
896
+ prefixWithinTolerance: armAPrefix.prefixWithinTolerance,
897
+ armA: {
898
+ command: armAExec.command,
899
+ exitCode: armAExec.exitCode,
900
+ wallMs: armAExec.wallMs,
901
+ prefix: armAPrefix,
902
+ failureSignatureMatch
903
+ },
904
+ armB: armBExec && armBRunner ? {
905
+ command: armBExec.command,
906
+ exitCode: armBExec.exitCode,
907
+ wallMs: armBExec.wallMs,
908
+ prefix: requirePrefix(armBRunner, "arm B"),
909
+ failureVanished: armBExec.exitCode === 0 && (signature && armBOutput ? !armBOutput.includes(signature) : true)
910
+ } : null,
911
+ timings: {
912
+ armAMs,
913
+ armBMs,
914
+ totalMs: Date.now() - totalStart
915
+ },
916
+ runIds: {
917
+ original: runId,
918
+ armA: armAResult.counterfactualRunId,
919
+ armB: armBRunId
920
+ }
921
+ };
922
+ mkdirSync(options.out, { recursive: true });
923
+ writeFileSync(join(options.out, "replay-verdict.json"), `${JSON.stringify(verdict, null, 2)}\n`);
924
+ writeFileSync(join(options.out, "report.md"), renderReport(verdict, target, armAExec, armBExec));
925
+ return verdict;
926
+ }
927
+ function requireExec(runner, arm) {
928
+ if (!runner.lastArm) throw new Error(`trajectory-replay: ${arm} finished without executing step k`);
929
+ return runner.lastArm;
930
+ }
931
+ function requirePrefix(runner, arm) {
932
+ if (!runner.lastPrefix) throw new Error(`trajectory-replay: ${arm} reported no prefix replay result`);
933
+ return runner.lastPrefix;
934
+ }
935
+ function renderReport(verdict, target, armA, armB) {
936
+ const lines = [];
937
+ lines.push(`# Replay verification — ${verdict.case}`);
938
+ lines.push("");
939
+ lines.push(`- image: \`${verdict.image}\` (driver: ${verdict.driver})`);
940
+ lines.push(`- error-critical step k = ${verdict.k}, cwd \`${verdict.cwd}\``);
941
+ lines.push(`- recorded returncode at k: ${verdict.recordedReturncode ?? "none recorded"}`);
942
+ lines.push(`- failure signature (${verdict.signatureBasis}): ${verdict.signature ? `\`${verdict.signature}\`` : "—"}`);
943
+ lines.push("");
944
+ lines.push(`## Prefix replay (steps 1..${verdict.k - 1})`);
945
+ lines.push("");
946
+ lines.push(`Executed ${verdict.prefixExecuted} steps; ${verdict.prefixConfirmed} confirmed against the recording, ${verdict.prefixDivergences.length} divergent (${verdict.prefixDivergencePct}%) — ${verdict.prefixReturncodeMismatches} returncode mismatches, ${verdict.prefixUnknownExpectations} with no recorded returncode to check.`);
947
+ lines.push("");
948
+ lines.push(`Within the 10% admission tolerance: **${verdict.prefixWithinTolerance}**.`);
949
+ if (verdict.prefixDivergences.length > 0) {
950
+ lines.push("");
951
+ lines.push("| step | kind | recorded rc | replayed exit |");
952
+ lines.push("| --- | --- | --- | --- |");
953
+ for (const d of verdict.prefixDivergences) lines.push(`| ${d.step} | ${d.kind} | ${d.expectedReturncode ?? "none recorded"} | ${d.actualExit} |`);
954
+ }
955
+ lines.push("");
956
+ lines.push("## Arm A — recorded step k, replayed");
957
+ lines.push("");
958
+ lines.push("```sh");
959
+ lines.push(armA.command);
960
+ lines.push("```");
961
+ lines.push("");
962
+ lines.push(`Exit ${armA.exitCode} in ${armA.wallMs}ms — failureSignatureMatch: **${verdict.armA.failureSignatureMatch}**`);
963
+ lines.push("");
964
+ lines.push("Recorded observation excerpt:");
965
+ lines.push("");
966
+ lines.push("```");
967
+ lines.push(excerpt(parseObservationOutput(target.observation)));
968
+ lines.push("```");
969
+ lines.push("");
970
+ lines.push("Replayed stdout+stderr excerpt:");
971
+ lines.push("");
972
+ lines.push("```");
973
+ lines.push(excerpt(`${armA.stdout}\n${armA.stderr}`.trim()));
974
+ lines.push("```");
975
+ if (armB && verdict.armB) {
976
+ lines.push("");
977
+ lines.push("## Arm B — corrected step k");
978
+ lines.push("");
979
+ lines.push("```sh");
980
+ lines.push(armB.command);
981
+ lines.push("```");
982
+ lines.push("");
983
+ lines.push(`Exit ${armB.exitCode} in ${armB.wallMs}ms — failureVanished: **${verdict.armB.failureVanished}** (prefix re-replayed in a fresh session: ${verdict.armB.prefix.prefixExecuted} steps, ${verdict.armB.prefix.prefixDivergences.length} divergences, ${verdict.armB.prefix.prefixDivergencePct}%)`);
984
+ lines.push("");
985
+ lines.push("Replayed stdout+stderr excerpt:");
986
+ lines.push("");
987
+ lines.push("```");
988
+ lines.push(excerpt(`${armB.stdout}\n${armB.stderr}`.trim()));
989
+ lines.push("```");
990
+ }
991
+ lines.push("");
992
+ lines.push("## Timings");
993
+ lines.push("");
994
+ lines.push(`arm A ${verdict.timings.armAMs}ms, arm B ${verdict.timings.armBMs ?? "—"}ms, total ${verdict.timings.totalMs}ms.`);
995
+ lines.push("");
996
+ return lines.join("\n");
997
+ }
998
+ //#endregion
999
+ //#region src/trajectory-replay/batch.ts
1000
+ /**
1001
+ * Batch replay verification across gold-labeled trajectory corpora.
1002
+ *
1003
+ * For every replayable case (a raw trajectory with a recorded image AND at
1004
+ * least one gold incorrect step) the batch:
1005
+ * 1. derives the replay-ready image through the injected `ImagePreparer`,
1006
+ * 2. replays the prefix and runs arm A — the recorded gold step k, and
1007
+ * 3. optionally generates a corrected command with one LLM call and runs
1008
+ * arm B in its own fresh session.
1009
+ *
1010
+ * Headline metrics:
1011
+ * replayability rate — fraction of replayable cases where the prefix
1012
+ * replays within the divergence tolerance AND arm A reproduces the
1013
+ * recorded returncode at k;
1014
+ * prefix fidelity — executed prefix steps and the share of them that did
1015
+ * not confirm the recording, split by kind. A corpus whose recordings
1016
+ * carry no returncodes shows up here as unknown-expectation steps, never
1017
+ * as a clean replay;
1018
+ * fix-flip rate — fraction of arm-B-executed cases where the failure
1019
+ * vanished (exit 0, signature absent).
1020
+ *
1021
+ * Image pulls and execs run strictly serially: pulls contend on disk and
1022
+ * registry bandwidth, and serial cases keep wall-time attribution per case
1023
+ * honest. Pull failures are report rows, never silent skips.
1024
+ */
1025
+ /** Deterministic PRNG for the fix-case sample; the seed lands in the report. */
1026
+ function mulberry32(seed) {
1027
+ let state = seed >>> 0;
1028
+ return () => {
1029
+ state = state + 1831565813 >>> 0;
1030
+ let t = state;
1031
+ t = Math.imul(t ^ t >>> 15, t | 1);
1032
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
1033
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
1034
+ };
1035
+ }
1036
+ function seededSample(items, size, seed) {
1037
+ const pool = [...items];
1038
+ const random = mulberry32(seed);
1039
+ for (let i = pool.length - 1; i > 0; i--) {
1040
+ const j = Math.floor(random() * (i + 1));
1041
+ [pool[i], pool[j]] = [pool[j], pool[i]];
1042
+ }
1043
+ return new Set(pool.slice(0, size));
1044
+ }
1045
+ function caseOutDirName(row) {
1046
+ return `${row.corpus}--${row.trajId}`.replaceAll(/[^A-Za-z0-9._-]/g, "_").slice(0, 180);
1047
+ }
1048
+ /**
1049
+ * Arm B standalone: fresh sandbox, prefix replay, corrected step k. Reuses
1050
+ * the same counterfactual scaffold as replayVerify without re-running arm A.
1051
+ */
1052
+ async function executeArmB(replayCase, fixCommand, backend, signature, stepTimeoutMs, prefixLimit, onProgress) {
1053
+ const store = new InMemoryTraceStore();
1054
+ const { runId } = await ingestRecordedTrajectory(store, replayCase.steps, replayCase.trajId);
1055
+ const runner = new SandboxCounterfactualRunner(backend, {
1056
+ cwd: replayCase.cwd,
1057
+ stepTimeoutMs,
1058
+ prefixLimit,
1059
+ onProgress
1060
+ });
1061
+ await runCounterfactual(store, runId, {
1062
+ kind: "custom",
1063
+ at: replayCase.k - 1,
1064
+ describe: "arm-B corrected step",
1065
+ apply: (step) => ({
1066
+ ...step,
1067
+ span: {
1068
+ ...step.span,
1069
+ args: { command: fixCommand }
1070
+ }
1071
+ })
1072
+ }, runner);
1073
+ const exec = runner.lastArm;
1074
+ if (!exec) throw new Error("arm B finished without executing the corrected step");
1075
+ const prefix = runner.lastPrefix;
1076
+ if (!prefix) throw new Error("arm B reported no prefix replay result");
1077
+ const output = `${exec.stdout}\n${exec.stderr}`;
1078
+ return {
1079
+ exitCode: exec.exitCode,
1080
+ prefixExecuted: prefix.prefixExecuted,
1081
+ prefixDivergences: prefix.prefixDivergences.length,
1082
+ prefixDivergencePct: prefix.prefixDivergencePct,
1083
+ failureVanished: exec.exitCode === 0 && (signature ? !output.includes(signature) : true),
1084
+ stdout: exec.stdout,
1085
+ stderr: exec.stderr
1086
+ };
1087
+ }
1088
+ function rate(numerator, denominator) {
1089
+ return denominator > 0 ? numerator / denominator : null;
1090
+ }
1091
+ async function runReplayBatch(options) {
1092
+ const onProgress = options.onProgress ?? (() => {});
1093
+ const enumeration = enumerateReplayableCases(options.corpora);
1094
+ let selected = enumeration.replayable;
1095
+ if (options.caseFilter) selected = selected.filter((c) => c.trajId.includes(options.caseFilter));
1096
+ if (options.caseLimit !== void 0) selected = selected.slice(0, options.caseLimit);
1097
+ onProgress(`enumerated ${enumeration.labelEntryCount} label entries → ${enumeration.replayable.length} replayable, ${enumeration.excluded.length} excluded; executing ${selected.length}`);
1098
+ const preparer = options.preparer ?? dockerImagePreparer();
1099
+ const backendFactory = options.backendFactory;
1100
+ mkdirSync(options.out, { recursive: true });
1101
+ const progressPath = join(options.out, "cases.jsonl");
1102
+ const rows = [];
1103
+ const pullFailures = [];
1104
+ const verdictByTraj = /* @__PURE__ */ new Map();
1105
+ for (const [index, replayCase] of selected.entries()) {
1106
+ const caseStart = Date.now();
1107
+ const label = `[${index + 1}/${selected.length}] ${replayCase.corpus}/${replayCase.trajId}`;
1108
+ onProgress(`${label}: preparing image ${replayCase.image}`);
1109
+ const preparation = await preparer.ensure(replayCase.image, replayCase.cwd);
1110
+ const base = {
1111
+ corpus: replayCase.corpus,
1112
+ trajId: replayCase.trajId,
1113
+ image: replayCase.image,
1114
+ cwd: replayCase.cwd,
1115
+ cwdSource: replayCase.cwdSource,
1116
+ k: replayCase.k,
1117
+ stepCount: replayCase.steps.length,
1118
+ goldIncorrectSteps: replayCase.goldIncorrectSteps,
1119
+ submitGoldsSkipped: replayCase.submitGoldsSkipped,
1120
+ recordedReturncodeAtK: replayCase.recordedReturncodeAtK
1121
+ };
1122
+ if (!preparation.succeeded) {
1123
+ pullFailures.push({
1124
+ corpus: replayCase.corpus,
1125
+ trajId: replayCase.trajId,
1126
+ image: replayCase.image,
1127
+ error: preparation.error
1128
+ });
1129
+ const row = {
1130
+ ...base,
1131
+ derivedImage: null,
1132
+ signature: null,
1133
+ status: "image-unavailable",
1134
+ error: preparation.error,
1135
+ imagePulled: false,
1136
+ imageBuilt: false,
1137
+ prefixExecuted: null,
1138
+ prefixDivergences: null,
1139
+ prefixDivergencePct: null,
1140
+ prefixConfirmed: null,
1141
+ prefixReturncodeMismatches: null,
1142
+ prefixUnknownExpectations: null,
1143
+ armAExit: null,
1144
+ armAReturncodeMatch: false,
1145
+ armASignatureMatch: false,
1146
+ replayed: false,
1147
+ fix: null,
1148
+ wallMs: Date.now() - caseStart
1149
+ };
1150
+ rows.push(row);
1151
+ appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
1152
+ onProgress(`${label}: image unavailable — ${preparation.error}`);
1153
+ continue;
1154
+ }
1155
+ const { derivedImage, pulled, built } = preparation.value;
1156
+ const stepTimeoutMs = options.stepTimeoutMs ?? replayCase.recordedStepTimeoutMs ?? 12e4;
1157
+ const caseOut = join(options.out, caseOutDirName(replayCase));
1158
+ onProgress(`${label}: arm A on ${derivedImage} (k=${replayCase.k}, timeout ${stepTimeoutMs}ms)`);
1159
+ try {
1160
+ const verdict = await replayVerify({
1161
+ stepsPath: replayCase.stepsPath,
1162
+ image: derivedImage,
1163
+ at: replayCase.k,
1164
+ cwd: replayCase.cwd,
1165
+ out: caseOut,
1166
+ caseId: replayCase.trajId,
1167
+ stepTimeoutMs,
1168
+ prefixLimit: options.prefixLimit,
1169
+ backend: backendFactory(derivedImage),
1170
+ onProgress: (message) => onProgress(`${label}: ${message}`)
1171
+ });
1172
+ verdictByTraj.set(replayCase.trajId, verdict);
1173
+ const returncodeMatch = verdict.recordedReturncode !== null && verdict.armA.exitCode === verdict.recordedReturncode;
1174
+ const row = {
1175
+ ...base,
1176
+ derivedImage,
1177
+ signature: verdict.signature,
1178
+ status: "ok",
1179
+ error: null,
1180
+ imagePulled: pulled,
1181
+ imageBuilt: built,
1182
+ prefixExecuted: verdict.prefixExecuted,
1183
+ prefixDivergences: verdict.prefixDivergences.length,
1184
+ prefixDivergencePct: verdict.prefixDivergencePct,
1185
+ prefixConfirmed: verdict.prefixConfirmed,
1186
+ prefixReturncodeMismatches: verdict.prefixReturncodeMismatches,
1187
+ prefixUnknownExpectations: verdict.prefixUnknownExpectations,
1188
+ armAExit: verdict.armA.exitCode,
1189
+ armAReturncodeMatch: returncodeMatch,
1190
+ armASignatureMatch: verdict.armA.failureSignatureMatch,
1191
+ replayed: verdict.prefixWithinTolerance && returncodeMatch,
1192
+ fix: null,
1193
+ wallMs: Date.now() - caseStart
1194
+ };
1195
+ rows.push(row);
1196
+ appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
1197
+ onProgress(`${label}: armA exit=${verdict.armA.exitCode} rcMatch=${returncodeMatch} divergences=${verdict.prefixDivergences.length}/${verdict.prefixExecuted} (${verdict.prefixDivergencePct}%, ${verdict.prefixUnknownExpectations} unknown-expectation)`);
1198
+ } catch (err) {
1199
+ const message = err instanceof Error ? err.message : String(err);
1200
+ const row = {
1201
+ ...base,
1202
+ derivedImage,
1203
+ signature: null,
1204
+ status: "replay-error",
1205
+ error: message.slice(0, 500),
1206
+ imagePulled: pulled,
1207
+ imageBuilt: built,
1208
+ prefixExecuted: null,
1209
+ prefixDivergences: null,
1210
+ prefixDivergencePct: null,
1211
+ prefixConfirmed: null,
1212
+ prefixReturncodeMismatches: null,
1213
+ prefixUnknownExpectations: null,
1214
+ armAExit: null,
1215
+ armAReturncodeMatch: false,
1216
+ armASignatureMatch: false,
1217
+ replayed: false,
1218
+ fix: null,
1219
+ wallMs: Date.now() - caseStart
1220
+ };
1221
+ rows.push(row);
1222
+ appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
1223
+ onProgress(`${label}: replay error — ${message.slice(0, 200)}`);
1224
+ }
1225
+ }
1226
+ let llm = null;
1227
+ if (options.fix !== "none") {
1228
+ const caller = options.fixCaller;
1229
+ if (!caller) throw new Error(`trajectory-replay: fix=${options.fix} requires a fixCaller`);
1230
+ const fixAttempts = options.fixAttempts ?? 3;
1231
+ const maxFixCases = options.maxFixCases ?? 30;
1232
+ const seed = options.seed ?? 17;
1233
+ const eligible = rows.filter((r) => r.status === "ok" && r.replayed);
1234
+ const sampled = eligible.length > maxFixCases ? seededSample(eligible, maxFixCases, seed) : new Set(eligible);
1235
+ if (eligible.length > maxFixCases) onProgress(`fix phase: ${eligible.length} eligible > cap ${maxFixCases}; seeded sample (seed=${seed})`);
1236
+ let calls = 0;
1237
+ let failures = 0;
1238
+ let promptTokens = 0;
1239
+ let completionTokens = 0;
1240
+ let callsWithoutUsage = 0;
1241
+ for (const row of eligible) {
1242
+ const index = rows.indexOf(row);
1243
+ if (!sampled.has(row)) {
1244
+ rows[index] = {
1245
+ ...row,
1246
+ fix: {
1247
+ attempted: false,
1248
+ sampledOut: true,
1249
+ command: null,
1250
+ llmError: null,
1251
+ usage: null,
1252
+ armBExit: null,
1253
+ armBPrefixExecuted: null,
1254
+ armBPrefixDivergences: null,
1255
+ armBPrefixDivergencePct: null,
1256
+ failureVanished: null,
1257
+ armBError: null,
1258
+ attempts: null,
1259
+ flippedAtAttempt: null
1260
+ }
1261
+ };
1262
+ continue;
1263
+ }
1264
+ const replayCase = selected.find((c) => c.trajId === row.trajId && c.corpus === row.corpus);
1265
+ const label = `fix ${row.corpus}/${row.trajId}`;
1266
+ const signature = verdictByTraj.get(row.trajId)?.signature ?? null;
1267
+ const stepTimeoutMs = options.stepTimeoutMs ?? replayCase.recordedStepTimeoutMs ?? 12e4;
1268
+ const promptInput = {
1269
+ taskStatement: replayCase.taskStatement,
1270
+ steps: replayCase.steps,
1271
+ k: replayCase.k
1272
+ };
1273
+ const caseOut = join(options.out, caseOutDirName(row));
1274
+ if (options.fix === "loop") {
1275
+ onProgress(`${label}: fix loop (budget ${fixAttempts} attempts)`);
1276
+ const result = await runFixLoop(caller, promptInput, async (command, attempt) => {
1277
+ onProgress(`${label}: attempt ${attempt} arm B — ${command.split("\n")[0].slice(0, 120)}`);
1278
+ const armB = await executeArmB(replayCase, command, backendFactory(row.derivedImage), signature, stepTimeoutMs, options.prefixLimit, (message) => onProgress(`${label}: attempt ${attempt}: ${message}`));
1279
+ mkdirSync(caseOut, { recursive: true });
1280
+ writeFileSync(join(caseOut, `armB-attempt${attempt}-result.json`), `${JSON.stringify({
1281
+ command,
1282
+ ...armB
1283
+ }, null, 2)}\n`);
1284
+ writeFileSync(join(caseOut, "armB-result.json"), `${JSON.stringify({
1285
+ command,
1286
+ attempt,
1287
+ ...armB
1288
+ }, null, 2)}\n`);
1289
+ return armB;
1290
+ }, {
1291
+ maxAttempts: fixAttempts,
1292
+ onProgress: (message) => onProgress(`${label}: ${message}`)
1293
+ });
1294
+ calls += result.llmCalls;
1295
+ failures += result.llmFailures;
1296
+ promptTokens += result.promptTokens;
1297
+ completionTokens += result.completionTokens;
1298
+ callsWithoutUsage += result.callsWithoutUsage;
1299
+ const usage = result.promptTokens + result.completionTokens > 0 ? {
1300
+ promptTokens: result.promptTokens,
1301
+ completionTokens: result.completionTokens
1302
+ } : null;
1303
+ const summary = (result.flippedAtAttempt !== null ? result.attempts.find((a) => a.attempt === result.flippedAtAttempt) : null) ?? [...result.attempts].reverse().find((a) => a.executed) ?? null;
1304
+ const lastRecord = result.attempts.at(-1) ?? null;
1305
+ rows[index] = {
1306
+ ...row,
1307
+ fix: summary ? {
1308
+ attempted: true,
1309
+ sampledOut: false,
1310
+ command: summary.command,
1311
+ llmError: null,
1312
+ usage,
1313
+ armBExit: summary.exitCode,
1314
+ armBPrefixExecuted: summary.prefixExecuted,
1315
+ armBPrefixDivergences: summary.prefixDivergences,
1316
+ armBPrefixDivergencePct: summary.prefixDivergencePct,
1317
+ failureVanished: summary.failureVanished,
1318
+ armBError: null,
1319
+ attempts: result.attempts,
1320
+ flippedAtAttempt: result.flippedAtAttempt
1321
+ } : result.aborted ? {
1322
+ attempted: true,
1323
+ sampledOut: false,
1324
+ command: lastRecord?.command ?? null,
1325
+ llmError: null,
1326
+ usage,
1327
+ armBExit: null,
1328
+ armBPrefixExecuted: null,
1329
+ armBPrefixDivergences: null,
1330
+ armBPrefixDivergencePct: null,
1331
+ failureVanished: null,
1332
+ armBError: lastRecord?.armBError ?? "sandbox error",
1333
+ attempts: result.attempts,
1334
+ flippedAtAttempt: null
1335
+ } : {
1336
+ attempted: true,
1337
+ sampledOut: false,
1338
+ command: null,
1339
+ llmError: lastRecord?.llmError ?? "no attempt produced a runnable fix",
1340
+ usage,
1341
+ armBExit: null,
1342
+ armBPrefixExecuted: null,
1343
+ armBPrefixDivergences: null,
1344
+ armBPrefixDivergencePct: null,
1345
+ failureVanished: null,
1346
+ armBError: null,
1347
+ attempts: result.attempts,
1348
+ flippedAtAttempt: null
1349
+ }
1350
+ };
1351
+ onProgress(`${label}: loop done — flipped=${result.flipped}` + (result.flippedAtAttempt !== null ? ` at attempt ${result.flippedAtAttempt}` : "") + ` (${result.llmCalls} calls, ${result.attempts.filter((a) => a.executed).length} arms)`);
1352
+ continue;
1353
+ }
1354
+ onProgress(`${label}: generating corrected command`);
1355
+ calls += 1;
1356
+ const generated = await generateFixCommand(caller, promptInput);
1357
+ if (!generated.succeeded) {
1358
+ failures += 1;
1359
+ rows[index] = {
1360
+ ...row,
1361
+ fix: {
1362
+ attempted: true,
1363
+ sampledOut: false,
1364
+ command: null,
1365
+ llmError: generated.error,
1366
+ usage: null,
1367
+ armBExit: null,
1368
+ armBPrefixExecuted: null,
1369
+ armBPrefixDivergences: null,
1370
+ armBPrefixDivergencePct: null,
1371
+ failureVanished: null,
1372
+ armBError: null,
1373
+ attempts: null,
1374
+ flippedAtAttempt: null
1375
+ }
1376
+ };
1377
+ onProgress(`${label}: LLM failed — ${generated.error.slice(0, 200)}`);
1378
+ continue;
1379
+ }
1380
+ if (generated.value.usage === null || generated.value.usage === void 0) callsWithoutUsage += 1;
1381
+ promptTokens += generated.value.usage?.promptTokens ?? 0;
1382
+ completionTokens += generated.value.usage?.completionTokens ?? 0;
1383
+ onProgress(`${label}: arm B — ${generated.value.command.split("\n")[0].slice(0, 120)}`);
1384
+ try {
1385
+ const armB = await executeArmB(replayCase, generated.value.command, backendFactory(row.derivedImage), signature, stepTimeoutMs, options.prefixLimit, (message) => onProgress(`${label}: ${message}`));
1386
+ rows[index] = {
1387
+ ...row,
1388
+ fix: {
1389
+ attempted: true,
1390
+ sampledOut: false,
1391
+ command: generated.value.command,
1392
+ llmError: null,
1393
+ usage: generated.value.usage,
1394
+ armBExit: armB.exitCode,
1395
+ armBPrefixExecuted: armB.prefixExecuted,
1396
+ armBPrefixDivergences: armB.prefixDivergences,
1397
+ armBPrefixDivergencePct: armB.prefixDivergencePct,
1398
+ failureVanished: armB.failureVanished,
1399
+ armBError: null,
1400
+ attempts: null,
1401
+ flippedAtAttempt: armB.failureVanished ? 1 : null
1402
+ }
1403
+ };
1404
+ mkdirSync(caseOut, { recursive: true });
1405
+ writeFileSync(join(caseOut, "armB-result.json"), `${JSON.stringify({
1406
+ command: generated.value.command,
1407
+ ...armB
1408
+ }, null, 2)}\n`);
1409
+ onProgress(`${label}: armB exit=${armB.exitCode} failureVanished=${armB.failureVanished}`);
1410
+ } catch (err) {
1411
+ const message = err instanceof Error ? err.message : String(err);
1412
+ rows[index] = {
1413
+ ...row,
1414
+ fix: {
1415
+ attempted: true,
1416
+ sampledOut: false,
1417
+ command: generated.value.command,
1418
+ llmError: null,
1419
+ usage: generated.value.usage,
1420
+ armBExit: null,
1421
+ armBPrefixExecuted: null,
1422
+ armBPrefixDivergences: null,
1423
+ armBPrefixDivergencePct: null,
1424
+ failureVanished: null,
1425
+ armBError: message.slice(0, 500),
1426
+ attempts: null,
1427
+ flippedAtAttempt: null
1428
+ }
1429
+ };
1430
+ onProgress(`${label}: arm B error — ${message.slice(0, 200)}`);
1431
+ }
1432
+ }
1433
+ llm = {
1434
+ model: options.fixModelLabel ?? "unknown",
1435
+ calls,
1436
+ failures,
1437
+ promptTokens,
1438
+ completionTokens,
1439
+ callsWithoutUsage
1440
+ };
1441
+ }
1442
+ const executedRows = rows.filter((r) => r.status === "ok");
1443
+ const replayed = executedRows.filter((r) => r.replayed);
1444
+ const sumOver = (pick) => executedRows.reduce((total, row) => total + (pick(row) ?? 0), 0);
1445
+ const executedSteps = sumOver((r) => r.prefixExecuted);
1446
+ const divergentSteps = sumOver((r) => r.prefixDivergences);
1447
+ const prefixFidelity = {
1448
+ executedSteps,
1449
+ divergentSteps,
1450
+ returncodeMismatches: sumOver((r) => r.prefixReturncodeMismatches),
1451
+ unknownExpectations: sumOver((r) => r.prefixUnknownExpectations),
1452
+ divergencePct: executedSteps > 0 ? Number((divergentSteps / executedSteps * 100).toFixed(1)) : null,
1453
+ tolerancePct: 10,
1454
+ casesWithinTolerance: executedRows.filter((r) => r.prefixDivergencePct !== null && r.prefixDivergencePct <= 10).length,
1455
+ casesExecuted: executedRows.length
1456
+ };
1457
+ const signatureStrict = executedRows.filter((r) => r.replayed && r.armASignatureMatch);
1458
+ const armBExecuted = rows.filter((r) => r.fix?.attempted && r.fix.failureVanished !== null);
1459
+ const flipped = armBExecuted.filter((r) => r.fix.failureVanished === true);
1460
+ const armBNonzeroRc = armBExecuted.filter((r) => r.recordedReturncodeAtK !== null && r.recordedReturncodeAtK !== 0);
1461
+ const flippedNonzeroRc = armBNonzeroRc.filter((r) => r.fix.failureVanished === true);
1462
+ const attempt1Executed = rows.filter((r) => r.fix?.attempts?.find((a) => a.attempt === 1)?.executed === true);
1463
+ const flippedAt1 = rows.filter((r) => r.fix?.flippedAtAttempt === 1);
1464
+ const flipsByAttempt = {};
1465
+ for (const row of rows) {
1466
+ const at = row.fix?.flippedAtAttempt;
1467
+ if (typeof at === "number") flipsByAttempt[String(at)] = (flipsByAttempt[String(at)] ?? 0) + 1;
1468
+ }
1469
+ const excludedByReason = {};
1470
+ for (const excluded of enumeration.excluded) excludedByReason[excluded.reason] = (excludedByReason[excluded.reason] ?? 0) + 1;
1471
+ const submitGoldsByCorpus = {};
1472
+ const submitEntry = (corpus) => submitGoldsByCorpus[corpus] ??= {
1473
+ submitOnlyCases: 0,
1474
+ goldsSkippedWithinReplayable: 0
1475
+ };
1476
+ for (const excluded of enumeration.excluded) if (excluded.reason === "gold-only-submit-step") submitEntry(excluded.corpus).submitOnlyCases += 1;
1477
+ for (const replayCase of enumeration.replayable) if (replayCase.submitGoldsSkipped > 0) submitEntry(replayCase.corpus).goldsSkippedWithinReplayable += replayCase.submitGoldsSkipped;
1478
+ const report = {
1479
+ generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
1480
+ corpora: options.corpora.map((c) => ({
1481
+ name: c.name,
1482
+ labelsPath: c.labelsPath,
1483
+ preparedDir: c.preparedDir
1484
+ })),
1485
+ totals: {
1486
+ labelEntries: enumeration.labelEntryCount,
1487
+ replayable: enumeration.replayable.length,
1488
+ executed: selected.length,
1489
+ excludedByReason,
1490
+ submitGoldsByCorpus
1491
+ },
1492
+ headline: {
1493
+ replayabilityRate: {
1494
+ numerator: replayed.length,
1495
+ denominator: selected.length,
1496
+ value: rate(replayed.length, selected.length)
1497
+ },
1498
+ signatureStrictRate: {
1499
+ numerator: signatureStrict.length,
1500
+ denominator: selected.length,
1501
+ value: rate(signatureStrict.length, selected.length)
1502
+ },
1503
+ prefixFidelity,
1504
+ fixFlipRate: options.fix !== "none" ? {
1505
+ numerator: flipped.length,
1506
+ denominator: armBExecuted.length,
1507
+ value: rate(flipped.length, armBExecuted.length)
1508
+ } : null,
1509
+ fixFlipRateNonzeroRc: options.fix !== "none" ? {
1510
+ numerator: flippedNonzeroRc.length,
1511
+ denominator: armBNonzeroRc.length,
1512
+ value: rate(flippedNonzeroRc.length, armBNonzeroRc.length)
1513
+ } : null,
1514
+ fixFlipAttempt1: options.fix === "loop" ? {
1515
+ numerator: flippedAt1.length,
1516
+ denominator: attempt1Executed.length,
1517
+ value: rate(flippedAt1.length, attempt1Executed.length)
1518
+ } : null,
1519
+ flipsByAttempt: options.fix === "loop" ? flipsByAttempt : null
1520
+ },
1521
+ llm,
1522
+ excluded: enumeration.excluded,
1523
+ pullFailures,
1524
+ cases: rows
1525
+ };
1526
+ writeFileSync(join(options.out, "batch-report.json"), `${JSON.stringify(report, null, 2)}\n`);
1527
+ writeFileSync(join(options.out, "batch-report.md"), renderBatchReport(report));
1528
+ return report;
1529
+ }
1530
+ function pct(value) {
1531
+ return value === null ? "—" : `${(value * 100).toFixed(1)}%`;
1532
+ }
1533
+ function renderBatchReport(report) {
1534
+ const lines = [];
1535
+ const { headline, totals } = report;
1536
+ lines.push("# Replay-verify batch report");
1537
+ lines.push("");
1538
+ lines.push(`Generated ${report.generatedAt}.`);
1539
+ lines.push("");
1540
+ lines.push("## Headline");
1541
+ lines.push("");
1542
+ const fidelity = headline.prefixFidelity;
1543
+ lines.push(`- **Replayability rate: ${pct(headline.replayabilityRate.value)}** (${headline.replayabilityRate.numerator}/${headline.replayabilityRate.denominator} replayable cases where the prefix replayed within ${fidelity.tolerancePct}% divergence AND arm A reproduced the recorded returncode at the gold step k).`);
1544
+ lines.push(`- Signature-strict rate: ${pct(headline.signatureStrictRate.value)} (${headline.signatureStrictRate.numerator}/${headline.signatureStrictRate.denominator}; additionally requires the recorded error substring in arm A output).`);
1545
+ lines.push(`- **Prefix divergence: ${fidelity.divergencePct === null ? "—" : `${fidelity.divergencePct}%`}** (${fidelity.divergentSteps}/${fidelity.executedSteps} executed prefix steps did not confirm the recording: ${fidelity.returncodeMismatches} returncode mismatches, ${fidelity.unknownExpectations} with no recorded returncode to check). ${fidelity.casesWithinTolerance}/${fidelity.casesExecuted} executed cases are within the ${fidelity.tolerancePct}% tolerance.`);
1546
+ if (headline.fixFlipRate) lines.push(`- **Fix-flip rate: ${pct(headline.fixFlipRate.value)}** (${headline.fixFlipRate.numerator}/${headline.fixFlipRate.denominator} arm-B-executed cases where the generated fix made the failure vanish).`);
1547
+ if (headline.fixFlipRateNonzeroRc) lines.push(`- Fix-flip rate on recorded-rc≠0 cases: ${pct(headline.fixFlipRateNonzeroRc.value)} (${headline.fixFlipRateNonzeroRc.numerator}/${headline.fixFlipRateNonzeroRc.denominator}; real recorded failures — a gold step recorded with rc 0 flips vacuously).`);
1548
+ if (headline.fixFlipAttempt1) lines.push(`- Fix-flip@1: ${pct(headline.fixFlipAttempt1.value)} (${headline.fixFlipAttempt1.numerator}/${headline.fixFlipAttempt1.denominator} cases whose attempt 1 executed — the one-shot-comparable number).`);
1549
+ if (headline.flipsByAttempt && Object.keys(headline.flipsByAttempt).length > 0) {
1550
+ const parts = Object.entries(headline.flipsByAttempt).sort((a, b) => Number(a[0]) - Number(b[0])).map(([attempt, count]) => `attempt ${attempt}: ${count}`);
1551
+ lines.push(`- Flips by attempt: ${parts.join(", ")}.`);
1552
+ }
1553
+ lines.push("");
1554
+ lines.push("## Enumeration");
1555
+ lines.push("");
1556
+ lines.push(`${totals.labelEntries} label entries across ${report.corpora.length} corpora → ${totals.replayable} replayable (SWE-style docker image + ≥1 gold incorrect step), ${totals.executed} executed.`);
1557
+ lines.push("");
1558
+ lines.push("| exclusion reason | count |");
1559
+ lines.push("| --- | --- |");
1560
+ for (const [reason, count] of Object.entries(totals.excludedByReason).sort((a, b) => b[1] - a[1])) lines.push(`| ${reason} | ${count} |`);
1561
+ const submitEntries = Object.entries(totals.submitGoldsByCorpus);
1562
+ if (submitEntries.length > 0) {
1563
+ lines.push("");
1564
+ lines.push("Submit-command golds are never counterfactual targets (a gold on the submit step marks a bad submit decision, not a failed command):");
1565
+ lines.push("");
1566
+ lines.push("| corpus | cases excluded (all golds = submit) | golds skipped within replayable cases |");
1567
+ lines.push("| --- | --- | --- |");
1568
+ for (const [corpus, stats] of submitEntries.sort((a, b) => a[0].localeCompare(b[0]))) lines.push(`| ${corpus} | ${stats.submitOnlyCases} | ${stats.goldsSkippedWithinReplayable} |`);
1569
+ }
1570
+ if (report.pullFailures.length > 0) {
1571
+ lines.push("");
1572
+ lines.push("## Image pull/build failures");
1573
+ lines.push("");
1574
+ lines.push("| corpus | trajectory | image | error |");
1575
+ lines.push("| --- | --- | --- | --- |");
1576
+ for (const failure of report.pullFailures) lines.push(`| ${failure.corpus} | ${failure.trajId} | \`${failure.image}\` | ${failure.error.replaceAll("|", "\\|")} |`);
1577
+ }
1578
+ lines.push("");
1579
+ lines.push("## Per-case results");
1580
+ lines.push("");
1581
+ lines.push("| corpus | trajectory | k | rc@k | prefix | confirmed | rc mismatch | unknown rc | div | div% | armA exit | rc match | sig match | replayed | fix | armB exit | armB div% | vanished | wall s |");
1582
+ lines.push(`| ${Array(19).fill("---").join(" | ")} |`);
1583
+ for (const row of report.cases) {
1584
+ const fix = row.fix;
1585
+ const fixCell = !fix ? "—" : fix.sampledOut ? "sampled-out" : fix.llmError ? "llm-failed" : fix.armBError ? "armB-error" : fix.attempts ? fix.flippedAtAttempt !== null ? `flip@${fix.flippedAtAttempt}` : `exhausted(${fix.attempts.length})` : "generated";
1586
+ lines.push([
1587
+ row.corpus,
1588
+ row.trajId.length > 48 ? `${row.trajId.slice(0, 45)}…` : row.trajId,
1589
+ row.k,
1590
+ row.recordedReturncodeAtK ?? "null",
1591
+ row.prefixExecuted ?? "—",
1592
+ row.prefixConfirmed ?? "—",
1593
+ row.prefixReturncodeMismatches ?? "—",
1594
+ row.prefixUnknownExpectations ?? "—",
1595
+ row.prefixDivergences ?? "—",
1596
+ row.prefixDivergencePct ?? "—",
1597
+ row.status === "ok" ? row.armAExit : row.status,
1598
+ row.armAReturncodeMatch ? "yes" : "no",
1599
+ row.armASignatureMatch ? "yes" : "no",
1600
+ row.replayed ? "**yes**" : "no",
1601
+ fixCell,
1602
+ fix?.armBExit ?? "—",
1603
+ fix?.armBPrefixDivergencePct ?? "—",
1604
+ fix?.failureVanished === null || fix === null ? "—" : fix.failureVanished ? "**yes**" : "no",
1605
+ (row.wallMs / 1e3).toFixed(1)
1606
+ ].join(" | "));
1607
+ }
1608
+ if (report.llm) {
1609
+ lines.push("");
1610
+ lines.push("## LLM fix generation");
1611
+ lines.push("");
1612
+ lines.push(`Model ${report.llm.model}: ${report.llm.calls} calls (${report.llm.failures} failed), ${report.llm.promptTokens} prompt + ${report.llm.completionTokens} completion tokens.`);
1613
+ }
1614
+ lines.push("");
1615
+ return lines.join("\n");
1616
+ }
1617
+ //#endregion
1618
+ //#region src/trajectory-replay/wire.ts
1619
+ /**
1620
+ * Finding-to-replay wire: the entry point that turns a cited incorrect-steps
1621
+ * finding into an executed replay proof.
1622
+ *
1623
+ * An analyst finding over a labeled trajectory corpus names its trajectory and
1624
+ * carries a subject of the form
1625
+ * `incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>`.
1626
+ * The wire maps that to a replay invocation: the executed step is the
1627
+ * finding's first incorrect step (which may differ from the gold label), the
1628
+ * image and cwd come from the trajectory's raw config, and — optionally — an
1629
+ * arm-B corrected command comes from the counterfactual fix generator.
1630
+ */
1631
+ const SUBJECT_PATTERN = /^incorrect-steps-(\d+)-(\d+)-(escaped|unescaped)-consequence-(\d+)$/;
1632
+ /** Null when the subject is not an incorrect-steps finding subject. */
1633
+ function parseIncorrectStepsSubject(subject) {
1634
+ const match = SUBJECT_PATTERN.exec(subject);
1635
+ if (!match) return null;
1636
+ return {
1637
+ firstStep: Number(match[1]),
1638
+ lastStep: Number(match[2]),
1639
+ escapeStatus: match[3],
1640
+ consequenceStep: Number(match[4])
1641
+ };
1642
+ }
1643
+ /**
1644
+ * Maps a finding onto replay resources, searching the given corpora for the
1645
+ * trajectory. Throws with the precise reason when the finding cannot be
1646
+ * replayed (malformed subject, unknown trajectory, non-replayable case, step
1647
+ * out of range) — the caller surfaces that reason instead of a proof.
1648
+ */
1649
+ function resolveFindingInvocation(finding, corpora) {
1650
+ const subject = parseIncorrectStepsSubject(finding.subject);
1651
+ if (!subject) throw new Error(`trajectory-replay: subject '${finding.subject}' is not incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>`);
1652
+ const failures = [];
1653
+ for (const corpus of corpora) {
1654
+ const resolution = resolveCaseResources(corpus, finding.trajId);
1655
+ if (resolution.resolved) {
1656
+ const at = subject.firstStep;
1657
+ if (!resolution.resources.steps.find((s) => s.step_id === at)) throw new Error(`trajectory-replay: finding step ${at} is outside ${finding.trajId} (${resolution.resources.steps.length} steps)`);
1658
+ return {
1659
+ resources: resolution.resources,
1660
+ subject,
1661
+ at
1662
+ };
1663
+ }
1664
+ failures.push(`${corpus.name}: ${resolution.reason}`);
1665
+ }
1666
+ throw new Error(`trajectory-replay: trajectory ${finding.trajId} is not replayable in any corpus — ${failures.join("; ")}`);
1667
+ }
1668
+ /**
1669
+ * Finding in, executed proof out. The image comes from the corpus resources;
1670
+ * the backend factory receives it as-is, so a factory backed by infrastructure
1671
+ * that needs a derived image must run an `ImagePreparer` first.
1672
+ */
1673
+ async function replayVerifyFinding(finding, options) {
1674
+ if (options.fixCaller && options.fixCommand !== void 0) throw new Error("trajectory-replay: pass fixCaller or fixCommand, not both");
1675
+ const invocation = resolveFindingInvocation(finding, options.corpora);
1676
+ let fixCommand = options.fixCommand ?? null;
1677
+ if (options.fixCaller) {
1678
+ const generated = await generateFixCommand(options.fixCaller, {
1679
+ taskStatement: invocation.resources.taskStatement,
1680
+ steps: invocation.resources.steps,
1681
+ k: invocation.at
1682
+ });
1683
+ if (!generated.succeeded) throw new Error(`trajectory-replay: fix generation failed — ${generated.error}`);
1684
+ fixCommand = generated.value.command;
1685
+ }
1686
+ const verdict = await replayVerify({
1687
+ stepsPath: invocation.resources.stepsPath,
1688
+ image: invocation.resources.image,
1689
+ at: invocation.at,
1690
+ fixCommand: fixCommand ?? void 0,
1691
+ cwd: invocation.resources.cwd,
1692
+ out: options.out,
1693
+ caseId: invocation.resources.trajId,
1694
+ stepTimeoutMs: options.stepTimeoutMs ?? invocation.resources.recordedStepTimeoutMs ?? void 0,
1695
+ prefixLimit: options.prefixLimit,
1696
+ backend: options.backendFactory(invocation.resources.image),
1697
+ onProgress: options.onProgress
1698
+ });
1699
+ return {
1700
+ invocation,
1701
+ fixCommand,
1702
+ verdict
1703
+ };
1704
+ }
1705
+ //#endregion
1706
+ //#region src/trajectory-replay/findings.ts
1707
+ /**
1708
+ * Proof-carrying findings: execute analyst findings as replays.
1709
+ *
1710
+ * An analyst finding is a cited claim ("step 12 is where the run went
1711
+ * wrong"). This module turns each finding into an executed verdict by
1712
+ * replaying the trajectory prefix and running the accused step (arm A),
1713
+ * optionally followed by a corrected step (arm B):
1714
+ *
1715
+ * `reproduced` arm A re-produced the recorded failure signature
1716
+ * (returncode + stable output substring).
1717
+ * `fix-flipped` arm A reproduced AND arm B's corrected command made the
1718
+ * failure vanish — the strongest per-finding proof.
1719
+ * `divergent` arm A executed but the recorded failure did NOT
1720
+ * reproduce; evidence against the finding (or against
1721
+ * replay fidelity — the receipt carries prefix
1722
+ * divergences so the reader can tell which).
1723
+ * `not-replayable` the finding could not be executed at all; the receipt
1724
+ * carries the precise reason (no step subject, unknown
1725
+ * trajectory, no recorded image, submit step, …).
1726
+ *
1727
+ * Verification is execution, not generation: no LLM is involved unless the
1728
+ * caller supplies a corrected command for arm B. A caller whose execution
1729
+ * environment must be reachable before proofs run passes `preflight`, which
1730
+ * fails loud instead of letting verification skip silently.
1731
+ *
1732
+ * Findings are matched by the shape the analyst product emits
1733
+ * (`AnalystFinding`): `subject` (`incorrect-step-<n>` or the wire's
1734
+ * `incorrect-steps-<f>-<l>-<escaped|unescaped>-consequence-<c>`),
1735
+ * `metadata.block_first_step`, and `trace://<traj>/…` evidence refs.
1736
+ */
1737
+ const SINGLE_STEP_SUBJECT = /^incorrect-step-(\d+)$/;
1738
+ /**
1739
+ * 1-based step the finding accuses, or null when the finding names none.
1740
+ * `metadata.block_first_step` wins over the subject: the analyst records the
1741
+ * block's first incorrect step there even when the subject names a later
1742
+ * step of the same block.
1743
+ */
1744
+ function findingReplayStep(finding) {
1745
+ const fromMetadata = finding.metadata?.block_first_step;
1746
+ if (typeof fromMetadata === "number" && Number.isInteger(fromMetadata) && fromMetadata >= 1) return fromMetadata;
1747
+ const subject = finding.subject ?? "";
1748
+ const wire = parseIncorrectStepsSubject(subject);
1749
+ if (wire) return wire.firstStep;
1750
+ const single = SINGLE_STEP_SUBJECT.exec(subject);
1751
+ if (single) return Number(single[1]);
1752
+ return null;
1753
+ }
1754
+ const TRACE_EVIDENCE_URI = /^trace:\/\/([^/]+)\//;
1755
+ /** Trajectory id from the finding's `trace://<traj>/…` evidence refs, or null. */
1756
+ function findingTrajectoryId(finding) {
1757
+ for (const ref of finding.evidence_refs ?? []) {
1758
+ if (typeof ref.uri !== "string") continue;
1759
+ const match = TRACE_EVIDENCE_URI.exec(ref.uri);
1760
+ if (match) return match[1];
1761
+ }
1762
+ return null;
1763
+ }
1764
+ function checkStep(steps, at) {
1765
+ const step = steps.find((s) => s.step_id === at);
1766
+ if (!step) return {
1767
+ ok: false,
1768
+ reason: `step ${at} is outside the trajectory (${steps.length} steps)`
1769
+ };
1770
+ if (isSubmitAction(step.action)) return {
1771
+ ok: false,
1772
+ reason: `step ${at} is the submit action — a submit decision has no executable failure to replay`
1773
+ };
1774
+ const recordedReturncode = parseRecordedReturncode(step.observation);
1775
+ if (recordedReturncode === null) return {
1776
+ ok: false,
1777
+ reason: `step ${at} recorded no returncode — there is no executable failure signature to reproduce`
1778
+ };
1779
+ return {
1780
+ ok: true,
1781
+ recordedReturncode
1782
+ };
1783
+ }
1784
+ /**
1785
+ * Decides whether one finding can be executed against the source, and with
1786
+ * what invocation. Never throws for a finding-shaped problem — every dead end
1787
+ * becomes a `not-replayable` reason the receipt can carry verbatim.
1788
+ */
1789
+ function resolveFindingReplayability(finding, source) {
1790
+ const at = findingReplayStep(finding);
1791
+ if (at === null) return {
1792
+ replayable: false,
1793
+ reason: `subject '${finding.subject ?? "(none)"}' names no trajectory step (expected incorrect-step-<n>, incorrect-steps-<f>-<l>-…, or metadata.block_first_step)`
1794
+ };
1795
+ if (source.kind === "direct") {
1796
+ const trajId = findingTrajectoryId(finding);
1797
+ if (trajId && source.caseId && trajId !== source.caseId) return {
1798
+ replayable: false,
1799
+ reason: `finding cites trajectory '${trajId}' but the supplied steps are case '${source.caseId}'`
1800
+ };
1801
+ let steps;
1802
+ try {
1803
+ steps = JSON.parse(readFileSync(source.stepsPath, "utf8"));
1804
+ } catch (err) {
1805
+ throw new Error(`verify-findings: cannot read steps file ${source.stepsPath} — ${err instanceof Error ? err.message : String(err)}`);
1806
+ }
1807
+ if (!Array.isArray(steps) || steps.length === 0) throw new Error(`verify-findings: ${source.stepsPath} is not a non-empty steps array`);
1808
+ const step = checkStep(steps, at);
1809
+ if (!step.ok) return {
1810
+ replayable: false,
1811
+ reason: step.reason
1812
+ };
1813
+ return {
1814
+ replayable: true,
1815
+ resolved: {
1816
+ caseId: source.caseId ?? trajId ?? source.stepsPath,
1817
+ stepsPath: source.stepsPath,
1818
+ image: source.image,
1819
+ cwd: source.cwd,
1820
+ at,
1821
+ recordedReturncode: step.recordedReturncode,
1822
+ recordedStepTimeoutMs: null
1823
+ }
1824
+ };
1825
+ }
1826
+ const trajId = findingTrajectoryId(finding);
1827
+ if (!trajId) return {
1828
+ replayable: false,
1829
+ reason: "finding carries no trace://<trajectory>/ evidence ref naming its trajectory"
1830
+ };
1831
+ const failures = [];
1832
+ for (const corpus of source.corpora) {
1833
+ const resolution = resolveCaseResources(corpus, trajId);
1834
+ if (!resolution.resolved) {
1835
+ failures.push(`${corpus.name}: ${resolution.reason}${resolution.detail ? ` (${resolution.detail})` : ""}`);
1836
+ continue;
1837
+ }
1838
+ const step = checkStep(resolution.resources.steps, at);
1839
+ if (!step.ok) return {
1840
+ replayable: false,
1841
+ reason: step.reason
1842
+ };
1843
+ return {
1844
+ replayable: true,
1845
+ resolved: {
1846
+ caseId: trajId,
1847
+ stepsPath: resolution.resources.stepsPath,
1848
+ image: resolution.resources.image,
1849
+ cwd: resolution.resources.cwd,
1850
+ at,
1851
+ recordedReturncode: step.recordedReturncode,
1852
+ recordedStepTimeoutMs: resolution.resources.recordedStepTimeoutMs
1853
+ }
1854
+ };
1855
+ }
1856
+ return {
1857
+ replayable: false,
1858
+ reason: `trajectory ${trajId} is not replayable in any corpus — ${failures.join("; ")}`
1859
+ };
1860
+ }
1861
+ /**
1862
+ * Arm A reproduced on a prefix the recording confirmed → the fix flipping it
1863
+ * beats plain reproduction; anything else diverged. A proof standing on a
1864
+ * prefix outside the divergence tolerance is divergent no matter what arm A
1865
+ * did: the state it ran against is not the recorded state.
1866
+ */
1867
+ function classifyVerdict(verdict) {
1868
+ if (!verdict.prefixWithinTolerance) return "divergent";
1869
+ if (!verdict.armA.failureSignatureMatch) return "divergent";
1870
+ if (verdict.armB?.failureVanished) return "fix-flipped";
1871
+ return "reproduced";
1872
+ }
1873
+ function receiptExecution(verdict) {
1874
+ return {
1875
+ image: verdict.image,
1876
+ cwd: verdict.cwd,
1877
+ recordedReturncode: verdict.recordedReturncode,
1878
+ signature: verdict.signature,
1879
+ signatureBasis: verdict.signatureBasis,
1880
+ armA: {
1881
+ command: verdict.armA.command,
1882
+ exitCode: verdict.armA.exitCode,
1883
+ wallMs: verdict.armA.wallMs,
1884
+ failureSignatureMatch: verdict.armA.failureSignatureMatch
1885
+ },
1886
+ armB: verdict.armB ? {
1887
+ command: verdict.armB.command,
1888
+ exitCode: verdict.armB.exitCode,
1889
+ wallMs: verdict.armB.wallMs,
1890
+ failureVanished: verdict.armB.failureVanished
1891
+ } : null,
1892
+ prefixExecuted: verdict.prefixExecuted,
1893
+ prefixDivergences: verdict.prefixDivergences.length,
1894
+ prefixDivergencePct: verdict.prefixDivergencePct,
1895
+ prefixReturncodeMismatches: verdict.prefixReturncodeMismatches,
1896
+ prefixUnknownExpectations: verdict.prefixUnknownExpectations,
1897
+ prefixWithinTolerance: verdict.prefixWithinTolerance,
1898
+ totalMs: verdict.timings.totalMs
1899
+ };
1900
+ }
1901
+ function writeReceipt(receiptDir, finding, verification, execution) {
1902
+ const receipt = {
1903
+ schema_version: "1.0.0",
1904
+ produced_at: (/* @__PURE__ */ new Date()).toISOString(),
1905
+ finding_id: verification.finding_id,
1906
+ analyst_id: finding.analyst_id ?? null,
1907
+ subject: verification.subject,
1908
+ claim: typeof finding.claim === "string" ? finding.claim.slice(0, 600) : null,
1909
+ trajectory_id: verification.trajectory_id,
1910
+ step: verification.step,
1911
+ verified: verification.verified,
1912
+ reason: verification.reason,
1913
+ execution,
1914
+ verdict_path: verification.verdict_path,
1915
+ deduplicated_with: verification.deduplicated_with
1916
+ };
1917
+ writeFileSync(join(receiptDir, "receipt.json"), `${JSON.stringify(receipt, null, 2)}\n`);
1918
+ }
1919
+ function receiptDirName(index, finding) {
1920
+ const id = typeof finding.finding_id === "string" && finding.finding_id.length > 0 ? finding.finding_id.replace(/[^A-Za-z0-9_-]/g, "_") : "finding";
1921
+ return `${String(index + 1).padStart(3, "0")}-${id}`;
1922
+ }
1923
+ /**
1924
+ * Verifies every finding against the source: resolves replayability, executes
1925
+ * one proof per distinct (case, step, fix) — findings accusing the same step
1926
+ * share the executed proof — and writes a receipt directory per finding plus a
1927
+ * run-level verifications.json.
1928
+ */
1929
+ async function verifyFindings(findings, options) {
1930
+ if (findings.length === 0) throw new Error("verify-findings: no findings to verify");
1931
+ mkdirSync(options.out, { recursive: true });
1932
+ const resolutions = findings.map((finding) => resolveFindingReplayability(finding, options.source));
1933
+ if (resolutions.some((resolution) => resolution.replayable) && options.preflight) await options.preflight();
1934
+ const preparer = options.source.kind === "corpus" ? options.source.preparer === void 0 ? dockerImagePreparer() : options.source.preparer : null;
1935
+ const preparedImages = /* @__PURE__ */ new Map();
1936
+ const executedByKey = /* @__PURE__ */ new Map();
1937
+ const verifications = [];
1938
+ let executions = 0;
1939
+ for (let index = 0; index < findings.length; index++) {
1940
+ const finding = findings[index];
1941
+ const resolution = resolutions[index];
1942
+ const receiptDir = join(options.out, receiptDirName(index, finding));
1943
+ mkdirSync(receiptDir, { recursive: true });
1944
+ const identity = {
1945
+ finding_id: finding.finding_id ?? null,
1946
+ subject: finding.subject ?? null,
1947
+ trajectory_id: findingTrajectoryId(finding)
1948
+ };
1949
+ if (!resolution.replayable) {
1950
+ const verification = {
1951
+ ...identity,
1952
+ step: findingReplayStep(finding),
1953
+ verified: "not-replayable",
1954
+ reason: resolution.reason,
1955
+ receipt: receiptDir,
1956
+ verdict_path: null,
1957
+ deduplicated_with: null
1958
+ };
1959
+ writeReceipt(receiptDir, finding, verification, null);
1960
+ verifications.push(verification);
1961
+ options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: not-replayable — ${resolution.reason}`);
1962
+ continue;
1963
+ }
1964
+ const resolved = resolution.resolved;
1965
+ const dedupeKey = `${resolved.caseId}::${resolved.at}::${options.fixCommand ?? ""}`;
1966
+ const prior = executedByKey.get(dedupeKey);
1967
+ if (prior) {
1968
+ const verification = {
1969
+ ...identity,
1970
+ step: resolved.at,
1971
+ verified: prior.verified,
1972
+ reason: prior.reason,
1973
+ receipt: receiptDir,
1974
+ verdict_path: prior.verdict_path,
1975
+ deduplicated_with: prior.receipt
1976
+ };
1977
+ const priorVerdict = prior.verdict_path ? JSON.parse(readFileSync(prior.verdict_path, "utf8")) : null;
1978
+ writeReceipt(receiptDir, finding, verification, priorVerdict ? receiptExecution(priorVerdict) : null);
1979
+ verifications.push(verification);
1980
+ options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: ${prior.verified} (shares proof with ${prior.finding_id ?? prior.receipt})`);
1981
+ continue;
1982
+ }
1983
+ let image = resolved.image;
1984
+ if (preparer) {
1985
+ const preparationKey = `${resolved.image}::${resolved.cwd}`;
1986
+ let preparation = preparedImages.get(preparationKey);
1987
+ if (!preparation) {
1988
+ const ensured = await preparer.ensure(resolved.image, resolved.cwd);
1989
+ preparation = ensured.succeeded ? {
1990
+ succeeded: true,
1991
+ image: ensured.value.derivedImage
1992
+ } : {
1993
+ succeeded: false,
1994
+ error: ensured.error
1995
+ };
1996
+ preparedImages.set(preparationKey, preparation);
1997
+ }
1998
+ if (!preparation.succeeded) {
1999
+ const verification = {
2000
+ ...identity,
2001
+ step: resolved.at,
2002
+ verified: "not-replayable",
2003
+ reason: `replay image could not be prepared — ${preparation.error}`,
2004
+ receipt: receiptDir,
2005
+ verdict_path: null,
2006
+ deduplicated_with: null
2007
+ };
2008
+ writeReceipt(receiptDir, finding, verification, null);
2009
+ verifications.push(verification);
2010
+ options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: not-replayable — image preparation failed`);
2011
+ continue;
2012
+ }
2013
+ image = preparation.image;
2014
+ }
2015
+ options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: executing arm A at step ${resolved.at} of ${resolved.caseId} on ${image}`);
2016
+ const verdict = await replayVerify({
2017
+ stepsPath: resolved.stepsPath,
2018
+ image,
2019
+ at: resolved.at,
2020
+ fixCommand: options.fixCommand,
2021
+ cwd: resolved.cwd,
2022
+ out: receiptDir,
2023
+ caseId: resolved.caseId,
2024
+ stepTimeoutMs: options.stepTimeoutMs ?? resolved.recordedStepTimeoutMs ?? void 0,
2025
+ prefixLimit: options.prefixLimit,
2026
+ backend: options.backendFactory(image),
2027
+ onProgress: options.onProgress
2028
+ });
2029
+ executions += 1;
2030
+ const verification = {
2031
+ ...identity,
2032
+ step: resolved.at,
2033
+ verified: classifyVerdict(verdict),
2034
+ reason: null,
2035
+ receipt: receiptDir,
2036
+ verdict_path: join(receiptDir, "replay-verdict.json"),
2037
+ deduplicated_with: null
2038
+ };
2039
+ executedByKey.set(dedupeKey, verification);
2040
+ writeReceipt(receiptDir, finding, verification, receiptExecution(verdict));
2041
+ verifications.push(verification);
2042
+ options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: ${verification.verified}`);
2043
+ }
2044
+ const counts = {
2045
+ reproduced: 0,
2046
+ "fix-flipped": 0,
2047
+ divergent: 0,
2048
+ "not-replayable": 0
2049
+ };
2050
+ for (const verification of verifications) counts[verification.verified] += 1;
2051
+ const run = {
2052
+ out: options.out,
2053
+ verifications,
2054
+ counts,
2055
+ executions
2056
+ };
2057
+ writeFileSync(join(options.out, "verifications.json"), `${JSON.stringify({
2058
+ schema_version: "1.0.0",
2059
+ ...run
2060
+ }, null, 2)}\n`);
2061
+ return run;
2062
+ }
2063
+ function verdictCell(verification) {
2064
+ switch (verification.verified) {
2065
+ case "reproduced": return "**VERIFIED** — reproduced";
2066
+ case "fix-flipped": return "**VERIFIED** — fix-flipped";
2067
+ case "divergent": return "DIVERGENT — recorded failure did not reproduce";
2068
+ case "not-replayable": return `UNVERIFIABLE — ${verification.reason ?? "no reason recorded"}`;
2069
+ }
2070
+ }
2071
+ /** Markdown section an analysis report appends when finding verification ran. */
2072
+ function renderVerifiedFindingsSection(run) {
2073
+ const lines = ["## Verified findings (executed replay)", ""];
2074
+ const total = run.verifications.length;
2075
+ lines.push(`${total} finding(s) → ${run.counts.reproduced} reproduced, ${run.counts["fix-flipped"]} fix-flipped, ${run.counts.divergent} divergent, ${run.counts["not-replayable"]} not replayable (${run.executions} execution(s); findings accusing the same step share one proof).`);
2076
+ lines.push("");
2077
+ lines.push("| Finding | Subject | Step | Verdict | Receipt |");
2078
+ lines.push("|---|---|---:|---|---|");
2079
+ for (const verification of run.verifications) {
2080
+ const shared = verification.deduplicated_with ? " (shared proof)" : "";
2081
+ lines.push(`| \`${verification.finding_id ?? "—"}\` | \`${verification.subject ?? "—"}\` | ${verification.step ?? "—"} | ${verdictCell(verification)} | \`${verification.receipt}\`${shared} |`);
2082
+ }
2083
+ lines.push("");
2084
+ lines.push("VERIFIED = the accused step was re-executed after replaying the trajectory prefix, and the recorded failure signature reproduced (fix-flipped: a corrected command additionally made it vanish). Each receipt directory carries receipt.json and, when executed, replay-verdict.json + report.md with real stdout/stderr.");
2085
+ lines.push("");
2086
+ return lines.join("\n");
2087
+ }
2088
+ /**
2089
+ * Accepts the two shapes findings travel in: a bare JSON array of analyst
2090
+ * findings, or an object with a `findings` array (e.g. an extracted
2091
+ * `observations[n]` from a result.json).
2092
+ */
2093
+ function readFindingsFile(path) {
2094
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
2095
+ const array = Array.isArray(parsed) ? parsed : parsed && typeof parsed === "object" && Array.isArray(parsed.findings) ? parsed.findings : null;
2096
+ if (!array) throw new Error(`${path} is neither a findings array nor an object with a findings array`);
2097
+ for (const entry of array) if (!entry || typeof entry !== "object") throw new Error(`${path}: every finding must be an object, got ${JSON.stringify(entry)}`);
2098
+ return array;
2099
+ }
2100
+ //#endregion
2101
+ export { PREFIX_DIVERGENCE_TOLERANCE_PCT, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, buildFixPrompt, buildRetryFixPrompt, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isSubmitAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, verifyFindings, wrapActionForExec };
2102
+
2103
+ //# sourceMappingURL=index.js.map