@tangle-network/agent-eval 0.144.6 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
  14. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
  22. package/dist/campaign-Tdy3h62h.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  160. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/charter.md +112 -0
  207. package/docs/experiment.md +104 -0
  208. package/docs/prime-analyst.md +1 -0
  209. package/docs/trace-analysis.md +26 -0
  210. package/docs/trace-repair-admission.md +194 -0
  211. package/docs/trace-repair-analyst-arms.md +121 -0
  212. package/docs/trace-repair-continuation.md +107 -0
  213. package/docs/trace-repair-grader.md +163 -0
  214. package/docs/trajectory-replay.md +110 -0
  215. package/docs/verification-strategies.md +103 -0
  216. package/package.json +19 -2
  217. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  218. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  219. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  220. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  221. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  222. package/dist/campaign-CXsdyym7.js.map +0 -1
  223. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  224. package/dist/index-4XwggC10.d.ts.map +0 -1
  225. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  226. package/dist/integrity-fdt8XPAv.js.map +0 -1
  227. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  228. package/dist/replay-Krvb114g.d.ts.map +0 -1
  229. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  230. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  231. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  232. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  233. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  234. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  235. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  236. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  237. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  238. package/dist/verdict-Dps8_okt.d.ts +0 -37
  239. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,3878 @@
1
+ import { c as ValidationError, n as CaptureIntegrityError } from "../errors-D-LKuDhb.js";
2
+ import { r as contentHash } from "../verdict-cache-BCcOh0kF.js";
3
+ import { E as pairedBootstrap } from "../statistics-ByxzSiOM.js";
4
+ import { d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage } from "../prime-protocol-BfSalTfR.js";
5
+ import { o as parseRecordedReturncode, r as deriveFailureSignature, t as wrapActionForExec } from "../exec-BLtYZdWo.js";
6
+ import { createHash } from "node:crypto";
7
+ import { spawn } from "node:child_process";
8
+ import { constants } from "node:os";
9
+ //#region src/trace-repair/mini-swe-scaffold.ts
10
+ /**
11
+ * The mini-swe-agent scaffold as the Terminal-Bench-2 trajectory corpus
12
+ * recorded it: one bash block per turn, one observation per command, and a
13
+ * sentinel command that ends the run.
14
+ *
15
+ * Every template here is byte-verified against
16
+ * `yoonholee/terminalbench-trajectories` (agent = `mini-swe-agent`, 6663 rows,
17
+ * one distinct system prompt across all of them). A continuation that renders
18
+ * different bytes puts the model in a different distribution than the prefix
19
+ * it inherits, so these strings are pinned, not configurable.
20
+ */
21
+ /** Whole-line marker that ends a run. The first output line must equal it and the command must exit 0. */
22
+ const SUBMIT_SENTINEL = "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT";
23
+ /** Outputs at or above this length are elided head+tail instead of shown whole. */
24
+ const OUTPUT_ELISION_THRESHOLD = 1e4;
25
+ /** Characters kept from each end of an elided output. */
26
+ const OUTPUT_ELISION_WINDOW = 5e3;
27
+ const MINI_SWE_SYSTEM_MESSAGE = `You are a helpful assistant that can interact with a computer.
28
+
29
+ Your response must contain exactly ONE bash code block with ONE command (or commands connected with && or ||).
30
+ Include a THOUGHT section before your command where you explain your reasoning process.
31
+ Format your response as shown in <format_example>.
32
+
33
+ <format_example>
34
+ Your reasoning and analysis here. Explain why you want to perform the action.
35
+
36
+ \`\`\`bash
37
+ your_command_here
38
+ \`\`\`
39
+ </format_example>
40
+
41
+ Failure to follow these rules will cause your response to be rejected.
42
+ `;
43
+ /** The second message of every recorded run: task plus workflow rules. */
44
+ function renderInstanceMessage(input) {
45
+ return `Please solve this issue: ${input.task}
46
+
47
+ You can execute bash commands and edit files to implement the necessary changes.
48
+
49
+ ## Recommended Workflow
50
+
51
+ This workflows should be done step-by-step so that you can iterate on your changes and any possible problems.
52
+
53
+ 1. Analyze the codebase by finding and reading relevant files
54
+ 2. Create a script to reproduce the issue
55
+ 3. Edit the source code to resolve the issue
56
+ 4. Verify your fix works by running your script again
57
+ 5. Test edge cases to ensure your fix is robust
58
+ 6. Submit your changes and finish your work by issuing the following command: \`echo ${SUBMIT_SENTINEL}\`.
59
+ Do not combine it with any other command. <important>After this command, you cannot continue working on this task.</important>
60
+
61
+ ## Important Rules
62
+
63
+ 1. Every response must contain exactly one action
64
+ 2. The action must be enclosed in triple backticks
65
+ 3. Directory or environment variable changes are not persistent. Every action is executed in a new subshell.
66
+ However, you can prefix any action with \`MY_ENV_VAR=MY_VALUE cd /path/to/working/dir && ...\` or write/load environment variables from files
67
+
68
+ <system_information>
69
+ ${input.systemInformation}
70
+ </system_information>
71
+
72
+ ## Formatting your response
73
+
74
+ Here is an example of a correct response:
75
+
76
+ <example_response>
77
+ THOUGHT: I need to understand the structure of the repository first. Let me check what files are in the current directory to get a better understanding of the codebase.
78
+
79
+ \`\`\`bash
80
+ ls -la
81
+ \`\`\`
82
+ </example_response>
83
+
84
+ ## Useful command examples
85
+
86
+ ### Create a new file:
87
+
88
+ \`\`\`bash
89
+ cat <<'EOF' > newfile.py
90
+ import numpy as np
91
+ hello = "world"
92
+ print(hello)
93
+ EOF
94
+ \`\`\`
95
+
96
+ ### Edit files with sed:\`\`\`bash
97
+ # Replace all occurrences
98
+ sed -i 's/old_string/new_string/g' filename.py
99
+
100
+ # Replace only first occurrence
101
+ sed -i 's/old_string/new_string/' filename.py
102
+
103
+ # Replace first occurrence on line 1
104
+ sed -i '1s/old_string/new_string/' filename.py
105
+
106
+ # Replace all occurrences in lines 1-10
107
+ sed -i '1,10s/old_string/new_string/g' filename.py
108
+ \`\`\`
109
+
110
+ ### View file content:
111
+
112
+ \`\`\`bash
113
+ # View specific lines with numbers
114
+ nl -ba filename.py | sed -n '10,20p'
115
+ \`\`\`
116
+
117
+ ### Any other command you want to run
118
+
119
+ \`\`\`bash
120
+ anything
121
+ \`\`\`
122
+ `;
123
+ }
124
+ const BASH_BLOCK = /```bash\n(.*?)\n```/gs;
125
+ /**
126
+ * Exactly one fenced bash block is an action; zero or many is a format error.
127
+ * The scaffold trims the command, so a block padded with blank lines executes
128
+ * the same command as an unpadded one.
129
+ */
130
+ function parseAction(assistantMessage) {
131
+ const blocks = [...assistantMessage.matchAll(BASH_BLOCK)].map((match) => match[1] ?? "");
132
+ if (blocks.length !== 1) return {
133
+ kind: "format-error",
134
+ actionCount: blocks.length
135
+ };
136
+ return {
137
+ kind: "action",
138
+ command: (blocks[0] ?? "").trim()
139
+ };
140
+ }
141
+ /**
142
+ * The observation the agent reads after a command. Short outputs are shown
143
+ * whole; long ones keep the first and last `OUTPUT_ELISION_WINDOW` characters
144
+ * with the dropped count between them.
145
+ */
146
+ function renderObservation(output) {
147
+ const head = output.exceptionInfo ? `<exception>${output.exceptionInfo}</exception>\n` : "";
148
+ const returncode = `<returncode>${output.returncode}</returncode>\n`;
149
+ if (output.output.length < 1e4) return `${head}${returncode}<output>\n${output.output}</output>`;
150
+ const elided = output.output.length - OUTPUT_ELISION_THRESHOLD;
151
+ return `${head}${returncode}<warning>\nThe output of your last command was too long.
152
+ Please try a different command that produces less output.
153
+ If you're looking at a file you can try use head, tail or sed to view a smaller number of lines selectively.
154
+ If you're using grep or find and it produced too much output, you can use a more selective search pattern.
155
+ If you really need to see something from the full command's output, you can redirect output to a file and then search in that file.
156
+ </warning><output_head>\n${output.output.slice(0, OUTPUT_ELISION_WINDOW)}\n</output_head>\n<elided_chars>\n${elided} characters elided\n</elided_chars>\n<output_tail>\n${output.output.slice(-5e3)}\n</output_tail>`;
157
+ }
158
+ /** The observation after the environment killed a command for exceeding its timeout. */
159
+ function renderTimeoutObservation(command, partialOutput) {
160
+ return `The last command <command>${command}</command> timed out and has been killed.\nThe output of the command was:\n <output>\n${partialOutput}\n</output>\nPlease try another command and make sure to avoid those requiring interactive input.`;
161
+ }
162
+ /** Substring `renderTimeoutObservation` always writes, whatever the command was. */
163
+ const TIMEOUT_OBSERVATION_MARKER = "timed out and has been killed";
164
+ /**
165
+ * True when the recording shows the environment killed this step at its
166
+ * wall-clock bound.
167
+ *
168
+ * Such a step carries no returncode, so no replay can confirm or contradict
169
+ * it. Callers use this to bound the replay of that step cheaply rather than to
170
+ * decide agreement.
171
+ */
172
+ function isRecordedTimeout(observation) {
173
+ return observation?.includes(TIMEOUT_OBSERVATION_MARKER) === true;
174
+ }
175
+ /** The observation after a turn that did not contain exactly one bash block. */
176
+ function renderFormatErrorObservation(actionCount) {
177
+ return `Please always provide EXACTLY ONE action in triple backticks, found ${actionCount} actions.\nIf you want to end the task, please issue the following command: \`echo ${SUBMIT_SENTINEL}\`\nwithout any other command.
178
+ Else, please format your response exactly as follows:
179
+
180
+ <response_example>
181
+ Here are some thoughts about why you want to perform the action.
182
+
183
+ \`\`\`bash
184
+ <action>
185
+ \`\`\`
186
+ </response_example>
187
+
188
+ Note: In rare cases, if you need to reference a similar format in your command, you might have
189
+ to proceed in two steps, first writing TRIPLEBACKTICKSBASH, then replacing them with \`\`\`bash.`;
190
+ }
191
+ /**
192
+ * The submission text when this output ends the run, `null` otherwise.
193
+ * A non-zero exit does not submit even when the sentinel is echoed, so an
194
+ * agent cannot end the run through a command that failed.
195
+ */
196
+ function submissionOf(output) {
197
+ if (output.returncode !== 0) return null;
198
+ const lines = output.output.replace(/^\s+/, "").split(/(?<=\n)/);
199
+ const first = lines[0];
200
+ if (first === void 0 || first.trim() !== "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT") return null;
201
+ return lines.slice(1).join("");
202
+ }
203
+ //#endregion
204
+ //#region src/trace-repair/action-budget.ts
205
+ /**
206
+ * The action budget an intervention must fit inside.
207
+ *
208
+ * The analyst answers with one action, applied at step k, drawn from the same
209
+ * action space the scaffold had: one shell command or one edit, at most 4 KB.
210
+ * An answer that buys a bigger action than the scaffold could take is not a
211
+ * counterfactual about the recorded run, so the budget is enforced before any
212
+ * container is opened and a violation never reaches the reproduction gate.
213
+ *
214
+ * "One action" is decided by counting top-level statements, not lines. A
215
+ * command list joined by `&&`, `||` or a pipe is one statement, because that
216
+ * is one thing the shell runs and one thing the scaffold could have typed.
217
+ * Two statements separated by a newline or `;` are two actions and are
218
+ * rejected. Heredoc bodies, comments and compound blocks (`if`, `for`,
219
+ * `while`, `until`, `case`, `{ … }`) are inside a statement, never separators.
220
+ */
221
+ /** The scaffold's own per-action budget, pre-registered for the campaign. */
222
+ const SCAFFOLD_INTERVENTION_BUDGET = Object.freeze({
223
+ maxBytes: 4096,
224
+ maxStatements: 1,
225
+ maxHeredocs: 1
226
+ });
227
+ /**
228
+ * Actions whose only effect is to consume a turn. They are rejected before a
229
+ * container opens: an intervention that changes nothing is measurably
230
+ * identical to the no-op control, and paying rollouts to rediscover that
231
+ * wastes the corpus.
232
+ */
233
+ const NO_OP_ACTIONS = Object.freeze([
234
+ ":",
235
+ "true",
236
+ "/bin/true",
237
+ "exit",
238
+ "exit 0"
239
+ ]);
240
+ const BLOCK_OPENERS = /* @__PURE__ */ new Set([
241
+ "if",
242
+ "for",
243
+ "while",
244
+ "until",
245
+ "case",
246
+ "select",
247
+ "do",
248
+ "then"
249
+ ]);
250
+ const BLOCK_CLOSERS = /* @__PURE__ */ new Set([
251
+ "fi",
252
+ "done",
253
+ "esac"
254
+ ]);
255
+ /**
256
+ * Split a shell script into top-level statements and count its heredocs.
257
+ *
258
+ * The scan tracks quoting, escapes, command substitution, brace and paren
259
+ * grouping, comments, compound-block keywords, and heredoc bodies. Anything
260
+ * it cannot resolve stays inside the current statement, so an unparseable
261
+ * action reads as one oversized statement and is rejected on bytes rather
262
+ * than silently accepted as one clean action.
263
+ */
264
+ function scanShellAction(script) {
265
+ const state = {
266
+ statements: [],
267
+ current: "",
268
+ heredocs: 0
269
+ };
270
+ let index = 0;
271
+ let parenDepth = 0;
272
+ let braceDepth = 0;
273
+ let blockDepth = 0;
274
+ let inSingle = false;
275
+ let inDouble = false;
276
+ let inBacktick = false;
277
+ let pendingHeredocs = [];
278
+ let word = "";
279
+ let atStatementStart = true;
280
+ const flushWord = () => {
281
+ if (word.length === 0) return;
282
+ if (BLOCK_OPENERS.has(word)) blockDepth += 1;
283
+ else if (BLOCK_CLOSERS.has(word)) blockDepth = Math.max(0, blockDepth - 1);
284
+ word = "";
285
+ };
286
+ const endStatement = () => {
287
+ flushWord();
288
+ const text = state.current.trim();
289
+ if (text.length > 0 && !isCommentOnly(text)) state.statements.push(text);
290
+ state.current = "";
291
+ atStatementStart = true;
292
+ };
293
+ while (index < script.length) {
294
+ const char = script[index];
295
+ if (inSingle) {
296
+ state.current += char;
297
+ if (char === "'") inSingle = false;
298
+ index += 1;
299
+ continue;
300
+ }
301
+ if (char === "\\" && index + 1 < script.length) {
302
+ const next = script[index + 1];
303
+ if (next === "\n" && !inDouble) {
304
+ state.current += char + next;
305
+ index += 2;
306
+ continue;
307
+ }
308
+ state.current += char + next;
309
+ index += 2;
310
+ continue;
311
+ }
312
+ if (inDouble) {
313
+ state.current += char;
314
+ if (char === "\"") inDouble = false;
315
+ index += 1;
316
+ continue;
317
+ }
318
+ if (char === "'") {
319
+ flushWord();
320
+ inSingle = true;
321
+ state.current += char;
322
+ index += 1;
323
+ atStatementStart = false;
324
+ continue;
325
+ }
326
+ if (char === "\"") {
327
+ flushWord();
328
+ inDouble = true;
329
+ state.current += char;
330
+ index += 1;
331
+ atStatementStart = false;
332
+ continue;
333
+ }
334
+ if (char === "`") {
335
+ inBacktick = !inBacktick;
336
+ state.current += char;
337
+ index += 1;
338
+ atStatementStart = false;
339
+ continue;
340
+ }
341
+ if (char === "#" && (atStatementStart || /\s/.test(script[index - 1] ?? " "))) {
342
+ const end = script.indexOf("\n", index);
343
+ const stop = end === -1 ? script.length : end;
344
+ state.current += script.slice(index, stop);
345
+ index = stop;
346
+ continue;
347
+ }
348
+ if (char === "<" && script[index + 1] === "<" && script[index + 2] !== "<" && script[index - 1] !== "<") {
349
+ const mark = readHeredocMark(script, index);
350
+ if (mark) {
351
+ pendingHeredocs.push(mark.mark);
352
+ state.heredocs += 1;
353
+ state.current += script.slice(index, mark.nextIndex);
354
+ index = mark.nextIndex;
355
+ atStatementStart = false;
356
+ continue;
357
+ }
358
+ }
359
+ if (char === "(") {
360
+ flushWord();
361
+ parenDepth += 1;
362
+ state.current += char;
363
+ index += 1;
364
+ atStatementStart = true;
365
+ continue;
366
+ }
367
+ if (char === ")") {
368
+ flushWord();
369
+ parenDepth = Math.max(0, parenDepth - 1);
370
+ state.current += char;
371
+ index += 1;
372
+ atStatementStart = false;
373
+ continue;
374
+ }
375
+ if (char === "{") {
376
+ flushWord();
377
+ braceDepth += 1;
378
+ state.current += char;
379
+ index += 1;
380
+ atStatementStart = true;
381
+ continue;
382
+ }
383
+ if (char === "}") {
384
+ flushWord();
385
+ braceDepth = Math.max(0, braceDepth - 1);
386
+ state.current += char;
387
+ index += 1;
388
+ atStatementStart = false;
389
+ continue;
390
+ }
391
+ if (char === "\n") {
392
+ flushWord();
393
+ if (pendingHeredocs.length > 0) {
394
+ const consumed = consumeHeredocBodies(script, index + 1, pendingHeredocs);
395
+ state.current += script.slice(index, consumed);
396
+ pendingHeredocs = [];
397
+ index = consumed;
398
+ continue;
399
+ }
400
+ if (parenDepth > 0 || braceDepth > 0 || blockDepth > 0 || inBacktick || endsWithContinuation(state.current)) {
401
+ state.current += char;
402
+ index += 1;
403
+ atStatementStart = true;
404
+ continue;
405
+ }
406
+ endStatement();
407
+ index += 1;
408
+ continue;
409
+ }
410
+ if (char === ";") {
411
+ flushWord();
412
+ if (parenDepth > 0 || braceDepth > 0 || blockDepth > 0 || inBacktick) {
413
+ state.current += char;
414
+ index += 1;
415
+ atStatementStart = true;
416
+ continue;
417
+ }
418
+ endStatement();
419
+ index += 1;
420
+ continue;
421
+ }
422
+ if (/\s/.test(char)) {
423
+ flushWord();
424
+ state.current += char;
425
+ index += 1;
426
+ continue;
427
+ }
428
+ if (char === "&" || char === "|") {
429
+ flushWord();
430
+ state.current += char;
431
+ index += 1;
432
+ atStatementStart = true;
433
+ continue;
434
+ }
435
+ word += char;
436
+ state.current += char;
437
+ atStatementStart = false;
438
+ index += 1;
439
+ }
440
+ endStatement();
441
+ return {
442
+ statements: state.statements,
443
+ heredocs: state.heredocs
444
+ };
445
+ }
446
+ function isCommentOnly(text) {
447
+ return text.split("\n").every((line) => line.trim().length === 0 || line.trim().startsWith("#"));
448
+ }
449
+ function endsWithContinuation(current) {
450
+ const trimmed = current.trimEnd();
451
+ return /(&&|\|\||\||&|\\)$/.test(trimmed);
452
+ }
453
+ function readHeredocMark(script, index) {
454
+ let cursor = index + 2;
455
+ let stripTabs = false;
456
+ if (script[cursor] === "-") {
457
+ stripTabs = true;
458
+ cursor += 1;
459
+ }
460
+ while (script[cursor] === " " || script[cursor] === " ") cursor += 1;
461
+ const quote = script[cursor];
462
+ if (quote === "'" || quote === "\"") {
463
+ const close = script.indexOf(quote, cursor + 1);
464
+ if (close === -1) return null;
465
+ return {
466
+ mark: {
467
+ delimiter: script.slice(cursor + 1, close),
468
+ stripTabs,
469
+ quoted: true
470
+ },
471
+ nextIndex: close + 1
472
+ };
473
+ }
474
+ const match = /^[A-Za-z_][A-Za-z0-9_]*/.exec(script.slice(cursor));
475
+ if (!match) return null;
476
+ return {
477
+ mark: {
478
+ delimiter: match[0],
479
+ stripTabs,
480
+ quoted: false
481
+ },
482
+ nextIndex: cursor + match[0].length
483
+ };
484
+ }
485
+ /** Consume every pending heredoc body; returns the offset just past the last
486
+ * terminator, or the end of the script when a terminator never arrives. */
487
+ function consumeHeredocBodies(script, from, marks) {
488
+ let cursor = from;
489
+ for (const mark of marks) {
490
+ let closed = false;
491
+ while (cursor < script.length) {
492
+ const lineEnd = script.indexOf("\n", cursor);
493
+ const stop = lineEnd === -1 ? script.length : lineEnd;
494
+ const line = script.slice(cursor, stop);
495
+ const candidate = mark.stripTabs ? line.replace(/^\t+/, "") : line;
496
+ cursor = lineEnd === -1 ? script.length : lineEnd + 1;
497
+ if (candidate === mark.delimiter) {
498
+ closed = true;
499
+ break;
500
+ }
501
+ }
502
+ if (!closed) return script.length;
503
+ }
504
+ return cursor;
505
+ }
506
+ /** Payload shape of an action: `edit` when it authors file content through a
507
+ * heredoc, `shell` otherwise. */
508
+ function classifyActionPayload(action) {
509
+ return scanShellAction(action).heredocs > 0 ? "edit" : "shell";
510
+ }
511
+ /**
512
+ * Measure an action against the budget.
513
+ *
514
+ * The budget bounds what the scaffold can execute: one top-level statement,
515
+ * one authored file, a byte cap, and neither a no-op nor a submit. Every
516
+ * rejection here is one of those.
517
+ *
518
+ * `declaredKind` is what the analyst called its own action. It is recorded
519
+ * beside the measured payload and never rejected on, because the scaffold runs
520
+ * the action identically either way — so rejecting the label scores an arm on
521
+ * how it described a repair rather than on the repair. A reader who wants the
522
+ * mismatch counts `declared` against `payload`.
523
+ */
524
+ function checkInterventionBudget(action, declaredKind, budget = SCAFFOLD_INTERVENTION_BUDGET) {
525
+ assertBudget(budget);
526
+ const scan = scanShellAction(action);
527
+ const payload = scan.heredocs > 0 ? "edit" : "shell";
528
+ const measurement = {
529
+ bytes: Buffer.byteLength(action, "utf8"),
530
+ statements: scan.statements.length,
531
+ heredocs: scan.heredocs,
532
+ payload,
533
+ declared: declaredKind
534
+ };
535
+ const reject = (violation, detail) => ({
536
+ admissible: false,
537
+ violation,
538
+ detail,
539
+ measurement
540
+ });
541
+ if (action.trim().length === 0) return reject("empty", "the intervention is empty");
542
+ if (measurement.bytes > budget.maxBytes) return reject("over-byte-cap", `${measurement.bytes} bytes exceeds the ${budget.maxBytes}-byte action budget`);
543
+ if (measurement.statements > budget.maxStatements) return reject("multiple-statements", `${measurement.statements} top-level statements exceeds the ${budget.maxStatements} the scaffold takes per action`);
544
+ if (measurement.heredocs > budget.maxHeredocs) return reject("multiple-heredocs", `${measurement.heredocs} heredocs exceeds the ${budget.maxHeredocs} one edit may author`);
545
+ if (action.includes("COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT")) return reject("submit-instead-of-repair", "the action submits the run instead of repairing it");
546
+ if (NO_OP_ACTIONS.includes(action.trim())) return reject("no-op-action", `"${action.trim()}" changes nothing`);
547
+ return {
548
+ admissible: true,
549
+ measurement
550
+ };
551
+ }
552
+ function assertBudget(budget) {
553
+ for (const field of [
554
+ "maxBytes",
555
+ "maxStatements",
556
+ "maxHeredocs"
557
+ ]) {
558
+ const value = budget[field];
559
+ if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`intervention budget ${field} must be a positive integer, got ${value}`);
560
+ }
561
+ }
562
+ /**
563
+ * Whitespace-insensitive comparison used to reject an intervention that is
564
+ * the recorded action again. Trailing whitespace and blank lines are the only
565
+ * differences a re-proposal can carry without changing what runs.
566
+ */
567
+ function normalizeActionForComparison(action) {
568
+ return action.split("\n").map((line) => line.trimEnd()).filter((line, index, lines) => line.length > 0 || index > 0 && index < lines.length - 1).join("\n").trim();
569
+ }
570
+ //#endregion
571
+ //#region src/trace-repair/admission-records.ts
572
+ /**
573
+ * The vocabulary of the TB-Repair admission pre-pass: what a corpus row is,
574
+ * which failure population it belongs to, why it can leave the funnel, and the
575
+ * denominator chain those exclusions add up to.
576
+ *
577
+ * Pure data and pure functions. The gate that spends containers on these types
578
+ * lives in `admission.ts`, and the artifact that publishes them lives in
579
+ * `admission-report.ts`.
580
+ */
581
+ /** A campaign scored rows the pre-pass did not admit, or dropped rows it did. */
582
+ var AdmissionDenominatorError = class extends CaptureIntegrityError {};
583
+ /** A row handed to the pre-pass carried a field only an analyst could produce. */
584
+ var AdmissionIndependenceError = class extends CaptureIntegrityError {};
585
+ const ADMISSION_ROW_KEYS = [
586
+ "rowId",
587
+ "taskName",
588
+ "recordedModel",
589
+ "recordedCommands",
590
+ "finalReturncode"
591
+ ];
592
+ /**
593
+ * Reject rows that carry anything beyond the recording.
594
+ *
595
+ * The check is a closed key list rather than a list of known analyst field
596
+ * names, because the failure to catch is "some new analyst output leaked into
597
+ * the gate", and only a closed shape catches the fields nobody thought of.
598
+ */
599
+ function assertAnalystIndependent(rows) {
600
+ const allowed = new Set(ADMISSION_ROW_KEYS);
601
+ for (const row of rows) {
602
+ const extra = Object.keys(row).filter((key) => !allowed.has(key));
603
+ if (extra.length > 0) throw new AdmissionIndependenceError(`admission row ${String(row.rowId)} carries analyst-side fields: ${extra.sort().join(", ")}`);
604
+ }
605
+ }
606
+ const ADMISSION_STRATA = [
607
+ "clean-exit",
608
+ "command-error",
609
+ "signal-kill"
610
+ ];
611
+ /** `null` when the recording holds no parseable final return code. */
612
+ function stratumOf(finalReturncode) {
613
+ if (finalReturncode === null || !Number.isInteger(finalReturncode)) return null;
614
+ if (finalReturncode === 0) return "clean-exit";
615
+ return finalReturncode < 0 ? "signal-kill" : "command-error";
616
+ }
617
+ const ADMISSION_EXCLUSION_ORDER = [
618
+ "no-recorded-commands",
619
+ "unparseable-final-returncode",
620
+ "stratum-not-admitted",
621
+ "task-oracle-uncertified",
622
+ "task-oracle-nondeterministic",
623
+ "prefix-replay-error",
624
+ "prefix-replay-empty",
625
+ "prefix-replay-truncated",
626
+ "prefix-divergence-above-threshold",
627
+ "end-state-oracle-error",
628
+ "end-state-tests-pass",
629
+ "no-fix-control-error",
630
+ "no-fix-control-rescued",
631
+ "no-op-control-error",
632
+ "no-op-control-rescued"
633
+ ];
634
+ /** Reasons decided from the recording alone, before a row can be stratified. */
635
+ const PRE_STRATUM_REASONS = ["no-recorded-commands", "unparseable-final-returncode"];
636
+ function isPreStratumReason(reason) {
637
+ return PRE_STRATUM_REASONS.includes(reason);
638
+ }
639
+ /** One line of prose per reason, for the rendered chain. */
640
+ const ADMISSION_EXCLUSION_MEANING = Object.freeze({
641
+ "no-recorded-commands": "the recording holds no command to substitute",
642
+ "unparseable-final-returncode": "the last observation carries no return code",
643
+ "stratum-not-admitted": "the row belongs to a population this campaign excluded",
644
+ "task-oracle-uncertified": "the task grader has no determinism certification on file",
645
+ "task-oracle-nondeterministic": "the task grader returns different verdicts on byte-identical state",
646
+ "prefix-replay-error": "the replay boundary failed, so divergence is unmeasured",
647
+ "prefix-replay-empty": "the replay executed no recorded step",
648
+ "prefix-replay-truncated": "the replay stopped short of the recorded end state",
649
+ "prefix-divergence-above-threshold": "replaying the prefix did not reproduce the recording",
650
+ "end-state-oracle-error": "the task grader failed, so the end state is unjudged",
651
+ "end-state-tests-pass": "the task tests pass on the recorded end state, so nothing failed",
652
+ "no-fix-control-error": "a no-fix rollout failed to run, so the control is unmeasured",
653
+ "no-fix-control-rescued": "the continuation policy passes the task with no intervention",
654
+ "no-op-control-error": "a no-op rollout failed to run, so the control is unmeasured",
655
+ "no-op-control-rescued": "an inert action plus continuation passes the task"
656
+ });
657
+ /**
658
+ * Build the funnel every campaign report publishes.
659
+ *
660
+ * Each stage names the reason, the rows that reached it, the rows it removed,
661
+ * and the rows that survived, so `input = admitted + sum(excluded)` can be read
662
+ * off the table instead of trusted.
663
+ */
664
+ function buildDenominatorChain(verdicts, admitStrata) {
665
+ const reasonTotals = emptyReasonTotals();
666
+ for (const verdict of verdicts) if (verdict.excludedBy !== null) reasonTotals[verdict.excludedBy] += 1;
667
+ const stratumReasons = ADMISSION_EXCLUSION_ORDER.filter((reason) => !isPreStratumReason(reason));
668
+ const artifact = {
669
+ version: 1,
670
+ overall: chainOf("all", verdicts, ADMISSION_EXCLUSION_ORDER),
671
+ byStratum: ADMISSION_STRATA.filter((stratum) => verdicts.some((verdict) => verdict.stratum === stratum)).map((stratum) => chainOf(stratum, verdicts.filter((verdict) => verdict.stratum === stratum), stratumReasons)),
672
+ reasonTotals,
673
+ unstratified: verdicts.filter((verdict) => verdict.stratum === null).length,
674
+ admitStrata: [...admitStrata]
675
+ };
676
+ assertChainReconciles(artifact);
677
+ return artifact;
678
+ }
679
+ function chainOf(scope, verdicts, reasons) {
680
+ let remaining = verdicts.length;
681
+ const stages = [];
682
+ for (const reason of reasons) {
683
+ const excluded = verdicts.filter((verdict) => verdict.excludedBy === reason).length;
684
+ const entering = remaining;
685
+ remaining -= excluded;
686
+ stages.push({
687
+ reason,
688
+ entering,
689
+ excluded,
690
+ remaining
691
+ });
692
+ }
693
+ return {
694
+ scope,
695
+ input: verdicts.length,
696
+ stages,
697
+ admitted: verdicts.filter((verdict) => verdict.admitted).length
698
+ };
699
+ }
700
+ function emptyReasonTotals() {
701
+ const totals = {};
702
+ for (const reason of ADMISSION_EXCLUSION_ORDER) totals[reason] = 0;
703
+ return totals;
704
+ }
705
+ /** A chain that does not add up is a broken denominator, so this throws. */
706
+ function assertChainReconciles(artifact) {
707
+ for (const chain of [artifact.overall, ...artifact.byStratum]) {
708
+ const excluded = chain.stages.reduce((total, stage) => total + stage.excluded, 0);
709
+ if (chain.input !== chain.admitted + excluded) throw new AdmissionDenominatorError(`denominator chain (${chain.scope}) does not reconcile: input ${chain.input} != admitted ${chain.admitted} + excluded ${excluded}`);
710
+ const last = chain.stages[chain.stages.length - 1];
711
+ if (last && last.remaining !== chain.admitted) throw new AdmissionDenominatorError(`denominator chain (${chain.scope}) ends at ${last.remaining} rows but reports ${chain.admitted} admitted`);
712
+ }
713
+ const stratumInputs = artifact.byStratum.reduce((total, chain) => total + chain.input, 0);
714
+ if (stratumInputs + artifact.unstratified !== artifact.overall.input) throw new AdmissionDenominatorError(`stratum inputs ${stratumInputs} plus ${artifact.unstratified} unstratified do not cover ${artifact.overall.input} input rows`);
715
+ }
716
+ //#endregion
717
+ //#region src/trace-repair/continuation-policy.ts
718
+ /** A rollout ran outside the pinned policy, so its evidence cannot be used. */
719
+ var ContinuationPolicyViolationError = class extends CaptureIntegrityError {};
720
+ /** Two arms did not run the same policy, so their difference is not the intervention. */
721
+ var ContinuationSymmetryError = class extends CaptureIntegrityError {};
722
+ /**
723
+ * Everything the policy fixes except the two a campaign must choose.
724
+ *
725
+ * `commandTimeoutSeconds` is 30 because the recorded runs used the scaffold's
726
+ * own 30-second environment timeout; a longer limit would let the continuation
727
+ * finish commands the recorded agent could not.
728
+ */
729
+ const CONTINUATION_POLICY_DEFAULTS = {
730
+ id: "tb-repair-continuation-v1",
731
+ stepBudget: 20,
732
+ temperature: 0,
733
+ maxTokens: 4096,
734
+ commandTimeoutSeconds: 30,
735
+ maxConsecutiveFormatErrors: 3,
736
+ networkMode: "none",
737
+ scaffold: "mini-swe-agent"
738
+ };
739
+ function definePinnedContinuationPolicy(input) {
740
+ const policy = {
741
+ ...CONTINUATION_POLICY_DEFAULTS,
742
+ ...input
743
+ };
744
+ if (!policy.model.trim()) throw new ValidationError("continuation policy requires a model id");
745
+ if (!Number.isInteger(policy.seed)) throw new ValidationError(`continuation policy seed must be an integer, got ${policy.seed}`);
746
+ requirePositiveInteger$1(policy.stepBudget, "stepBudget");
747
+ requirePositiveInteger$1(policy.maxTokens, "maxTokens");
748
+ requirePositiveInteger$1(policy.commandTimeoutSeconds, "commandTimeoutSeconds");
749
+ requirePositiveInteger$1(policy.maxConsecutiveFormatErrors, "maxConsecutiveFormatErrors");
750
+ if (!Number.isFinite(policy.temperature) || policy.temperature < 0) throw new ValidationError(`continuation policy temperature must be a non-negative number, got ${policy.temperature}`);
751
+ return Object.freeze(policy);
752
+ }
753
+ function requirePositiveInteger$1(value, field) {
754
+ if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`continuation policy ${field} must be a positive integer, got ${value}`);
755
+ }
756
+ /**
757
+ * Hash over the policy and the scaffold text it renders. A changed template
758
+ * changes the digest, so rollouts recorded before and after an edit cannot be
759
+ * pooled by accident.
760
+ */
761
+ function continuationPolicyDigest(policy) {
762
+ return contentHash({
763
+ policy: { ...policy },
764
+ systemMessage: MINI_SWE_SYSTEM_MESSAGE,
765
+ instanceMessage: renderInstanceMessage({
766
+ task: "<task>",
767
+ systemInformation: "<system>"
768
+ }),
769
+ formatErrorObservation: renderFormatErrorObservation(0),
770
+ timeoutObservation: renderTimeoutObservation("<command>", "<output>"),
771
+ observation: renderObservation({
772
+ returncode: 0,
773
+ output: ""
774
+ })
775
+ });
776
+ }
777
+ /**
778
+ * Per-rollout seed. It reads the policy seed, the row, and the rollout index —
779
+ * deliberately not the arm, so paired rollouts across arms draw identically.
780
+ */
781
+ function continuationSeed(policySeed, rowId, rolloutIndex) {
782
+ const input = `${policySeed}:${rowId}:${rolloutIndex}`;
783
+ let hash = 2166136261;
784
+ for (let i = 0; i < input.length; i += 1) {
785
+ hash ^= input.charCodeAt(i);
786
+ hash = Math.imul(hash, 16777619) >>> 0;
787
+ }
788
+ return hash >>> 1;
789
+ }
790
+ /**
791
+ * Run the scaffold forward for `rollouts` independent continuations.
792
+ *
793
+ * Each rollout gets its own environment from the factory, because a rollout
794
+ * mutates the container it runs in and the next one must start from the same
795
+ * state, not from the previous rollout's leftovers.
796
+ */
797
+ async function runContinuation(options) {
798
+ const { policy, arm, rowId, prefix, rollouts, model, environments } = options;
799
+ const clock = options.clock ?? Date.now;
800
+ requirePositiveInteger$1(rollouts, "rollouts");
801
+ assertPrefix(prefix);
802
+ const policyDigest = continuationPolicyDigest(policy);
803
+ const records = [];
804
+ for (let index = 0; index < rollouts; index += 1) records.push(await runOneRollout({
805
+ policy,
806
+ policyDigest,
807
+ arm,
808
+ rowId,
809
+ index,
810
+ prefix,
811
+ model,
812
+ environments,
813
+ clock
814
+ }));
815
+ return records;
816
+ }
817
+ function assertPrefix(prefix) {
818
+ if (prefix.length < 2) throw new ValidationError("continuation prefix needs the system and task messages");
819
+ if (prefix[0]?.role !== "system" || prefix[1]?.role !== "user") throw new ValidationError("continuation prefix must start with a system then a user message");
820
+ if (prefix[prefix.length - 1]?.role !== "user") throw new ValidationError("continuation prefix must end on a user message; an assistant turn with no observation means the replay left an action unanswered");
821
+ }
822
+ async function runOneRollout(input) {
823
+ const { policy, arm, rowId, index, clock } = input;
824
+ const seed = continuationSeed(policy.seed, rowId, index);
825
+ const startedMs = clock();
826
+ const environment = await input.environments.create({
827
+ rowId,
828
+ arm,
829
+ rolloutIndex: index
830
+ });
831
+ try {
832
+ const description = await environment.describe();
833
+ if (description.networkMode !== policy.networkMode) throw new ContinuationPolicyViolationError(`continuation requires network mode "${policy.networkMode}", container ${environment.containerRef} reports "${description.networkMode}"`);
834
+ const messages = [...input.prefix];
835
+ const steps = [];
836
+ let exitStatus = "step-budget-exhausted";
837
+ let submission = null;
838
+ let terminalError;
839
+ let consecutiveFormatErrors = 0;
840
+ for (let step = 1; step <= policy.stepBudget; step += 1) {
841
+ const callStartedMs = clock();
842
+ let response;
843
+ try {
844
+ response = await input.model({
845
+ model: policy.model,
846
+ messages: [...messages],
847
+ seed,
848
+ temperature: policy.temperature,
849
+ maxTokens: policy.maxTokens
850
+ });
851
+ } catch (error) {
852
+ exitStatus = "model-error";
853
+ terminalError = errorMessage(error);
854
+ break;
855
+ }
856
+ const call = {
857
+ servedModel: response.servedModel,
858
+ seed,
859
+ latencyMs: clock() - callStartedMs,
860
+ usage: response.usage,
861
+ costUsd: response.costUsd,
862
+ finishReason: response.finishReason ?? null,
863
+ contentChars: response.content.length
864
+ };
865
+ messages.push({
866
+ role: "assistant",
867
+ content: response.content
868
+ });
869
+ const parsed = parseAction(response.content);
870
+ if (parsed.kind === "format-error") {
871
+ consecutiveFormatErrors += 1;
872
+ const observation = renderFormatErrorObservation(parsed.actionCount);
873
+ messages.push({
874
+ role: "user",
875
+ content: observation
876
+ });
877
+ steps.push({
878
+ step,
879
+ assistantMessage: response.content,
880
+ action: null,
881
+ observation,
882
+ execution: null,
883
+ model: call
884
+ });
885
+ if (consecutiveFormatErrors >= policy.maxConsecutiveFormatErrors) {
886
+ exitStatus = "repeated-format-error";
887
+ break;
888
+ }
889
+ continue;
890
+ }
891
+ consecutiveFormatErrors = 0;
892
+ const execStartedMs = clock();
893
+ let execution;
894
+ try {
895
+ execution = await environment.exec(parsed.command, { timeoutSeconds: policy.commandTimeoutSeconds });
896
+ } catch (error) {
897
+ exitStatus = "environment-error";
898
+ terminalError = errorMessage(error);
899
+ steps.push({
900
+ step,
901
+ assistantMessage: response.content,
902
+ action: parsed.command,
903
+ observation: null,
904
+ execution: null,
905
+ model: call,
906
+ error: errorMessage(error)
907
+ });
908
+ break;
909
+ }
910
+ const execRecord = {
911
+ command: parsed.command,
912
+ returncode: execution.returncode,
913
+ timedOut: execution.timedOut,
914
+ outputChars: execution.output.length,
915
+ durationMs: clock() - execStartedMs
916
+ };
917
+ if (execution.timedOut) {
918
+ const observation = renderTimeoutObservation(parsed.command, execution.output);
919
+ messages.push({
920
+ role: "user",
921
+ content: observation
922
+ });
923
+ steps.push({
924
+ step,
925
+ assistantMessage: response.content,
926
+ action: parsed.command,
927
+ observation,
928
+ execution: execRecord,
929
+ model: call
930
+ });
931
+ continue;
932
+ }
933
+ const submitted = submissionOf(execution);
934
+ if (submitted !== null) {
935
+ exitStatus = "submitted";
936
+ submission = submitted;
937
+ steps.push({
938
+ step,
939
+ assistantMessage: response.content,
940
+ action: parsed.command,
941
+ observation: null,
942
+ execution: execRecord,
943
+ model: call
944
+ });
945
+ break;
946
+ }
947
+ const observation = renderObservation(execution);
948
+ messages.push({
949
+ role: "user",
950
+ content: observation
951
+ });
952
+ steps.push({
953
+ step,
954
+ assistantMessage: response.content,
955
+ action: parsed.command,
956
+ observation,
957
+ execution: execRecord,
958
+ model: call
959
+ });
960
+ }
961
+ const endedMs = clock();
962
+ const rollout = {
963
+ rolloutId: `${rowId}:${arm}:${index}`,
964
+ arm,
965
+ rowId,
966
+ index,
967
+ seed,
968
+ policyDigest: input.policyDigest,
969
+ environmentId: input.environments.id,
970
+ containerRef: environment.containerRef,
971
+ environment: description,
972
+ steps,
973
+ exitStatus,
974
+ submission,
975
+ usage: totalUsage(steps),
976
+ costProvenance: totalCost(steps),
977
+ wallMs: endedMs - startedMs,
978
+ startedAt: new Date(startedMs).toISOString(),
979
+ endedAt: new Date(endedMs).toISOString()
980
+ };
981
+ return terminalError === void 0 ? rollout : {
982
+ ...rollout,
983
+ terminalError
984
+ };
985
+ } finally {
986
+ await environment.dispose();
987
+ }
988
+ }
989
+ function errorMessage(error) {
990
+ return error instanceof Error ? error.message : String(error);
991
+ }
992
+ /**
993
+ * Sum only what the provider reported. A call with no usage raises
994
+ * `callsWithUsage` short of `calls` and clears `captured`, so a partially
995
+ * reported rollout can never read as a fully measured one.
996
+ */
997
+ function totalUsage(steps) {
998
+ let input = 0;
999
+ let output = 0;
1000
+ let reasoning = 0;
1001
+ let cached = 0;
1002
+ let cacheWrite = 0;
1003
+ let sawReasoning = false;
1004
+ let sawCached = false;
1005
+ let sawCacheWrite = false;
1006
+ let callsWithUsage = 0;
1007
+ for (const step of steps) {
1008
+ const usage = step.model.usage;
1009
+ if (!usage) continue;
1010
+ callsWithUsage += 1;
1011
+ input += usage.input;
1012
+ output += usage.output;
1013
+ if (usage.reasoning !== void 0) {
1014
+ reasoning += usage.reasoning;
1015
+ sawReasoning = true;
1016
+ }
1017
+ if (usage.cached !== void 0) {
1018
+ cached += usage.cached;
1019
+ sawCached = true;
1020
+ }
1021
+ if (usage.cacheWrite !== void 0) {
1022
+ cacheWrite += usage.cacheWrite;
1023
+ sawCacheWrite = true;
1024
+ }
1025
+ }
1026
+ const totals = {
1027
+ calls: steps.length,
1028
+ callsWithUsage,
1029
+ captured: steps.length > 0 && callsWithUsage === steps.length,
1030
+ input,
1031
+ output
1032
+ };
1033
+ if (sawReasoning) totals.reasoning = reasoning;
1034
+ if (sawCached) totals.cached = cached;
1035
+ if (sawCacheWrite) totals.cacheWrite = cacheWrite;
1036
+ return totals;
1037
+ }
1038
+ /**
1039
+ * One unpriced call makes the rollout's cost unknown. Summing the rest would
1040
+ * report a number smaller than what was spent.
1041
+ */
1042
+ function totalCost(steps) {
1043
+ if (steps.length === 0) return {
1044
+ kind: "uncaptured",
1045
+ usd: null
1046
+ };
1047
+ let usd = 0;
1048
+ for (const step of steps) {
1049
+ if (step.model.costUsd === null) return {
1050
+ kind: "uncaptured",
1051
+ usd: null
1052
+ };
1053
+ usd += step.model.costUsd;
1054
+ }
1055
+ return {
1056
+ kind: "observed",
1057
+ usd
1058
+ };
1059
+ }
1060
+ /**
1061
+ * Prove the arms ran the same policy. Rollouts paired by row and index must
1062
+ * carry the same policy digest and the same seed; anything else means the
1063
+ * measured difference includes a policy change, not only the intervention.
1064
+ */
1065
+ function assertArmSymmetry(rollouts) {
1066
+ const digests = new Set(rollouts.map((rollout) => rollout.policyDigest));
1067
+ if (digests.size > 1) throw new ContinuationSymmetryError(`arms ran different policies: ${[...digests].sort().join(", ")}`);
1068
+ const seeds = /* @__PURE__ */ new Map();
1069
+ for (const rollout of rollouts) {
1070
+ const key = `${rollout.rowId}:${rollout.index}`;
1071
+ const seen = seeds.get(key);
1072
+ if (!seen) {
1073
+ seeds.set(key, {
1074
+ seed: rollout.seed,
1075
+ arm: rollout.arm
1076
+ });
1077
+ continue;
1078
+ }
1079
+ if (seen.seed !== rollout.seed) throw new ContinuationSymmetryError(`paired rollouts ${key} drew different seeds: ${seen.arm}=${seen.seed}, ${rollout.arm}=${rollout.seed}`);
1080
+ }
1081
+ }
1082
+ //#endregion
1083
+ //#region src/trace-repair/control-policy.ts
1084
+ /**
1085
+ * The control policy admission screens under, declared rather than assumed.
1086
+ *
1087
+ * Admission conditions 3 and 4 ask whether a row is rescued by continuing from
1088
+ * the recorded end state with no intervention, and by continuing after an
1089
+ * action that changes nothing. Both are questions about a policy, and neither
1090
+ * is answerable without knowing what that policy is allowed to do.
1091
+ *
1092
+ * One number decides whether the question can be answered at all. Condition 2
1093
+ * has already graded the recorded end state and found it failing. A control
1094
+ * rollout that makes no model call executes no command, so the container it
1095
+ * grades holds those same bytes — the ones already graded as failing. Under
1096
+ * such a policy a control pass is not a rescue; it is the task's own grader
1097
+ * answering differently about identical state. The condition cannot fire for
1098
+ * the reason it exists, and every row walks through it.
1099
+ *
1100
+ * So the policy is a required, hashed parameter that lands on every admission
1101
+ * decision, and a configuration whose control cannot reach the outcome it
1102
+ * screens for is refused where it is configured rather than passed silently.
1103
+ */
1104
+ /** The declared control cannot produce the outcome the criteria screen for. */
1105
+ var UncalibratedControlError = class extends CaptureIntegrityError {};
1106
+ const CONTROL_SCREENING_MODES = ["enforced", "declared-inert"];
1107
+ /**
1108
+ * A control rollout changes the graded state only by executing something, and
1109
+ * it executes only what a model call asks for. At a zero budget it grades the
1110
+ * bytes it was handed.
1111
+ */
1112
+ function controlCanRescue(stepBudget) {
1113
+ return stepBudget > 0;
1114
+ }
1115
+ function defineControlPolicy(input) {
1116
+ if (!input.id.trim()) throw new ValidationError("control policy requires an id");
1117
+ if (!input.scaffold.trim()) throw new ValidationError("control policy requires a scaffold name");
1118
+ if (!Number.isInteger(input.stepBudget) || input.stepBudget < 0) throw new ValidationError(`control policy stepBudget must be a non-negative integer, got ${input.stepBudget}`);
1119
+ if (!Number.isInteger(input.commandTimeoutSeconds) || input.commandTimeoutSeconds <= 0) throw new ValidationError(`control policy commandTimeoutSeconds must be a positive integer, got ${input.commandTimeoutSeconds}`);
1120
+ if (input.stepBudget === 0 && input.model !== null) throw new ValidationError(`control policy ${input.id} declares model ${input.model} at a zero step budget; a policy that makes no model call must record model: null`);
1121
+ if (input.stepBudget > 0 && !input.model?.trim()) throw new ValidationError(`control policy ${input.id} allows ${input.stepBudget} model call(s) but names no model`);
1122
+ const declaration = {
1123
+ id: input.id,
1124
+ stepBudget: input.stepBudget,
1125
+ scaffold: input.scaffold,
1126
+ model: input.model,
1127
+ commandTimeoutSeconds: input.commandTimeoutSeconds
1128
+ };
1129
+ return Object.freeze({
1130
+ ...declaration,
1131
+ digest: contentHash(declaration),
1132
+ canRescue: controlCanRescue(input.stepBudget)
1133
+ });
1134
+ }
1135
+ /**
1136
+ * Refuse a configuration whose control and screening mode contradict.
1137
+ *
1138
+ * Both directions are faults, and both are silent without this. A screening
1139
+ * control that cannot act passes every row through a condition it can never
1140
+ * fire. A control declared inert that can in fact act hides a real screen
1141
+ * behind a label that says nothing was screened.
1142
+ */
1143
+ function assertControlCalibrated(policy, screening) {
1144
+ if (!CONTROL_SCREENING_MODES.includes(screening)) throw new ValidationError(`unknown control screening mode: ${screening}`);
1145
+ const canRescue = controlCanRescue(policy.stepBudget);
1146
+ if (screening === "enforced" && !canRescue) throw new UncalibratedControlError(`control policy ${policy.id} (digest ${policy.digest}) allows ${policy.stepBudget} model call(s) per rollout, so a control rollout executes no command and grades the same bytes the end-state check already graded as failing. Under it conditions 3 and 4 can only fire on a grader that disagrees with itself, so they screen nothing. Give the control a step budget of at least 1, or set controlScreening to 'declared-inert' and read a control pass as the oracle flip it is.`);
1147
+ if (screening === "declared-inert" && canRescue) throw new UncalibratedControlError(`control policy ${policy.id} (digest ${policy.digest}) allows ${policy.stepBudget} model call(s) per rollout, so it can rescue a row, but the criteria declare it inert. Set controlScreening to 'enforced' so a control pass is recorded as a rescue.`);
1148
+ }
1149
+ //#endregion
1150
+ //#region src/trace-repair/admission.ts
1151
+ const ADMISSION_CONFIG_DEFAULTS = Object.freeze({
1152
+ maxPrefixDivergence: .1,
1153
+ controlRollouts: 3,
1154
+ admitStrata: Object.freeze(["clean-exit", "command-error"]),
1155
+ inertAction: "true",
1156
+ controlScreening: "enforced",
1157
+ concurrency: 1
1158
+ });
1159
+ function resolveAdmissionConfig(input = {}) {
1160
+ const config = {
1161
+ ...ADMISSION_CONFIG_DEFAULTS,
1162
+ ...input
1163
+ };
1164
+ if (!Number.isFinite(config.maxPrefixDivergence)) throw new ValidationError(`admission maxPrefixDivergence must be a number, got ${config.maxPrefixDivergence}`);
1165
+ if (config.maxPrefixDivergence < 0 || config.maxPrefixDivergence > 1) throw new ValidationError(`admission maxPrefixDivergence must be a share between 0 and 1, got ${config.maxPrefixDivergence}`);
1166
+ requirePositiveInteger(config.controlRollouts, "controlRollouts");
1167
+ requirePositiveInteger(config.concurrency, "concurrency");
1168
+ if (config.admitStrata.length === 0) throw new ValidationError("admission admitStrata must name at least one stratum");
1169
+ for (const stratum of config.admitStrata) if (!ADMISSION_STRATA.includes(stratum)) throw new ValidationError(`admission admitStrata holds an unknown stratum: ${stratum}`);
1170
+ if (config.inertAction.trim().length === 0) throw new ValidationError("admission inertAction must be a non-empty command");
1171
+ if (!CONTROL_SCREENING_MODES.includes(config.controlScreening)) throw new ValidationError(`admission controlScreening must be one of ${CONTROL_SCREENING_MODES.join(", ")}, got ${config.controlScreening}`);
1172
+ return Object.freeze({
1173
+ ...config,
1174
+ admitStrata: Object.freeze([...config.admitStrata])
1175
+ });
1176
+ }
1177
+ function requirePositiveInteger(value, field) {
1178
+ if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`admission ${field} must be a positive integer, got ${value}`);
1179
+ }
1180
+ /**
1181
+ * The recorded command the no-op control replaces, drawn per rollout.
1182
+ *
1183
+ * The draw reads the policy seed, the row, and the rollout index, so it is
1184
+ * reproducible from the artifact and cannot depend on when the pre-pass ran.
1185
+ */
1186
+ function noOpInjectionStep(policySeed, rowId, rolloutIndex, recordedCommands) {
1187
+ requirePositiveInteger(recordedCommands, "recordedCommands");
1188
+ const input = `no-op:${policySeed}:${rowId}:${rolloutIndex}`;
1189
+ let hash = 2166136261;
1190
+ for (let i = 0; i < input.length; i += 1) {
1191
+ hash ^= input.charCodeAt(i);
1192
+ hash = Math.imul(hash, 16777619) >>> 0;
1193
+ }
1194
+ return (hash >>> 1) % recordedCommands + 1;
1195
+ }
1196
+ /**
1197
+ * Run the pre-pass over every row and publish the denominator it produced.
1198
+ *
1199
+ * Nothing here reads an analyst output. The report is frozen, and a campaign
1200
+ * proves it measured this denominator by passing the report back to
1201
+ * `assertDenominatorIntact`.
1202
+ */
1203
+ async function runAdmission(options) {
1204
+ const { rows, policy, replayer, oracle, controls, taskOracles } = options;
1205
+ const config = resolveAdmissionConfig(options.config);
1206
+ const clock = options.clock ?? Date.now;
1207
+ assertAnalystIndependent(rows);
1208
+ assertUniqueRowIds(rows);
1209
+ const policyDigest = continuationPolicyDigest(policy);
1210
+ assertControlCalibrated({
1211
+ id: policy.id,
1212
+ digest: policyDigest,
1213
+ stepBudget: policy.stepBudget
1214
+ }, config.controlScreening);
1215
+ const verdicts = await mapOrdered(rows, config.concurrency, (row) => admitRow$1({
1216
+ row,
1217
+ policy,
1218
+ policyDigest,
1219
+ replayer,
1220
+ oracle,
1221
+ controls,
1222
+ taskOracles,
1223
+ config
1224
+ }));
1225
+ const strata = groupAdmittedByStratum(verdicts);
1226
+ const provenance = {
1227
+ replayerId: replayer.id,
1228
+ oracleId: oracle.id,
1229
+ controlRunnerId: controls.id,
1230
+ policyId: policy.id,
1231
+ policyModel: policy.model,
1232
+ policySeed: policy.seed,
1233
+ policyDigest,
1234
+ policyStepBudget: policy.stepBudget,
1235
+ controlScreening: config.controlScreening,
1236
+ certifiedTasks: Object.freeze(Object.fromEntries([...taskOracles].map(([task, verdict]) => [task, verdict.flipRate])))
1237
+ };
1238
+ return Object.freeze({
1239
+ config,
1240
+ provenance,
1241
+ rows: Object.freeze(verdicts),
1242
+ strata,
1243
+ chain: buildDenominatorChain(verdicts, config.admitStrata),
1244
+ controlCost: summarizeControlCost(verdicts),
1245
+ digest: contentHash({
1246
+ config,
1247
+ provenance,
1248
+ admitted: ADMISSION_STRATA.map((stratum) => ({
1249
+ stratum,
1250
+ rowIds: strata[stratum]
1251
+ }))
1252
+ }),
1253
+ generatedAt: new Date(clock()).toISOString()
1254
+ });
1255
+ }
1256
+ function assertUniqueRowIds(rows) {
1257
+ const seen = /* @__PURE__ */ new Set();
1258
+ for (const row of rows) {
1259
+ if (typeof row.rowId !== "string" || row.rowId.length === 0) throw new ValidationError("admission row requires a non-empty rowId");
1260
+ if (seen.has(row.rowId)) throw new ValidationError(`admission received rowId ${row.rowId} twice`);
1261
+ seen.add(row.rowId);
1262
+ }
1263
+ }
1264
+ async function admitRow$1(input) {
1265
+ const { row, config } = input;
1266
+ const checks = [];
1267
+ const rollouts = [];
1268
+ const summaries = [];
1269
+ const certification = input.taskOracles.get(row.taskName) ?? null;
1270
+ const base = {
1271
+ rowId: row.rowId,
1272
+ taskName: row.taskName,
1273
+ recordedModel: row.recordedModel,
1274
+ recordedCommands: row.recordedCommands,
1275
+ finalReturncode: row.finalReturncode,
1276
+ controlPolicyDigest: input.policyDigest,
1277
+ controlScreening: config.controlScreening,
1278
+ oracleFlipRate: certification === null ? null : certification.flipRate
1279
+ };
1280
+ const finish = (stratum, reason, errorDetail = null) => ({
1281
+ ...base,
1282
+ stratum,
1283
+ admitted: reason === null,
1284
+ excludedBy: reason,
1285
+ errorDetail,
1286
+ checks,
1287
+ rollouts: summaries
1288
+ });
1289
+ if (!Number.isInteger(row.recordedCommands) || row.recordedCommands < 1) return finish(null, "no-recorded-commands");
1290
+ const stratum = stratumOf(row.finalReturncode);
1291
+ if (stratum === null) return finish(null, "unparseable-final-returncode");
1292
+ checks.push({
1293
+ check: "stratum",
1294
+ stratum
1295
+ });
1296
+ if (!config.admitStrata.includes(stratum)) return finish(stratum, "stratum-not-admitted");
1297
+ if (certification === null) return finish(stratum, "task-oracle-uncertified");
1298
+ checks.push({
1299
+ check: "task-oracle",
1300
+ stable: certification.stable,
1301
+ flipRate: certification.flipRate,
1302
+ replicates: certification.replicates
1303
+ });
1304
+ if (!certification.stable) return finish(stratum, "task-oracle-nondeterministic", certification.detail);
1305
+ const replay = await input.replayer.replay(row);
1306
+ if (!replay.succeeded) return finish(stratum, "prefix-replay-error", replay.error);
1307
+ const { prefixExecuted, prefixDivergences } = replay.value;
1308
+ if (!Number.isInteger(prefixExecuted) || prefixExecuted < 1) return finish(stratum, "prefix-replay-empty");
1309
+ if (prefixExecuted < row.recordedCommands) return finish(stratum, "prefix-replay-truncated");
1310
+ const divergenceRatio = prefixDivergences.length / prefixExecuted;
1311
+ checks.push({
1312
+ check: "prefix-replay",
1313
+ prefixExecuted,
1314
+ divergences: prefixDivergences.length,
1315
+ divergenceRatio
1316
+ });
1317
+ if (divergenceRatio > config.maxPrefixDivergence) return finish(stratum, "prefix-divergence-above-threshold");
1318
+ const endState = await input.oracle.grade(row);
1319
+ if (!endState.succeeded) return finish(stratum, "end-state-oracle-error", endState.error);
1320
+ checks.push({
1321
+ check: "end-state-tests",
1322
+ passed: endState.value.passed,
1323
+ reward: endState.value.reward
1324
+ });
1325
+ if (endState.value.passed) return finish(stratum, "end-state-tests-pass");
1326
+ for (const arm of CONTROL_ARMS) {
1327
+ const control = await runControlArm({
1328
+ ...input,
1329
+ arm
1330
+ });
1331
+ checks.push(control.record);
1332
+ rollouts.push(...control.rollouts);
1333
+ summaries.push(...control.summaries);
1334
+ if (control.outcome === "error") return finish(stratum, CONTROL_EXCLUSIONS[arm].error, control.error);
1335
+ if (control.outcome === "rescued") return finish(stratum, CONTROL_EXCLUSIONS[arm].rescued);
1336
+ }
1337
+ assertArmSymmetry(rollouts);
1338
+ return finish(stratum, null);
1339
+ }
1340
+ const CONTROL_ARMS = ["no-fix-control", "no-op-control"];
1341
+ const CONTROL_EXCLUSIONS = Object.freeze({
1342
+ "no-fix-control": {
1343
+ error: "no-fix-control-error",
1344
+ rescued: "no-fix-control-rescued"
1345
+ },
1346
+ "no-op-control": {
1347
+ error: "no-op-control-error",
1348
+ rescued: "no-op-control-rescued"
1349
+ }
1350
+ });
1351
+ /**
1352
+ * Run one control arm until it is decided.
1353
+ *
1354
+ * A rollout that passes decides the arm at once, so the remaining rollouts go
1355
+ * unpaid. A boundary failure decides it too, and as an error: counting an
1356
+ * unmeasured rollout as a failure would admit a row nobody verified.
1357
+ */
1358
+ async function runControlArm(input) {
1359
+ const { row, arm, config, policy } = input;
1360
+ const rollouts = [];
1361
+ const summaries = [];
1362
+ const injections = [];
1363
+ let passes = 0;
1364
+ let error = null;
1365
+ for (let index = 0; index < config.controlRollouts; index += 1) {
1366
+ const injection = arm === "no-op-control" ? {
1367
+ step: noOpInjectionStep(policy.seed, row.rowId, index, row.recordedCommands),
1368
+ action: config.inertAction
1369
+ } : null;
1370
+ if (injection) injections.push(injection);
1371
+ const result = await input.controls.run({
1372
+ row,
1373
+ arm,
1374
+ rolloutIndex: index,
1375
+ injection
1376
+ });
1377
+ if (!result.succeeded) {
1378
+ error = result.error;
1379
+ break;
1380
+ }
1381
+ const { rollout, tests } = result.value;
1382
+ assertRolloutMatchesRequest(rollout, {
1383
+ row,
1384
+ arm,
1385
+ index
1386
+ });
1387
+ rollouts.push(rollout);
1388
+ summaries.push({
1389
+ arm,
1390
+ index: rollout.index,
1391
+ seed: rollout.seed,
1392
+ policyDigest: rollout.policyDigest,
1393
+ exitStatus: rollout.exitStatus,
1394
+ testsPassed: tests.passed,
1395
+ costUsd: rollout.costProvenance.usd
1396
+ });
1397
+ if (tests.passed) {
1398
+ passes += 1;
1399
+ break;
1400
+ }
1401
+ }
1402
+ const record = {
1403
+ check: "control",
1404
+ arm,
1405
+ rolloutsRun: rollouts.length,
1406
+ passes,
1407
+ injections
1408
+ };
1409
+ if (error !== null) return {
1410
+ outcome: "error",
1411
+ error,
1412
+ record,
1413
+ rollouts,
1414
+ summaries
1415
+ };
1416
+ if (passes > 0) return {
1417
+ outcome: "rescued",
1418
+ error: null,
1419
+ record,
1420
+ rollouts,
1421
+ summaries
1422
+ };
1423
+ return {
1424
+ outcome: "failed-every-rollout",
1425
+ error: null,
1426
+ record,
1427
+ rollouts,
1428
+ summaries
1429
+ };
1430
+ }
1431
+ function assertRolloutMatchesRequest(rollout, request) {
1432
+ if (rollout.arm !== request.arm) throw new AdmissionDenominatorError(`control runner returned a ${rollout.arm} rollout for the ${request.arm} arm`);
1433
+ if (rollout.rowId !== request.row.rowId) throw new AdmissionDenominatorError(`control runner returned a rollout for row ${rollout.rowId} while deciding ${request.row.rowId}`);
1434
+ if (rollout.index !== request.index) throw new AdmissionDenominatorError(`control runner returned rollout index ${rollout.index} for requested index ${request.index}`);
1435
+ }
1436
+ /**
1437
+ * Model cost of every control rollout that ran.
1438
+ *
1439
+ * One unpriced rollout makes the total uncaptured, because summing the priced
1440
+ * ones reports less than the pre-pass spent. Zero rollouts is an observed zero:
1441
+ * nothing ran, so nothing is missing.
1442
+ */
1443
+ function summarizeControlCost(verdicts) {
1444
+ let usd = 0;
1445
+ for (const verdict of verdicts) for (const rollout of verdict.rollouts) {
1446
+ if (rollout.costUsd === null) return {
1447
+ kind: "uncaptured",
1448
+ usd: null
1449
+ };
1450
+ usd += rollout.costUsd;
1451
+ }
1452
+ return {
1453
+ kind: "observed",
1454
+ usd
1455
+ };
1456
+ }
1457
+ function groupAdmittedByStratum(verdicts) {
1458
+ const grouped = {
1459
+ "clean-exit": [],
1460
+ "command-error": [],
1461
+ "signal-kill": []
1462
+ };
1463
+ for (const verdict of verdicts) {
1464
+ if (!verdict.admitted || verdict.stratum === null) continue;
1465
+ grouped[verdict.stratum].push(verdict.rowId);
1466
+ }
1467
+ return Object.freeze({
1468
+ "clean-exit": Object.freeze(grouped["clean-exit"]),
1469
+ "command-error": Object.freeze(grouped["command-error"]),
1470
+ "signal-kill": Object.freeze(grouped["signal-kill"])
1471
+ });
1472
+ }
1473
+ /** Bounded worker pool that keeps results in input order. */
1474
+ async function mapOrdered(items, concurrency, fn) {
1475
+ const results = new Array(items.length);
1476
+ let next = 0;
1477
+ const workerCount = Math.max(1, Math.min(concurrency, items.length));
1478
+ const workers = Array.from({ length: workerCount }, async () => {
1479
+ for (;;) {
1480
+ const index = next;
1481
+ next += 1;
1482
+ if (index >= items.length) return;
1483
+ const item = items[index];
1484
+ if (item === void 0) return;
1485
+ results[index] = await fn(item);
1486
+ }
1487
+ });
1488
+ await Promise.all(workers);
1489
+ return results;
1490
+ }
1491
+ /** Admitted row ids in one stratum. No call returns them pooled. */
1492
+ function admittedRowIds(report, stratum) {
1493
+ return report.strata[stratum];
1494
+ }
1495
+ function admittedCount(report) {
1496
+ return ADMISSION_STRATA.reduce((total, stratum) => total + report.strata[stratum].length, 0);
1497
+ }
1498
+ /**
1499
+ * Prove a campaign measured the denominator admission published.
1500
+ *
1501
+ * Three ways a denominator moves after the fact, each rejected here: scoring a
1502
+ * row that was never admitted, scoring a row that was never sampled, and
1503
+ * dropping a sampled row instead of scoring it. The third is the one an analyst
1504
+ * can cause on its own — declining a row it cannot solve — so a sampled row
1505
+ * with no outcome is an error, not a smaller `n`.
1506
+ */
1507
+ function assertDenominatorIntact(input) {
1508
+ const { report, strata, sampled, scored } = input;
1509
+ for (const stratum of strata) if (!ADMISSION_STRATA.includes(stratum)) throw new ValidationError(`unknown stratum in denominator check: ${stratum}`);
1510
+ assertNoDuplicates(sampled, "sampled");
1511
+ assertNoDuplicates(scored, "scored");
1512
+ const eligible = new Set(strata.flatMap((stratum) => [...report.strata[stratum]]));
1513
+ const notAdmitted = sampled.filter((rowId) => !eligible.has(rowId));
1514
+ if (notAdmitted.length > 0) throw new AdmissionDenominatorError(`${notAdmitted.length} sampled row(s) are not admitted in strata ${strata.join(", ")}: ${preview(notAdmitted)}`);
1515
+ const sampledSet = new Set(sampled);
1516
+ const unsampled = scored.filter((rowId) => !sampledSet.has(rowId));
1517
+ if (unsampled.length > 0) throw new AdmissionDenominatorError(`${unsampled.length} scored row(s) were never sampled: ${preview(unsampled)}`);
1518
+ const scoredSet = new Set(scored);
1519
+ const dropped = sampled.filter((rowId) => !scoredSet.has(rowId));
1520
+ if (dropped.length > 0) throw new AdmissionDenominatorError(`denominator shrank by ${dropped.length} row(s): sampled but never scored: ${preview(dropped)}`);
1521
+ }
1522
+ /** A row counted twice raises `n` without measuring anything twice. */
1523
+ function assertNoDuplicates(rowIds, field) {
1524
+ const seen = /* @__PURE__ */ new Set();
1525
+ const duplicates = rowIds.filter((rowId) => {
1526
+ if (seen.has(rowId)) return true;
1527
+ seen.add(rowId);
1528
+ return false;
1529
+ });
1530
+ if (duplicates.length > 0) throw new AdmissionDenominatorError(`${field} lists ${duplicates.length} duplicate row id(s): ${preview([...new Set(duplicates)])}`);
1531
+ }
1532
+ function preview(rowIds) {
1533
+ const shown = rowIds.slice(0, 5).join(", ");
1534
+ return rowIds.length > 5 ? `${shown}, +${rowIds.length - 5} more` : shown;
1535
+ }
1536
+ //#endregion
1537
+ //#region src/trace-repair/admission-contract.ts
1538
+ /**
1539
+ * Admission: the executed checks a row must pass before an analyst is allowed
1540
+ * to see it, and the brand that proves it did.
1541
+ *
1542
+ * Every check is analyst-independent. None of them reads a finding, a label,
1543
+ * or a k, and all of them are anchored at the recorded end state — the point
1544
+ * the agent actually stopped at — so the same evidence admits a row no matter
1545
+ * which step an analyst later blames:
1546
+ *
1547
+ * oracle-determinism the task's own suite returns one verdict on identical
1548
+ * bytes, so its answer is about the state
1549
+ * prefix-fidelity the recorded trajectory replays with at most 10 % of
1550
+ * its steps diverging from their recorded returncode
1551
+ * end-state-fails the held-out suite fails on the recorded end state
1552
+ * no-fix-control 3 of 3 continuations from the end state fail
1553
+ * no-op-control 3 of 3 continuations from the end state, after an
1554
+ * action that changes nothing, fail
1555
+ *
1556
+ * The two controls are what make `Delta-repair` a difference rather than a
1557
+ * rate, and what stops a row where the agent was one free step from success
1558
+ * from counting as a repair the analyst caused. They only do that under a
1559
+ * control that can act: `control-policy.ts` holds the declaration, and the
1560
+ * criteria name which reading of a control pass applies.
1561
+ *
1562
+ * The determinism check sits in front of all of it. Every downstream check
1563
+ * reads the same suite, so a suite whose verdict is not a function of the state
1564
+ * makes each of them a coin flip rather than a measurement.
1565
+ *
1566
+ * This module owns the CONTRACT, not the execution. `runAdmission` in
1567
+ * `./admission` executes the checks against real containers and hands the
1568
+ * measured evidence to `admitRow`, which decides and brands. Splitting it that
1569
+ * way keeps the decision auditable from the recorded numbers alone: a reviewer
1570
+ * can re-derive every admission from the evidence file without re-running a
1571
+ * container.
1572
+ */
1573
+ const TB_REPAIR_ADMISSION_CRITERIA = Object.freeze({
1574
+ maxPrefixDivergenceRatio: .1,
1575
+ controlRollouts: 3,
1576
+ controlScreening: "enforced"
1577
+ });
1578
+ /**
1579
+ * Decide admission from executed evidence.
1580
+ *
1581
+ * Pure: it opens no container and calls no model. Every threshold it applies
1582
+ * is in `criteria`, and every number it reads is in `evidence`, so an
1583
+ * admission decision is reproducible from the recorded evidence alone.
1584
+ *
1585
+ * It throws, rather than rejecting, when the criteria and the declared control
1586
+ * contradict. A contradiction there is a property of the configuration and not
1587
+ * of the row, so it must stop the run instead of producing one verdict per row
1588
+ * that reads as if a check had been applied.
1589
+ */
1590
+ function admitRow(evidence, criteria = TB_REPAIR_ADMISSION_CRITERIA) {
1591
+ assertCriteria(criteria);
1592
+ assertControlCalibrated(evidence.controlPolicy, criteria.controlScreening);
1593
+ if (evidence.oracleDeterminism.taskName !== evidence.taskName) throw new ValidationError(`row ${evidence.rowId} is from task ${evidence.taskName} but carries the oracle certification for ${evidence.oracleDeterminism.taskName}`);
1594
+ const screening = {
1595
+ controlPolicy: evidence.controlPolicy,
1596
+ controlScreening: criteria.controlScreening,
1597
+ controlPolicyDigest: evidence.controlPolicy.digest,
1598
+ taskName: evidence.oracleDeterminism.taskName,
1599
+ oracleStable: evidence.oracleDeterminism.stable,
1600
+ oracleFlipRate: evidence.oracleDeterminism.flipRate
1601
+ };
1602
+ const reject = (rejection, detail) => ({
1603
+ admitted: false,
1604
+ rowId: evidence.rowId,
1605
+ screening,
1606
+ rejection,
1607
+ detail
1608
+ });
1609
+ const oracle = evidence.oracleDeterminism;
1610
+ if (!oracle.stable) return reject("task-oracle-nondeterministic", `task ${oracle.taskName} graded byte-identical state inconsistently: flip rate ${(oracle.flipRate * 100).toFixed(1)} % over ${oracle.replicates} replicates (${oracle.detail}). Every check below reads that suite, so none of them measures state.`);
1611
+ if (evidence.steps.length === 0) return reject("empty-trajectory", "the row records no steps");
1612
+ const { stepsReplayed, divergences } = evidence.prefixFidelity;
1613
+ if (stepsReplayed <= 0) return reject("empty-trajectory", "no recorded step was replayed");
1614
+ const prefixDivergenceRatio = divergences / stepsReplayed;
1615
+ if (prefixDivergenceRatio > criteria.maxPrefixDivergenceRatio) return reject("prefix-divergence-too-high", `${divergences}/${stepsReplayed} replayed steps diverged (${(prefixDivergenceRatio * 100).toFixed(1)} %), above the ${(criteria.maxPrefixDivergenceRatio * 100).toFixed(1)} % ceiling`);
1616
+ if (evidence.endStatePassed) return reject("end-state-already-passes", "the held-out suite passes on the recorded end state, so the row records no failure to repair");
1617
+ for (const [name, arm, rescued] of [[
1618
+ "no-fix control",
1619
+ evidence.noFixControl,
1620
+ "no-fix-control-passed"
1621
+ ], [
1622
+ "no-op control",
1623
+ evidence.noOpControl,
1624
+ "no-op-control-passed"
1625
+ ]]) {
1626
+ if (arm.rollouts !== criteria.controlRollouts) return reject("control-rollouts-short", `${name} ran ${arm.rollouts} rollouts, the criteria pre-register ${criteria.controlRollouts}`);
1627
+ if (arm.policyDigest !== evidence.controlPolicy.digest) return reject("control-policy-mismatch", `${name} reported policy ${arm.policyDigest} but the criteria screen under ${evidence.controlPolicy.id} (${evidence.controlPolicy.digest}); the arm did not run the declared control`);
1628
+ if (arm.passes !== 0) return criteria.controlScreening === "enforced" ? reject(rescued, `${name} passed ${arm.passes}/${arm.rollouts}; the row is repairable by continuing alone`) : reject("control-passed-on-identical-state", `${name} passed ${arm.passes}/${arm.rollouts} under ${evidence.controlPolicy.id}, which executes no command, so it graded the same bytes the end-state check graded as failing. The task's certification reports flip rate ${(oracle.flipRate * 100).toFixed(1)} %; this row is a further flip, not a rescue.`);
1629
+ }
1630
+ return {
1631
+ admitted: true,
1632
+ screening,
1633
+ row: {
1634
+ rowId: evidence.rowId,
1635
+ taskName: evidence.taskName,
1636
+ image: evidence.image,
1637
+ cwd: evidence.cwd,
1638
+ taskStatement: evidence.taskStatement,
1639
+ steps: evidence.steps,
1640
+ criteria,
1641
+ prefixDivergenceRatio,
1642
+ suiteDigest: evidence.suiteDigest,
1643
+ controlPolicy: evidence.controlPolicy,
1644
+ controlScreening: criteria.controlScreening,
1645
+ policyDigest: evidence.controlPolicy.digest,
1646
+ controlRate: evidence.noFixControl.passes / evidence.noFixControl.rollouts,
1647
+ controlRollouts: evidence.noFixControl.rollouts
1648
+ }
1649
+ };
1650
+ }
1651
+ function assertCriteria(criteria) {
1652
+ const { maxPrefixDivergenceRatio, controlRollouts } = criteria;
1653
+ if (!(maxPrefixDivergenceRatio >= 0 && maxPrefixDivergenceRatio <= 1)) throw new ValidationError(`admission maxPrefixDivergenceRatio must be within [0,1], got ${maxPrefixDivergenceRatio}`);
1654
+ if (!Number.isInteger(controlRollouts) || controlRollouts <= 0) throw new ValidationError(`admission controlRollouts must be a positive integer, got ${controlRollouts}`);
1655
+ }
1656
+ //#endregion
1657
+ //#region src/trace-repair/admission-report.ts
1658
+ /** Plain JSON. `JSON.stringify` of this object is the machine-readable artifact. */
1659
+ function admissionArtifact(report) {
1660
+ assertChainReconciles(report.chain);
1661
+ return {
1662
+ version: 1,
1663
+ kind: "tb-repair-admission",
1664
+ generatedAt: report.generatedAt,
1665
+ digest: report.digest,
1666
+ config: report.config,
1667
+ provenance: report.provenance,
1668
+ chain: report.chain,
1669
+ admitted: {
1670
+ "clean-exit": report.strata["clean-exit"],
1671
+ "command-error": report.strata["command-error"],
1672
+ "signal-kill": report.strata["signal-kill"]
1673
+ },
1674
+ controlCost: {
1675
+ kind: report.controlCost.kind,
1676
+ usd: report.controlCost.usd
1677
+ },
1678
+ rows: report.rows
1679
+ };
1680
+ }
1681
+ /** Markdown for the campaign report. Every number here is also in the artifact. */
1682
+ function renderAdmissionReport(artifact, options = {}) {
1683
+ const lines = [];
1684
+ lines.push("# TB-Repair admission", "");
1685
+ lines.push(`${artifact.chain.overall.admitted} of ${artifact.chain.overall.input} rows admitted. Digest \`${artifact.digest}\`.`, "");
1686
+ lines.push(...provenanceSection(artifact));
1687
+ lines.push(...chainSection(artifact.chain.overall, "Denominator chain"));
1688
+ for (const chain of artifact.chain.byStratum) lines.push(...chainSection(chain, `Denominator chain — ${chain.scope}`));
1689
+ lines.push(...stratumSection(artifact));
1690
+ lines.push(...reasonSection(artifact.chain.reasonTotals));
1691
+ const rowLimit = options.rowLimit ?? 0;
1692
+ if (rowLimit > 0) lines.push(...rowSection(artifact.rows, rowLimit));
1693
+ return `${lines.join("\n").trimEnd()}\n`;
1694
+ }
1695
+ function provenanceSection(artifact) {
1696
+ const { provenance, config, controlCost } = artifact;
1697
+ const cost = controlCost.usd === null ? "uncaptured" : `$${controlCost.usd.toFixed(4)}`;
1698
+ return [
1699
+ "## Provenance",
1700
+ "",
1701
+ "| field | value |",
1702
+ "| --- | --- |",
1703
+ `| generated | ${artifact.generatedAt} |`,
1704
+ `| prefix replayer | \`${provenance.replayerId}\` |`,
1705
+ `| end-state oracle | \`${provenance.oracleId}\` |`,
1706
+ `| control runner | \`${provenance.controlRunnerId}\` |`,
1707
+ `| continuation policy | \`${provenance.policyId}\` |`,
1708
+ `| policy model | \`${provenance.policyModel}\` |`,
1709
+ `| policy seed | ${provenance.policySeed} |`,
1710
+ `| policy digest | \`${provenance.policyDigest}\` |`,
1711
+ `| policy step budget | ${provenance.policyStepBudget} model call(s) per control rollout |`,
1712
+ `| control screening | \`${provenance.controlScreening}\` |`,
1713
+ `| certified task oracles | ${formatCertifiedTasks(provenance.certifiedTasks)} |`,
1714
+ `| max prefix divergence | ${formatShare(config.maxPrefixDivergence)} |`,
1715
+ `| control rollouts per arm | ${config.controlRollouts} |`,
1716
+ `| inert action | \`${config.inertAction}\` |`,
1717
+ `| strata admitted | ${config.admitStrata.join(", ")} |`,
1718
+ `| control model cost | ${cost} (${controlCost.kind}) |`,
1719
+ ""
1720
+ ];
1721
+ }
1722
+ function formatCertifiedTasks(certified) {
1723
+ const entries = Object.entries(certified).sort(([a], [b]) => a.localeCompare(b));
1724
+ if (entries.length === 0) return "none";
1725
+ return entries.map(([task, flipRate]) => `${task} (flip ${formatShare(flipRate)})`).join(", ");
1726
+ }
1727
+ function chainSection(chain, heading) {
1728
+ const lines = [
1729
+ `## ${heading}`,
1730
+ "",
1731
+ "| stage | exclusion reason | entering | excluded | remaining |",
1732
+ "| --- | --- | --- | --- | --- |"
1733
+ ];
1734
+ chain.stages.forEach((stage, index) => {
1735
+ lines.push(`| ${index + 1} | \`${stage.reason}\` | ${stage.entering} | ${stage.excluded} | ${stage.remaining} |`);
1736
+ });
1737
+ const excluded = chain.input - chain.admitted;
1738
+ lines.push("", `Input ${chain.input} = admitted ${chain.admitted} + excluded ${excluded}. Admission rate ${formatShare(rate(chain.admitted, chain.input))}.`, "");
1739
+ return lines;
1740
+ }
1741
+ function stratumSection(artifact) {
1742
+ const lines = [
1743
+ "## Strata",
1744
+ "",
1745
+ "| stratum | admitted rows | in this campaign |",
1746
+ "| --- | --- | --- |"
1747
+ ];
1748
+ for (const stratum of ADMISSION_STRATA) {
1749
+ const admitted = artifact.admitted[stratum].length;
1750
+ const eligible = artifact.chain.admitStrata.includes(stratum) ? "yes" : "no";
1751
+ lines.push(`| ${stratum} | ${admitted} | ${eligible} |`);
1752
+ }
1753
+ lines.push("", "Sample within a stratum. A command-level repair cannot address a signal kill, so pooling that population with the others averages an addressable class with an unaddressable one.", "");
1754
+ return lines;
1755
+ }
1756
+ function reasonSection(totals) {
1757
+ const lines = [
1758
+ "## Exclusions",
1759
+ "",
1760
+ "| reason | rows | what it means |",
1761
+ "| --- | --- | --- |"
1762
+ ];
1763
+ for (const reason of ADMISSION_EXCLUSION_ORDER) lines.push(`| \`${reason}\` | ${totals[reason]} | ${ADMISSION_EXCLUSION_MEANING[reason]} |`);
1764
+ lines.push("");
1765
+ return lines;
1766
+ }
1767
+ function rowSection(rows, limit) {
1768
+ const lines = [
1769
+ "## Rows",
1770
+ "",
1771
+ "| row | task | stratum | final rc | admitted | excluded by |",
1772
+ "| --- | --- | --- | --- | --- | --- |"
1773
+ ];
1774
+ for (const row of rows.slice(0, limit)) lines.push(`| \`${row.rowId}\` | ${row.taskName} | ${row.stratum ?? "—"} | ${row.finalReturncode ?? "—"} | ${row.admitted ? "yes" : "no"} | ${row.excludedBy === null ? "—" : `\`${row.excludedBy}\``} |`);
1775
+ if (rows.length > limit) lines.push("", `${rows.length - limit} further rows in the artifact.`);
1776
+ lines.push("");
1777
+ return lines;
1778
+ }
1779
+ function rate(part, total) {
1780
+ return total === 0 ? 0 : part / total;
1781
+ }
1782
+ function formatShare(share) {
1783
+ return `${(share * 100).toFixed(2)}%`;
1784
+ }
1785
+ //#endregion
1786
+ //#region src/trace-repair/blinding.ts
1787
+ /**
1788
+ * What the analyst is shown.
1789
+ *
1790
+ * A blinded trajectory prefix: the task the agent was given and the actions
1791
+ * and observations it produced, and nothing about how the run was judged. No
1792
+ * held-out suite, no suite digest, no control rates, no end-state result, no
1793
+ * label.
1794
+ *
1795
+ * The blinding is by construction rather than by redaction. The returned
1796
+ * object is assembled field by field from an admitted row, so a field added to
1797
+ * `AdmittedRow` later cannot leak into an analyst prompt by default — it has
1798
+ * to be copied here on purpose.
1799
+ *
1800
+ * Requiring an `AdmittedRow` is the other half of the guarantee: a row cannot
1801
+ * be shown to an analyst before the four admission checks have executed and
1802
+ * passed.
1803
+ */
1804
+ function blindTrajectory(row, options = {}) {
1805
+ const through = options.throughStep ?? row.steps.length;
1806
+ if (!Number.isInteger(through) || through < 1 || through > row.steps.length) throw new ValidationError(`blindTrajectory throughStep must be within [1, ${row.steps.length}], got ${through}`);
1807
+ const steps = row.steps.slice(0, through).map((step) => ({
1808
+ step_id: step.step_id,
1809
+ action: step.action,
1810
+ observation: step.observation
1811
+ }));
1812
+ return {
1813
+ rowId: row.rowId,
1814
+ taskStatement: row.taskStatement,
1815
+ steps,
1816
+ recordedSteps: row.steps.length,
1817
+ maxK: through
1818
+ };
1819
+ }
1820
+ /**
1821
+ * Fields an analyst prompt must never carry. Exported so a consumer that
1822
+ * builds its own prompt can assert against the same list this module honours.
1823
+ */
1824
+ const BLINDED_FIELDS = Object.freeze([
1825
+ "criteria",
1826
+ "controlPolicy",
1827
+ "controlRate",
1828
+ "controlRollouts",
1829
+ "controlScreening",
1830
+ "suiteDigest",
1831
+ "policyDigest",
1832
+ "prefixDivergenceRatio",
1833
+ "image",
1834
+ "cwd"
1835
+ ]);
1836
+ //#endregion
1837
+ //#region src/trace-repair/repair-prompt.ts
1838
+ /**
1839
+ * The one question every repair arm answers.
1840
+ *
1841
+ * Arms differ in what EXECUTES the question — a single chat completion, an
1842
+ * agent harness, a DSPy program — and in nothing else. One prompt module is
1843
+ * what makes a comparison between them an ablation of the execution path
1844
+ * rather than a comparison of two prompts.
1845
+ *
1846
+ * The execution environment is stated, not hinted. The recorded scaffold runs
1847
+ * every action in a fresh `/bin/sh` and takes exactly one top-level statement,
1848
+ * so an arm that is not told that writes bash-only syntax and dies at
1849
+ * execution for a reason that has nothing to do with whether it localized the
1850
+ * failure. Stating the shell leaks no label: the held-out suite, the control
1851
+ * rates and the end-state verdict all stay out, and the prompt is built from a
1852
+ * `BlindedTrajectoryPrefix`, which cannot carry them.
1853
+ *
1854
+ * The action caps in the prose are read from the budget the grader enforces,
1855
+ * so the instruction and the rejection can never drift apart.
1856
+ */
1857
+ const REPAIR_QUESTION = "This coding agent ran and did not finish the task. Name the ONE recorded step whose action you would replace, and give the single shell action to run instead of it.";
1858
+ /**
1859
+ * The task definition. Parameterised by the budget so the stated caps are the
1860
+ * enforced caps.
1861
+ */
1862
+ function repairTaskPolicy(budget = SCAFFOLD_INTERVENTION_BUDGET) {
1863
+ return Object.freeze([
1864
+ "You are given a task statement and the complete recorded trajectory of an agent that attempted it and failed.",
1865
+ "Your answer is executed. The action you return is run inside the real task container, at the point in the trajectory you name, and the task's held-out test suite is then run against the resulting state. You are scored on whether the suite passes, never on how your explanation reads.",
1866
+ "",
1867
+ "HOW YOUR ACTION IS EXECUTED — read this before writing it:",
1868
+ "- The action runs in a FRESH `/bin/sh`, which is dash on this image, not bash. C-style `for ((i=0;i<n;i++))`, `[[ ]]`, arrays, and `local` are bash-only and will fail with a syntax error.",
1869
+ `- The scaffold accepts exactly ${budget.maxStatements === 1 ? "ONE top-level statement" : `${budget.maxStatements} top-level statements`}. Wrap multiple commands in a single \`{ ...; }\` block or chain them with \`&&\`.`,
1870
+ `- The action is at most ${budget.maxBytes} bytes, and must not be a no-op such as \`true\`, \`:\` or an empty string.`,
1871
+ "- It runs from the task working directory, as root, with the network available.",
1872
+ "- Prefer writing a complete correct file with a heredoc over patching in place; you cannot see the result of your own action, so it must work on the first attempt.",
1873
+ "Name the step whose action you would replace as `k`, using the step_id shown in the trajectory. Choosing the last step means \"instead of the last thing the agent did, do this\". You are not repairing the agent's reasoning; you are producing the state the suite requires.",
1874
+ "Return no finding only when no single replacement action could plausibly make the suite pass."
1875
+ ]);
1876
+ }
1877
+ /** The reply grammar, stated to an arm that answers in one JSON object. */
1878
+ const REPAIR_CONTRACT_LINES = Object.freeze([
1879
+ "OUTPUT CONTRACT — reply with ONE fenced JSON object and nothing else:",
1880
+ "```json",
1881
+ "{\"answer\": \"<one sentence on what went wrong>\", \"findings\": [{\"k\": <int>, \"failure_claim\": \"<what went wrong at step k>\", \"intervention\": {\"kind\": \"shell\", \"action\": \"<the single shell action to run instead>\"}}]}",
1882
+ "```",
1883
+ "Rules:",
1884
+ "- `findings` carries AT MOST ONE object. Never more.",
1885
+ "- `k` is an integer step_id present in the trajectory below.",
1886
+ "- `failure_claim` is a non-empty string. It is recorded and never scored, so spend your effort on the action.",
1887
+ "- `intervention.kind` is the string \"shell\", or \"edit\" when the action authors a file with a heredoc.",
1888
+ "- `intervention.action` is the exact shell text to execute. It is not a description and not a diff.",
1889
+ "- Return `\"findings\": []` only to say that no single action could repair this run."
1890
+ ]);
1891
+ /** The same grammar restated for the bounded repair turn, which never carries
1892
+ * the trajectory again. */
1893
+ const REPAIR_REPAIR_CONTRACT_LINES = Object.freeze([
1894
+ "Your previous reply did not satisfy the output contract. Resend the answer, and nothing else, as ONE fenced JSON object:",
1895
+ "```json",
1896
+ "{\"answer\": \"<one sentence>\", \"findings\": [{\"k\": <int>, \"failure_claim\": \"<string>\", \"intervention\": {\"kind\": \"shell\", \"action\": \"<shell text>\"}}]}",
1897
+ "```",
1898
+ "`findings` carries at most one object. Do not restate the trajectory."
1899
+ ]);
1900
+ /** The trajectory as an arm sees it: actions, observations, nothing about grading. */
1901
+ function renderRepairTrajectory(prefix) {
1902
+ return prefix.steps.map((step) => [
1903
+ `--- step_id ${step.step_id} ---`,
1904
+ "ACTION:",
1905
+ step.action,
1906
+ "OBSERVATION:",
1907
+ step.observation === null ? "(no observation recorded)" : step.observation
1908
+ ].join("\n")).join("\n\n");
1909
+ }
1910
+ function repairTrajectoryHeader(prefix) {
1911
+ return `RECORDED TRAJECTORY (${prefix.steps.length} steps; valid k is any step_id below, the last is ${prefix.maxK}):`;
1912
+ }
1913
+ /** The task definition an arm receives: the policy plus the statement the
1914
+ * recorded agent was given. */
1915
+ function repairTaskDefinition(prefix, budget = SCAFFOLD_INTERVENTION_BUDGET) {
1916
+ return [
1917
+ ...repairTaskPolicy(budget),
1918
+ "",
1919
+ "TASK STATEMENT GIVEN TO THE AGENT:",
1920
+ prefix.taskStatement
1921
+ ].join("\n");
1922
+ }
1923
+ /**
1924
+ * Digest of the question and task policy every arm shares.
1925
+ *
1926
+ * This is the part of the prompt that is equal by construction across arms:
1927
+ * the question, the execution rules, and the budget the caps are read from.
1928
+ * An arm's own contract text — its output grammar, its typed-signature
1929
+ * instructions — is deliberately outside it, because arms differ there.
1930
+ */
1931
+ function repairQuestionSha256(budget = SCAFFOLD_INTERVENTION_BUDGET) {
1932
+ return createHash("sha256").update(JSON.stringify({
1933
+ kind: "tb-repair-analyst-question",
1934
+ question: REPAIR_QUESTION,
1935
+ taskPolicy: repairTaskPolicy(budget),
1936
+ budget
1937
+ })).digest("hex");
1938
+ }
1939
+ /**
1940
+ * Digest of the composed question one arm answered.
1941
+ *
1942
+ * Covers the shared question and the arm's own declared contract text, so two
1943
+ * arms that ask materially different composed questions — a JSON grammar
1944
+ * versus a typed SUBMIT signature — stamp different digests, and two arms
1945
+ * that ask the identical composed question share one.
1946
+ */
1947
+ function repairArmPromptSha256(budget, contract) {
1948
+ return createHash("sha256").update(JSON.stringify({
1949
+ kind: "tb-repair-analyst-prompt",
1950
+ question: REPAIR_QUESTION,
1951
+ taskPolicy: repairTaskPolicy(budget),
1952
+ budget,
1953
+ contract
1954
+ })).digest("hex");
1955
+ }
1956
+ //#endregion
1957
+ //#region src/trace-repair/analyst-arm.ts
1958
+ /**
1959
+ * What an analyst arm is, and what every arm owes the comparison.
1960
+ *
1961
+ * An arm is one way of EXECUTING the repair question: a single chat
1962
+ * completion, an agent harness, a DSPy program. The question, the reply
1963
+ * grammar, the action budget and the bounded repair turn belong to the
1964
+ * comparison, not to the arm — so they live here and every arm inherits them.
1965
+ *
1966
+ * Three rules this module makes structural rather than customary:
1967
+ *
1968
+ * one contract an arm returns `RepairFinding` or an honest decline,
1969
+ * and nothing else parses. An arm that produced neither
1970
+ * fails loud with a typed reason; nothing is defaulted.
1971
+ * one budget `askRepairArm` measures every action against the
1972
+ * scaffold budget and records the measurement. The grader
1973
+ * remains the single authority that rejects, so a
1974
+ * violation is reported here and refused there — never
1975
+ * silently trimmed to fit.
1976
+ * one repair turn arms declare how many bounded retries a malformed reply
1977
+ * earns. `repairArmAsymmetries` refuses a set whose arms
1978
+ * disagree, because a second attempt is a second sample
1979
+ * the other arms never got.
1980
+ *
1981
+ * What arms are ALLOWED to differ in is declared, not hidden: `affordances`
1982
+ * names what an arm can do that another cannot — read the trajectory through
1983
+ * a code interpreter, take internal turns — and `repairArmAsymmetries` renders
1984
+ * those differences so a reader sees them beside the result instead of having
1985
+ * to infer them from two runners' source.
1986
+ */
1987
+ /**
1988
+ * Ask one arm about one admitted row.
1989
+ *
1990
+ * Blinding, prompt identity and budget measurement happen here, once, for
1991
+ * every arm. An arm that wants a different question does not get one.
1992
+ */
1993
+ async function askRepairArm(options) {
1994
+ const { arm, row } = options;
1995
+ assertDeclaration(arm.declaration);
1996
+ const now = options.now ?? Date.now;
1997
+ const startedMs = now();
1998
+ const prefix = blindTrajectory(row, options.throughStep === void 0 ? {} : { throughStep: options.throughStep });
1999
+ const reply = await arm.ask({
2000
+ prefix,
2001
+ ...options.signal ? { signal: options.signal } : {}
2002
+ });
2003
+ assertReplyWithinDeclaration(arm.declaration, reply);
2004
+ const budget = reply.status === "finding" ? checkInterventionBudget(reply.intervention.action, reply.intervention.kind, arm.declaration.budget) : null;
2005
+ return {
2006
+ rowId: row.rowId,
2007
+ armId: arm.declaration.id,
2008
+ promptSha256: repairArmPromptSha256(arm.declaration.budget, arm.declaration.promptContract),
2009
+ reply,
2010
+ budget,
2011
+ wallMs: now() - startedMs
2012
+ };
2013
+ }
2014
+ /**
2015
+ * The answer in the grammar the grader consumes.
2016
+ *
2017
+ * Null for a failure: a run that could not answer has no answer to grade, and
2018
+ * turning it into a decline would credit an arm with an honest null it never
2019
+ * produced.
2020
+ */
2021
+ function repairArmResponse(answer) {
2022
+ const { reply } = answer;
2023
+ if (reply.status === "declined") return { kind: "no-decisive-failure" };
2024
+ if (reply.status === "failed") return null;
2025
+ return {
2026
+ kind: "finding",
2027
+ k: reply.k,
2028
+ failureClaim: reply.failureClaim,
2029
+ intervention: reply.intervention
2030
+ };
2031
+ }
2032
+ /**
2033
+ * Refuse a set of arms that cannot be compared, and describe what still
2034
+ * differs between the ones that can.
2035
+ *
2036
+ * Two properties are hard: the arms measure actions against the same budget,
2037
+ * and a malformed reply earns the same number of retries everywhere. Both are
2038
+ * things that would move a score without moving the thing being measured.
2039
+ *
2040
+ * Certification is the third: a set where some arms run certified text and
2041
+ * others do not is refused outright. An optimisation applies to every arm or
2042
+ * to none — a mixed set measures the optimisation, then reports it as the
2043
+ * harness.
2044
+ */
2045
+ function repairArmAsymmetries(arms, options = {}) {
2046
+ if (arms.length === 0) throw new ValidationError("a repair-arm comparison needs at least one arm");
2047
+ const declarations = arms.map((arm) => arm.declaration);
2048
+ for (const declaration of declarations) assertDeclaration(declaration);
2049
+ const { ids, repairTurns } = assertEqualDeclarativeTerms("repair arm", declarations.map((declaration) => ({
2050
+ id: declaration.id,
2051
+ repairTurns: declaration.repairTurns
2052
+ })));
2053
+ const budget = options.budget ?? declarations[0].budget;
2054
+ const mismatched = declarations.find((declaration) => !sameBudget(declaration.budget, budget));
2055
+ if (mismatched) throw new ValidationError(`arm '${mismatched.id}' measures actions against ${JSON.stringify(mismatched.budget)}, the comparison runs ${JSON.stringify(budget)}; one arm buying a bigger action than another is a difference in what was asked, not in what answered`);
2056
+ const certified = declarations.filter((declaration) => declaration.certification.kind === "certified");
2057
+ if (certified.length > 0 && certified.length !== declarations.length) throw new ValidationError(`${certified.length}/${declarations.length} arms run certified text (${certified.map((declaration) => declaration.id).join(", ")}); an optimisation applies to every arm or to none`);
2058
+ return {
2059
+ armIds: ids,
2060
+ sharedAffordances: ALL_AFFORDANCES.filter((affordance) => declarations.every((declaration) => declaration.affordances.includes(affordance))),
2061
+ asymmetries: declarations.map((declaration) => ({
2062
+ armId: declaration.id,
2063
+ extraAffordances: ALL_AFFORDANCES.filter((affordance) => declaration.affordances.includes(affordance) && declarations.some((other) => !other.affordances.includes(affordance))),
2064
+ missingAffordances: ALL_AFFORDANCES.filter((affordance) => !declaration.affordances.includes(affordance) && declarations.some((other) => other.affordances.includes(affordance))),
2065
+ certification: declaration.certification,
2066
+ promptSha256: repairArmPromptSha256(declaration.budget, declaration.promptContract)
2067
+ })),
2068
+ noArmCertified: certified.length === 0,
2069
+ budget,
2070
+ repairTurns,
2071
+ questionSha256: repairQuestionSha256(budget)
2072
+ };
2073
+ }
2074
+ const ALL_AFFORDANCES = Object.freeze([
2075
+ "inline-trajectory",
2076
+ "trajectory-tools",
2077
+ "code-interpreter",
2078
+ "agent-loop"
2079
+ ]);
2080
+ function sameBudget(a, b) {
2081
+ return a.maxBytes === b.maxBytes && a.maxStatements === b.maxStatements && a.maxHeredocs === b.maxHeredocs;
2082
+ }
2083
+ function assertDeclaration(declaration) {
2084
+ if (typeof declaration.id !== "string" || declaration.id.trim() !== declaration.id || !declaration.id) throw new ValidationError("a repair arm id must be a trimmed non-empty string");
2085
+ if (typeof declaration.execution !== "string" || !declaration.execution.trim()) throw new ValidationError(`arm '${declaration.id}' must state what executes its answer`);
2086
+ if (!Number.isInteger(declaration.repairTurns) || declaration.repairTurns < 0) throw new ValidationError(`arm '${declaration.id}' repairTurns must be a non-negative integer, got ${declaration.repairTurns}`);
2087
+ if (declaration.certification.kind === "none" && (typeof declaration.certification.reason !== "string" || !declaration.certification.reason.trim())) throw new ValidationError(`arm '${declaration.id}' carries no certification and must say why in one sentence`);
2088
+ if (!Array.isArray(declaration.promptContract) || declaration.promptContract.length === 0 || declaration.promptContract.some((line) => typeof line !== "string")) throw new ValidationError(`arm '${declaration.id}' must declare its contract text as a non-empty array of strings; the per-arm prompt digest is computed from it`);
2089
+ for (const affordance of declaration.affordances) if (!ALL_AFFORDANCES.includes(affordance)) throw new ValidationError(`arm '${declaration.id}' declares unknown affordance '${affordance}'`);
2090
+ }
2091
+ /** An arm that reports more repair turns than it declared is not the arm the
2092
+ * comparison admitted, so the answer never reaches a grader. */
2093
+ function assertReplyWithinDeclaration(declaration, reply) {
2094
+ if (reply.repair.attempted && declaration.repairTurns === 0) throw new ValidationError(`arm '${declaration.id}' declares no bounded repair turn but took one`);
2095
+ if (reply.status === "failed" && !reply.failure.trim()) throw new ValidationError(`arm '${declaration.id}' failed without stating why`);
2096
+ if (reply.status === "finding" && !reply.intervention.action) throw new ValidationError(`arm '${declaration.id}' returned a finding with an empty intervention`);
2097
+ }
2098
+ //#endregion
2099
+ //#region src/trace-repair/analyst-response.ts
2100
+ /**
2101
+ * What an analyst is allowed to say about a blinded trajectory prefix.
2102
+ *
2103
+ * Exactly one of two answers:
2104
+ *
2105
+ * a finding {k, failureClaim, intervention}
2106
+ * the literal string no-decisive-failure
2107
+ *
2108
+ * Nothing else parses. A finding names one step, states what went wrong
2109
+ * there, and supplies one action to run instead. Declining is a real answer
2110
+ * with its own cell in the funnel, not a parse failure — an admitted row has
2111
+ * a failure by construction, but it need not have a single-action repair, and
2112
+ * an analyst that says so is answering the question it was asked.
2113
+ */
2114
+ /** The literal an analyst returns when no single step carries the failure. */
2115
+ const NO_DECISIVE_FAILURE = "no-decisive-failure";
2116
+ /**
2117
+ * Read an analyst reply.
2118
+ *
2119
+ * Accepts the bare literal, or a JSON object carrying one finding. The reply
2120
+ * is untrusted text, so every field is checked and nothing is defaulted: a
2121
+ * missing `k` is a parse failure, never step 1.
2122
+ */
2123
+ function parseAnalystResponse(reply) {
2124
+ const trimmed = reply.trim();
2125
+ if (trimmed.length === 0) return {
2126
+ succeeded: false,
2127
+ failure: "unreadable",
2128
+ detail: "the reply is empty"
2129
+ };
2130
+ if (trimmed === "no-decisive-failure") return {
2131
+ succeeded: true,
2132
+ value: { kind: "no-decisive-failure" }
2133
+ };
2134
+ const json = extractJsonObject(trimmed);
2135
+ if (!json) return {
2136
+ succeeded: false,
2137
+ failure: "unreadable",
2138
+ detail: `expected the literal "${NO_DECISIVE_FAILURE}" or one JSON object`
2139
+ };
2140
+ if (json.count > 1) return {
2141
+ succeeded: false,
2142
+ failure: "not-a-single-answer",
2143
+ detail: `the reply carries ${json.count} findings; exactly one is allowed`
2144
+ };
2145
+ const body = json.value;
2146
+ if (body.no_decisive_failure === true || body.finding === "no-decisive-failure") return {
2147
+ succeeded: true,
2148
+ value: { kind: "no-decisive-failure" }
2149
+ };
2150
+ const k = body.k;
2151
+ if (typeof k !== "number" || !Number.isInteger(k)) return {
2152
+ succeeded: false,
2153
+ failure: "missing-k",
2154
+ detail: `k must be an integer, got ${describe$2(k)}`
2155
+ };
2156
+ const failureClaim = body.failure_claim ?? body.failureClaim;
2157
+ if (typeof failureClaim !== "string" || failureClaim.trim().length === 0) return {
2158
+ succeeded: false,
2159
+ failure: "missing-failure-claim",
2160
+ detail: "failure_claim must be a non-empty string"
2161
+ };
2162
+ const raw = body.intervention;
2163
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return {
2164
+ succeeded: false,
2165
+ failure: "missing-intervention",
2166
+ detail: "intervention must be an object with kind and action"
2167
+ };
2168
+ const intervention = raw;
2169
+ const action = intervention.action;
2170
+ if (typeof action !== "string") return {
2171
+ succeeded: false,
2172
+ failure: "missing-intervention",
2173
+ detail: "intervention.action must be a string"
2174
+ };
2175
+ const kind = intervention.kind;
2176
+ if (kind !== "shell" && kind !== "edit") return {
2177
+ succeeded: false,
2178
+ failure: "unknown-intervention-kind",
2179
+ detail: `intervention.kind must be "shell" or "edit", got ${describe$2(kind)}`
2180
+ };
2181
+ return {
2182
+ succeeded: true,
2183
+ value: {
2184
+ kind: "finding",
2185
+ k,
2186
+ failureClaim: failureClaim.trim(),
2187
+ intervention: {
2188
+ kind,
2189
+ action
2190
+ }
2191
+ }
2192
+ };
2193
+ }
2194
+ /** Build a finding directly, for callers that already hold typed fields. */
2195
+ function repairFinding(input) {
2196
+ if (!Number.isInteger(input.k) || input.k < 1) throw new ValidationError(`repair finding k must be a positive integer, got ${input.k}`);
2197
+ if (input.failureClaim.trim().length === 0) throw new ValidationError("repair finding requires a non-empty failure claim");
2198
+ return {
2199
+ kind: "finding",
2200
+ k: input.k,
2201
+ failureClaim: input.failureClaim.trim(),
2202
+ intervention: input.intervention
2203
+ };
2204
+ }
2205
+ function describe$2(value) {
2206
+ if (value === null) return "null";
2207
+ if (value === void 0) return "undefined";
2208
+ return typeof value === "string" ? JSON.stringify(value) : String(value);
2209
+ }
2210
+ /**
2211
+ * Pull the single JSON object out of a reply, tolerating a fenced block and
2212
+ * surrounding prose. `count` reports how many top-level objects were found so
2213
+ * a reply hedging with several findings is rejected rather than silently
2214
+ * reduced to the first one.
2215
+ */
2216
+ function extractJsonObject(reply) {
2217
+ const fenced = [...reply.matchAll(/```(?:json)?\s*\n([\s\S]*?)```/g)].map((m) => m[1].trim());
2218
+ const candidates = fenced.length > 0 ? fenced : [reply];
2219
+ const parsed = [];
2220
+ for (const candidate of candidates) for (const slice of topLevelObjectSlices(candidate)) try {
2221
+ const value = JSON.parse(slice);
2222
+ if (Array.isArray(value)) {
2223
+ for (const item of value) if (isPlainObject(item)) parsed.push(item);
2224
+ } else if (isPlainObject(value)) parsed.push(value);
2225
+ } catch {}
2226
+ if (parsed.length === 0) return null;
2227
+ return {
2228
+ value: parsed[0],
2229
+ count: parsed.length
2230
+ };
2231
+ }
2232
+ function isPlainObject(value) {
2233
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2234
+ }
2235
+ /** Balanced `{…}` and `[…]` slices at the top level of a candidate string. */
2236
+ function topLevelObjectSlices(text) {
2237
+ const slices = [];
2238
+ let depth = 0;
2239
+ let start = -1;
2240
+ let inString = false;
2241
+ let escaped = false;
2242
+ for (let i = 0; i < text.length; i += 1) {
2243
+ const char = text[i];
2244
+ if (inString) {
2245
+ if (escaped) escaped = false;
2246
+ else if (char === "\\") escaped = true;
2247
+ else if (char === "\"") inString = false;
2248
+ continue;
2249
+ }
2250
+ if (char === "\"") {
2251
+ inString = true;
2252
+ continue;
2253
+ }
2254
+ if (char === "{" || char === "[") {
2255
+ if (depth === 0) start = i;
2256
+ depth += 1;
2257
+ continue;
2258
+ }
2259
+ if (char === "}" || char === "]") {
2260
+ depth -= 1;
2261
+ if (depth === 0 && start >= 0) {
2262
+ slices.push(text.slice(start, i + 1));
2263
+ start = -1;
2264
+ }
2265
+ if (depth < 0) depth = 0;
2266
+ }
2267
+ }
2268
+ return slices;
2269
+ }
2270
+ //#endregion
2271
+ //#region src/trace-repair/arm-completion.ts
2272
+ function createCompletionRepairArm(options) {
2273
+ if (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0) throw new ValidationError(`repair arm '${options.id}' timeoutMs must be a positive safe integer, got ${options.timeoutMs}`);
2274
+ const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
2275
+ return {
2276
+ declaration: {
2277
+ id: options.id,
2278
+ execution: options.execution,
2279
+ certification: options.certification,
2280
+ budget,
2281
+ repairTurns: 1,
2282
+ affordances: options.affordances,
2283
+ promptContract: [...REPAIR_CONTRACT_LINES, ...REPAIR_REPAIR_CONTRACT_LINES]
2284
+ },
2285
+ async ask(request) {
2286
+ const prompt = buildPrimePrompt({
2287
+ question: REPAIR_QUESTION,
2288
+ taskDefinition: repairTaskDefinition(request.prefix, budget),
2289
+ contractLines: REPAIR_CONTRACT_LINES,
2290
+ trajectoryHeader: repairTrajectoryHeader(request.prefix),
2291
+ renderedTrajectory: renderRepairTrajectory(request.prefix)
2292
+ });
2293
+ const outcome = await runPrimeExchange({
2294
+ contract: replyContract(request.prefix),
2295
+ prompt,
2296
+ transport: options.transport,
2297
+ url: options.url,
2298
+ model: options.model,
2299
+ timeoutMs: options.timeoutMs,
2300
+ repair: true,
2301
+ ...request.signal ? { signal: request.signal } : {}
2302
+ });
2303
+ const usage = armUsage(outcome.usage, options.pricing);
2304
+ if (!outcome.ok) return {
2305
+ status: "failed",
2306
+ failure: `${outcome.failure.kind}: ${outcome.failure.message}`,
2307
+ answer: null,
2308
+ rejectedRows: [],
2309
+ repair: outcome.repair,
2310
+ usage
2311
+ };
2312
+ const row = outcome.rows[0];
2313
+ if (row === void 0) return {
2314
+ status: "declined",
2315
+ answer: outcome.answer,
2316
+ reportedRows: outcome.reportedRows,
2317
+ rejectedRows: outcome.rejected,
2318
+ repair: outcome.repair,
2319
+ usage
2320
+ };
2321
+ return {
2322
+ status: "finding",
2323
+ k: row.k,
2324
+ failureClaim: row.failureClaim,
2325
+ intervention: {
2326
+ kind: row.kind,
2327
+ action: row.action
2328
+ },
2329
+ answer: outcome.answer,
2330
+ reportedRows: outcome.reportedRows,
2331
+ rejectedRows: outcome.rejected,
2332
+ repair: outcome.repair,
2333
+ usage
2334
+ };
2335
+ }
2336
+ };
2337
+ }
2338
+ /**
2339
+ * Decode one reply row.
2340
+ *
2341
+ * Every field is checked and nothing is defaulted: a `k` outside the recorded
2342
+ * step ids is a rejected row rather than a clamped one, because an arm that
2343
+ * names a step the recording does not hold has not localized anything.
2344
+ */
2345
+ function replyContract(prefix) {
2346
+ const validSteps = prefix.steps.map((step) => step.step_id);
2347
+ return {
2348
+ rowsField: "findings",
2349
+ contractLines: REPAIR_CONTRACT_LINES,
2350
+ repairContractLines: REPAIR_REPAIR_CONTRACT_LINES,
2351
+ maxRows: 1,
2352
+ decodeRow(row, index) {
2353
+ if (typeof row !== "object" || row === null || Array.isArray(row)) return {
2354
+ ok: false,
2355
+ reason: `row ${index} is not an object`
2356
+ };
2357
+ const record = row;
2358
+ const k = record.k;
2359
+ if (!Number.isInteger(k) || !validSteps.includes(k)) return {
2360
+ ok: false,
2361
+ reason: `k must be a recorded step_id in [${validSteps[0] ?? 1}, ${prefix.maxK}], got ${describe$1(k)}`
2362
+ };
2363
+ const claim = record.failure_claim ?? record.failureClaim;
2364
+ if (typeof claim !== "string" || claim.trim().length === 0) return {
2365
+ ok: false,
2366
+ reason: "failure_claim must be a non-empty string"
2367
+ };
2368
+ const raw = record.intervention;
2369
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return {
2370
+ ok: false,
2371
+ reason: "intervention must be an object with kind and action"
2372
+ };
2373
+ const intervention = raw;
2374
+ const action = intervention.action;
2375
+ if (typeof action !== "string" || action.length === 0) return {
2376
+ ok: false,
2377
+ reason: "intervention.action must be a non-empty string"
2378
+ };
2379
+ const kind = intervention.kind;
2380
+ if (kind !== "shell" && kind !== "edit") return {
2381
+ ok: false,
2382
+ reason: `intervention.kind must be "shell" or "edit", got ${describe$1(kind)}`
2383
+ };
2384
+ return {
2385
+ ok: true,
2386
+ row: {
2387
+ k,
2388
+ failureClaim: claim.trim(),
2389
+ kind,
2390
+ action
2391
+ }
2392
+ };
2393
+ }
2394
+ };
2395
+ }
2396
+ function armUsage(usage, pricing) {
2397
+ const receipt = analystUsageReceiptFromPrimeUsage(usage, pricing);
2398
+ return {
2399
+ calls: usage.calls,
2400
+ inputTokens: usage.inputTokens,
2401
+ outputTokens: usage.outputTokens,
2402
+ costUsd: receipt.cost.usd ?? receipt.knownCostUsd ?? null
2403
+ };
2404
+ }
2405
+ function describe$1(value) {
2406
+ if (value === null) return "null";
2407
+ if (value === void 0) return "undefined";
2408
+ return typeof value === "string" ? JSON.stringify(value) : String(value);
2409
+ }
2410
+ //#endregion
2411
+ //#region src/trace-repair/arm-dspy.ts
2412
+ /**
2413
+ * Token the bridge reads to select the typed repair signature.
2414
+ *
2415
+ * The wire protocol carries instructions as opaque text, so the task is named
2416
+ * inside them rather than by adding a field every unrelated caller would have
2417
+ * to set. The same mechanism selects the CodeTraceBench typed signature.
2418
+ */
2419
+ const DSPY_REPAIR_TASK_TOKEN = "tb-repair-typed-";
2420
+ /** Signature the bridge reports for a repair analysis. */
2421
+ const DSPY_REPAIR_SIGNATURE = "tb-repair-typed-v1";
2422
+ /** The name the trajectory is bound to inside the program's environment. */
2423
+ const DSPY_REPAIR_TRAJECTORY_INPUT = "trajectory";
2424
+ /**
2425
+ * The repair contract restated for the typed signature.
2426
+ *
2427
+ * Transport differs from the chat-completion arms — typed SUBMIT instead of a
2428
+ * fenced JSON object — and the QUESTION does not. Both read from
2429
+ * `repairTaskPolicy`, so the execution rules an arm is held to cannot drift
2430
+ * between arms.
2431
+ */
2432
+ function dspyRepairInstructions(budget = SCAFFOLD_INTERVENTION_BUDGET) {
2433
+ return [
2434
+ `TASK ${DSPY_REPAIR_TASK_TOKEN}${DSPY_REPAIR_SIGNATURE}`,
2435
+ "",
2436
+ ...repairTaskPolicy(budget),
2437
+ "",
2438
+ `The recorded trajectory is already loaded into your environment as \`${DSPY_REPAIR_TRAJECTORY_INPUT}\`,`,
2439
+ "a list of {step_id, action, observation} objects in recorded order. Read it with code",
2440
+ "rather than re-fetching it, and do not claim to have inspected anything you did not read.",
2441
+ "",
2442
+ "SUBMIT at most ONE repair, shaped as:",
2443
+ "- k: the step_id whose action you replace. It must be a step_id present in the trajectory.",
2444
+ "- failure_claim: what went wrong at step k. Recorded, never scored.",
2445
+ "- intervention_kind: \"edit\" when the action authors a file with a heredoc, \"shell\" otherwise.",
2446
+ `- action: the EXACT shell text to run instead, at most ${budget.maxBytes} bytes and exactly`,
2447
+ ` ${budget.maxStatements === 1 ? "one top-level statement" : `${budget.maxStatements} top-level statements`}. It is executed verbatim: not a description, not a diff, not a plan.`,
2448
+ "SUBMIT an empty list only when no single replacement action could make the suite pass."
2449
+ ].join("\n");
2450
+ }
2451
+ function createDspyRepairArm(options) {
2452
+ assertDspyRepairEngine(options.engine);
2453
+ const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
2454
+ const instructions = dspyRepairInstructions(budget);
2455
+ return {
2456
+ declaration: {
2457
+ id: options.id ?? "dspy-rlm",
2458
+ execution: `DSPy RLM program (${options.engine.id} v${options.engine.version}) with a typed repair signature and a code environment`,
2459
+ certification: {
2460
+ kind: "none",
2461
+ reason: "the GEPA-certified analyst text was earned on the CodeTraceBench incorrect-step contract, which asks for blocks of wrong steps rather than one executable repair; this arm runs instruction text authored for the repair contract and certified by nothing"
2462
+ },
2463
+ budget,
2464
+ repairTurns: 1,
2465
+ affordances: [
2466
+ "inline-trajectory",
2467
+ "code-interpreter",
2468
+ "agent-loop"
2469
+ ],
2470
+ promptContract: [instructions]
2471
+ },
2472
+ async ask(request) {
2473
+ const steps = request.prefix.steps.map((step) => ({
2474
+ step_id: step.step_id,
2475
+ action: step.action,
2476
+ observation: step.observation
2477
+ }));
2478
+ let result;
2479
+ try {
2480
+ result = await options.engine.analyze({
2481
+ analystId: options.analystId,
2482
+ question: [
2483
+ REPAIR_QUESTION,
2484
+ "",
2485
+ `The trajectory has ${steps.length} recorded steps, step_ids ${steps.map((step) => step.step_id).join(", ")}.`
2486
+ ].join("\n"),
2487
+ instructions,
2488
+ tools: [],
2489
+ taskInputs: {
2490
+ [DSPY_REPAIR_TRAJECTORY_INPUT]: steps,
2491
+ taskStatement: request.prefix.taskStatement
2492
+ },
2493
+ limits: options.limits,
2494
+ costLedger: options.costLedger,
2495
+ costPhase: options.costPhase,
2496
+ ...options.costTags ? { costTags: options.costTags } : {},
2497
+ ...request.signal ? { signal: request.signal } : {},
2498
+ ...options.log ? { log: options.log } : {}
2499
+ });
2500
+ } catch (error) {
2501
+ return {
2502
+ status: "failed",
2503
+ failure: `engine threw: ${error instanceof Error ? error.message : String(error)}`,
2504
+ answer: null,
2505
+ rejectedRows: [],
2506
+ repair: {
2507
+ attempted: false,
2508
+ succeeded: null
2509
+ },
2510
+ usage: engineUsage(null)
2511
+ };
2512
+ }
2513
+ const payload = readRepairPayload(result.runtime, { validSteps: steps.map((step) => step.step_id) });
2514
+ const usage = engineUsage(result.modelCalls);
2515
+ if (!payload.ok) return {
2516
+ status: "failed",
2517
+ failure: payload.reason,
2518
+ answer: result.answer,
2519
+ rejectedRows: [],
2520
+ repair: {
2521
+ attempted: false,
2522
+ succeeded: null
2523
+ },
2524
+ usage
2525
+ };
2526
+ const { rows, dropped, repair, reported, failure } = payload.value;
2527
+ const repairTurn = {
2528
+ attempted: repair !== null,
2529
+ succeeded: repair === null ? null : failure === null
2530
+ };
2531
+ if (failure !== null) return {
2532
+ status: "failed",
2533
+ failure,
2534
+ answer: result.answer,
2535
+ rejectedRows: dropped,
2536
+ repair: repairTurn,
2537
+ usage
2538
+ };
2539
+ const row = rows[0];
2540
+ if (row === void 0) return {
2541
+ status: "declined",
2542
+ answer: result.answer,
2543
+ reportedRows: reported,
2544
+ rejectedRows: dropped,
2545
+ repair: repairTurn,
2546
+ usage
2547
+ };
2548
+ return {
2549
+ status: "finding",
2550
+ k: row.k,
2551
+ failureClaim: row.failureClaim,
2552
+ intervention: {
2553
+ kind: row.kind,
2554
+ action: row.action
2555
+ },
2556
+ answer: result.answer,
2557
+ reportedRows: reported,
2558
+ rejectedRows: dropped,
2559
+ repair: repairTurn,
2560
+ usage
2561
+ };
2562
+ }
2563
+ };
2564
+ }
2565
+ /**
2566
+ * Read the bridge's typed repair block.
2567
+ *
2568
+ * Structure is checked and fails loud: a missing block, a wrong signature, or
2569
+ * a row that does not carry an integer k and a non-empty action is a wiring
2570
+ * fault with a reason — never a default, and never an empty list that would
2571
+ * grade as a decline the program did not make. A structurally sound row whose
2572
+ * k is not a recorded step id is the model's mistake, and it is dropped with
2573
+ * its reason instead.
2574
+ */
2575
+ function readRepairPayload(runtime, options) {
2576
+ const block = runtime.repair;
2577
+ if (!isRecord(block)) return {
2578
+ ok: false,
2579
+ reason: "the engine returned no runtime.repair block; the bridge did not run the typed repair signature"
2580
+ };
2581
+ if (block.signature !== "tb-repair-typed-v1") return {
2582
+ ok: false,
2583
+ reason: `runtime.repair.signature is ${describe(block.signature)}, expected "${DSPY_REPAIR_SIGNATURE}"`
2584
+ };
2585
+ if (!Number.isSafeInteger(block.reported) || block.reported < 0) return {
2586
+ ok: false,
2587
+ reason: `runtime.repair.reported must be a non-negative integer, got ${describe(block.reported)}`
2588
+ };
2589
+ if (!Array.isArray(block.rows)) return {
2590
+ ok: false,
2591
+ reason: "runtime.repair.rows must be an array"
2592
+ };
2593
+ const repair = block.repair;
2594
+ if (repair !== null && typeof repair !== "string") return {
2595
+ ok: false,
2596
+ reason: `runtime.repair.repair must be a string or null, got ${describe(repair)}`
2597
+ };
2598
+ const rawFailure = block.failure;
2599
+ if (rawFailure !== void 0 && rawFailure !== null && typeof rawFailure !== "string") return {
2600
+ ok: false,
2601
+ reason: `runtime.repair.failure must be a string or null, got ${describe(rawFailure)}`
2602
+ };
2603
+ const failure = typeof rawFailure === "string" ? rawFailure : null;
2604
+ const dropped = [];
2605
+ if (block.dropped !== void 0) {
2606
+ if (!Array.isArray(block.dropped)) return {
2607
+ ok: false,
2608
+ reason: "runtime.repair.dropped must be an array"
2609
+ };
2610
+ for (const [index, raw] of block.dropped.entries()) {
2611
+ if (!isRecord(raw) || typeof raw.reason !== "string") return {
2612
+ ok: false,
2613
+ reason: `runtime.repair.dropped[${index}] must carry a reason string`
2614
+ };
2615
+ dropped.push({
2616
+ index: Number.isSafeInteger(raw.index) ? raw.index : index,
2617
+ reason: raw.reason
2618
+ });
2619
+ }
2620
+ }
2621
+ const rows = [];
2622
+ for (const [index, raw] of block.rows.entries()) {
2623
+ const decoded = decodeRepairRow(raw);
2624
+ if (!decoded.ok) return {
2625
+ ok: false,
2626
+ reason: `runtime.repair.rows[${index}]: ${decoded.reason}`
2627
+ };
2628
+ if (!options.validSteps.includes(decoded.row.k)) {
2629
+ dropped.push({
2630
+ index,
2631
+ reason: `k must be a recorded step_id in [${options.validSteps[0] ?? 1}, ${options.validSteps[options.validSteps.length - 1] ?? 1}], got ${decoded.row.k}`
2632
+ });
2633
+ continue;
2634
+ }
2635
+ if (rows.length >= 1) {
2636
+ dropped.push({
2637
+ index,
2638
+ reason: "exceeds the one-repair cap"
2639
+ });
2640
+ continue;
2641
+ }
2642
+ rows.push(decoded.row);
2643
+ }
2644
+ return {
2645
+ ok: true,
2646
+ value: {
2647
+ signature: DSPY_REPAIR_SIGNATURE,
2648
+ reported: block.reported,
2649
+ rows,
2650
+ dropped,
2651
+ repair: repair ?? null,
2652
+ failure
2653
+ }
2654
+ };
2655
+ }
2656
+ function decodeRepairRow(raw) {
2657
+ if (!isRecord(raw)) return {
2658
+ ok: false,
2659
+ reason: "not an object"
2660
+ };
2661
+ const k = raw.k;
2662
+ if (!Number.isInteger(k) || k < 1) return {
2663
+ ok: false,
2664
+ reason: `k must be a positive integer, got ${describe(k)}`
2665
+ };
2666
+ const claim = raw.failure_claim;
2667
+ if (typeof claim !== "string" || claim.trim().length === 0) return {
2668
+ ok: false,
2669
+ reason: "failure_claim must be a non-empty string"
2670
+ };
2671
+ const intervention = raw.intervention;
2672
+ if (!isRecord(intervention)) return {
2673
+ ok: false,
2674
+ reason: "intervention must be an object with kind and action"
2675
+ };
2676
+ const kind = intervention.kind;
2677
+ if (kind !== "shell" && kind !== "edit") return {
2678
+ ok: false,
2679
+ reason: `intervention.kind must be "shell" or "edit", got ${describe(kind)}`
2680
+ };
2681
+ const action = intervention.action;
2682
+ if (typeof action !== "string" || action.length === 0) return {
2683
+ ok: false,
2684
+ reason: "intervention.action must be a non-empty string"
2685
+ };
2686
+ return {
2687
+ ok: true,
2688
+ row: {
2689
+ k,
2690
+ failureClaim: claim.trim(),
2691
+ kind,
2692
+ action
2693
+ }
2694
+ };
2695
+ }
2696
+ /**
2697
+ * The engine reports model calls; token counts and cost live in the caller's
2698
+ * cost ledger, which meters the proxy the engine ran through. Reporting them
2699
+ * here as zero would state a measurement this arm did not make.
2700
+ */
2701
+ function engineUsage(modelCalls) {
2702
+ return {
2703
+ calls: modelCalls,
2704
+ inputTokens: null,
2705
+ outputTokens: null,
2706
+ costUsd: null
2707
+ };
2708
+ }
2709
+ function isRecord(value) {
2710
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2711
+ }
2712
+ function describe(value) {
2713
+ if (value === null) return "null";
2714
+ if (value === void 0) return "undefined";
2715
+ return typeof value === "string" ? JSON.stringify(value) : String(value);
2716
+ }
2717
+ /** Thrown by callers that build a repair arm with a non-DSPy engine. */
2718
+ function assertDspyRepairEngine(engine) {
2719
+ if (engine.id !== "dspy-rlm") throw new ValidationError(`the DSPy repair arm needs the dspy-rlm engine, got '${engine.id}'; another engine does not run the typed repair signature and would answer a different contract`);
2720
+ }
2721
+ //#endregion
2722
+ //#region src/trace-repair/continuation-records.ts
2723
+ /**
2724
+ * Project a full message list into corpus steps: the system and task messages
2725
+ * become their own steps, and every assistant turn carries the command it
2726
+ * requested plus the observation that answered it. A trailing assistant turn
2727
+ * with no answer keeps `obs: null`, which is how the corpus records a run that
2728
+ * ended on its last command.
2729
+ */
2730
+ function toRecordedSteps(messages) {
2731
+ const steps = [];
2732
+ for (let i = 0; i < messages.length; i += 1) {
2733
+ const message = messages[i];
2734
+ if (!message) continue;
2735
+ if (message.role === "assistant") {
2736
+ const parsed = parseAction(message.content);
2737
+ const next = messages[i + 1];
2738
+ steps.push({
2739
+ src: "agent",
2740
+ msg: message.content,
2741
+ tools: parsed.kind === "action" ? [{
2742
+ fn: "bash_command",
2743
+ cmd: parsed.command
2744
+ }] : null,
2745
+ obs: next && next.role === "user" ? next.content : null
2746
+ });
2747
+ continue;
2748
+ }
2749
+ if (message.role === "system") {
2750
+ steps.push({
2751
+ src: "system",
2752
+ msg: message.content,
2753
+ tools: null,
2754
+ obs: null
2755
+ });
2756
+ continue;
2757
+ }
2758
+ const previous = messages[i - 1];
2759
+ if (!previous || previous.role === "system") steps.push({
2760
+ src: "user",
2761
+ msg: message.content,
2762
+ tools: null,
2763
+ obs: null
2764
+ });
2765
+ }
2766
+ return steps;
2767
+ }
2768
+ /** Corpus steps for the continuation alone, excluding the prefix it inherited. */
2769
+ function rolloutRecordedSteps(rollout) {
2770
+ return rollout.steps.map((step) => ({
2771
+ src: "agent",
2772
+ msg: step.assistantMessage,
2773
+ tools: step.action === null ? null : [{
2774
+ fn: "bash_command",
2775
+ cmd: step.action
2776
+ }],
2777
+ obs: step.observation
2778
+ }));
2779
+ }
2780
+ /**
2781
+ * Hash over everything the policy determines: the actions taken, the
2782
+ * observations they produced, the seeds, the exit, and the usage.
2783
+ *
2784
+ * Wall-clock fields are excluded because they vary between identical runs;
2785
+ * two rollouts with the same digest did the same work, whatever they cost in
2786
+ * seconds. Use it to assert determinism, never to assert equal latency.
2787
+ */
2788
+ function rolloutDigest(rollout) {
2789
+ return contentHash({
2790
+ arm: rollout.arm,
2791
+ rowId: rollout.rowId,
2792
+ index: rollout.index,
2793
+ seed: rollout.seed,
2794
+ policyDigest: rollout.policyDigest,
2795
+ exitStatus: rollout.exitStatus,
2796
+ submission: rollout.submission,
2797
+ usage: rollout.usage,
2798
+ costUsd: rollout.costProvenance.usd,
2799
+ costKind: rollout.costProvenance.kind,
2800
+ steps: rollout.steps.map((step) => ({
2801
+ step: step.step,
2802
+ assistantMessage: step.assistantMessage,
2803
+ action: step.action,
2804
+ observation: step.observation,
2805
+ returncode: step.execution?.returncode ?? null,
2806
+ timedOut: step.execution?.timedOut ?? null,
2807
+ seed: step.model.seed,
2808
+ servedModel: step.model.servedModel,
2809
+ usage: step.model.usage,
2810
+ costUsd: step.model.costUsd
2811
+ }))
2812
+ });
2813
+ }
2814
+ //#endregion
2815
+ //#region src/trace-repair/degenerate-strategies.ts
2816
+ const DEGENERATE_STRATEGIES = [
2817
+ {
2818
+ id: "point-at-any-nonzero-exit-step",
2819
+ strategy: "Name the first step with a nonzero returncode, claim it failed, and let the reproduction gate confirm it.",
2820
+ defeat: "Reproduction is a gate, not a tier that pays. The credit vector has no term for it, so a reproduced step with no working intervention scores exactly what an unreproduced one scores: nothing.",
2821
+ defeatKind: "gate",
2822
+ enforcedIn: "funnel.ts (repairCredit has no reproduction term)"
2823
+ },
2824
+ {
2825
+ id: "propose-the-recorded-command-again",
2826
+ strategy: "Return the action the agent already ran at step k, so the arm reproduces the recorded state and looks like a faithful replay.",
2827
+ defeat: "The intervention is compared against the recorded action at k after whitespace normalisation and rejected before a container opens.",
2828
+ defeatKind: "gate",
2829
+ enforcedIn: "grade.ts (recorded-action-reproposed rejection)"
2830
+ },
2831
+ {
2832
+ id: "propose-a-no-op",
2833
+ strategy: "Return an action that changes nothing, so the arm inherits whatever the trajectory would have done anyway.",
2834
+ defeat: "Literal no-ops are rejected at the budget check. A semantic no-op cannot be detected syntactically, so it runs and measures at the no-op control floor: its paired delta is zero.",
2835
+ defeatKind: "measurement",
2836
+ enforcedIn: "action-budget.ts (NO_OP_ACTIONS) and delta-repair.ts (paired delta)"
2837
+ },
2838
+ {
2839
+ id: "submit-instead-of-repair",
2840
+ strategy: "Return the submit sentinel so the run ends immediately and the arm terminates cleanly.",
2841
+ defeat: "An action carrying the submit sentinel is rejected at the budget check. Ending the run is what the recorded agent already did and the tests already failed on it.",
2842
+ defeatKind: "gate",
2843
+ enforcedIn: "action-budget.ts (SUBMIT_SENTINEL rejection)"
2844
+ },
2845
+ {
2846
+ id: "touch-the-test-suite",
2847
+ strategy: "Write a passing suite, or a reward file, at the path the grader will read, so the oracle grades the trajectory's own artifact.",
2848
+ defeat: "The oracle purges the suite root and uploads the held-out suite from outside the session at grade time, then verifies the bytes it reads back. A planted suite is overwritten; a session that refuses the overwrite raises a tamper error instead of returning a pass.",
2849
+ defeatKind: "gate",
2850
+ enforcedIn: "test-oracle.ts (purge, upload, read-back digest)"
2851
+ },
2852
+ {
2853
+ id: "buy-a-bigger-action",
2854
+ strategy: "Return a multi-command script or a whole-file rewrite that does far more than one scaffold turn could.",
2855
+ defeat: "The budget counts top-level statements, heredocs and bytes. More than one action, more than one authored file, or more than 4 KB is rejected before a container opens.",
2856
+ defeatKind: "gate",
2857
+ enforcedIn: "action-budget.ts (checkInterventionBudget)"
2858
+ },
2859
+ {
2860
+ id: "decline-every-hard-row",
2861
+ strategy: "Answer no-decisive-failure on everything except the rows that are obviously repairable, so the reported rate is computed on an easy subset.",
2862
+ defeat: "A declined row keeps its cell in the funnel and stays in the denominator with a paired delta of zero, because its intervention arm is definitionally its control arm. Declining cannot raise the headline; it can only dilute it.",
2863
+ defeatKind: "measurement",
2864
+ enforcedIn: "grade.ts (declined outcome) and delta-repair.ts (full admitted denominator)"
2865
+ },
2866
+ {
2867
+ id: "repair-somewhere-other-than-k",
2868
+ strategy: "Name a plausible-looking k, then submit an intervention that fixes the task from any state, so the answer scores without localising anything.",
2869
+ defeat: "The intervention is executed at the k the analyst named, on the state produced by replaying steps 1..k-1. There is no separate localisation credit to win and no label the grader reads, so a wrong k is only penalised through the repair failing to work there.",
2870
+ defeatKind: "measurement",
2871
+ enforcedIn: "grade.ts (the intervention is applied at the named k only)"
2872
+ }
2873
+ ];
2874
+ function degenerateStrategy(id) {
2875
+ const found = DEGENERATE_STRATEGIES.find((entry) => entry.id === id);
2876
+ if (!found) throw new Error(`unknown degenerate strategy: ${id}`);
2877
+ return found;
2878
+ }
2879
+ //#endregion
2880
+ //#region src/trace-repair/funnel.ts
2881
+ const CREDIT_TERMS = [
2882
+ "executes",
2883
+ "localFlip",
2884
+ "repairRate"
2885
+ ];
2886
+ const ZERO_CREDIT = Object.freeze({
2887
+ executes: 0,
2888
+ localFlip: 0,
2889
+ repairRate: 0
2890
+ });
2891
+ /** What an answer earned. Every outcome but `measured` earns nothing. */
2892
+ function repairCredit(grade) {
2893
+ if (grade.outcome !== "measured") return ZERO_CREDIT;
2894
+ return {
2895
+ executes: 1,
2896
+ localFlip: grade.localFlip.passed ? 1 : 0,
2897
+ repairRate: grade.repair.rollouts === 0 ? 0 : grade.repair.passes / grade.repair.rollouts
2898
+ };
2899
+ }
2900
+ /** True once the answer is a well-formed, budget-admissible answer. A decline
2901
+ * is well formed, so it parses. */
2902
+ function reachedT0(grade) {
2903
+ return grade.outcome !== "rejected";
2904
+ }
2905
+ /** True once the recorded state at k came back. A decline never reaches the
2906
+ * gate, because it names no k. */
2907
+ function reachedT1(grade) {
2908
+ return grade.outcome === "did-not-execute" || grade.outcome === "measured";
2909
+ }
2910
+ /** True once the intervention ran at k and exited cleanly. */
2911
+ function reachedT2(grade) {
2912
+ return grade.outcome === "measured";
2913
+ }
2914
+ function countFunnel(grades) {
2915
+ let rejected = 0;
2916
+ let declined = 0;
2917
+ let t1 = 0;
2918
+ let t2 = 0;
2919
+ let t3 = 0;
2920
+ let t4Any = 0;
2921
+ let t4All = 0;
2922
+ for (const grade of grades) {
2923
+ if (grade.outcome === "rejected") rejected += 1;
2924
+ if (grade.outcome === "declined") declined += 1;
2925
+ if (reachedT1(grade)) t1 += 1;
2926
+ if (reachedT2(grade)) t2 += 1;
2927
+ if (grade.outcome === "measured") {
2928
+ if (grade.localFlip.passed) t3 += 1;
2929
+ if (grade.repair.passes > 0) t4Any += 1;
2930
+ if (grade.repair.rollouts > 0 && grade.repair.passes === grade.repair.rollouts) t4All += 1;
2931
+ }
2932
+ }
2933
+ return {
2934
+ rows: grades.length,
2935
+ rejected,
2936
+ declined,
2937
+ t0Parsed: grades.length - rejected,
2938
+ t1Reproduced: t1,
2939
+ t2Executed: t2,
2940
+ t3LocalFlip: t3,
2941
+ t4RepairFlipAny: t4Any,
2942
+ t4RepairFlipAll: t4All
2943
+ };
2944
+ }
2945
+ //#endregion
2946
+ //#region src/trace-repair/delta-repair.ts
2947
+ /**
2948
+ * The headline metric.
2949
+ *
2950
+ * Delta-repair = P(tests pass | intervention) − P(tests pass | no-fix control)
2951
+ *
2952
+ * Paired per row, then bootstrapped over rows. Paired because the rows differ
2953
+ * enormously from each other and not at all between arms: the same trajectory,
2954
+ * the same image, the same held-out suite, the same continuation policy, and
2955
+ * one difference — the state the continuation starts from.
2956
+ *
2957
+ * Every admitted row stays in the denominator, including the ones an analyst
2958
+ * declined and the ones whose answer was rejected. Those rows contribute a
2959
+ * paired difference of exactly zero, because with no intervention to run their
2960
+ * arm IS their control arm. Declining is therefore free of error and free of
2961
+ * reward, which is what makes it an honest answer rather than a way to pick an
2962
+ * easy subset.
2963
+ *
2964
+ * The report never collapses to this one number. The funnel counts, the rates
2965
+ * on measured rows alone, the per-row table and the threats travel with it,
2966
+ * because a difference of means computed on a corpus that was admitted on the
2967
+ * control failing is conditional on that admission and the reader has to see
2968
+ * it to price it.
2969
+ */
2970
+ function deltaRepair(rowResults, options = {}) {
2971
+ if (rowResults.length === 0) throw new ValidationError("deltaRepair needs at least one graded row");
2972
+ const seen = /* @__PURE__ */ new Set();
2973
+ for (const row of rowResults) {
2974
+ if (seen.has(row.rowId)) throw new ValidationError(`deltaRepair received row ${row.rowId} twice`);
2975
+ seen.add(row.rowId);
2976
+ }
2977
+ const funnel = countFunnel(rowResults.map((row) => row.grade));
2978
+ const control = rowResults.map((row) => row.controlRate);
2979
+ const intervention = rowResults.map((row) => row.interventionRate);
2980
+ const deltaAll = interval(control, intervention, options);
2981
+ const measured = rowResults.filter((row) => row.grade.outcome === "measured");
2982
+ const measuredOnly = measured.length === 0 ? emptyInterval(options) : interval(measured.map((row) => row.controlRate), measured.map((row) => row.interventionRate), options);
2983
+ return {
2984
+ rows: rowResults.length,
2985
+ funnel,
2986
+ interventionRate: mean(intervention),
2987
+ controlRate: mean(control),
2988
+ deltaRepair: deltaAll,
2989
+ measuredOnly,
2990
+ measuredRows: measured.length,
2991
+ rowResults,
2992
+ threats: collectThreats(rowResults, funnel, deltaAll)
2993
+ };
2994
+ }
2995
+ function interval(control, intervention, options) {
2996
+ const result = pairedBootstrap([...control], [...intervention], {
2997
+ statistic: "mean",
2998
+ seed: options.seed,
2999
+ resamples: options.resamples,
3000
+ confidence: options.confidence
3001
+ });
3002
+ return {
3003
+ n: result.n,
3004
+ mean: result.mean,
3005
+ median: result.median,
3006
+ low: result.low,
3007
+ high: result.high,
3008
+ confidence: result.confidence,
3009
+ resamples: result.resamples,
3010
+ gateEligible: result.gateEligible
3011
+ };
3012
+ }
3013
+ function emptyInterval(options) {
3014
+ return {
3015
+ n: 0,
3016
+ mean: 0,
3017
+ median: 0,
3018
+ low: 0,
3019
+ high: 0,
3020
+ confidence: options.confidence ?? .95,
3021
+ resamples: options.resamples ?? 2e3,
3022
+ gateEligible: false
3023
+ };
3024
+ }
3025
+ function collectThreats(rowResults, funnel, delta) {
3026
+ const threats = [{
3027
+ id: "control-position-asymmetry",
3028
+ statement: "The no-fix control continues from the recorded end state, while the intervention arm continues from step k and has to redo the work the recording did after k inside the same step budget. The arms are matched on policy and budget, not on position.",
3029
+ direction: "understates"
3030
+ }];
3031
+ const inert = rowResults.filter((row) => row.controlScreening === "declared-inert");
3032
+ if (inert.length > 0) threats.push({
3033
+ id: "control-cannot-rescue",
3034
+ statement: `${inert.length}/${rowResults.length} rows were screened under a control that makes no model call, so its rollouts graded the same bytes the end-state check graded as failing. A control rate of zero on those rows is a restatement of the end-state check, not a measurement of what continuing alone can repair.`,
3035
+ direction: "unknown"
3036
+ });
3037
+ if (inert.length === 0 && rowResults.every((row) => row.controlRate === 0)) threats.push({
3038
+ id: "admission-conditions-on-control-failure",
3039
+ statement: "Every row was admitted on its no-fix control failing every rollout, so the control rate is zero everywhere and Delta-repair equals the intervention rate. The estimate is conditional on that admission and does not describe rows the control can already repair.",
3040
+ direction: "unknown"
3041
+ });
3042
+ if (!delta.gateEligible) threats.push({
3043
+ id: "bootstrap-below-min-n",
3044
+ statement: `The interval covers ${delta.n} paired rows, below the 20 where a percentile bootstrap holds its nominal error rate. Read it as descriptive spread; a promotion must not turn on it.`,
3045
+ direction: "unknown"
3046
+ });
3047
+ const divergent = rowResults.filter((row) => measuredPrefixDivergences(row.grade) > 0).length;
3048
+ if (divergent > 0) threats.push({
3049
+ id: "prefix-divergence-present",
3050
+ statement: `${divergent}/${rowResults.length} rows replayed at least one prefix step to a different exit code than the recording. The state the intervention landed on is close to the recorded one, not identical to it.`,
3051
+ direction: "unknown"
3052
+ });
3053
+ const withFailures = rowResults.filter((row) => row.grade.outcome === "measured" && row.grade.repair.interventionFailures > 0).length;
3054
+ if (withFailures > 0) threats.push({
3055
+ id: "intervention-failures-present",
3056
+ statement: `${withFailures} rows had at least one rollout where the intervention failed to run after it had already run cleanly in the local-flip session. Those rollouts count as non-passes.`,
3057
+ direction: "understates"
3058
+ });
3059
+ if (delta.low === delta.high) threats.push({
3060
+ id: "zero-variance-interval",
3061
+ statement: "Every resample produced the same statistic, so the interval has zero width. That is an absence of variation in the data, not certainty about the effect.",
3062
+ direction: "unknown"
3063
+ });
3064
+ if (funnel.declined + funnel.rejected > funnel.rows / 2) threats.push({
3065
+ id: "declines-carry-the-denominator",
3066
+ statement: `${funnel.declined} declined and ${funnel.rejected} rejected of ${funnel.rows} rows contribute a paired delta of zero. The headline is dominated by rows where no intervention ran.`,
3067
+ direction: "understates"
3068
+ });
3069
+ return threats;
3070
+ }
3071
+ function measuredPrefixDivergences(grade) {
3072
+ if (grade.outcome === "measured" || grade.outcome === "did-not-execute") return grade.execution.prefix.divergences;
3073
+ if (grade.outcome === "not-reproduced" && grade.reproduction.basis !== "no-recorded-observation") return grade.reproduction.prefix.divergences;
3074
+ return 0;
3075
+ }
3076
+ function mean(values) {
3077
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
3078
+ }
3079
+ /** Markdown report: provenance, the funnel, every per-row column, the
3080
+ * distribution, and the threats. */
3081
+ function renderDeltaRepairReport(report) {
3082
+ const lines = [];
3083
+ const pct = (value) => `${(value * 100).toFixed(1)} %`;
3084
+ lines.push("# Delta-repair");
3085
+ lines.push("");
3086
+ lines.push(`**${signed(report.deltaRepair.mean)}** paired mean over ${report.rows} admitted rows (${(report.deltaRepair.confidence * 100).toFixed(0)} % CI ${signed(report.deltaRepair.low)} … ${signed(report.deltaRepair.high)}, ${report.deltaRepair.resamples} resamples, gate-eligible: ${report.deltaRepair.gateEligible}).`);
3087
+ lines.push("");
3088
+ lines.push(`P(tests pass | intervention) = ${pct(report.interventionRate)}; P(tests pass | no-fix control) = ${pct(report.controlRate)}.`);
3089
+ lines.push("");
3090
+ lines.push("## Funnel");
3091
+ lines.push("");
3092
+ lines.push("| cell | rows | share |");
3093
+ lines.push("| --- | --- | --- |");
3094
+ const f = report.funnel;
3095
+ const share = (value) => pct(f.rows === 0 ? 0 : value / f.rows);
3096
+ for (const [label, value] of [
3097
+ ["admitted rows", f.rows],
3098
+ ["t0 parsed", f.t0Parsed],
3099
+ ["t1 reproduced (gate, pays nothing)", f.t1Reproduced],
3100
+ ["t2 intervention executes", f.t2Executed],
3101
+ ["t3 local flip", f.t3LocalFlip],
3102
+ ["t4 repair flip (any rollout)", f.t4RepairFlipAny],
3103
+ ["t4 repair flip (every rollout)", f.t4RepairFlipAll],
3104
+ ["no-decisive-failure", f.declined],
3105
+ ["rejected", f.rejected]
3106
+ ]) lines.push(`| ${label} | ${value} | ${share(value)} |`);
3107
+ lines.push("");
3108
+ lines.push("## Rows");
3109
+ lines.push("");
3110
+ lines.push("| row | outcome | k | reproduction basis | reproduced | intervention exit | local flip | repair passes | rollouts | intervention failures | prefix steps | prefix divergences | P(int) | P(ctl) | delta | wall ms |");
3111
+ lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
3112
+ for (const row of report.rowResults) lines.push(`| ${rowCells(row).join(" | ")} |`);
3113
+ lines.push("");
3114
+ lines.push("## Measured rows only");
3115
+ lines.push("");
3116
+ lines.push(`${report.measuredRows} rows reached t2. Paired mean ${signed(report.measuredOnly.mean)} (CI ${signed(report.measuredOnly.low)} … ${signed(report.measuredOnly.high)}). Conditional on the analyst answering, so it is not the headline.`);
3117
+ lines.push("");
3118
+ lines.push("## Threats");
3119
+ lines.push("");
3120
+ for (const threat of report.threats) lines.push(`- **${threat.id}** (${threat.direction} the effect) — ${threat.statement}`);
3121
+ lines.push("");
3122
+ return lines.join("\n");
3123
+ }
3124
+ function rowCells(row) {
3125
+ const grade = row.grade;
3126
+ const reproduction = grade.outcome === "not-reproduced" || grade.outcome === "did-not-execute" || grade.outcome === "measured" ? grade.reproduction : null;
3127
+ const execution = grade.outcome === "did-not-execute" || grade.outcome === "measured" ? grade.execution : null;
3128
+ const repair = grade.outcome === "measured" ? grade.repair : null;
3129
+ const prefix = execution?.prefix ?? (reproduction && reproduction.basis !== "no-recorded-observation" ? reproduction.prefix : null);
3130
+ const rejection = grade.outcome === "rejected" ? `rejected: ${grade.rejection.reason}` : grade.outcome;
3131
+ return [
3132
+ row.rowId,
3133
+ rejection,
3134
+ grade.outcome === "rejected" || grade.outcome === "declined" ? "—" : String(grade.k),
3135
+ reproduction ? reproduction.basis : "—",
3136
+ reproduction ? String(reproduction.reproduced) : "—",
3137
+ execution ? String(execution.exitCode) : "—",
3138
+ grade.outcome === "measured" ? String(grade.localFlip.passed) : "—",
3139
+ repair ? String(repair.passes) : "—",
3140
+ repair ? String(repair.rollouts) : "0",
3141
+ repair ? String(repair.interventionFailures) : "—",
3142
+ prefix ? String(prefix.stepsReplayed) : "—",
3143
+ prefix ? String(prefix.divergences) : "—",
3144
+ row.interventionRate.toFixed(3),
3145
+ row.controlRate.toFixed(3),
3146
+ signed(row.delta),
3147
+ String(row.wallMs)
3148
+ ];
3149
+ }
3150
+ function signed(value) {
3151
+ const points = value * 100;
3152
+ return `${points >= 0 ? "+" : ""}${points.toFixed(1)} pp`;
3153
+ }
3154
+ //#endregion
3155
+ //#region src/trace-repair/docker-environment.ts
3156
+ /**
3157
+ * Docker-backed continuation environment.
3158
+ *
3159
+ * The runner refuses any container whose network mode is not `none`, so this
3160
+ * builds containers that way and reports the mode it reads back from the
3161
+ * daemon rather than the mode it asked for. A container created elsewhere and
3162
+ * attached here is described the same way, so an environment that quietly kept
3163
+ * its network is rejected instead of producing evidence.
3164
+ *
3165
+ * Commands are bounded twice. `timeout` inside the container is the primary
3166
+ * bound, because killing the host-side `docker exec` client leaves the process
3167
+ * it started running in the container, where it would keep writing files under
3168
+ * later steps. The host-side kill is a backstop for a daemon that stops
3169
+ * answering. The container must therefore provide `timeout`; `describe()`
3170
+ * checks for it and refuses the container when it is absent.
3171
+ */
3172
+ /**
3173
+ * Spawns a process, merges stdout and stderr the way the scaffold reads them,
3174
+ * and kills the whole process group on timeout so a killed command leaves no
3175
+ * children running in the container's namespace.
3176
+ */
3177
+ const nodeProcessRunner = (request) => new Promise((resolve, reject) => {
3178
+ const [command, ...args] = request.argv;
3179
+ if (!command) {
3180
+ reject(new ValidationError("process runner requires a command"));
3181
+ return;
3182
+ }
3183
+ const child = spawn(command, args, {
3184
+ detached: true,
3185
+ stdio: [
3186
+ "ignore",
3187
+ "pipe",
3188
+ "pipe"
3189
+ ]
3190
+ });
3191
+ const chunks = [];
3192
+ let timedOut = false;
3193
+ child.stdout.setEncoding("utf8");
3194
+ child.stderr.setEncoding("utf8");
3195
+ child.stdout.on("data", (chunk) => chunks.push(chunk));
3196
+ child.stderr.on("data", (chunk) => chunks.push(chunk));
3197
+ const timer = request.timeoutSeconds === void 0 ? void 0 : setTimeout(() => {
3198
+ timedOut = true;
3199
+ if (child.pid === void 0) return;
3200
+ try {
3201
+ process.kill(-child.pid, "SIGKILL");
3202
+ } catch (error) {
3203
+ if (error.code !== "ESRCH") reject(error);
3204
+ }
3205
+ }, request.timeoutSeconds * 1e3);
3206
+ child.on("error", (error) => {
3207
+ if (timer) clearTimeout(timer);
3208
+ reject(error);
3209
+ });
3210
+ child.on("close", (code, signal) => {
3211
+ if (timer) clearTimeout(timer);
3212
+ const output = chunks.join("");
3213
+ if (code !== null) {
3214
+ resolve({
3215
+ output,
3216
+ exitCode: code,
3217
+ timedOut
3218
+ });
3219
+ return;
3220
+ }
3221
+ const signalNumber = signal === null ? void 0 : constants.signals[signal];
3222
+ if (signalNumber === void 0) {
3223
+ reject(new ValidationError(`process closed with no exit code and no known signal: ${signal}`));
3224
+ return;
3225
+ }
3226
+ resolve({
3227
+ output,
3228
+ exitCode: -signalNumber,
3229
+ timedOut
3230
+ });
3231
+ });
3232
+ });
3233
+ const DEFAULT_INTERPRETER = ["bash", "-lc"];
3234
+ const DEFAULT_EXECUTABLE = "docker";
3235
+ /** Seconds `timeout` waits after SIGTERM before sending SIGKILL. */
3236
+ const KILL_GRACE_SECONDS = 5;
3237
+ /** Extra seconds the host-side backstop waits after the in-container bound. */
3238
+ const BACKSTOP_SECONDS = 10;
3239
+ /** `timeout` reports this when it stopped the command. */
3240
+ const TIMEOUT_EXIT_CODE = 124;
3241
+ /**
3242
+ * Arguments that create a container the policy accepts. `--network none` is
3243
+ * not optional: a continuation with network access could install what the
3244
+ * recorded run could not, and the arms would no longer differ only by the
3245
+ * intervention.
3246
+ */
3247
+ function dockerRunArgs(input) {
3248
+ return [
3249
+ input.executable ?? DEFAULT_EXECUTABLE,
3250
+ "run",
3251
+ "-d",
3252
+ "--name",
3253
+ input.name,
3254
+ "--network",
3255
+ "none",
3256
+ "-w",
3257
+ input.cwd,
3258
+ "--rm",
3259
+ input.image,
3260
+ "sleep",
3261
+ input.containerLifetime ?? "2h"
3262
+ ];
3263
+ }
3264
+ function createDockerContinuationEnvironment(options) {
3265
+ if (!options.containerRef.trim()) throw new ValidationError("docker continuation environment requires a container reference");
3266
+ const executable = options.executable ?? DEFAULT_EXECUTABLE;
3267
+ const interpreter = options.interpreter ?? DEFAULT_INTERPRETER;
3268
+ const envEntries = Object.entries(options.env ?? {});
3269
+ return {
3270
+ containerRef: options.containerRef,
3271
+ async describe() {
3272
+ const inspect = await options.runProcess({ argv: [
3273
+ executable,
3274
+ "inspect",
3275
+ "--format",
3276
+ "{{.HostConfig.NetworkMode}} {{.Config.Image}}",
3277
+ options.containerRef
3278
+ ] });
3279
+ if (inspect.exitCode !== 0) throw new ValidationError(`docker inspect failed for ${options.containerRef} (exit ${inspect.exitCode}): ${inspect.output.trim()}`);
3280
+ const [networkMode, image] = inspect.output.trim().split(" ");
3281
+ if (!networkMode) throw new ValidationError(`docker inspect returned no network mode for ${options.containerRef}`);
3282
+ if ((await options.runProcess({ argv: [
3283
+ executable,
3284
+ "exec",
3285
+ options.containerRef,
3286
+ ...interpreter,
3287
+ "command -v timeout"
3288
+ ] })).exitCode !== 0) throw new ValidationError(`container ${options.containerRef} provides no \`timeout\`, so a long command cannot be bounded inside it`);
3289
+ return image ? {
3290
+ networkMode,
3291
+ image
3292
+ } : { networkMode };
3293
+ },
3294
+ async exec(command, execOptions) {
3295
+ const argv = [
3296
+ executable,
3297
+ "exec",
3298
+ "-w",
3299
+ options.cwd
3300
+ ];
3301
+ for (const [key, value] of envEntries) argv.push("-e", `${key}=${value}`);
3302
+ argv.push(options.containerRef, "timeout", `--kill-after=${KILL_GRACE_SECONDS}s`, `${execOptions.timeoutSeconds}s`, ...interpreter, command);
3303
+ const result = await options.runProcess({
3304
+ argv,
3305
+ timeoutSeconds: execOptions.timeoutSeconds + KILL_GRACE_SECONDS + BACKSTOP_SECONDS
3306
+ });
3307
+ const timedOut = result.timedOut || result.exitCode === TIMEOUT_EXIT_CODE;
3308
+ return {
3309
+ output: result.output,
3310
+ returncode: result.exitCode,
3311
+ timedOut
3312
+ };
3313
+ },
3314
+ async dispose() {
3315
+ if (!options.removeOnDispose) return;
3316
+ await options.runProcess({ argv: [
3317
+ executable,
3318
+ "rm",
3319
+ "-f",
3320
+ options.containerRef
3321
+ ] });
3322
+ }
3323
+ };
3324
+ }
3325
+ //#endregion
3326
+ //#region src/trace-repair/grade.ts
3327
+ /**
3328
+ * Grade one analyst answer about one admitted row.
3329
+ *
3330
+ * The whole funnel runs here, in the order that spends the least before it
3331
+ * knows: parse and budget checks open no container, the reproduction gate
3332
+ * opens one, the local flip opens one more, and only then does the repair arm
3333
+ * pay for its rollouts.
3334
+ *
3335
+ * Two properties a reviewer should be able to check by reading this file:
3336
+ *
3337
+ * The intervention is applied at the k the analyst named, on the state
3338
+ * produced by replaying steps 1..k-1, and nowhere else. There is no search
3339
+ * over nearby steps, no credit for being close, and no label in scope — the
3340
+ * grader never receives one. A wrong k therefore scores zero the only honest
3341
+ * way it can: the repair has to work where the analyst said the failure was.
3342
+ *
3343
+ * The row must arrive branded by `admitRow`. That is the type-level form of
3344
+ * "admission runs before any analyst sees a row": there is no signature here
3345
+ * that accepts an unadmitted row.
3346
+ */
3347
+ /** Rollouts of the repair arm disagreed with the controls about the policy. */
3348
+ var RepairArmSymmetryError = class extends CaptureIntegrityError {};
3349
+ const DEFAULT_STEP_TIMEOUT_MS = 3e5;
3350
+ const OUTPUT_EXCERPT_CHARS = 4e3;
3351
+ async function gradeRepairRow(options) {
3352
+ const startedMs = Date.now();
3353
+ const grade = await produceGrade(options);
3354
+ return finish(options.row, grade, startedMs);
3355
+ }
3356
+ function finish(row, grade, startedMs) {
3357
+ const credit = repairCredit(grade);
3358
+ const interventionRate = grade.outcome === "measured" ? credit.repairRate : row.controlRate;
3359
+ return {
3360
+ rowId: row.rowId,
3361
+ grade,
3362
+ credit,
3363
+ interventionRate,
3364
+ controlRate: row.controlRate,
3365
+ controlRollouts: row.controlRollouts,
3366
+ controlScreening: row.controlScreening,
3367
+ controlPolicyDigest: row.policyDigest,
3368
+ repairRollouts: grade.outcome === "measured" ? grade.repair.rollouts : 0,
3369
+ delta: interventionRate - row.controlRate,
3370
+ wallMs: Date.now() - startedMs
3371
+ };
3372
+ }
3373
+ async function produceGrade(options) {
3374
+ const { row, response } = options;
3375
+ if (response.kind === "no-decisive-failure") return { outcome: "declined" };
3376
+ const target = targetStep(row, response);
3377
+ if (!target.ok) return target.grade;
3378
+ const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
3379
+ const check = checkInterventionBudget(response.intervention.action, response.intervention.kind, budget);
3380
+ if (!check.admissible) return {
3381
+ outcome: "rejected",
3382
+ rejection: {
3383
+ source: "budget",
3384
+ reason: check.violation,
3385
+ detail: check.detail,
3386
+ measurement: check.measurement
3387
+ }
3388
+ };
3389
+ if (normalizeActionForComparison(response.intervention.action) === normalizeActionForComparison(target.step.action)) return {
3390
+ outcome: "rejected",
3391
+ rejection: {
3392
+ source: "target",
3393
+ reason: "recorded-action-reproposed",
3394
+ detail: `the intervention is the action already recorded at step ${response.k}`
3395
+ }
3396
+ };
3397
+ const reproduction = await runReproductionGate(options, response, target.step);
3398
+ if (!reproduction.reproduced) return {
3399
+ outcome: "not-reproduced",
3400
+ k: response.k,
3401
+ reproduction
3402
+ };
3403
+ const local = await runLocalFlip(options, response);
3404
+ if (local.kind === "did-not-execute") return {
3405
+ outcome: "did-not-execute",
3406
+ k: response.k,
3407
+ reproduction,
3408
+ execution: local.execution
3409
+ };
3410
+ const repair = await runRepairArm(options, response);
3411
+ return {
3412
+ outcome: "measured",
3413
+ k: response.k,
3414
+ reproduction,
3415
+ execution: local.execution,
3416
+ localFlip: local.tests,
3417
+ repair
3418
+ };
3419
+ }
3420
+ function targetStep(row, finding) {
3421
+ if (finding.k < 1 || finding.k > row.steps.length) return {
3422
+ ok: false,
3423
+ grade: {
3424
+ outcome: "rejected",
3425
+ rejection: {
3426
+ source: "target",
3427
+ reason: "k-out-of-range",
3428
+ detail: `k=${finding.k} is outside [1, ${row.steps.length}]`
3429
+ }
3430
+ }
3431
+ };
3432
+ const step = row.steps[finding.k - 1];
3433
+ if (step.step_id !== finding.k) throw new ValidationError(`trace-repair: row ${row.rowId} steps[${finding.k - 1}].step_id=${step.step_id} != ${finding.k}; an admitted row must carry 1-based contiguous step ids`);
3434
+ return {
3435
+ ok: true,
3436
+ step
3437
+ };
3438
+ }
3439
+ async function runReproductionGate(options, finding, target) {
3440
+ const { row } = options;
3441
+ if (target.observation === null) {
3442
+ options.onProgress?.(`row ${row.rowId}: step ${finding.k} recorded no observation; the reproduction gate passes vacuously`);
3443
+ return {
3444
+ basis: "no-recorded-observation",
3445
+ reproduced: true
3446
+ };
3447
+ }
3448
+ const recordedReturncode = parseRecordedReturncode(target.observation);
3449
+ if (recordedReturncode === null) throw new ValidationError(`trace-repair: row ${row.rowId} step ${finding.k} carries an observation with no <returncode>; the corpus row is malformed and must not have been admitted`);
3450
+ const signature = deriveFailureSignature(target.observation);
3451
+ const session = await options.sessions.open({
3452
+ rowId: row.rowId,
3453
+ image: row.image,
3454
+ arm: "reproduce",
3455
+ rolloutIndex: 0
3456
+ });
3457
+ try {
3458
+ const prefix = await replayPrefix(session, options, finding.k);
3459
+ const result = await session.exec(wrapActionForExec(target.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
3460
+ const output = `${result.stdout}\n${result.stderr}`;
3461
+ const signatureObserved = signature === null ? null : output.includes(signature);
3462
+ const reproduced = !result.timedOut && result.exitCode === recordedReturncode && signatureObserved !== false;
3463
+ options.onProgress?.(`row ${row.rowId}: reproduction at step ${finding.k} exit=${result.exitCode} (recorded ${recordedReturncode}) reproduced=${reproduced}`);
3464
+ return {
3465
+ basis: signature === null ? "returncode-only" : "returncode+output-substring",
3466
+ reproduced,
3467
+ recordedReturncode,
3468
+ observedExitCode: result.exitCode,
3469
+ signature,
3470
+ signatureObserved,
3471
+ prefix
3472
+ };
3473
+ } finally {
3474
+ await session.close();
3475
+ }
3476
+ }
3477
+ async function runLocalFlip(options, finding) {
3478
+ const { row } = options;
3479
+ const session = await options.sessions.open({
3480
+ rowId: row.rowId,
3481
+ image: row.image,
3482
+ arm: "local-flip",
3483
+ rolloutIndex: 0
3484
+ });
3485
+ try {
3486
+ const prefix = await replayPrefix(session, options, finding.k);
3487
+ const startedMs = Date.now();
3488
+ const result = await session.exec(wrapActionForExec(finding.intervention.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
3489
+ const output = `${result.stdout}\n${result.stderr}`.trim();
3490
+ const execution = {
3491
+ command: finding.intervention.action,
3492
+ exitCode: result.exitCode,
3493
+ timedOut: result.timedOut,
3494
+ wallMs: Date.now() - startedMs,
3495
+ output: excerpt(output),
3496
+ prefix
3497
+ };
3498
+ if (result.timedOut || result.exitCode !== 0) {
3499
+ options.onProgress?.(`row ${row.rowId}: the intervention did not run at step ${finding.k} (exit=${result.exitCode} timedOut=${result.timedOut})`);
3500
+ return {
3501
+ kind: "did-not-execute",
3502
+ execution
3503
+ };
3504
+ }
3505
+ const tests = await gradeTests(options, session, "local-flip", 0);
3506
+ options.onProgress?.(`row ${row.rowId}: local flip=${tests.passed} after the intervention at step ${finding.k}`);
3507
+ return {
3508
+ kind: "measured",
3509
+ execution,
3510
+ tests
3511
+ };
3512
+ } finally {
3513
+ await session.close();
3514
+ }
3515
+ }
3516
+ async function runRepairArm(options, finding) {
3517
+ const { row } = options;
3518
+ const rollouts = options.repairRollouts ?? row.controlRollouts;
3519
+ if (!Number.isInteger(rollouts) || rollouts <= 0) throw new ValidationError(`repairRollouts must be a positive integer, got ${rollouts}`);
3520
+ const evidence = [];
3521
+ let passes = 0;
3522
+ let interventionFailures = 0;
3523
+ let policyDigest = null;
3524
+ for (let index = 0; index < rollouts; index += 1) {
3525
+ const session = await options.sessions.open({
3526
+ rowId: row.rowId,
3527
+ image: row.image,
3528
+ arm: "intervention",
3529
+ rolloutIndex: index
3530
+ });
3531
+ try {
3532
+ await replayPrefix(session, options, finding.k);
3533
+ const result = await session.exec(wrapActionForExec(finding.intervention.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
3534
+ if (result.timedOut || result.exitCode !== 0) {
3535
+ interventionFailures += 1;
3536
+ evidence.push({
3537
+ rolloutIndex: index,
3538
+ status: "intervention-failed",
3539
+ interventionExitCode: result.exitCode,
3540
+ timedOut: result.timedOut
3541
+ });
3542
+ continue;
3543
+ }
3544
+ const continuation = await options.continuation({
3545
+ rowId: row.rowId,
3546
+ arm: "intervention",
3547
+ rolloutIndex: index,
3548
+ session,
3549
+ steps: row.steps,
3550
+ k: finding.k,
3551
+ injected: {
3552
+ action: finding.intervention.action,
3553
+ returncode: result.exitCode,
3554
+ output: `${result.stdout}\n${result.stderr}`.trim(),
3555
+ timedOut: result.timedOut
3556
+ },
3557
+ taskStatement: row.taskStatement
3558
+ });
3559
+ assertPolicySymmetry(row, continuation, index);
3560
+ policyDigest = continuation.policyDigest;
3561
+ const tests = await gradeTests(options, session, "intervention", index);
3562
+ if (tests.passed) passes += 1;
3563
+ evidence.push({
3564
+ rolloutIndex: index,
3565
+ status: "completed",
3566
+ interventionExitCode: result.exitCode,
3567
+ continuation,
3568
+ tests
3569
+ });
3570
+ options.onProgress?.(`row ${row.rowId}: repair rollout ${index} tests=${tests.passed} after ${continuation.steps} continuation steps (${continuation.exitStatus})`);
3571
+ } finally {
3572
+ await session.close();
3573
+ }
3574
+ }
3575
+ return {
3576
+ rollouts,
3577
+ passes,
3578
+ interventionFailures,
3579
+ policyDigest,
3580
+ rolloutEvidence: evidence
3581
+ };
3582
+ }
3583
+ function assertPolicySymmetry(row, continuation, index) {
3584
+ if (continuation.policyDigest !== row.policyDigest) throw new RepairArmSymmetryError(`row ${row.rowId} repair rollout ${index} ran policy ${continuation.policyDigest} but its controls ran ${row.policyDigest}; the difference between the arms would not be the intervention`);
3585
+ }
3586
+ async function gradeTests(options, session, arm, rolloutIndex) {
3587
+ const outcome = await options.oracle.grade(session, {
3588
+ rowId: options.row.rowId,
3589
+ arm,
3590
+ rolloutIndex
3591
+ });
3592
+ if (outcome.suiteDigest !== options.row.suiteDigest) throw new RepairArmSymmetryError(`row ${options.row.rowId} was admitted against suite ${options.row.suiteDigest} but the ${arm} arm graded against ${outcome.suiteDigest}; the arms did not answer the same question`);
3593
+ return {
3594
+ passed: outcome.passed,
3595
+ exitCode: outcome.exitCode,
3596
+ timedOut: outcome.timedOut,
3597
+ suiteDigest: outcome.suiteDigest
3598
+ };
3599
+ }
3600
+ /** Replay recorded steps 1..k-1 to rebuild the state the intervention lands
3601
+ * on. Divergence is counted and reported, never repaired. */
3602
+ async function replayPrefix(session, options, k) {
3603
+ const { row } = options;
3604
+ const timeoutMs = options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS;
3605
+ const recordedTimeoutMs = options.recordedTimeoutStepMs ?? timeoutMs;
3606
+ const startedMs = Date.now();
3607
+ let divergences = 0;
3608
+ let stepsReplayed = 0;
3609
+ for (const step of row.steps.slice(0, k - 1)) {
3610
+ const bound = isRecordedTimeout(step.observation) ? recordedTimeoutMs : timeoutMs;
3611
+ const result = await session.exec(wrapActionForExec(step.action, row.cwd), bound);
3612
+ stepsReplayed += 1;
3613
+ const recorded = parseRecordedReturncode(step.observation);
3614
+ if (recorded !== null && recorded !== result.exitCode) divergences += 1;
3615
+ }
3616
+ return {
3617
+ stepsReplayed,
3618
+ divergences,
3619
+ wallMs: Date.now() - startedMs
3620
+ };
3621
+ }
3622
+ function excerpt(text) {
3623
+ if (text.length <= OUTPUT_EXCERPT_CHARS) return text;
3624
+ const half = Math.floor(OUTPUT_EXCERPT_CHARS / 2);
3625
+ return `${text.slice(0, half)}\n… [${text.length - OUTPUT_EXCERPT_CHARS} chars elided] …\n${text.slice(-2e3)}`;
3626
+ }
3627
+ //#endregion
3628
+ //#region src/trace-repair/oracle-determinism.ts
3629
+ /**
3630
+ * Whether a task's own grader is a function of the container state.
3631
+ *
3632
+ * A repair benchmark uses the task's held-out suite as ground truth: the suite
3633
+ * says the state before the intervention fails and the state after it passes,
3634
+ * and the difference is the measurement. That reading needs the suite to answer
3635
+ * the same way twice on the same bytes. A suite that asserts on wall clock does
3636
+ * not: the same container passes or fails by chance, so a control arm can
3637
+ * "rescue" a row nothing touched, and an intervention arm can lose a repair it
3638
+ * made.
3639
+ *
3640
+ * The check is direct. Grade the same container N times with nothing written
3641
+ * between the runs, pool the replicates by the state they graded, and read the
3642
+ * minority share.
3643
+ *
3644
+ * It reads that share per ASSERTION, not per suite, wherever the suite reports
3645
+ * assertions. A pass/fail reward is a conjunction over many assertions, so a
3646
+ * suite whose timing assertions each flip independently can still return the
3647
+ * same reward on every replicate of a state that sits far from the threshold —
3648
+ * and then return a coin flip on the states a campaign actually grades, which
3649
+ * sit near it. Per-assertion counting sees the flip at the anchor; reward
3650
+ * counting does not. A state whose replicates carried no assertion report falls
3651
+ * back to the reward and records that it did, so a coarser measurement is
3652
+ * visible rather than assumed equivalent.
3653
+ *
3654
+ * Replicates are also run under machine contention, because unanimity on an
3655
+ * idle box proves nothing about a threshold the load never approached.
3656
+ * Replicates on one state pool across both loads, so a verdict that moves when
3657
+ * the machine gets busy reads as the flip it is.
3658
+ */
3659
+ /** A task graded a repair by something other than the state under test. */
3660
+ var NondeterministicOracleError = class extends CaptureIntegrityError {};
3661
+ /** The unit a whole-suite verdict is counted under when nothing finer exists. */
3662
+ const SUITE_REWARD_UNIT = "suite-reward";
3663
+ /** Replicates one group needs before a flip is measurable at all. */
3664
+ const MIN_ORACLE_REPLICATES = 2;
3665
+ /**
3666
+ * Reduce measured replicates to a verdict.
3667
+ *
3668
+ * Pure: it opens no container. The tool that runs the replicates hands the
3669
+ * counts here, so the rule that decides stability is one rule and a reviewer
3670
+ * can re-derive any verdict from the recorded replicates.
3671
+ */
3672
+ function oracleDeterminism(evidence) {
3673
+ if (evidence.groups.length === 0) throw new ValidationError(`oracle determinism for ${evidence.taskName} received no replicate group`);
3674
+ for (const group of evidence.groups) if (group.replicates.length < 2) throw new ValidationError(`oracle determinism for ${evidence.taskName} received ${group.replicates.length} ${group.load} replicate(s) on the ${group.state} state; 2 are needed before a flip can be observed`);
3675
+ const byState = [];
3676
+ for (const state of ["unsolved", "solved"]) {
3677
+ const groups = evidence.groups.filter((group) => group.state === state);
3678
+ if (groups.length === 0) continue;
3679
+ byState.push(stateVerdict(state, groups));
3680
+ }
3681
+ const flipRate = byState.reduce((worst, state) => Math.max(worst, state.flipRate), 0);
3682
+ const replicates = byState.reduce((total, state) => total + state.replicates, 0);
3683
+ return {
3684
+ taskName: evidence.taskName,
3685
+ image: evidence.image,
3686
+ suiteDigest: evidence.suiteDigest,
3687
+ stable: flipRate === 0,
3688
+ replicates,
3689
+ flipRate,
3690
+ byState,
3691
+ measuredAt: evidence.measuredAt,
3692
+ detail: byState.map(describeState).join("; ")
3693
+ };
3694
+ }
3695
+ function stateVerdict(state, groups) {
3696
+ const replicates = groups.flatMap((group) => [...group.replicates]);
3697
+ const passes = replicates.filter((replicate) => replicate.passed).length;
3698
+ const rates = groups.map((group) => group.replicates.filter((r) => r.passed).length / group.replicates.length);
3699
+ const perAssertion = replicates.every((replicate) => replicate.assertions !== null);
3700
+ const flipped = [];
3701
+ const assertionSetUnstable = [];
3702
+ if (perAssertion) {
3703
+ const outcomes = /* @__PURE__ */ new Map();
3704
+ for (const replicate of replicates) for (const assertion of replicate.assertions ?? []) {
3705
+ const seen = outcomes.get(assertion.id);
3706
+ if (seen) seen.push(assertion.passed);
3707
+ else outcomes.set(assertion.id, [assertion.passed]);
3708
+ }
3709
+ for (const [id, results] of [...outcomes].sort(([a], [b]) => a.localeCompare(b))) {
3710
+ if (results.length !== replicates.length) assertionSetUnstable.push(id);
3711
+ const unitPasses = results.filter(Boolean).length;
3712
+ const unitFails = results.length - unitPasses;
3713
+ const minority = Math.min(unitPasses, unitFails);
3714
+ const absent = replicates.length - results.length;
3715
+ if (minority + absent > 0) flipped.push({
3716
+ unit: id,
3717
+ passes: unitPasses,
3718
+ fails: unitFails + absent,
3719
+ flipRate: (minority + absent) / replicates.length
3720
+ });
3721
+ }
3722
+ } else {
3723
+ const minority = Math.min(passes, replicates.length - passes);
3724
+ if (minority > 0) flipped.push({
3725
+ unit: SUITE_REWARD_UNIT,
3726
+ passes,
3727
+ fails: replicates.length - passes,
3728
+ flipRate: minority / replicates.length
3729
+ });
3730
+ }
3731
+ return {
3732
+ state,
3733
+ replicates: replicates.length,
3734
+ passes,
3735
+ fails: replicates.length - passes,
3736
+ granularity: perAssertion ? "per-assertion" : "reward",
3737
+ flipRate: flipped.reduce((worst, unit) => Math.max(worst, unit.flipRate), 0),
3738
+ flipped,
3739
+ assertionSetUnstable,
3740
+ loadSensitive: new Set(rates).size > 1,
3741
+ rewardsObserved: [...new Set(replicates.map((r) => r.reward ?? "NO_REWARD_FILE"))].sort()
3742
+ };
3743
+ }
3744
+ function describeState(state) {
3745
+ const base = `${state.state}: ${state.passes}/${state.replicates} suite pass, ${state.granularity} counting`;
3746
+ if (state.flipped.length === 0) return `${base}, no flip`;
3747
+ const worst = state.flipped.reduce((a, b) => b.flipRate > a.flipRate ? b : a);
3748
+ return `${base}, ${state.flipped.length} unit(s) flipped, worst ${worst.unit} ${worst.passes}/${worst.passes + worst.fails} pass` + (state.loadSensitive ? " (load-sensitive)" : "");
3749
+ }
3750
+ function parseTaskOracleRegistry(document) {
3751
+ if (typeof document !== "object" || document === null) throw new ValidationError("task oracle registry must be an object");
3752
+ const { version, measurements } = document;
3753
+ if (version !== 1) throw new ValidationError(`task oracle registry version must be 1, got ${String(version)}`);
3754
+ if (!Array.isArray(measurements)) throw new ValidationError("task oracle registry needs a measurements array");
3755
+ return taskOracleRegistry(measurements.map((evidence) => oracleDeterminism(evidence)));
3756
+ }
3757
+ function taskOracleRegistry(verdicts) {
3758
+ const registry = /* @__PURE__ */ new Map();
3759
+ for (const verdict of verdicts) {
3760
+ if (registry.has(verdict.taskName)) throw new ValidationError(`task oracle registry received ${verdict.taskName} twice; a task has one certification`);
3761
+ registry.set(verdict.taskName, verdict);
3762
+ }
3763
+ return registry;
3764
+ }
3765
+ /** Throws unless the task graded the same bytes the same way every replicate. */
3766
+ function assertDeterministicOracle(verdict) {
3767
+ if (verdict.stable) return;
3768
+ throw new NondeterministicOracleError(`task ${verdict.taskName} graded byte-identical state inconsistently: flip rate ${(verdict.flipRate * 100).toFixed(1)} % over ${verdict.replicates} replicates (${verdict.detail}). Its verdict is not a function of the state, so it cannot carry ground truth for an intervention study.`);
3769
+ }
3770
+ //#endregion
3771
+ //#region src/trace-repair/test-oracle.ts
3772
+ /**
3773
+ * The held-out suite, injected from outside the box at grade time.
3774
+ *
3775
+ * A repair is only measurable if the thing that decides pass or fail is out
3776
+ * of the trajectory's reach. Terminal-Bench gets that by uploading the suite
3777
+ * into the container at grade time, after the agent has stopped; a suite the
3778
+ * agent planted is overwritten before it is ever read. `injectedTestOracle`
3779
+ * reproduces that property and then proves it per call:
3780
+ *
3781
+ * 1. purge the suite root, so a planted extra file cannot survive
3782
+ * 2. upload every suite file from outside the session
3783
+ * 3. read the bytes back from inside and hash them
3784
+ * 4. refuse to grade when the read-back digest is not the uploaded digest
3785
+ *
3786
+ * Step 4 is why the property is asserted rather than assumed. A container
3787
+ * that silently drops the upload, or a filesystem trick that serves different
3788
+ * bytes to the reader, raises `TestSuiteTamperedError` instead of returning a
3789
+ * result. An oracle that cannot prove what it graded reports nothing.
3790
+ */
3791
+ /** The suite the oracle read back is not the suite it uploaded. */
3792
+ var TestSuiteTamperedError = class extends CaptureIntegrityError {};
3793
+ /** The oracle could not place or run the suite, so it graded nothing. */
3794
+ var TestOracleError = class extends CaptureIntegrityError {};
3795
+ const DEFAULT_UPLOAD_TIMEOUT_MS = 6e4;
3796
+ const DEFAULT_COMMAND_TIMEOUT_MS = 9e5;
3797
+ /**
3798
+ * Content digest of the suite: sha256 over each path and its bytes, in path
3799
+ * order. Two suites with the same digest are the same suite.
3800
+ */
3801
+ function testSuiteDigest(files) {
3802
+ const hash = createHash("sha256");
3803
+ for (const file of [...files].sort((a, b) => a.path.localeCompare(b.path))) {
3804
+ hash.update(file.path);
3805
+ hash.update("\0");
3806
+ hash.update(Buffer.from(file.contents, "utf8"));
3807
+ hash.update("\0");
3808
+ }
3809
+ return hash.digest("hex");
3810
+ }
3811
+ function injectedTestOracle(options) {
3812
+ assertOracleOptions(options);
3813
+ const uploadTimeoutMs = options.uploadTimeoutMs ?? DEFAULT_UPLOAD_TIMEOUT_MS;
3814
+ const commandTimeoutMs = options.commandTimeoutMs ?? DEFAULT_COMMAND_TIMEOUT_MS;
3815
+ const expectedDigest = testSuiteDigest(options.files);
3816
+ return { async grade(session, context) {
3817
+ const where = `${context.rowId}/${context.arm}#${context.rolloutIndex}`;
3818
+ for (const directory of options.purge ?? []) await mustSucceed(session, `rm -rf ${shellQuote(directory)}`, uploadTimeoutMs, `purge ${directory} in ${where}`);
3819
+ for (const file of options.files) {
3820
+ const directory = parentDirectory(file.path);
3821
+ if (directory) await mustSucceed(session, `mkdir -p ${shellQuote(directory)}`, uploadTimeoutMs, `mkdir ${directory} in ${where}`);
3822
+ await mustSucceed(session, `printf %s ${shellQuote(Buffer.from(file.contents, "utf8").toString("base64"))} | base64 -d > ${shellQuote(file.path)}`, uploadTimeoutMs, `upload ${file.path} in ${where}`);
3823
+ if (file.mode) await mustSucceed(session, `chmod ${file.mode} ${shellQuote(file.path)}`, uploadTimeoutMs, `chmod ${file.path} in ${where}`);
3824
+ }
3825
+ const readBack = [];
3826
+ for (const file of options.files) {
3827
+ const result = await mustSucceed(session, `base64 < ${shellQuote(file.path)} | tr -d '\\n'`, uploadTimeoutMs, `read back ${file.path} in ${where}`);
3828
+ readBack.push({
3829
+ path: file.path,
3830
+ contents: Buffer.from(result.stdout.trim(), "base64").toString("utf8")
3831
+ });
3832
+ }
3833
+ const observedDigest = testSuiteDigest(readBack);
3834
+ if (observedDigest !== expectedDigest) throw new TestSuiteTamperedError(`test suite in ${where} does not match the suite uploaded from outside (expected ${expectedDigest}, read back ${observedDigest}); the graded result is discarded`);
3835
+ const run = await session.exec(options.command, commandTimeoutMs);
3836
+ return {
3837
+ passed: run.exitCode === 0 && !run.timedOut,
3838
+ exitCode: run.exitCode,
3839
+ output: `${run.stdout}\n${run.stderr}`.trim(),
3840
+ suiteDigest: observedDigest,
3841
+ timedOut: run.timedOut
3842
+ };
3843
+ } };
3844
+ }
3845
+ /**
3846
+ * Run a setup command that must succeed. A failed upload is an oracle
3847
+ * failure, never a failed test: reporting it as a failing suite would turn a
3848
+ * broken container into evidence against the intervention.
3849
+ */
3850
+ async function mustSucceed(session, command, timeoutMs, what) {
3851
+ const result = await session.exec(command, timeoutMs);
3852
+ if (result.timedOut) throw new TestOracleError(`test oracle timed out while it tried to ${what}`);
3853
+ if (result.exitCode !== 0) throw new TestOracleError(`test oracle failed to ${what}: exit ${result.exitCode}\n${result.stderr.trim()}`);
3854
+ return result;
3855
+ }
3856
+ function assertOracleOptions(options) {
3857
+ if (options.files.length === 0) throw new ValidationError("injectedTestOracle requires at least one suite file");
3858
+ const seen = /* @__PURE__ */ new Set();
3859
+ for (const file of options.files) {
3860
+ if (!file.path.startsWith("/")) throw new ValidationError(`suite file path must be absolute, got "${file.path}"`);
3861
+ if (seen.has(file.path)) throw new ValidationError(`suite file ${file.path} is listed twice`);
3862
+ seen.add(file.path);
3863
+ }
3864
+ if (options.command.trim().length === 0) throw new ValidationError("injectedTestOracle requires a suite command");
3865
+ for (const directory of options.purge ?? []) if (!directory.startsWith("/") || directory.trim() === "/") throw new ValidationError(`purge path must be an absolute directory below the root, got "${directory}"`);
3866
+ }
3867
+ function parentDirectory(path) {
3868
+ const index = path.lastIndexOf("/");
3869
+ if (index <= 0) return null;
3870
+ return path.slice(0, index);
3871
+ }
3872
+ function shellQuote(value) {
3873
+ return `'${value.replaceAll("'", `'\\''`)}'`;
3874
+ }
3875
+ //#endregion
3876
+ export { ADMISSION_CONFIG_DEFAULTS, ADMISSION_EXCLUSION_MEANING, ADMISSION_EXCLUSION_ORDER, ADMISSION_ROW_KEYS, ADMISSION_STRATA, AdmissionDenominatorError, AdmissionIndependenceError, BLINDED_FIELDS, CONTINUATION_POLICY_DEFAULTS, CONTROL_SCREENING_MODES, CREDIT_TERMS, ContinuationPolicyViolationError, ContinuationSymmetryError, DEGENERATE_STRATEGIES, DSPY_REPAIR_SIGNATURE, DSPY_REPAIR_TASK_TOKEN, DSPY_REPAIR_TRAJECTORY_INPUT, MINI_SWE_SYSTEM_MESSAGE, MIN_ORACLE_REPLICATES, NO_DECISIVE_FAILURE, NO_OP_ACTIONS, NondeterministicOracleError, OUTPUT_ELISION_THRESHOLD, OUTPUT_ELISION_WINDOW, REPAIR_CONTRACT_LINES, REPAIR_QUESTION, REPAIR_REPAIR_CONTRACT_LINES, RepairArmSymmetryError, SCAFFOLD_INTERVENTION_BUDGET, SUBMIT_SENTINEL, SUITE_REWARD_UNIT, TB_REPAIR_ADMISSION_CRITERIA, TestOracleError, TestSuiteTamperedError, UncalibratedControlError, admissionArtifact, admitRow, admittedCount, admittedRowIds, askRepairArm, assertAnalystIndependent, assertArmSymmetry, assertChainReconciles, assertControlCalibrated, assertDenominatorIntact, assertDeterministicOracle, assertDspyRepairEngine, blindTrajectory, buildDenominatorChain, checkInterventionBudget, classifyActionPayload, continuationPolicyDigest, continuationSeed, controlCanRescue, countFunnel, createCompletionRepairArm, createDockerContinuationEnvironment, createDspyRepairArm, defineControlPolicy, definePinnedContinuationPolicy, degenerateStrategy, deltaRepair, dockerRunArgs, dspyRepairInstructions, gradeRepairRow, injectedTestOracle, isPreStratumReason, isRecordedTimeout, noOpInjectionStep, nodeProcessRunner, normalizeActionForComparison, oracleDeterminism, parseAction, parseAnalystResponse, parseTaskOracleRegistry, reachedT0, reachedT1, reachedT2, readRepairPayload, renderAdmissionReport, renderDeltaRepairReport, renderFormatErrorObservation, renderInstanceMessage, renderObservation, renderRepairTrajectory, renderTimeoutObservation, repairArmAsymmetries, repairArmPromptSha256, repairArmResponse, repairCredit, repairFinding, repairQuestionSha256, repairTaskDefinition, repairTaskPolicy, repairTrajectoryHeader, resolveAdmissionConfig, rolloutDigest, rolloutRecordedSteps, runAdmission, runContinuation, scanShellAction, stratumOf, submissionOf, taskOracleRegistry, testSuiteDigest, toRecordedSteps, totalCost, totalUsage };
3877
+
3878
+ //# sourceMappingURL=index.js.map