@tangle-network/agent-eval 0.144.6 → 0.144.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
  14. package/dist/benchmark-command-BKENp2s5.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
  22. package/dist/campaign--HVSuvV0.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
  160. package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/campaign-proposers.md +5 -0
  207. package/docs/charter.md +112 -0
  208. package/docs/experiment.md +104 -0
  209. package/docs/prime-analyst.md +1 -0
  210. package/docs/trace-analysis.md +26 -0
  211. package/docs/trace-repair-admission.md +194 -0
  212. package/docs/trace-repair-analyst-arms.md +121 -0
  213. package/docs/trace-repair-continuation.md +107 -0
  214. package/docs/trace-repair-grader.md +163 -0
  215. package/docs/trajectory-replay.md +110 -0
  216. package/docs/verification-strategies.md +103 -0
  217. package/package.json +19 -2
  218. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  219. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  220. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  221. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  222. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  223. package/dist/campaign-CXsdyym7.js.map +0 -1
  224. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  225. package/dist/index-4XwggC10.d.ts.map +0 -1
  226. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  227. package/dist/integrity-fdt8XPAv.js.map +0 -1
  228. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  229. package/dist/replay-Krvb114g.d.ts.map +0 -1
  230. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  231. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  232. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  233. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  234. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  235. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  236. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  237. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  238. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  239. package/dist/verdict-Dps8_okt.d.ts +0 -37
  240. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,2102 @@
1
+ import { r as CaptureIntegrityError } from "../errors-CKPfb2aH.js";
2
+ import { b as CustomTokenPricing, c as CostLedgerHandle, p as CostProvenance } from "../cost-ledger-Bv_e8XHY.js";
3
+ import { u as RunTokenUsage } from "../run-record-DdSa93_W.js";
4
+ import { a as TraceAnalystLimits, n as TraceAnalysisEngine } from "../engine-nB64f48I.js";
5
+ import { t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
6
+ import { t as RecordedTrajectoryStep } from "../steps-BArUxhna.js";
7
+ //#region src/trace-repair/mini-swe-scaffold.d.ts
8
+ /**
9
+ * The mini-swe-agent scaffold as the Terminal-Bench-2 trajectory corpus
10
+ * recorded it: one bash block per turn, one observation per command, and a
11
+ * sentinel command that ends the run.
12
+ *
13
+ * Every template here is byte-verified against
14
+ * `yoonholee/terminalbench-trajectories` (agent = `mini-swe-agent`, 6663 rows,
15
+ * one distinct system prompt across all of them). A continuation that renders
16
+ * different bytes puts the model in a different distribution than the prefix
17
+ * it inherits, so these strings are pinned, not configurable.
18
+ */
19
+ /** Whole-line marker that ends a run. The first output line must equal it and the command must exit 0. */
20
+ declare const SUBMIT_SENTINEL = "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT";
21
+ /** Outputs at or above this length are elided head+tail instead of shown whole. */
22
+ declare const OUTPUT_ELISION_THRESHOLD = 10000;
23
+ /** Characters kept from each end of an elided output. */
24
+ declare const OUTPUT_ELISION_WINDOW = 5000;
25
+ declare const MINI_SWE_SYSTEM_MESSAGE = "You are a helpful assistant that can interact with a computer.\n\nYour response must contain exactly ONE bash code block with ONE command (or commands connected with && or ||).\nInclude a THOUGHT section before your command where you explain your reasoning process.\nFormat your response as shown in <format_example>.\n\n<format_example>\nYour reasoning and analysis here. Explain why you want to perform the action.\n\n```bash\nyour_command_here\n```\n</format_example>\n\nFailure to follow these rules will cause your response to be rejected.\n";
26
+ interface InstanceMessageInput {
27
+ /** The task statement, substituted verbatim. */
28
+ task: string;
29
+ /**
30
+ * `uname`-style line the recorded runs rendered from the host running the
31
+ * scaffold, e.g. `Linux 6.8.0-86-generic #87-Ubuntu SMP ... x86_64`.
32
+ */
33
+ systemInformation: string;
34
+ }
35
+ /** The second message of every recorded run: task plus workflow rules. */
36
+ declare function renderInstanceMessage(input: InstanceMessageInput): string;
37
+ type ParsedAction = {
38
+ kind: 'action';
39
+ command: string;
40
+ } | {
41
+ kind: 'format-error';
42
+ actionCount: number;
43
+ };
44
+ /**
45
+ * Exactly one fenced bash block is an action; zero or many is a format error.
46
+ * The scaffold trims the command, so a block padded with blank lines executes
47
+ * the same command as an unpadded one.
48
+ */
49
+ declare function parseAction(assistantMessage: string): ParsedAction;
50
+ interface CommandOutput {
51
+ returncode: number;
52
+ output: string;
53
+ /** Set when the environment itself failed rather than the command. */
54
+ exceptionInfo?: string;
55
+ }
56
+ /**
57
+ * The observation the agent reads after a command. Short outputs are shown
58
+ * whole; long ones keep the first and last `OUTPUT_ELISION_WINDOW` characters
59
+ * with the dropped count between them.
60
+ */
61
+ declare function renderObservation(output: CommandOutput): string;
62
+ /** The observation after the environment killed a command for exceeding its timeout. */
63
+ declare function renderTimeoutObservation(command: string, partialOutput: string): string;
64
+ /**
65
+ * True when the recording shows the environment killed this step at its
66
+ * wall-clock bound.
67
+ *
68
+ * Such a step carries no returncode, so no replay can confirm or contradict
69
+ * it. Callers use this to bound the replay of that step cheaply rather than to
70
+ * decide agreement.
71
+ */
72
+ declare function isRecordedTimeout(observation: string | null): boolean;
73
+ /** The observation after a turn that did not contain exactly one bash block. */
74
+ declare function renderFormatErrorObservation(actionCount: number): string;
75
+ /**
76
+ * The submission text when this output ends the run, `null` otherwise.
77
+ * A non-zero exit does not submit even when the sentinel is echoed, so an
78
+ * agent cannot end the run through a command that failed.
79
+ */
80
+ declare function submissionOf(output: CommandOutput): string | null;
81
+ //#endregion
82
+ //#region src/trace-repair/action-budget.d.ts
83
+ /**
84
+ * The action budget an intervention must fit inside.
85
+ *
86
+ * The analyst answers with one action, applied at step k, drawn from the same
87
+ * action space the scaffold had: one shell command or one edit, at most 4 KB.
88
+ * An answer that buys a bigger action than the scaffold could take is not a
89
+ * counterfactual about the recorded run, so the budget is enforced before any
90
+ * container is opened and a violation never reaches the reproduction gate.
91
+ *
92
+ * "One action" is decided by counting top-level statements, not lines. A
93
+ * command list joined by `&&`, `||` or a pipe is one statement, because that
94
+ * is one thing the shell runs and one thing the scaffold could have typed.
95
+ * Two statements separated by a newline or `;` are two actions and are
96
+ * rejected. Heredoc bodies, comments and compound blocks (`if`, `for`,
97
+ * `while`, `until`, `case`, `{ … }`) are inside a statement, never separators.
98
+ */
99
+ /**
100
+ * Payload shape of an action.
101
+ *
102
+ * `edit` authors file content inline through a heredoc; `shell` runs a
103
+ * command. The split matters because the two carry different amounts of
104
+ * information for the same byte count, and a report that pools them hides it.
105
+ */
106
+ type ActionPayloadKind = 'shell' | 'edit';
107
+ interface InterventionBudget {
108
+ /** Hard cap on the UTF-8 byte length of the action. */
109
+ readonly maxBytes: number;
110
+ /** Top-level statements the action may contain. One means one action. */
111
+ readonly maxStatements: number;
112
+ /** Heredoc redirections an `edit` may contain. One means one file. */
113
+ readonly maxHeredocs: number;
114
+ }
115
+ /** The scaffold's own per-action budget, pre-registered for the campaign. */
116
+ declare const SCAFFOLD_INTERVENTION_BUDGET: InterventionBudget;
117
+ /**
118
+ * Actions whose only effect is to consume a turn. They are rejected before a
119
+ * container opens: an intervention that changes nothing is measurably
120
+ * identical to the no-op control, and paying rollouts to rediscover that
121
+ * wastes the corpus.
122
+ */
123
+ declare const NO_OP_ACTIONS: readonly string[];
124
+ /**
125
+ * Split a shell script into top-level statements and count its heredocs.
126
+ *
127
+ * The scan tracks quoting, escapes, command substitution, brace and paren
128
+ * grouping, comments, compound-block keywords, and heredoc bodies. Anything
129
+ * it cannot resolve stays inside the current statement, so an unparseable
130
+ * action reads as one oversized statement and is rejected on bytes rather
131
+ * than silently accepted as one clean action.
132
+ */
133
+ declare function scanShellAction(script: string): {
134
+ statements: string[];
135
+ heredocs: number;
136
+ };
137
+ /** Payload shape of an action: `edit` when it authors file content through a
138
+ * heredoc, `shell` otherwise. */
139
+ declare function classifyActionPayload(action: string): ActionPayloadKind;
140
+ /** Why an action fails the budget. Every value is a rejection the report
141
+ * counts by name; none of them is a scoring judgement. */
142
+ type BudgetViolation = 'empty' | 'over-byte-cap' | 'multiple-statements' | 'multiple-heredocs' | 'no-op-action' | 'submit-instead-of-repair';
143
+ interface BudgetMeasurement {
144
+ bytes: number;
145
+ statements: number;
146
+ heredocs: number;
147
+ /** What the action IS, measured from its text. */
148
+ payload: ActionPayloadKind;
149
+ /** What the analyst SAID it was. Recorded, never a rejection. */
150
+ declared: ActionPayloadKind;
151
+ }
152
+ type BudgetCheck = {
153
+ readonly admissible: true;
154
+ readonly measurement: BudgetMeasurement;
155
+ } | {
156
+ readonly admissible: false;
157
+ readonly violation: BudgetViolation;
158
+ readonly detail: string;
159
+ readonly measurement: BudgetMeasurement;
160
+ };
161
+ /**
162
+ * Measure an action against the budget.
163
+ *
164
+ * The budget bounds what the scaffold can execute: one top-level statement,
165
+ * one authored file, a byte cap, and neither a no-op nor a submit. Every
166
+ * rejection here is one of those.
167
+ *
168
+ * `declaredKind` is what the analyst called its own action. It is recorded
169
+ * beside the measured payload and never rejected on, because the scaffold runs
170
+ * the action identically either way — so rejecting the label scores an arm on
171
+ * how it described a repair rather than on the repair. A reader who wants the
172
+ * mismatch counts `declared` against `payload`.
173
+ */
174
+ declare function checkInterventionBudget(action: string, declaredKind: ActionPayloadKind, budget?: InterventionBudget): BudgetCheck;
175
+ /**
176
+ * Whitespace-insensitive comparison used to reject an intervention that is
177
+ * the recorded action again. Trailing whitespace and blank lines are the only
178
+ * differences a re-proposal can carry without changing what runs.
179
+ */
180
+ declare function normalizeActionForComparison(action: string): string;
181
+ //#endregion
182
+ //#region src/trace-repair/continuation-records.d.ts
183
+ /** Which arm produced a rollout. The policy is identical across all three. */
184
+ type ContinuationArm = 'intervention' | 'no-fix-control' | 'no-op-control';
185
+ /** Chat message in the scaffold's own vocabulary. */
186
+ interface ContinuationMessage {
187
+ role: 'system' | 'user' | 'assistant';
188
+ content: string;
189
+ }
190
+ /** Why a rollout stopped. */
191
+ type ContinuationExitStatus = 'submitted' | 'step-budget-exhausted' | 'repeated-format-error' | 'model-error' | 'environment-error';
192
+ interface ContinuationExecRecord {
193
+ command: string;
194
+ returncode: number;
195
+ /** True when the environment killed the command at the policy timeout. */
196
+ timedOut: boolean;
197
+ outputChars: number;
198
+ durationMs: number;
199
+ }
200
+ interface ContinuationModelCall {
201
+ /** Model id the provider reported serving, which can differ from the requested id. */
202
+ servedModel: string;
203
+ seed: number;
204
+ latencyMs: number;
205
+ /** `null` when the provider reported no usage. Never a zero-filled stand-in. */
206
+ usage: RunTokenUsage | null;
207
+ /** `null` when neither the provider nor local pricing produced an amount. */
208
+ costUsd: number | null;
209
+ finishReason: string | null;
210
+ contentChars: number;
211
+ }
212
+ interface ContinuationStepRecord {
213
+ /** 1-based index within the continuation, not within the whole trajectory. */
214
+ step: number;
215
+ assistantMessage: string;
216
+ /** `null` when the turn held zero or several bash blocks. */
217
+ action: string | null;
218
+ /**
219
+ * `null` when the command that ran ended the rollout: the submission
220
+ * sentinel gets no observation, which is how the corpus records a finished
221
+ * run, and a failed environment call produces none either.
222
+ */
223
+ observation: string | null;
224
+ /** `null` when no command ran, so a format error is never read as an exit-0 command. */
225
+ execution: ContinuationExecRecord | null;
226
+ model: ContinuationModelCall;
227
+ /** Environment failure that ended the rollout at this step. */
228
+ error?: string;
229
+ }
230
+ interface ContinuationUsageTotals {
231
+ /**
232
+ * Recorded model calls. A call that threw produced no step, so it is absent
233
+ * here and named in `terminalError` instead.
234
+ */
235
+ calls: number;
236
+ /** Calls that carried usage. Below `calls` means the totals cover part of the rollout. */
237
+ callsWithUsage: number;
238
+ /** True only when every call reported usage. */
239
+ captured: boolean;
240
+ input: number;
241
+ output: number;
242
+ reasoning?: number;
243
+ cached?: number;
244
+ cacheWrite?: number;
245
+ }
246
+ interface ContinuationEnvironmentDescription {
247
+ /** Docker network mode of the container. The policy admits `none` only. */
248
+ networkMode: string;
249
+ image?: string;
250
+ }
251
+ interface ContinuationRollout {
252
+ rolloutId: string;
253
+ arm: ContinuationArm;
254
+ /** Corpus row the continuation belongs to. Pairs rollouts across arms. */
255
+ rowId: string;
256
+ /** 0-based rollout index within the arm. */
257
+ index: number;
258
+ /** Seed handed to the model for every call in this rollout. */
259
+ seed: number;
260
+ /** Hash over the policy and scaffold templates. Equal across arms by construction. */
261
+ policyDigest: string;
262
+ environmentId: string;
263
+ containerRef: string;
264
+ environment: ContinuationEnvironmentDescription;
265
+ steps: ContinuationStepRecord[];
266
+ exitStatus: ContinuationExitStatus;
267
+ /** Text after the sentinel line, present only on `submitted`. */
268
+ submission: string | null;
269
+ usage: ContinuationUsageTotals;
270
+ costProvenance: CostProvenance;
271
+ wallMs: number;
272
+ startedAt: string;
273
+ endedAt: string;
274
+ /** Message from the model or environment failure that ended the rollout. */
275
+ terminalError?: string;
276
+ }
277
+ /** A tool call in corpus shape. */
278
+ interface RecordedToolCall {
279
+ fn: 'bash_command';
280
+ cmd: string;
281
+ }
282
+ /** One trajectory step in corpus shape. */
283
+ interface RecordedStep {
284
+ src: 'system' | 'user' | 'agent';
285
+ msg: string;
286
+ tools: RecordedToolCall[] | null;
287
+ obs: string | null;
288
+ }
289
+ /**
290
+ * Project a full message list into corpus steps: the system and task messages
291
+ * become their own steps, and every assistant turn carries the command it
292
+ * requested plus the observation that answered it. A trailing assistant turn
293
+ * with no answer keeps `obs: null`, which is how the corpus records a run that
294
+ * ended on its last command.
295
+ */
296
+ declare function toRecordedSteps(messages: readonly ContinuationMessage[]): RecordedStep[];
297
+ /** Corpus steps for the continuation alone, excluding the prefix it inherited. */
298
+ declare function rolloutRecordedSteps(rollout: ContinuationRollout): RecordedStep[];
299
+ /**
300
+ * Hash over everything the policy determines: the actions taken, the
301
+ * observations they produced, the seeds, the exit, and the usage.
302
+ *
303
+ * Wall-clock fields are excluded because they vary between identical runs;
304
+ * two rollouts with the same digest did the same work, whatever they cost in
305
+ * seconds. Use it to assert determinism, never to assert equal latency.
306
+ */
307
+ declare function rolloutDigest(rollout: ContinuationRollout): string;
308
+ //#endregion
309
+ //#region src/trace-repair/control-policy.d.ts
310
+ /** The declared control cannot produce the outcome the criteria screen for. */
311
+ declare class UncalibratedControlError extends CaptureIntegrityError {}
312
+ /**
313
+ * What the criteria expect of the control.
314
+ *
315
+ * - `enforced` — conditions 3 and 4 decide admission, so the control must be
316
+ * able to rescue a row.
317
+ * - `declared-inert` — the campaign has pinned a control that cannot rescue
318
+ * anything (a calibration run comparing arms at equal depth, for instance).
319
+ * Conditions 3 and 4 still run, but a pass then means the task's grader
320
+ * disagreed with itself about identical state, and the row leaves as that.
321
+ */
322
+ type ControlScreening = 'enforced' | 'declared-inert';
323
+ declare const CONTROL_SCREENING_MODES: readonly ControlScreening[];
324
+ interface ControlPolicyInput {
325
+ /** Stable name for the frozen configuration. */
326
+ readonly id: string;
327
+ /** Model calls one control rollout may make after the arm's own treatment. */
328
+ readonly stepBudget: number;
329
+ /** Scaffold the rollout runs under. */
330
+ readonly scaffold: string;
331
+ /** Requested model id, or `null` when the budget calls no model. */
332
+ readonly model: string | null;
333
+ /** Per-command wall-clock limit inside the container. */
334
+ readonly commandTimeoutSeconds: number;
335
+ }
336
+ /**
337
+ * The part of a control the calibration rule reads.
338
+ *
339
+ * Both admission paths declare a control in their own vocabulary — the pure
340
+ * contract takes a `ControlPolicy`, the executing pre-pass runs a pinned
341
+ * continuation policy — and both are checked by one rule against this shape,
342
+ * so the two cannot drift into disagreeing about what a usable control is.
343
+ */
344
+ interface ControlCapability {
345
+ readonly id: string;
346
+ readonly digest: string;
347
+ /** Model calls one control rollout may make after the arm's own treatment. */
348
+ readonly stepBudget: number;
349
+ }
350
+ interface ControlPolicy extends ControlPolicyInput, ControlCapability {
351
+ /** Hash over the whole declaration. A hand-written label cannot stand in for
352
+ * it, so two runs cannot share a digest while differing in step budget. */
353
+ readonly digest: string;
354
+ /**
355
+ * Whether a rollout under this policy can reach a state the recorded end
356
+ * state was not already in. False at a zero step budget: no model call means
357
+ * no command, which means the graded bytes are the ones condition 2 read.
358
+ */
359
+ readonly canRescue: boolean;
360
+ }
361
+ /**
362
+ * A control rollout changes the graded state only by executing something, and
363
+ * it executes only what a model call asks for. At a zero budget it grades the
364
+ * bytes it was handed.
365
+ */
366
+ declare function controlCanRescue(stepBudget: number): boolean;
367
+ declare function defineControlPolicy(input: ControlPolicyInput): ControlPolicy;
368
+ /**
369
+ * Refuse a configuration whose control and screening mode contradict.
370
+ *
371
+ * Both directions are faults, and both are silent without this. A screening
372
+ * control that cannot act passes every row through a condition it can never
373
+ * fire. A control declared inert that can in fact act hides a real screen
374
+ * behind a label that says nothing was screened.
375
+ */
376
+ declare function assertControlCalibrated(policy: ControlCapability, screening: ControlScreening): void;
377
+ //#endregion
378
+ //#region src/trace-repair/admission-records.d.ts
379
+ /** A campaign scored rows the pre-pass did not admit, or dropped rows it did. */
380
+ declare class AdmissionDenominatorError extends CaptureIntegrityError {}
381
+ /** A row handed to the pre-pass carried a field only an analyst could produce. */
382
+ declare class AdmissionIndependenceError extends CaptureIntegrityError {}
383
+ /**
384
+ * One corpus row, described by what the recording holds and nothing else.
385
+ *
386
+ * The field list is closed on purpose: `assertAnalystIndependent` rejects any
387
+ * extra key, which is what stops a finding, a step index, or a proposed
388
+ * intervention from reaching the gate that fixes the denominator.
389
+ */
390
+ interface AdmissionRow {
391
+ /** Stable identifier for the recorded trajectory. Duplicates are rejected. */
392
+ rowId: string;
393
+ /** Terminal-Bench-2 task the trajectory ran. */
394
+ taskName: string;
395
+ /** Model that produced the recorded trajectory. */
396
+ recordedModel: string;
397
+ /**
398
+ * Recorded shell commands. An action can be substituted at any one of them,
399
+ * and a replay must execute all of them to reach the recorded end state.
400
+ */
401
+ recordedCommands: number;
402
+ /**
403
+ * Return code of the trajectory's last recorded observation. Negative values
404
+ * are signal kills and are real. `null` means the recording did not parse,
405
+ * which excludes the row instead of defaulting it.
406
+ */
407
+ finalReturncode: number | null;
408
+ }
409
+ declare const ADMISSION_ROW_KEYS: readonly string[];
410
+ /**
411
+ * Reject rows that carry anything beyond the recording.
412
+ *
413
+ * The check is a closed key list rather than a list of known analyst field
414
+ * names, because the failure to catch is "some new analyst output leaked into
415
+ * the gate", and only a closed shape catches the fields nobody thought of.
416
+ */
417
+ declare function assertAnalystIndependent(rows: readonly AdmissionRow[]): void;
418
+ /**
419
+ * Failure population a row belongs to.
420
+ *
421
+ * - `clean-exit` — the last command exited 0 and the tests still fail, so the
422
+ * agent stopped believing it was done. The assay measured 87.13% of admitted
423
+ * failures here.
424
+ * - `command-error` — the last command exited non-zero.
425
+ * - `signal-kill` — the last command was killed by a signal, recorded as a
426
+ * negative return code. Substituting one command does not address a timeout,
427
+ * so this population stays separate from the other two.
428
+ */
429
+ type AdmissionStratum = 'clean-exit' | 'command-error' | 'signal-kill';
430
+ declare const ADMISSION_STRATA: readonly AdmissionStratum[];
431
+ /** `null` when the recording holds no parseable final return code. */
432
+ declare function stratumOf(finalReturncode: number | null): AdmissionStratum | null;
433
+ /**
434
+ * Why a row left the funnel.
435
+ *
436
+ * `ADMISSION_EXCLUSION_ORDER` is the order the gate applies the checks, and it
437
+ * is cost-ordered: everything decidable from the recording runs before the
438
+ * first container, and the control rollouts run last.
439
+ */
440
+ type AdmissionExclusionReason = 'no-recorded-commands' | 'unparseable-final-returncode' | 'stratum-not-admitted' | 'task-oracle-uncertified' | 'task-oracle-nondeterministic' | 'prefix-replay-error' | 'prefix-replay-empty' | 'prefix-replay-truncated' | 'prefix-divergence-above-threshold' | 'end-state-oracle-error' | 'end-state-tests-pass' | 'no-fix-control-error' | 'no-fix-control-rescued' | 'no-op-control-error' | 'no-op-control-rescued';
441
+ declare const ADMISSION_EXCLUSION_ORDER: readonly AdmissionExclusionReason[];
442
+ declare function isPreStratumReason(reason: AdmissionExclusionReason): boolean;
443
+ /** One line of prose per reason, for the rendered chain. */
444
+ declare const ADMISSION_EXCLUSION_MEANING: Readonly<Record<AdmissionExclusionReason, string>>;
445
+ type AdmissionControlArm = 'no-fix-control' | 'no-op-control';
446
+ /** The inert action the no-op control substitutes, and where it goes. */
447
+ interface AdmissionNoOpInjection {
448
+ /** 1-based recorded command the inert action replaces. */
449
+ step: number;
450
+ action: string;
451
+ }
452
+ type AdmissionCheckRecord = {
453
+ check: 'stratum';
454
+ stratum: AdmissionStratum;
455
+ } | {
456
+ check: 'prefix-replay';
457
+ prefixExecuted: number;
458
+ divergences: number;
459
+ divergenceRatio: number;
460
+ } | {
461
+ check: 'task-oracle';
462
+ stable: boolean;
463
+ flipRate: number;
464
+ replicates: number;
465
+ } | {
466
+ check: 'end-state-tests';
467
+ passed: boolean;
468
+ reward: number | null;
469
+ } | {
470
+ check: 'control';
471
+ arm: AdmissionControlArm;
472
+ rolloutsRun: number;
473
+ passes: number;
474
+ injections: AdmissionNoOpInjection[];
475
+ };
476
+ /** What one control rollout did, small enough to serialize for every row. */
477
+ interface AdmissionRolloutSummary {
478
+ arm: AdmissionControlArm;
479
+ index: number;
480
+ seed: number;
481
+ policyDigest: string;
482
+ exitStatus: ContinuationRollout['exitStatus'];
483
+ testsPassed: boolean;
484
+ /** `null` when the rollout carried an unpriced model call. */
485
+ costUsd: number | null;
486
+ }
487
+ interface AdmissionRowVerdict {
488
+ rowId: string;
489
+ taskName: string;
490
+ recordedModel: string;
491
+ recordedCommands: number;
492
+ finalReturncode: number | null;
493
+ /** `null` only together with a pre-stratum exclusion reason. */
494
+ stratum: AdmissionStratum | null;
495
+ /** The control this row was screened under, and how a control pass reads.
496
+ * Recorded on every verdict, including rows excluded before a control ran,
497
+ * so a reader never has to open the runner's source to learn what screened
498
+ * the row. */
499
+ controlPolicyDigest: string;
500
+ controlScreening: ControlScreening;
501
+ /** The task's certified oracle flip rate. Zero on a certified-stable task,
502
+ * `null` when the task carried no certification. */
503
+ oracleFlipRate: number | null;
504
+ admitted: boolean;
505
+ /** `null` when the row is admitted. */
506
+ excludedBy: AdmissionExclusionReason | null;
507
+ /** Message from the boundary that failed, on the four `*-error` reasons. */
508
+ errorDetail: string | null;
509
+ /** Checks that ran, in order. A check never reached is absent, not zero. */
510
+ checks: AdmissionCheckRecord[];
511
+ rollouts: AdmissionRolloutSummary[];
512
+ }
513
+ interface DenominatorStage {
514
+ reason: AdmissionExclusionReason;
515
+ entering: number;
516
+ excluded: number;
517
+ remaining: number;
518
+ }
519
+ interface DenominatorChain {
520
+ /** `all` counts every input row; a stratum chain starts after stratification. */
521
+ scope: 'all' | AdmissionStratum;
522
+ input: number;
523
+ stages: DenominatorStage[];
524
+ admitted: number;
525
+ }
526
+ interface DenominatorChainArtifact {
527
+ version: 1;
528
+ overall: DenominatorChain;
529
+ byStratum: DenominatorChain[];
530
+ reasonTotals: Record<AdmissionExclusionReason, number>;
531
+ /** Rows excluded before a stratum could be assigned. */
532
+ unstratified: number;
533
+ /** Strata this run admitted. Others can only reach `stratum-not-admitted`. */
534
+ admitStrata: readonly AdmissionStratum[];
535
+ }
536
+ /**
537
+ * Build the funnel every campaign report publishes.
538
+ *
539
+ * Each stage names the reason, the rows that reached it, the rows it removed,
540
+ * and the rows that survived, so `input = admitted + sum(excluded)` can be read
541
+ * off the table instead of trusted.
542
+ */
543
+ declare function buildDenominatorChain(verdicts: readonly AdmissionRowVerdict[], admitStrata: readonly AdmissionStratum[]): DenominatorChainArtifact;
544
+ /** A chain that does not add up is a broken denominator, so this throws. */
545
+ declare function assertChainReconciles(artifact: DenominatorChainArtifact): void;
546
+ //#endregion
547
+ //#region src/trace-repair/continuation-policy.d.ts
548
+ /** A rollout ran outside the pinned policy, so its evidence cannot be used. */
549
+ declare class ContinuationPolicyViolationError extends CaptureIntegrityError {}
550
+ /** Two arms did not run the same policy, so their difference is not the intervention. */
551
+ declare class ContinuationSymmetryError extends CaptureIntegrityError {}
552
+ interface PinnedContinuationPolicy {
553
+ /** Stable name for the frozen configuration, recorded on every rollout. */
554
+ readonly id: string;
555
+ /** Requested model id. The served id is recorded per call and can differ. */
556
+ readonly model: string;
557
+ /** Seed root. Per-rollout seeds derive from it and never from the arm. */
558
+ readonly seed: number;
559
+ /** Model calls allowed per rollout. */
560
+ readonly stepBudget: number;
561
+ readonly temperature: number;
562
+ readonly maxTokens: number;
563
+ /** Per-command wall-clock limit inside the container. */
564
+ readonly commandTimeoutSeconds: number;
565
+ /** Consecutive unparseable turns that end a rollout. */
566
+ readonly maxConsecutiveFormatErrors: number;
567
+ /** The only admissible container network mode. */
568
+ readonly networkMode: 'none';
569
+ readonly scaffold: 'mini-swe-agent';
570
+ }
571
+ /**
572
+ * Everything the policy fixes except the two a campaign must choose.
573
+ *
574
+ * `commandTimeoutSeconds` is 30 because the recorded runs used the scaffold's
575
+ * own 30-second environment timeout; a longer limit would let the continuation
576
+ * finish commands the recorded agent could not.
577
+ */
578
+ declare const CONTINUATION_POLICY_DEFAULTS: {
579
+ readonly id: 'tb-repair-continuation-v1';
580
+ readonly stepBudget: 20;
581
+ readonly temperature: 0;
582
+ readonly maxTokens: 4096;
583
+ readonly commandTimeoutSeconds: 30;
584
+ readonly maxConsecutiveFormatErrors: 3;
585
+ readonly networkMode: 'none';
586
+ readonly scaffold: 'mini-swe-agent';
587
+ };
588
+ interface DefineContinuationPolicyInput extends Partial<Omit<PinnedContinuationPolicy, 'networkMode' | 'scaffold'>> {
589
+ /** Required: a campaign pins one model, and no default can stand in for it. */
590
+ model: string;
591
+ /** Required: a default seed would make two campaigns silently share a draw. */
592
+ seed: number;
593
+ }
594
+ declare function definePinnedContinuationPolicy(input: DefineContinuationPolicyInput): PinnedContinuationPolicy;
595
+ /**
596
+ * Hash over the policy and the scaffold text it renders. A changed template
597
+ * changes the digest, so rollouts recorded before and after an edit cannot be
598
+ * pooled by accident.
599
+ */
600
+ declare function continuationPolicyDigest(policy: PinnedContinuationPolicy): string;
601
+ /**
602
+ * Per-rollout seed. It reads the policy seed, the row, and the rollout index —
603
+ * deliberately not the arm, so paired rollouts across arms draw identically.
604
+ */
605
+ declare function continuationSeed(policySeed: number, rowId: string, rolloutIndex: number): number;
606
+ interface ContinuationModelRequest {
607
+ model: string;
608
+ messages: ContinuationMessage[];
609
+ seed: number;
610
+ temperature: number;
611
+ maxTokens: number;
612
+ }
613
+ interface ContinuationModelResponse {
614
+ content: string;
615
+ /** Model id the provider reported serving. */
616
+ servedModel: string;
617
+ /** `null` when the provider omitted usage. Callers must not substitute zeros. */
618
+ usage: RunTokenUsage | null;
619
+ /** `null` when no amount is available from the provider or local pricing. */
620
+ costUsd: number | null;
621
+ finishReason?: string | null;
622
+ }
623
+ type ContinuationModel = (request: ContinuationModelRequest) => Promise<ContinuationModelResponse>;
624
+ interface ContinuationExecResult {
625
+ output: string;
626
+ returncode: number;
627
+ /** True when the environment killed the command at the policy timeout. */
628
+ timedOut: boolean;
629
+ /** Set when the environment itself failed rather than the command. */
630
+ exceptionInfo?: string;
631
+ }
632
+ interface ContinuationEnvironment {
633
+ /** Container id or equivalent handle, recorded for provenance. */
634
+ containerRef: string;
635
+ describe(): Promise<ContinuationEnvironmentDescription>;
636
+ exec(command: string, options: {
637
+ timeoutSeconds: number;
638
+ }): Promise<ContinuationExecResult>;
639
+ dispose(): Promise<void>;
640
+ }
641
+ interface ContinuationEnvironmentRequest {
642
+ rowId: string;
643
+ arm: ContinuationArm;
644
+ rolloutIndex: number;
645
+ }
646
+ interface ContinuationEnvironmentFactory {
647
+ id: string;
648
+ /**
649
+ * A container restored to the post-step-k state for one rollout. The arm is
650
+ * passed because the state under test differs by arm; the policy applied to
651
+ * that state does not.
652
+ */
653
+ create(request: ContinuationEnvironmentRequest): Promise<ContinuationEnvironment>;
654
+ }
655
+ interface RunContinuationOptions {
656
+ policy: PinnedContinuationPolicy;
657
+ arm: ContinuationArm;
658
+ /** Corpus row under repair. Pairs rollouts across arms. */
659
+ rowId: string;
660
+ /**
661
+ * Messages through step k, as the replay layer rebuilt them: the pinned
662
+ * system and task messages, then every replayed turn with the observation
663
+ * the replay itself produced.
664
+ */
665
+ prefix: readonly ContinuationMessage[];
666
+ /** Rollouts to run for this arm. */
667
+ rollouts: number;
668
+ model: ContinuationModel;
669
+ environments: ContinuationEnvironmentFactory;
670
+ /** Epoch milliseconds. Injected so tests can assert on records without wall-clock noise. */
671
+ clock?: () => number;
672
+ }
673
+ /**
674
+ * Run the scaffold forward for `rollouts` independent continuations.
675
+ *
676
+ * Each rollout gets its own environment from the factory, because a rollout
677
+ * mutates the container it runs in and the next one must start from the same
678
+ * state, not from the previous rollout's leftovers.
679
+ */
680
+ declare function runContinuation(options: RunContinuationOptions): Promise<ContinuationRollout[]>;
681
+ /**
682
+ * Sum only what the provider reported. A call with no usage raises
683
+ * `callsWithUsage` short of `calls` and clears `captured`, so a partially
684
+ * reported rollout can never read as a fully measured one.
685
+ */
686
+ declare function totalUsage(steps: readonly ContinuationStepRecord[]): ContinuationUsageTotals;
687
+ /**
688
+ * One unpriced call makes the rollout's cost unknown. Summing the rest would
689
+ * report a number smaller than what was spent.
690
+ */
691
+ declare function totalCost(steps: readonly ContinuationStepRecord[]): CostProvenance;
692
+ /**
693
+ * Prove the arms ran the same policy. Rollouts paired by row and index must
694
+ * carry the same policy digest and the same seed; anything else means the
695
+ * measured difference includes a policy change, not only the intervention.
696
+ */
697
+ declare function assertArmSymmetry(rollouts: readonly ContinuationRollout[]): void;
698
+ //#endregion
699
+ //#region src/trace-repair/oracle-determinism.d.ts
700
+ /** A task graded a repair by something other than the state under test. */
701
+ declare class NondeterministicOracleError extends CaptureIntegrityError {}
702
+ /** The two states certification produces: the published image, and that image
703
+ * after the task's own reference solution ran. */
704
+ type OracleStateLabel = 'unsolved' | 'solved';
705
+ /** Machine load the replicate ran under. */
706
+ type OracleLoad = 'idle' | 'contended';
707
+ /** The unit a whole-suite verdict is counted under when nothing finer exists. */
708
+ declare const SUITE_REWARD_UNIT = "suite-reward";
709
+ interface OracleAssertionResult {
710
+ /** Test identifier the suite reported, verbatim. */
711
+ readonly id: string;
712
+ readonly passed: boolean;
713
+ }
714
+ interface OracleReplicate {
715
+ /** 0-based index within its group. */
716
+ readonly index: number;
717
+ /**
718
+ * Raw reward the task's grader wrote, verbatim. `null` when it wrote none —
719
+ * a missing reward and a zero reward are different failures, and neither is
720
+ * folded into the other.
721
+ */
722
+ readonly reward: string | null;
723
+ readonly passed: boolean;
724
+ readonly wallMs: number;
725
+ /** Per-assertion verdicts the suite reported, or `null` when it reported
726
+ * none. Never an empty array standing in for "the suite said nothing". */
727
+ readonly assertions: readonly OracleAssertionResult[] | null;
728
+ }
729
+ interface OracleReplicateGroup {
730
+ readonly state: OracleStateLabel;
731
+ readonly load: OracleLoad;
732
+ /**
733
+ * Replicates that graded one container with nothing written between them.
734
+ * Two or more, because a single grading measures no flip.
735
+ */
736
+ readonly replicates: readonly OracleReplicate[];
737
+ }
738
+ interface OracleDeterminismEvidence {
739
+ readonly taskName: string;
740
+ /** Image the replicates ran on, by registry manifest digest. An image with no
741
+ * published digest, such as one loaded from an archive, records its tag; the
742
+ * absent `@sha256:` is what says so. */
743
+ readonly image: string;
744
+ /**
745
+ * Digest of the suite bytes every replicate graded against, in whatever
746
+ * scheme the measuring tool uses. Provenance only: it identifies the suite a
747
+ * certification is about and is never compared against a digest produced by
748
+ * a different scheme.
749
+ */
750
+ readonly suiteDigest: string;
751
+ readonly groups: readonly OracleReplicateGroup[];
752
+ readonly measuredAt: string;
753
+ }
754
+ /** One thing that did not agree with itself across replicates of a state. */
755
+ interface OracleFlippedUnit {
756
+ /** An assertion id, or `suite-reward` when the suite reported no assertions. */
757
+ readonly unit: string;
758
+ readonly passes: number;
759
+ readonly fails: number;
760
+ readonly flipRate: number;
761
+ }
762
+ interface OracleStateVerdict {
763
+ readonly state: OracleStateLabel;
764
+ readonly replicates: number;
765
+ /** Replicates whose whole-suite reward passed. */
766
+ readonly passes: number;
767
+ readonly fails: number;
768
+ /** What the flip rate was counted over. `reward` is the coarser reading. */
769
+ readonly granularity: 'per-assertion' | 'reward';
770
+ /** Largest minority share over the counted units. */
771
+ readonly flipRate: number;
772
+ readonly flipped: readonly OracleFlippedUnit[];
773
+ /** Assertion ids some replicates reported and others did not. The suite's
774
+ * own shape moved, which is a flip in itself. */
775
+ readonly assertionSetUnstable: readonly string[];
776
+ /** Loads whose whole-suite pass rate on this state differed. */
777
+ readonly loadSensitive: boolean;
778
+ /** Every distinct raw reward the grader wrote, sorted. */
779
+ readonly rewardsObserved: readonly string[];
780
+ }
781
+ interface OracleDeterminismVerdict {
782
+ readonly taskName: string;
783
+ readonly image: string;
784
+ readonly suiteDigest: string;
785
+ /** True only when every counted unit agreed with itself on every state. */
786
+ readonly stable: boolean;
787
+ readonly replicates: number;
788
+ /** Largest per-unit minority share over every state. Zero on a stable task. */
789
+ readonly flipRate: number;
790
+ readonly byState: readonly OracleStateVerdict[];
791
+ readonly measuredAt: string;
792
+ /** One line naming what was measured, for an artifact a human reads. */
793
+ readonly detail: string;
794
+ }
795
+ /** Replicates one group needs before a flip is measurable at all. */
796
+ declare const MIN_ORACLE_REPLICATES = 2;
797
+ /**
798
+ * Reduce measured replicates to a verdict.
799
+ *
800
+ * Pure: it opens no container. The tool that runs the replicates hands the
801
+ * counts here, so the rule that decides stability is one rule and a reviewer
802
+ * can re-derive any verdict from the recorded replicates.
803
+ */
804
+ declare function oracleDeterminism(evidence: OracleDeterminismEvidence): OracleDeterminismVerdict;
805
+ /**
806
+ * Certified verdicts by task name.
807
+ *
808
+ * A task with no entry is uncertified, which is not the same as unstable: one
809
+ * says the check has not run, the other says it ran and the task failed it.
810
+ * Both stop a row, with different reasons.
811
+ */
812
+ type TaskOracleRegistry = ReadonlyMap<string, OracleDeterminismVerdict>;
813
+ /**
814
+ * What a checked-in certification file holds.
815
+ *
816
+ * It stores the measured replicates, not the verdicts they imply. The verdict
817
+ * is re-derived on read by the same rule a campaign enforces, so a file cannot
818
+ * declare a task stable — only the replicates can, and a reviewer can recount
819
+ * them.
820
+ */
821
+ interface TaskOracleRegistryDocument {
822
+ readonly version: 1;
823
+ readonly measurements: readonly OracleDeterminismEvidence[];
824
+ }
825
+ declare function parseTaskOracleRegistry(document: unknown): TaskOracleRegistry;
826
+ declare function taskOracleRegistry(verdicts: readonly OracleDeterminismVerdict[]): TaskOracleRegistry;
827
+ /** Throws unless the task graded the same bytes the same way every replicate. */
828
+ declare function assertDeterministicOracle(verdict: OracleDeterminismVerdict): void;
829
+ //#endregion
830
+ //#region src/trace-repair/admission.d.ts
831
+ /**
832
+ * Result of an external call. Callers read `succeeded` before `value`: a
833
+ * boundary failure is an exclusion with its own reason, never a false verdict.
834
+ */
835
+ type AdmissionOutcome<T> = {
836
+ succeeded: true;
837
+ value: T;
838
+ } | {
839
+ succeeded: false;
840
+ error: string;
841
+ };
842
+ interface AdmissionDivergence {
843
+ /** 1-based recorded step. */
844
+ step: number;
845
+ expectedReturncode: number | null;
846
+ actualExit: number;
847
+ }
848
+ interface AdmissionPrefixReplay {
849
+ /** Recorded steps the replay actually executed. */
850
+ prefixExecuted: number;
851
+ /** Steps whose replayed exit differs from the recorded return code. */
852
+ prefixDivergences: readonly AdmissionDivergence[];
853
+ }
854
+ interface AdmissionPrefixReplayer {
855
+ /** Lands in the artifact. A fake and a container replayer must not share one. */
856
+ id: string;
857
+ /** Replay the recorded prefix in the task's pinned image and report divergence. */
858
+ replay(row: AdmissionRow): Promise<AdmissionOutcome<AdmissionPrefixReplay>>;
859
+ }
860
+ interface AdmissionTestVerdict {
861
+ passed: boolean;
862
+ /** Reward the task's own grader reported. `null` when it reported none. */
863
+ reward: number | null;
864
+ }
865
+ interface AdmissionEndStateOracle {
866
+ id: string;
867
+ /** Run the task's held-out tests against the recorded end state. */
868
+ grade(row: AdmissionRow): Promise<AdmissionOutcome<AdmissionTestVerdict>>;
869
+ }
870
+ interface AdmissionControlRequest {
871
+ row: AdmissionRow;
872
+ arm: AdmissionControlArm;
873
+ /** 0-based rollout index within the arm. */
874
+ rolloutIndex: number;
875
+ /** Present for `no-op-control` only; `null` for `no-fix-control`. */
876
+ injection: AdmissionNoOpInjection | null;
877
+ }
878
+ interface AdmissionControlObservation {
879
+ /** The task's held-out tests, run after the continuation stopped. */
880
+ tests: AdmissionTestVerdict;
881
+ /** The rollout the pinned policy produced. Its digest and seed prove symmetry. */
882
+ rollout: ContinuationRollout;
883
+ }
884
+ interface AdmissionControlRunner {
885
+ id: string;
886
+ /**
887
+ * Restore the row's post-trajectory state, apply the arm's treatment, run the
888
+ * pinned continuation policy forward, then grade the container it left.
889
+ */
890
+ run(request: AdmissionControlRequest): Promise<AdmissionOutcome<AdmissionControlObservation>>;
891
+ }
892
+ interface AdmissionConfigInput {
893
+ /** Share of replayed prefix steps allowed to diverge. Default 0.10. */
894
+ maxPrefixDivergence?: number;
895
+ /** Rollouts each control arm must fail. Default 3. */
896
+ controlRollouts?: number;
897
+ /**
898
+ * Strata a campaign accepts. The default omits `signal-kill`, because
899
+ * substituting one command does not address a timeout. The excluded count
900
+ * stays visible in the chain instead of disappearing.
901
+ */
902
+ admitStrata?: readonly AdmissionStratum[];
903
+ /** Shell action the no-op control substitutes. Default `true`. */
904
+ inertAction?: string;
905
+ /**
906
+ * How a control pass is read. Default `enforced`, which requires a control
907
+ * that can act; `resolveAdmissionConfig` refuses the pairing that cannot.
908
+ */
909
+ controlScreening?: ControlScreening;
910
+ /** Rows processed at once. Output order always follows input order. Default 1. */
911
+ concurrency?: number;
912
+ }
913
+ interface AdmissionConfig {
914
+ readonly maxPrefixDivergence: number;
915
+ readonly controlRollouts: number;
916
+ readonly admitStrata: readonly AdmissionStratum[];
917
+ readonly inertAction: string;
918
+ readonly controlScreening: ControlScreening;
919
+ readonly concurrency: number;
920
+ }
921
+ declare const ADMISSION_CONFIG_DEFAULTS: AdmissionConfig;
922
+ declare function resolveAdmissionConfig(input?: AdmissionConfigInput): AdmissionConfig;
923
+ /**
924
+ * The recorded command the no-op control replaces, drawn per rollout.
925
+ *
926
+ * The draw reads the policy seed, the row, and the rollout index, so it is
927
+ * reproducible from the artifact and cannot depend on when the pre-pass ran.
928
+ */
929
+ declare function noOpInjectionStep(policySeed: number, rowId: string, rolloutIndex: number, recordedCommands: number): number;
930
+ interface AdmissionProvenance {
931
+ replayerId: string;
932
+ oracleId: string;
933
+ controlRunnerId: string;
934
+ policyId: string;
935
+ policyModel: string;
936
+ policySeed: number;
937
+ policyDigest: string;
938
+ /** Model calls a control rollout may make. Zero cannot screen anything, and
939
+ * `resolveAdmissionConfig` refuses that pairing before a container opens. */
940
+ policyStepBudget: number;
941
+ controlScreening: ControlScreening;
942
+ /** Tasks whose oracle carried a determinism certification, and their
943
+ * measured flip rates. */
944
+ certifiedTasks: Readonly<Record<string, number>>;
945
+ }
946
+ interface AdmissionReport {
947
+ config: AdmissionConfig;
948
+ provenance: AdmissionProvenance;
949
+ /** Every input row, in input order, admitted or not. */
950
+ rows: readonly AdmissionRowVerdict[];
951
+ /** Admitted row ids per stratum. There is no pooled list; pooling is explicit. */
952
+ strata: Readonly<Record<AdmissionStratum, readonly string[]>>;
953
+ chain: DenominatorChainArtifact;
954
+ /** Model cost of the control rollouts. One unpriced rollout makes it uncaptured. */
955
+ controlCost: CostProvenance;
956
+ /** Hash over the admitted ids, the config, and the provenance. */
957
+ digest: string;
958
+ generatedAt: string;
959
+ }
960
+ interface RunAdmissionOptions {
961
+ rows: readonly AdmissionRow[];
962
+ /** The policy the campaign's arms run under. Its seed drives the no-op draw. */
963
+ policy: PinnedContinuationPolicy;
964
+ replayer: AdmissionPrefixReplayer;
965
+ oracle: AdmissionEndStateOracle;
966
+ controls: AdmissionControlRunner;
967
+ /**
968
+ * Determinism certifications by task name. Required: every check below reads
969
+ * the task's own suite, so a suite whose verdict is not a function of the
970
+ * state makes each of them a coin flip. A task with no entry is excluded as
971
+ * uncertified rather than assumed stable.
972
+ */
973
+ taskOracles: TaskOracleRegistry;
974
+ config?: AdmissionConfigInput;
975
+ /** Epoch milliseconds. Injected so a report can be compared exactly in a test. */
976
+ clock?: () => number;
977
+ }
978
+ /**
979
+ * Run the pre-pass over every row and publish the denominator it produced.
980
+ *
981
+ * Nothing here reads an analyst output. The report is frozen, and a campaign
982
+ * proves it measured this denominator by passing the report back to
983
+ * `assertDenominatorIntact`.
984
+ */
985
+ declare function runAdmission(options: RunAdmissionOptions): Promise<AdmissionReport>;
986
+ /** Admitted row ids in one stratum. No call returns them pooled. */
987
+ declare function admittedRowIds(report: AdmissionReport, stratum: AdmissionStratum): readonly string[];
988
+ declare function admittedCount(report: AdmissionReport): number;
989
+ interface DenominatorIntactInput {
990
+ report: AdmissionReport;
991
+ /** Strata the campaign drew from. Rows outside them are not expected. */
992
+ strata: readonly AdmissionStratum[];
993
+ /** Rows the campaign sampled from the admitted set. */
994
+ sampled: readonly string[];
995
+ /** Rows the campaign produced an outcome for, including `no-decisive-failure`. */
996
+ scored: readonly string[];
997
+ }
998
+ /**
999
+ * Prove a campaign measured the denominator admission published.
1000
+ *
1001
+ * Three ways a denominator moves after the fact, each rejected here: scoring a
1002
+ * row that was never admitted, scoring a row that was never sampled, and
1003
+ * dropping a sampled row instead of scoring it. The third is the one an analyst
1004
+ * can cause on its own — declining a row it cannot solve — so a sampled row
1005
+ * with no outcome is an error, not a smaller `n`.
1006
+ */
1007
+ declare function assertDenominatorIntact(input: DenominatorIntactInput): void;
1008
+ //#endregion
1009
+ //#region src/trace-repair/admission-contract.d.ts
1010
+ /** Pre-registered admission thresholds. */
1011
+ interface AdmissionCriteria {
1012
+ /** Share of replayed prefix steps allowed to diverge from their recorded
1013
+ * returncode. Above it the container state under test is not the recorded
1014
+ * one, so nothing measured on it is about the recorded run. */
1015
+ readonly maxPrefixDivergenceRatio: number;
1016
+ /** Control rollouts each control arm must run. */
1017
+ readonly controlRollouts: number;
1018
+ /** How a control pass is read. `enforced` requires a control that can act. */
1019
+ readonly controlScreening: ControlScreening;
1020
+ }
1021
+ declare const TB_REPAIR_ADMISSION_CRITERIA: AdmissionCriteria;
1022
+ interface PrefixFidelityEvidence {
1023
+ /** Recorded steps that were replayed. */
1024
+ readonly stepsReplayed: number;
1025
+ /** Replayed steps whose exit code differed from the recorded returncode. */
1026
+ readonly divergences: number;
1027
+ }
1028
+ interface ControlArmEvidence {
1029
+ readonly rollouts: number;
1030
+ /** Rollouts whose held-out suite passed. Admission requires zero. */
1031
+ readonly passes: number;
1032
+ /** Policy digest every rollout reported. An arm that ran something other
1033
+ * than the declared control is not the control the criteria describe. */
1034
+ readonly policyDigest: string;
1035
+ }
1036
+ interface AdmissionEvidence {
1037
+ readonly rowId: string;
1038
+ /** Terminal-Bench-2 task the trajectory ran. It selects the certification
1039
+ * below, and a certification for another task is refused rather than read. */
1040
+ readonly taskName: string;
1041
+ /** Published image the trajectory was recorded against. */
1042
+ readonly image: string;
1043
+ /** Working directory the scaffold ran every action from. */
1044
+ readonly cwd: string;
1045
+ /** Task statement handed to the recorded agent. */
1046
+ readonly taskStatement: string;
1047
+ readonly steps: readonly RecordedTrajectoryStep[];
1048
+ /**
1049
+ * The task's certified oracle determinism. Required: a campaign that has not
1050
+ * measured whether its ground truth is a function of the state cannot
1051
+ * construct evidence, which is where that omission should stop.
1052
+ */
1053
+ readonly oracleDeterminism: OracleDeterminismVerdict;
1054
+ /** The control both arms ran, declared and hashed. */
1055
+ readonly controlPolicy: ControlPolicy;
1056
+ readonly prefixFidelity: PrefixFidelityEvidence;
1057
+ /** Held-out suite result on the recorded end state. Admission requires a fail. */
1058
+ readonly endStatePassed: boolean;
1059
+ /** Digest of the suite that produced every result above. */
1060
+ readonly suiteDigest: string;
1061
+ readonly noFixControl: ControlArmEvidence;
1062
+ readonly noOpControl: ControlArmEvidence;
1063
+ }
1064
+ /** Which pre-registered check refused the row. */
1065
+ type AdmissionRejection = 'task-oracle-nondeterministic' | 'prefix-divergence-too-high' | 'end-state-already-passes' | 'no-fix-control-passed' | 'no-op-control-passed' | 'control-passed-on-identical-state' | 'control-rollouts-short' | 'control-policy-mismatch' | 'empty-trajectory';
1066
+ declare const ADMITTED: unique symbol;
1067
+ /**
1068
+ * A row that passed every admission check.
1069
+ *
1070
+ * The brand is a phantom: it exists in the type and never in the value, and
1071
+ * the symbol that names it is not exported, so `AdmittedRow` cannot be
1072
+ * constructed anywhere but here. Both `blindTrajectory` and `gradeRepairRow`
1073
+ * require one. That is the structural form of "admission runs before any
1074
+ * analyst sees a row": there is no signature that accepts an unadmitted row.
1075
+ */
1076
+ interface AdmittedRow {
1077
+ readonly [ADMITTED]: true;
1078
+ readonly rowId: string;
1079
+ readonly taskName: string;
1080
+ readonly image: string;
1081
+ readonly cwd: string;
1082
+ readonly taskStatement: string;
1083
+ readonly steps: readonly RecordedTrajectoryStep[];
1084
+ readonly criteria: AdmissionCriteria;
1085
+ readonly prefixDivergenceRatio: number;
1086
+ readonly suiteDigest: string;
1087
+ /** The control the row was screened under, carried whole so a later reader
1088
+ * never has to open the runner's source to learn what screened it. */
1089
+ readonly controlPolicy: ControlPolicy;
1090
+ readonly controlScreening: ControlScreening;
1091
+ /** Policy digest both controls ran under. Every intervention rollout must
1092
+ * report the same one or the arms are not comparable. */
1093
+ readonly policyDigest: string;
1094
+ /** Control rate the paired delta subtracts. Zero by admission, carried
1095
+ * explicitly so the estimator never hardcodes it. */
1096
+ readonly controlRate: number;
1097
+ readonly controlRollouts: number;
1098
+ }
1099
+ /** Carried by every decision, admitted or not, so an artifact answers "which
1100
+ * control screened this row" without reading the runner's source. */
1101
+ interface AdmissionScreeningRecord {
1102
+ readonly controlPolicy: ControlPolicy;
1103
+ readonly controlScreening: ControlScreening;
1104
+ readonly controlPolicyDigest: string;
1105
+ readonly taskName: string;
1106
+ readonly oracleStable: boolean;
1107
+ readonly oracleFlipRate: number;
1108
+ }
1109
+ type AdmissionDecision = {
1110
+ readonly admitted: true;
1111
+ readonly screening: AdmissionScreeningRecord;
1112
+ readonly row: AdmittedRow;
1113
+ } | {
1114
+ readonly admitted: false;
1115
+ readonly rowId: string;
1116
+ readonly screening: AdmissionScreeningRecord;
1117
+ readonly rejection: AdmissionRejection;
1118
+ readonly detail: string;
1119
+ };
1120
+ /**
1121
+ * Decide admission from executed evidence.
1122
+ *
1123
+ * Pure: it opens no container and calls no model. Every threshold it applies
1124
+ * is in `criteria`, and every number it reads is in `evidence`, so an
1125
+ * admission decision is reproducible from the recorded evidence alone.
1126
+ *
1127
+ * It throws, rather than rejecting, when the criteria and the declared control
1128
+ * contradict. A contradiction there is a property of the configuration and not
1129
+ * of the row, so it must stop the run instead of producing one verdict per row
1130
+ * that reads as if a check had been applied.
1131
+ */
1132
+ declare function admitRow(evidence: AdmissionEvidence, criteria?: AdmissionCriteria): AdmissionDecision;
1133
+ //#endregion
1134
+ //#region src/trace-repair/admission-report.d.ts
1135
+ interface AdmissionArtifact {
1136
+ version: 1;
1137
+ kind: 'tb-repair-admission';
1138
+ generatedAt: string;
1139
+ /** Hash over the admitted ids, the config, and the provenance. */
1140
+ digest: string;
1141
+ config: AdmissionConfig;
1142
+ provenance: AdmissionProvenance;
1143
+ chain: DenominatorChainArtifact;
1144
+ /** Admitted row ids per stratum, the only form the campaign may sample from. */
1145
+ admitted: Record<AdmissionStratum, readonly string[]>;
1146
+ controlCost: {
1147
+ kind: string;
1148
+ usd: number | null;
1149
+ };
1150
+ /** Every input row, admitted or not, with the stratum and reason it carried. */
1151
+ rows: readonly AdmissionRowVerdict[];
1152
+ }
1153
+ /** Plain JSON. `JSON.stringify` of this object is the machine-readable artifact. */
1154
+ declare function admissionArtifact(report: AdmissionReport): AdmissionArtifact;
1155
+ interface RenderAdmissionOptions {
1156
+ /** Rows listed in the per-row appendix. Default 0, which omits the appendix. */
1157
+ rowLimit?: number;
1158
+ }
1159
+ /** Markdown for the campaign report. Every number here is also in the artifact. */
1160
+ declare function renderAdmissionReport(artifact: AdmissionArtifact, options?: RenderAdmissionOptions): string;
1161
+ //#endregion
1162
+ //#region src/trace-repair/analyst-response.d.ts
1163
+ /** The literal an analyst returns when no single step carries the failure. */
1164
+ declare const NO_DECISIVE_FAILURE = "no-decisive-failure";
1165
+ interface RepairIntervention {
1166
+ /** What the action is: a shell command, or an edit that authors file content. */
1167
+ readonly kind: ActionPayloadKind;
1168
+ /** The action itself, exactly as it will be executed at step k. */
1169
+ readonly action: string;
1170
+ }
1171
+ interface RepairFinding {
1172
+ readonly kind: 'finding';
1173
+ /** 1-based step the analyst blames. The repair must work here; there is no
1174
+ * credit for naming a step and repairing elsewhere. */
1175
+ readonly k: number;
1176
+ /** What the analyst says went wrong at step k. Recorded, never scored: a
1177
+ * claim graded by another model would put a judge inside the metric. */
1178
+ readonly failureClaim: string;
1179
+ readonly intervention: RepairIntervention;
1180
+ }
1181
+ interface RepairDeclined {
1182
+ readonly kind: 'no-decisive-failure';
1183
+ }
1184
+ type AnalystResponse = RepairFinding | RepairDeclined;
1185
+ /** Why a response does not parse. Counted by name in the funnel. */
1186
+ type ResponseParseFailure = 'unreadable' | 'not-a-single-answer' | 'missing-k' | 'missing-failure-claim' | 'missing-intervention' | 'unknown-intervention-kind';
1187
+ type ParseAnalystResponseOutcome = {
1188
+ readonly succeeded: true;
1189
+ readonly value: AnalystResponse;
1190
+ } | {
1191
+ readonly succeeded: false;
1192
+ readonly failure: ResponseParseFailure;
1193
+ readonly detail: string;
1194
+ };
1195
+ /**
1196
+ * Read an analyst reply.
1197
+ *
1198
+ * Accepts the bare literal, or a JSON object carrying one finding. The reply
1199
+ * is untrusted text, so every field is checked and nothing is defaulted: a
1200
+ * missing `k` is a parse failure, never step 1.
1201
+ */
1202
+ declare function parseAnalystResponse(reply: string): ParseAnalystResponseOutcome;
1203
+ /** Build a finding directly, for callers that already hold typed fields. */
1204
+ declare function repairFinding(input: {
1205
+ k: number;
1206
+ failureClaim: string;
1207
+ intervention: RepairIntervention;
1208
+ }): RepairFinding;
1209
+ //#endregion
1210
+ //#region src/trace-repair/blinding.d.ts
1211
+ interface BlindedStep {
1212
+ readonly step_id: number;
1213
+ readonly action: string;
1214
+ readonly observation: string | null;
1215
+ }
1216
+ interface BlindedTrajectoryPrefix {
1217
+ readonly rowId: string;
1218
+ /** The task statement the recorded agent was given. */
1219
+ readonly taskStatement: string;
1220
+ readonly steps: readonly BlindedStep[];
1221
+ /** Steps the recording holds. Equal to `steps.length` unless the prefix was
1222
+ * deliberately truncated. */
1223
+ readonly recordedSteps: number;
1224
+ /** Largest k the analyst may name for this prefix. */
1225
+ readonly maxK: number;
1226
+ }
1227
+ interface BlindTrajectoryOptions {
1228
+ /** Truncate the prefix after this 1-based step. Defaults to the whole
1229
+ * recording, which is what the recorded agent itself saw before it stopped. */
1230
+ readonly throughStep?: number;
1231
+ }
1232
+ declare function blindTrajectory(row: AdmittedRow, options?: BlindTrajectoryOptions): BlindedTrajectoryPrefix;
1233
+ /**
1234
+ * Fields an analyst prompt must never carry. Exported so a consumer that
1235
+ * builds its own prompt can assert against the same list this module honours.
1236
+ */
1237
+ declare const BLINDED_FIELDS: readonly string[];
1238
+ //#endregion
1239
+ //#region src/trace-repair/analyst-arm.d.ts
1240
+ /**
1241
+ * Something an arm can do that another arm cannot.
1242
+ *
1243
+ * Declared per arm and reported beside the result. This list is the vocabulary
1244
+ * of admissible difference: an arm whose advantage is not nameable here is not
1245
+ * comparable to the others, and that is a design decision to take deliberately
1246
+ * rather than a field to widen.
1247
+ */
1248
+ type RepairArmAffordance =
1249
+ /** The whole trajectory arrives in the prompt text. */
1250
+ 'inline-trajectory' |
1251
+ /** The trajectory is read through bounded tools instead of the prompt. */
1252
+ 'trajectory-tools' |
1253
+ /** The arm may run code while it reasons. */
1254
+ 'code-interpreter' |
1255
+ /** The arm takes internal turns of its own before answering. */
1256
+ 'agent-loop';
1257
+ /**
1258
+ * Whether the arm's instruction text carries a certification, and from where.
1259
+ *
1260
+ * On the repair task nothing is certified: the one certified analyst artifact
1261
+ * this repository holds was earned on a different benchmark under a different
1262
+ * output contract. `kind: 'none'` with a reason is therefore the expected
1263
+ * value, and it is recorded rather than omitted so a reader never has to
1264
+ * assume.
1265
+ */
1266
+ type RepairArmCertification = {
1267
+ readonly kind: 'none';
1268
+ readonly reason: string;
1269
+ } | {
1270
+ readonly kind: 'certified';
1271
+ /** The artifact whose text this arm runs. */
1272
+ readonly artifact: string;
1273
+ /** The benchmark the certification was earned on. */
1274
+ readonly benchmark: string;
1275
+ /** The output contract it was certified for. */
1276
+ readonly contract: string;
1277
+ };
1278
+ interface RepairArmDeclaration {
1279
+ readonly id: string;
1280
+ /** What actually answers, in one plain sentence. Recorded so a report never
1281
+ * has to infer the execution path from the runner's source. */
1282
+ readonly execution: string;
1283
+ readonly certification: RepairArmCertification;
1284
+ /** The action budget the arm's answers are measured against. */
1285
+ readonly budget: InterventionBudget;
1286
+ /** Bounded retries a structurally malformed reply earns. */
1287
+ readonly repairTurns: number;
1288
+ readonly affordances: readonly RepairArmAffordance[];
1289
+ /** The arm's own instruction and output-grammar text, exactly as the arm
1290
+ * composes it into the question. It enters the per-arm prompt digest, so a
1291
+ * contract change — a reworded grammar, a new signature version — changes
1292
+ * the digest of every answer the arm stamps. */
1293
+ readonly promptContract: readonly string[];
1294
+ }
1295
+ /** What an arm is handed. The type carries only the blinded prefix, so a
1296
+ * grading field is unreachable rather than merely unread: the admitted row
1297
+ * stays with `askRepairArm`, which does the blinding and the bookkeeping. */
1298
+ interface RepairArmRequest {
1299
+ readonly prefix: BlindedTrajectoryPrefix;
1300
+ readonly signal?: AbortSignal;
1301
+ }
1302
+ /** Token accounting exactly as the execution path reported it. A side nobody
1303
+ * measured stays null rather than becoming a zero. */
1304
+ interface RepairArmUsage {
1305
+ readonly calls: number | null;
1306
+ readonly inputTokens: number | null;
1307
+ readonly outputTokens: number | null;
1308
+ readonly costUsd: number | null;
1309
+ }
1310
+ interface RepairArmRepairTurn {
1311
+ readonly attempted: boolean;
1312
+ /** Null when no repair turn ran, or when it never returned a reply. */
1313
+ readonly succeeded: boolean | null;
1314
+ }
1315
+ /** A row an arm reported and the pipeline refused, counted by reason. */
1316
+ interface RepairArmRejectedRow {
1317
+ readonly index: number;
1318
+ readonly reason: string;
1319
+ }
1320
+ /**
1321
+ * What an arm hands back.
1322
+ *
1323
+ * `finding` and `declined` are the two answers the grader consumes. `failed`
1324
+ * is everything else, and it always carries a reason: an arm that could not
1325
+ * answer must say so rather than return an empty finding that grades as a
1326
+ * decline it never made.
1327
+ */
1328
+ type RepairArmReply = {
1329
+ readonly status: 'finding';
1330
+ readonly k: number;
1331
+ readonly failureClaim: string;
1332
+ readonly intervention: RepairIntervention;
1333
+ readonly answer: string | null;
1334
+ readonly reportedRows: number;
1335
+ readonly rejectedRows: readonly RepairArmRejectedRow[];
1336
+ readonly repair: RepairArmRepairTurn;
1337
+ readonly usage: RepairArmUsage;
1338
+ } | {
1339
+ readonly status: 'declined';
1340
+ readonly answer: string | null;
1341
+ readonly reportedRows: number;
1342
+ readonly rejectedRows: readonly RepairArmRejectedRow[];
1343
+ readonly repair: RepairArmRepairTurn;
1344
+ readonly usage: RepairArmUsage;
1345
+ } | {
1346
+ readonly status: 'failed';
1347
+ readonly failure: string;
1348
+ readonly answer: string | null;
1349
+ readonly rejectedRows: readonly RepairArmRejectedRow[];
1350
+ readonly repair: RepairArmRepairTurn;
1351
+ readonly usage: RepairArmUsage;
1352
+ };
1353
+ interface RepairArm {
1354
+ readonly declaration: RepairArmDeclaration;
1355
+ ask(request: RepairArmRequest): Promise<RepairArmReply>;
1356
+ }
1357
+ /**
1358
+ * The budget verdict on an arm's action, measured before any container opens.
1359
+ *
1360
+ * Recorded, never enforced here: `gradeRepairRow` refuses an inadmissible
1361
+ * action, and one authority is the point. Measuring it at answer time is what
1362
+ * lets a report say what an arm spent its bytes on without paying for a
1363
+ * rollout to find out.
1364
+ */
1365
+ type RepairArmBudgetRecord = {
1366
+ readonly admissible: true;
1367
+ readonly measurement: BudgetMeasurement;
1368
+ } | {
1369
+ readonly admissible: false;
1370
+ readonly violation: BudgetViolation;
1371
+ readonly detail: string;
1372
+ readonly measurement: BudgetMeasurement;
1373
+ };
1374
+ interface RepairArmAnswer {
1375
+ readonly rowId: string;
1376
+ readonly armId: string;
1377
+ /** Digest of the composed question the arm answered. */
1378
+ readonly promptSha256: string;
1379
+ readonly reply: RepairArmReply;
1380
+ /** Present only for a finding. */
1381
+ readonly budget: RepairArmBudgetRecord | null;
1382
+ readonly wallMs: number;
1383
+ }
1384
+ interface AskRepairArmOptions {
1385
+ readonly arm: RepairArm;
1386
+ readonly row: AdmittedRow;
1387
+ /** Truncate the prefix after this 1-based step. Defaults to the whole recording. */
1388
+ readonly throughStep?: number;
1389
+ readonly signal?: AbortSignal;
1390
+ /** Injectable clock so a test can assert a recorded duration. */
1391
+ readonly now?: () => number;
1392
+ }
1393
+ /**
1394
+ * Ask one arm about one admitted row.
1395
+ *
1396
+ * Blinding, prompt identity and budget measurement happen here, once, for
1397
+ * every arm. An arm that wants a different question does not get one.
1398
+ */
1399
+ declare function askRepairArm(options: AskRepairArmOptions): Promise<RepairArmAnswer>;
1400
+ /**
1401
+ * The answer in the grammar the grader consumes.
1402
+ *
1403
+ * Null for a failure: a run that could not answer has no answer to grade, and
1404
+ * turning it into a decline would credit an arm with an honest null it never
1405
+ * produced.
1406
+ */
1407
+ declare function repairArmResponse(answer: RepairArmAnswer): AnalystResponse | null;
1408
+ /** One arm's difference from the set, as a report renders it. */
1409
+ interface RepairArmAsymmetry {
1410
+ readonly armId: string;
1411
+ /** Affordances this arm has that at least one other arm lacks. */
1412
+ readonly extraAffordances: readonly RepairArmAffordance[];
1413
+ /** Affordances at least one other arm has and this one lacks. */
1414
+ readonly missingAffordances: readonly RepairArmAffordance[];
1415
+ readonly certification: RepairArmCertification;
1416
+ /** Digest of the composed question this arm answers: the shared question
1417
+ * plus the arm's own contract text. Two arms with the same digest asked the
1418
+ * identical composed question; two arms with different digests did not. */
1419
+ readonly promptSha256: string;
1420
+ }
1421
+ interface RepairArmAsymmetryReport {
1422
+ readonly armIds: readonly string[];
1423
+ /** Affordances every arm has. Not an asymmetry; recorded so a reader can see
1424
+ * the floor the comparison runs on. */
1425
+ readonly sharedAffordances: readonly RepairArmAffordance[];
1426
+ readonly asymmetries: readonly RepairArmAsymmetry[];
1427
+ /** True when no arm carries a certification. The honest state on a task
1428
+ * whose graders are newer than every certified artifact. */
1429
+ readonly noArmCertified: boolean;
1430
+ readonly budget: InterventionBudget;
1431
+ readonly repairTurns: number;
1432
+ /** Digest of the question and task policy every arm shares. Per-arm
1433
+ * composed-question digests live on each asymmetry entry. */
1434
+ readonly questionSha256: string;
1435
+ }
1436
+ /**
1437
+ * Refuse a set of arms that cannot be compared, and describe what still
1438
+ * differs between the ones that can.
1439
+ *
1440
+ * Two properties are hard: the arms measure actions against the same budget,
1441
+ * and a malformed reply earns the same number of retries everywhere. Both are
1442
+ * things that would move a score without moving the thing being measured.
1443
+ *
1444
+ * Certification is the third: a set where some arms run certified text and
1445
+ * others do not is refused outright. An optimisation applies to every arm or
1446
+ * to none — a mixed set measures the optimisation, then reports it as the
1447
+ * harness.
1448
+ */
1449
+ declare function repairArmAsymmetries(arms: readonly RepairArm[], options?: {
1450
+ readonly budget?: InterventionBudget;
1451
+ }): RepairArmAsymmetryReport;
1452
+ //#endregion
1453
+ //#region src/trace-repair/arm-completion.d.ts
1454
+ interface CompletionRepairArmOptions {
1455
+ readonly id: string;
1456
+ /** What actually answers the POST, in one plain sentence. */
1457
+ readonly execution: string;
1458
+ /** Full endpoint, e.g. `http://127.0.0.1:4200/v1/chat/completions`. */
1459
+ readonly url: string;
1460
+ readonly model: string;
1461
+ /** Injected so a test drives the arm without a network. */
1462
+ readonly transport: PrimeBridgeTransport;
1463
+ /** Deadline for ONE turn. */
1464
+ readonly timeoutMs: number;
1465
+ /** Rates used to price the reported tokens. */
1466
+ readonly pricing: CustomTokenPricing;
1467
+ readonly budget?: InterventionBudget;
1468
+ /** What this arm can do that another cannot. */
1469
+ readonly affordances: readonly RepairArmAffordance[];
1470
+ readonly certification: RepairArmCertification;
1471
+ }
1472
+ declare function createCompletionRepairArm(options: CompletionRepairArmOptions): RepairArm;
1473
+ //#endregion
1474
+ //#region src/trace-repair/arm-dspy.d.ts
1475
+ /**
1476
+ * Token the bridge reads to select the typed repair signature.
1477
+ *
1478
+ * The wire protocol carries instructions as opaque text, so the task is named
1479
+ * inside them rather than by adding a field every unrelated caller would have
1480
+ * to set. The same mechanism selects the CodeTraceBench typed signature.
1481
+ */
1482
+ declare const DSPY_REPAIR_TASK_TOKEN = "tb-repair-typed-";
1483
+ /** Signature the bridge reports for a repair analysis. */
1484
+ declare const DSPY_REPAIR_SIGNATURE = "tb-repair-typed-v1";
1485
+ /** The name the trajectory is bound to inside the program's environment. */
1486
+ declare const DSPY_REPAIR_TRAJECTORY_INPUT = "trajectory";
1487
+ /**
1488
+ * The repair contract restated for the typed signature.
1489
+ *
1490
+ * Transport differs from the chat-completion arms — typed SUBMIT instead of a
1491
+ * fenced JSON object — and the QUESTION does not. Both read from
1492
+ * `repairTaskPolicy`, so the execution rules an arm is held to cannot drift
1493
+ * between arms.
1494
+ */
1495
+ declare function dspyRepairInstructions(budget?: InterventionBudget): string;
1496
+ interface DspyRepairArmOptions {
1497
+ /** The DSPy engine. Injected, so a test drives the arm with no subprocess. */
1498
+ readonly engine: TraceAnalysisEngine;
1499
+ readonly limits: TraceAnalystLimits;
1500
+ readonly costLedger: CostLedgerHandle;
1501
+ readonly costPhase: string;
1502
+ readonly analystId: string;
1503
+ readonly costTags?: Record<string, string>;
1504
+ readonly budget?: InterventionBudget;
1505
+ readonly id?: string;
1506
+ readonly log?: (message: string, fields?: Record<string, unknown>) => void;
1507
+ }
1508
+ declare function createDspyRepairArm(options: DspyRepairArmOptions): RepairArm;
1509
+ interface DspyRepairRow {
1510
+ readonly k: number;
1511
+ readonly failureClaim: string;
1512
+ readonly kind: ActionPayloadKind;
1513
+ readonly action: string;
1514
+ }
1515
+ interface DspyRepairPayload {
1516
+ readonly signature: string;
1517
+ readonly reported: number;
1518
+ readonly rows: readonly DspyRepairRow[];
1519
+ readonly dropped: readonly RepairArmRejectedRow[];
1520
+ /** How the bounded structured re-read recovered the typed field, or null
1521
+ * when the field arrived typed and no re-read was needed. */
1522
+ readonly repair: string | null;
1523
+ /** The bridge's own account of a typed field it could not read — a reply
1524
+ * that stayed malformed after the bounded re-read. The analysis completed,
1525
+ * so the answer text survives beside it; null on a readable reply. */
1526
+ readonly failure: string | null;
1527
+ }
1528
+ type ReadRepairPayload = {
1529
+ readonly ok: true;
1530
+ readonly value: DspyRepairPayload;
1531
+ } | {
1532
+ readonly ok: false;
1533
+ readonly reason: string;
1534
+ };
1535
+ interface ReadRepairPayloadOptions {
1536
+ /** The step ids the blinded prefix holds. A row naming any other k is the
1537
+ * model's mistake, not the bridge's, so it lands in `dropped` — the same
1538
+ * rejected-row path the chat-completion arms take — never in a finding the
1539
+ * grader would refuse as out of range. */
1540
+ readonly validSteps: readonly number[];
1541
+ }
1542
+ /**
1543
+ * Read the bridge's typed repair block.
1544
+ *
1545
+ * Structure is checked and fails loud: a missing block, a wrong signature, or
1546
+ * a row that does not carry an integer k and a non-empty action is a wiring
1547
+ * fault with a reason — never a default, and never an empty list that would
1548
+ * grade as a decline the program did not make. A structurally sound row whose
1549
+ * k is not a recorded step id is the model's mistake, and it is dropped with
1550
+ * its reason instead.
1551
+ */
1552
+ declare function readRepairPayload(runtime: Record<string, unknown>, options: ReadRepairPayloadOptions): ReadRepairPayload;
1553
+ /** Thrown by callers that build a repair arm with a non-DSPy engine. */
1554
+ declare function assertDspyRepairEngine(engine: TraceAnalysisEngine): void;
1555
+ //#endregion
1556
+ //#region src/trace-repair/degenerate-strategies.d.ts
1557
+ /**
1558
+ * The strategies that would score without repairing anything, named here so a
1559
+ * reviewer can check each one against the mechanism that defeats it and a
1560
+ * test can be required for each.
1561
+ *
1562
+ * Two kinds of defeat, and the difference matters:
1563
+ *
1564
+ * gate the grader refuses the answer before spending a rollout
1565
+ * measurement the answer runs and measures the same as its control
1566
+ *
1567
+ * A gate is cheap and certain. A measurement defeat is the honest one where
1568
+ * no syntactic rule can decide — a semantic no-op looks like a repair until
1569
+ * the tests disagree with it — and it costs rollouts to establish.
1570
+ */
1571
+ type DegenerateDefeatKind = 'gate' | 'measurement';
1572
+ interface DegenerateStrategy {
1573
+ readonly id: string;
1574
+ /** What the analyst does to score without repairing. */
1575
+ readonly strategy: string;
1576
+ /** The mechanism that removes the reward. */
1577
+ readonly defeat: string;
1578
+ readonly defeatKind: DegenerateDefeatKind;
1579
+ /** Module that carries the mechanism. */
1580
+ readonly enforcedIn: string;
1581
+ }
1582
+ declare const DEGENERATE_STRATEGIES: readonly [{
1583
+ readonly id: 'point-at-any-nonzero-exit-step';
1584
+ readonly strategy: 'Name the first step with a nonzero returncode, claim it failed, and let the reproduction gate confirm it.';
1585
+ readonly defeat: 'Reproduction is a gate, not a tier that pays. The credit vector has no term for it, so a reproduced step with no working intervention scores exactly what an unreproduced one scores: nothing.';
1586
+ readonly defeatKind: 'gate';
1587
+ readonly enforcedIn: 'funnel.ts (repairCredit has no reproduction term)';
1588
+ }, {
1589
+ readonly id: 'propose-the-recorded-command-again';
1590
+ readonly strategy: 'Return the action the agent already ran at step k, so the arm reproduces the recorded state and looks like a faithful replay.';
1591
+ readonly defeat: 'The intervention is compared against the recorded action at k after whitespace normalisation and rejected before a container opens.';
1592
+ readonly defeatKind: 'gate';
1593
+ readonly enforcedIn: 'grade.ts (recorded-action-reproposed rejection)';
1594
+ }, {
1595
+ readonly id: 'propose-a-no-op';
1596
+ readonly strategy: 'Return an action that changes nothing, so the arm inherits whatever the trajectory would have done anyway.';
1597
+ readonly defeat: 'Literal no-ops are rejected at the budget check. A semantic no-op cannot be detected syntactically, so it runs and measures at the no-op control floor: its paired delta is zero.';
1598
+ readonly defeatKind: 'measurement';
1599
+ readonly enforcedIn: 'action-budget.ts (NO_OP_ACTIONS) and delta-repair.ts (paired delta)';
1600
+ }, {
1601
+ readonly id: 'submit-instead-of-repair';
1602
+ readonly strategy: 'Return the submit sentinel so the run ends immediately and the arm terminates cleanly.';
1603
+ readonly defeat: 'An action carrying the submit sentinel is rejected at the budget check. Ending the run is what the recorded agent already did and the tests already failed on it.';
1604
+ readonly defeatKind: 'gate';
1605
+ readonly enforcedIn: 'action-budget.ts (SUBMIT_SENTINEL rejection)';
1606
+ }, {
1607
+ readonly id: 'touch-the-test-suite';
1608
+ readonly strategy: "Write a passing suite, or a reward file, at the path the grader will read, so the oracle grades the trajectory's own artifact.";
1609
+ readonly defeat: 'The oracle purges the suite root and uploads the held-out suite from outside the session at grade time, then verifies the bytes it reads back. A planted suite is overwritten; a session that refuses the overwrite raises a tamper error instead of returning a pass.';
1610
+ readonly defeatKind: 'gate';
1611
+ readonly enforcedIn: 'test-oracle.ts (purge, upload, read-back digest)';
1612
+ }, {
1613
+ readonly id: 'buy-a-bigger-action';
1614
+ readonly strategy: 'Return a multi-command script or a whole-file rewrite that does far more than one scaffold turn could.';
1615
+ readonly defeat: 'The budget counts top-level statements, heredocs and bytes. More than one action, more than one authored file, or more than 4 KB is rejected before a container opens.';
1616
+ readonly defeatKind: 'gate';
1617
+ readonly enforcedIn: 'action-budget.ts (checkInterventionBudget)';
1618
+ }, {
1619
+ readonly id: 'decline-every-hard-row';
1620
+ readonly strategy: 'Answer no-decisive-failure on everything except the rows that are obviously repairable, so the reported rate is computed on an easy subset.';
1621
+ readonly defeat: 'A declined row keeps its cell in the funnel and stays in the denominator with a paired delta of zero, because its intervention arm is definitionally its control arm. Declining cannot raise the headline; it can only dilute it.';
1622
+ readonly defeatKind: 'measurement';
1623
+ readonly enforcedIn: 'grade.ts (declined outcome) and delta-repair.ts (full admitted denominator)';
1624
+ }, {
1625
+ readonly id: 'repair-somewhere-other-than-k';
1626
+ readonly strategy: 'Name a plausible-looking k, then submit an intervention that fixes the task from any state, so the answer scores without localising anything.';
1627
+ readonly defeat: 'The intervention is executed at the k the analyst named, on the state produced by replaying steps 1..k-1. There is no separate localisation credit to win and no label the grader reads, so a wrong k is only penalised through the repair failing to work there.';
1628
+ readonly defeatKind: 'measurement';
1629
+ readonly enforcedIn: 'grade.ts (the intervention is applied at the named k only)';
1630
+ }];
1631
+ type DegenerateStrategyId = (typeof DEGENERATE_STRATEGIES)[number]['id'];
1632
+ declare function degenerateStrategy(id: DegenerateStrategyId): DegenerateStrategy;
1633
+ //#endregion
1634
+ //#region src/trace-repair/ports.d.ts
1635
+ /**
1636
+ * Which arm a session serves. The arm names the container state under test,
1637
+ * never a different policy: `Delta-repair` only measures the intervention when
1638
+ * every arm is continued the same way.
1639
+ */
1640
+ type RepairArm$1 = 'reproduce' | 'local-flip' | 'intervention' | 'no-fix-control' | 'no-op-control' | 'end-state';
1641
+ interface RepairExecResult {
1642
+ exitCode: number;
1643
+ stdout: string;
1644
+ stderr: string;
1645
+ /** True when the environment killed the command at its wall-clock limit. */
1646
+ timedOut: boolean;
1647
+ }
1648
+ interface RepairSession {
1649
+ /** Container id or equivalent handle, recorded for provenance. */
1650
+ readonly ref: string;
1651
+ exec(command: string, timeoutMs: number): Promise<RepairExecResult>;
1652
+ close(): Promise<void>;
1653
+ }
1654
+ interface RepairSessionRequest {
1655
+ rowId: string;
1656
+ /** Image the trajectory was recorded against. Never a locally rebuilt one:
1657
+ * unpinned apt and pip installs drift a rebuild away from the recording. */
1658
+ image: string;
1659
+ arm: RepairArm$1;
1660
+ /** 0-based index within the arm. Every rollout gets its own environment. */
1661
+ rolloutIndex: number;
1662
+ }
1663
+ interface RepairSessionFactory {
1664
+ /** One fresh environment per call; the grader closes it. */
1665
+ open(request: RepairSessionRequest): Promise<RepairSession>;
1666
+ }
1667
+ interface TestOracleContext {
1668
+ rowId: string;
1669
+ arm: RepairArm$1;
1670
+ rolloutIndex: number;
1671
+ }
1672
+ interface TestOracleOutcome {
1673
+ /** True only when the suite command exited 0. Never defaulted. */
1674
+ passed: boolean;
1675
+ exitCode: number;
1676
+ output: string;
1677
+ /** Digest of the suite as read back from inside the container after upload. */
1678
+ suiteDigest: string;
1679
+ timedOut: boolean;
1680
+ }
1681
+ /**
1682
+ * The held-out suite.
1683
+ *
1684
+ * A conforming oracle uploads the suite from outside the session at grade
1685
+ * time and verifies what it reads back, so a trajectory that plants its own
1686
+ * passing suite is overwritten rather than believed. `injectedTestOracle`
1687
+ * implements that; see `test-oracle.ts`.
1688
+ */
1689
+ interface TestOracle {
1690
+ grade(session: RepairSession, context: TestOracleContext): Promise<TestOracleOutcome>;
1691
+ }
1692
+ interface RepairContinuationRequest {
1693
+ rowId: string;
1694
+ arm: RepairArm$1;
1695
+ rolloutIndex: number;
1696
+ /** Environment already restored to the state this arm continues from. */
1697
+ session: RepairSession;
1698
+ /** Recorded steps the continuation must treat as already taken. */
1699
+ steps: readonly RecordedTrajectoryStep[];
1700
+ /** 1-based step the injected action replaced; null for a control that
1701
+ * continues from the recorded end state. */
1702
+ k: number | null;
1703
+ /**
1704
+ * The action that ran in place of step k and the raw result it produced.
1705
+ * Null for the no-fix control, which injects nothing. The runner renders it
1706
+ * into the scaffold's own observation grammar, so the grader never emits a
1707
+ * second copy of that format.
1708
+ */
1709
+ injected: InjectedAction | null;
1710
+ taskStatement: string;
1711
+ }
1712
+ interface InjectedAction {
1713
+ action: string;
1714
+ returncode: number;
1715
+ output: string;
1716
+ timedOut: boolean;
1717
+ }
1718
+ interface RepairContinuationOutcome {
1719
+ /** Frozen configuration the rollout ran under. */
1720
+ policyId: string;
1721
+ /** Hash over the policy and its scaffold templates. Equal across arms by
1722
+ * construction; the grader rejects a row whose arms disagree. */
1723
+ policyDigest: string;
1724
+ /** Model calls the rollout made. */
1725
+ steps: number;
1726
+ /** Why the rollout stopped, in the continuation layer's own vocabulary. */
1727
+ exitStatus: string;
1728
+ submitted: boolean;
1729
+ }
1730
+ /** Runs the pinned continuation policy forward from a prepared session. */
1731
+ type RepairContinuationRunner = (request: RepairContinuationRequest) => Promise<RepairContinuationOutcome>;
1732
+ //#endregion
1733
+ //#region src/trace-repair/funnel.d.ts
1734
+ /** Why an answer never reached the reproduction gate. */
1735
+ type RepairRejection = {
1736
+ readonly source: 'parse';
1737
+ readonly reason: ResponseParseFailure;
1738
+ readonly detail: string;
1739
+ } | {
1740
+ readonly source: 'budget';
1741
+ readonly reason: BudgetViolation;
1742
+ readonly detail: string;
1743
+ readonly measurement: BudgetMeasurement;
1744
+ } | {
1745
+ readonly source: 'target';
1746
+ readonly reason: 'k-out-of-range' | 'recorded-action-reproposed';
1747
+ readonly detail: string;
1748
+ };
1749
+ interface PrefixReplayEvidence {
1750
+ /** Recorded steps 1..k-1 executed to rebuild the state at k. */
1751
+ readonly stepsReplayed: number;
1752
+ /** Replayed steps whose exit code differed from the recorded returncode. */
1753
+ readonly divergences: number;
1754
+ readonly wallMs: number;
1755
+ }
1756
+ /**
1757
+ * What the reproduction gate observed at k.
1758
+ *
1759
+ * `signature` and `signatureObserved` are null together when the recorded
1760
+ * observation carried no error line; the gate then rests on the returncode
1761
+ * alone and says so through `basis`.
1762
+ */
1763
+ type ReproductionEvidence = {
1764
+ /** The step recorded no observation, so there is no state to reproduce.
1765
+ * The gate passes vacuously and no container is opened for it. */
1766
+ readonly basis: 'no-recorded-observation';
1767
+ readonly reproduced: true;
1768
+ } | {
1769
+ readonly basis: 'returncode-only' | 'returncode+output-substring';
1770
+ readonly reproduced: boolean;
1771
+ readonly recordedReturncode: number;
1772
+ readonly observedExitCode: number;
1773
+ readonly signature: string | null;
1774
+ readonly signatureObserved: boolean | null;
1775
+ readonly prefix: PrefixReplayEvidence;
1776
+ };
1777
+ interface InterventionExecution {
1778
+ readonly command: string;
1779
+ readonly exitCode: number;
1780
+ readonly timedOut: boolean;
1781
+ readonly wallMs: number;
1782
+ readonly output: string;
1783
+ readonly prefix: PrefixReplayEvidence;
1784
+ }
1785
+ interface TestRunEvidence {
1786
+ readonly passed: boolean;
1787
+ readonly exitCode: number;
1788
+ readonly timedOut: boolean;
1789
+ /** Digest of the suite the oracle read back from inside the container. */
1790
+ readonly suiteDigest: string;
1791
+ }
1792
+ /**
1793
+ * One repair rollout.
1794
+ *
1795
+ * A rollout whose intervention failed to run is kept and counted as a
1796
+ * non-pass rather than dropped. Dropping it would quietly remove the rollouts
1797
+ * where the environment behaved worst, which is the direction that flatters
1798
+ * the intervention.
1799
+ */
1800
+ type RepairRolloutEvidence = {
1801
+ readonly rolloutIndex: number;
1802
+ readonly status: 'completed';
1803
+ readonly interventionExitCode: number;
1804
+ readonly continuation: RepairContinuationOutcome;
1805
+ readonly tests: TestRunEvidence;
1806
+ } | {
1807
+ readonly rolloutIndex: number;
1808
+ readonly status: 'intervention-failed';
1809
+ readonly interventionExitCode: number;
1810
+ readonly timedOut: boolean;
1811
+ };
1812
+ interface RepairArmEvidence {
1813
+ /** Rollouts attempted. The denominator of `repairRate`. */
1814
+ readonly rollouts: number;
1815
+ /** Rollouts whose held-out suite passed. */
1816
+ readonly passes: number;
1817
+ /** Rollouts where the intervention did not run, counted as non-passes. */
1818
+ readonly interventionFailures: number;
1819
+ /** Policy digest every completed rollout reported, checked against the
1820
+ * controls. Null when no rollout completed. */
1821
+ readonly policyDigest: string | null;
1822
+ readonly rolloutEvidence: readonly RepairRolloutEvidence[];
1823
+ }
1824
+ /**
1825
+ * The deepest state an answer reached.
1826
+ *
1827
+ * Only `measured` carries a repair, so a gate that closed has no field a score
1828
+ * could read.
1829
+ */
1830
+ type RepairGrade = {
1831
+ readonly outcome: 'rejected';
1832
+ readonly rejection: RepairRejection;
1833
+ } | {
1834
+ readonly outcome: 'declined';
1835
+ } | {
1836
+ readonly outcome: 'not-reproduced';
1837
+ readonly k: number;
1838
+ readonly reproduction: ReproductionEvidence;
1839
+ } | {
1840
+ readonly outcome: 'did-not-execute';
1841
+ readonly k: number;
1842
+ readonly reproduction: ReproductionEvidence;
1843
+ readonly execution: InterventionExecution;
1844
+ } | {
1845
+ readonly outcome: 'measured';
1846
+ readonly k: number;
1847
+ readonly reproduction: ReproductionEvidence;
1848
+ readonly execution: InterventionExecution;
1849
+ readonly localFlip: TestRunEvidence;
1850
+ readonly repair: RepairArmEvidence;
1851
+ };
1852
+ /**
1853
+ * Everything an answer earns.
1854
+ *
1855
+ * Three numeric terms and no reproduction term. A vector rather than a scalar
1856
+ * because the three measure different things: whether the action ran, whether
1857
+ * it fixed the task outright, and how often it led to a fix once the agent
1858
+ * carried on.
1859
+ */
1860
+ interface RepairCredit {
1861
+ readonly executes: 0 | 1;
1862
+ readonly localFlip: 0 | 1;
1863
+ /** Share of repair rollouts whose held-out suite passed. */
1864
+ readonly repairRate: number;
1865
+ }
1866
+ declare const CREDIT_TERMS: readonly ["executes", "localFlip", "repairRate"];
1867
+ /** What an answer earned. Every outcome but `measured` earns nothing. */
1868
+ declare function repairCredit(grade: RepairGrade): RepairCredit;
1869
+ /** True once the answer is a well-formed, budget-admissible answer. A decline
1870
+ * is well formed, so it parses. */
1871
+ declare function reachedT0(grade: RepairGrade): boolean;
1872
+ /** True once the recorded state at k came back. A decline never reaches the
1873
+ * gate, because it names no k. */
1874
+ declare function reachedT1(grade: RepairGrade): boolean;
1875
+ /** True once the intervention ran at k and exited cleanly. */
1876
+ declare function reachedT2(grade: RepairGrade): boolean;
1877
+ interface RepairFunnelCounts {
1878
+ readonly rows: number;
1879
+ readonly rejected: number;
1880
+ readonly declined: number;
1881
+ readonly t0Parsed: number;
1882
+ readonly t1Reproduced: number;
1883
+ readonly t2Executed: number;
1884
+ readonly t3LocalFlip: number;
1885
+ /** Rows where at least one repair rollout passed the held-out suite. */
1886
+ readonly t4RepairFlipAny: number;
1887
+ /** Rows where every repair rollout passed. */
1888
+ readonly t4RepairFlipAll: number;
1889
+ }
1890
+ declare function countFunnel(grades: readonly RepairGrade[]): RepairFunnelCounts;
1891
+ //#endregion
1892
+ //#region src/trace-repair/grade.d.ts
1893
+ /** Rollouts of the repair arm disagreed with the controls about the policy. */
1894
+ declare class RepairArmSymmetryError extends CaptureIntegrityError {}
1895
+ interface GradeRepairOptions {
1896
+ readonly row: AdmittedRow;
1897
+ readonly response: AnalystResponse;
1898
+ readonly sessions: RepairSessionFactory;
1899
+ readonly oracle: TestOracle;
1900
+ readonly continuation: RepairContinuationRunner;
1901
+ /** Repair rollouts for the intervention arm. Defaults to the row's own
1902
+ * control rollout count so the arms are matched. */
1903
+ readonly repairRollouts?: number;
1904
+ readonly budget?: InterventionBudget;
1905
+ /** Wall-clock limit for one replayed or injected action. */
1906
+ readonly stepTimeoutMs?: number;
1907
+ /**
1908
+ * Wall-clock limit for replaying a step the recording itself killed.
1909
+ *
1910
+ * Such a step carries no returncode, so the replay agrees with the recording
1911
+ * no matter how long it waits. Bounding it separately cuts what a prefix
1912
+ * costs without changing what it counts. Defaults to `stepTimeoutMs`.
1913
+ */
1914
+ readonly recordedTimeoutStepMs?: number;
1915
+ readonly onProgress?: (message: string) => void;
1916
+ }
1917
+ interface RepairRowResult {
1918
+ readonly rowId: string;
1919
+ readonly grade: RepairGrade;
1920
+ readonly credit: RepairCredit;
1921
+ /** P(tests pass | intervention). Equals `controlRate` when no intervention
1922
+ * arm ran, because the row's state is then the control's state. */
1923
+ readonly interventionRate: number;
1924
+ readonly controlRate: number;
1925
+ readonly controlRollouts: number;
1926
+ /** How the row's control pass was read. Travels with the result so a report
1927
+ * can say whether a control rate of zero was screened or merely inert. */
1928
+ readonly controlScreening: ControlScreening;
1929
+ readonly controlPolicyDigest: string;
1930
+ readonly repairRollouts: number;
1931
+ /** The row's paired contribution to Delta-repair. */
1932
+ readonly delta: number;
1933
+ readonly wallMs: number;
1934
+ }
1935
+ declare function gradeRepairRow(options: GradeRepairOptions): Promise<RepairRowResult>;
1936
+ //#endregion
1937
+ //#region src/trace-repair/delta-repair.d.ts
1938
+ interface DeltaRepairOptions {
1939
+ /** Deterministic resampling seed. Derived from the deltas when omitted. */
1940
+ readonly seed?: number;
1941
+ readonly resamples?: number;
1942
+ readonly confidence?: number;
1943
+ }
1944
+ interface DeltaRepairInterval {
1945
+ readonly n: number;
1946
+ readonly mean: number;
1947
+ readonly median: number;
1948
+ readonly low: number;
1949
+ readonly high: number;
1950
+ readonly confidence: number;
1951
+ readonly resamples: number;
1952
+ /** False below the pair count where a percentile interval carries its
1953
+ * nominal error rate. A gate must not turn on a false one. */
1954
+ readonly gateEligible: boolean;
1955
+ }
1956
+ type RepairThreatId = 'control-position-asymmetry' | 'control-cannot-rescue' | 'admission-conditions-on-control-failure' | 'bootstrap-below-min-n' | 'prefix-divergence-present' | 'intervention-failures-present' | 'zero-variance-interval' | 'declines-carry-the-denominator';
1957
+ interface RepairThreat {
1958
+ readonly id: RepairThreatId;
1959
+ readonly statement: string;
1960
+ /** Which way the threat pushes the headline if it is real. */
1961
+ readonly direction: 'understates' | 'overstates' | 'unknown';
1962
+ }
1963
+ interface DeltaRepairReport {
1964
+ readonly rows: number;
1965
+ readonly funnel: RepairFunnelCounts;
1966
+ /** Mean over every admitted row. */
1967
+ readonly interventionRate: number;
1968
+ readonly controlRate: number;
1969
+ readonly deltaRepair: DeltaRepairInterval;
1970
+ /** The same estimate restricted to rows that reached t2, where an
1971
+ * intervention actually ran. Conditional on the analyst answering, so it is
1972
+ * never the headline. */
1973
+ readonly measuredOnly: DeltaRepairInterval;
1974
+ readonly measuredRows: number;
1975
+ readonly rowResults: readonly RepairRowResult[];
1976
+ readonly threats: readonly RepairThreat[];
1977
+ }
1978
+ declare function deltaRepair(rowResults: readonly RepairRowResult[], options?: DeltaRepairOptions): DeltaRepairReport;
1979
+ /** Markdown report: provenance, the funnel, every per-row column, the
1980
+ * distribution, and the threats. */
1981
+ declare function renderDeltaRepairReport(report: DeltaRepairReport): string;
1982
+ //#endregion
1983
+ //#region src/trace-repair/docker-environment.d.ts
1984
+ interface ProcessRequest {
1985
+ argv: string[];
1986
+ /** Kills the process group when exceeded. */
1987
+ timeoutSeconds?: number;
1988
+ }
1989
+ interface ProcessResult {
1990
+ /** stdout and stderr interleaved, as the scaffold reads them. */
1991
+ output: string;
1992
+ exitCode: number;
1993
+ /** True when the process was killed for exceeding `timeoutSeconds`. */
1994
+ timedOut: boolean;
1995
+ }
1996
+ /** Runs a process on the host. Injected so the environment is testable without a daemon. */
1997
+ type ProcessRunner = (request: ProcessRequest) => Promise<ProcessResult>;
1998
+ /**
1999
+ * Spawns a process, merges stdout and stderr the way the scaffold reads them,
2000
+ * and kills the whole process group on timeout so a killed command leaves no
2001
+ * children running in the container's namespace.
2002
+ */
2003
+ declare const nodeProcessRunner: ProcessRunner;
2004
+ interface DockerContinuationEnvironmentOptions {
2005
+ /** Container id or name holding the post-step-k state. */
2006
+ containerRef: string;
2007
+ /** Working directory for every command. */
2008
+ cwd: string;
2009
+ /** Environment variables set on every command. */
2010
+ env?: Record<string, string>;
2011
+ /** Interpreter the scaffold runs commands through. */
2012
+ interpreter?: string[];
2013
+ /** `docker` unless a compatible client is used. */
2014
+ executable?: string;
2015
+ runProcess: ProcessRunner;
2016
+ /** Removes the container on dispose. Leave false when the caller owns its lifecycle. */
2017
+ removeOnDispose?: boolean;
2018
+ }
2019
+ /**
2020
+ * Arguments that create a container the policy accepts. `--network none` is
2021
+ * not optional: a continuation with network access could install what the
2022
+ * recorded run could not, and the arms would no longer differ only by the
2023
+ * intervention.
2024
+ */
2025
+ declare function dockerRunArgs(input: {
2026
+ image: string;
2027
+ name: string;
2028
+ cwd: string;
2029
+ containerLifetime?: string;
2030
+ executable?: string;
2031
+ }): string[];
2032
+ declare function createDockerContinuationEnvironment(options: DockerContinuationEnvironmentOptions): ContinuationEnvironment;
2033
+ //#endregion
2034
+ //#region src/trace-repair/repair-prompt.d.ts
2035
+ declare const REPAIR_QUESTION = "This coding agent ran and did not finish the task. Name the ONE recorded step whose action you would replace, and give the single shell action to run instead of it.";
2036
+ /**
2037
+ * The task definition. Parameterised by the budget so the stated caps are the
2038
+ * enforced caps.
2039
+ */
2040
+ declare function repairTaskPolicy(budget?: InterventionBudget): readonly string[];
2041
+ /** The reply grammar, stated to an arm that answers in one JSON object. */
2042
+ declare const REPAIR_CONTRACT_LINES: readonly string[];
2043
+ /** The same grammar restated for the bounded repair turn, which never carries
2044
+ * the trajectory again. */
2045
+ declare const REPAIR_REPAIR_CONTRACT_LINES: readonly string[];
2046
+ /** The trajectory as an arm sees it: actions, observations, nothing about grading. */
2047
+ declare function renderRepairTrajectory(prefix: BlindedTrajectoryPrefix): string;
2048
+ declare function repairTrajectoryHeader(prefix: BlindedTrajectoryPrefix): string;
2049
+ /** The task definition an arm receives: the policy plus the statement the
2050
+ * recorded agent was given. */
2051
+ declare function repairTaskDefinition(prefix: BlindedTrajectoryPrefix, budget?: InterventionBudget): string;
2052
+ /**
2053
+ * Digest of the question and task policy every arm shares.
2054
+ *
2055
+ * This is the part of the prompt that is equal by construction across arms:
2056
+ * the question, the execution rules, and the budget the caps are read from.
2057
+ * An arm's own contract text — its output grammar, its typed-signature
2058
+ * instructions — is deliberately outside it, because arms differ there.
2059
+ */
2060
+ declare function repairQuestionSha256(budget?: InterventionBudget): string;
2061
+ /**
2062
+ * Digest of the composed question one arm answered.
2063
+ *
2064
+ * Covers the shared question and the arm's own declared contract text, so two
2065
+ * arms that ask materially different composed questions — a JSON grammar
2066
+ * versus a typed SUBMIT signature — stamp different digests, and two arms
2067
+ * that ask the identical composed question share one.
2068
+ */
2069
+ declare function repairArmPromptSha256(budget: InterventionBudget, contract: readonly string[]): string;
2070
+ //#endregion
2071
+ //#region src/trace-repair/test-oracle.d.ts
2072
+ /** The suite the oracle read back is not the suite it uploaded. */
2073
+ declare class TestSuiteTamperedError extends CaptureIntegrityError {}
2074
+ /** The oracle could not place or run the suite, so it graded nothing. */
2075
+ declare class TestOracleError extends CaptureIntegrityError {}
2076
+ interface TestSuiteFile {
2077
+ /** Absolute path inside the container. */
2078
+ readonly path: string;
2079
+ readonly contents: string;
2080
+ /** Octal mode applied after upload, e.g. '0755' for a runner script. */
2081
+ readonly mode?: string;
2082
+ }
2083
+ interface InjectedTestOracleOptions {
2084
+ /** Every file of the held-out suite. Uploaded on every grade call. */
2085
+ readonly files: readonly TestSuiteFile[];
2086
+ /** Command that runs the suite. Exit 0 is the only pass. */
2087
+ readonly command: string;
2088
+ /** Absolute directories removed before upload. A planted file that the
2089
+ * upload does not overwrite dies here. */
2090
+ readonly purge?: readonly string[];
2091
+ readonly uploadTimeoutMs?: number;
2092
+ readonly commandTimeoutMs?: number;
2093
+ }
2094
+ /**
2095
+ * Content digest of the suite: sha256 over each path and its bytes, in path
2096
+ * order. Two suites with the same digest are the same suite.
2097
+ */
2098
+ declare function testSuiteDigest(files: readonly TestSuiteFile[]): string;
2099
+ declare function injectedTestOracle(options: InjectedTestOracleOptions): TestOracle;
2100
+ //#endregion
2101
+ export { ADMISSION_CONFIG_DEFAULTS, ADMISSION_EXCLUSION_MEANING, ADMISSION_EXCLUSION_ORDER, ADMISSION_ROW_KEYS, ADMISSION_STRATA, type ActionPayloadKind, type AdmissionArtifact, type AdmissionCheckRecord, type AdmissionConfig, type AdmissionConfigInput, type AdmissionControlArm, type AdmissionControlObservation, type AdmissionControlRequest, type AdmissionControlRunner, type AdmissionCriteria, type AdmissionDecision, AdmissionDenominatorError, type AdmissionDivergence, type AdmissionEndStateOracle, type AdmissionEvidence, type AdmissionExclusionReason, AdmissionIndependenceError, type AdmissionNoOpInjection, type AdmissionOutcome, type AdmissionPrefixReplay, type AdmissionPrefixReplayer, type AdmissionProvenance, type AdmissionRejection, type AdmissionReport, type AdmissionRolloutSummary, type AdmissionRow, type AdmissionRowVerdict, type AdmissionScreeningRecord, type AdmissionStratum, type AdmissionTestVerdict, type AdmittedRow, type AnalystResponse, type AskRepairArmOptions, BLINDED_FIELDS, type BlindTrajectoryOptions, type BlindedStep, type BlindedTrajectoryPrefix, type BudgetCheck, type BudgetMeasurement, type BudgetViolation, CONTINUATION_POLICY_DEFAULTS, CONTROL_SCREENING_MODES, CREDIT_TERMS, type CommandOutput, type CompletionRepairArmOptions, type ContinuationArm, type ContinuationEnvironment, type ContinuationEnvironmentDescription, type ContinuationEnvironmentFactory, type ContinuationEnvironmentRequest, type ContinuationExecRecord, type ContinuationExecResult, type ContinuationExitStatus, type ContinuationMessage, type ContinuationModel, type ContinuationModelCall, type ContinuationModelRequest, type ContinuationModelResponse, ContinuationPolicyViolationError, type ContinuationRollout, type ContinuationStepRecord, ContinuationSymmetryError, type ContinuationUsageTotals, type ControlArmEvidence, type ControlCapability, type ControlPolicy, type ControlPolicyInput, type ControlScreening, DEGENERATE_STRATEGIES, DSPY_REPAIR_SIGNATURE, DSPY_REPAIR_TASK_TOKEN, DSPY_REPAIR_TRAJECTORY_INPUT, type DefineContinuationPolicyInput, type DegenerateDefeatKind, type DegenerateStrategy, type DegenerateStrategyId, type DeltaRepairInterval, type DeltaRepairOptions, type DeltaRepairReport, type DenominatorChain, type DenominatorChainArtifact, type DenominatorIntactInput, type DenominatorStage, type DockerContinuationEnvironmentOptions, type DspyRepairArmOptions, type GradeRepairOptions, type InjectedAction, type InjectedTestOracleOptions, type InterventionBudget, type InterventionExecution, MINI_SWE_SYSTEM_MESSAGE, MIN_ORACLE_REPLICATES, NO_DECISIVE_FAILURE, NO_OP_ACTIONS, NondeterministicOracleError, OUTPUT_ELISION_THRESHOLD, OUTPUT_ELISION_WINDOW, type OracleAssertionResult, type OracleDeterminismEvidence, type OracleDeterminismVerdict, type OracleFlippedUnit, type OracleLoad, type OracleReplicate, type OracleReplicateGroup, type OracleStateLabel, type OracleStateVerdict, type ParseAnalystResponseOutcome, type ParsedAction, type PinnedContinuationPolicy, type PrefixFidelityEvidence, type PrefixReplayEvidence, type ProcessRequest, type ProcessResult, type ProcessRunner, REPAIR_CONTRACT_LINES, REPAIR_QUESTION, REPAIR_REPAIR_CONTRACT_LINES, type ReadRepairPayloadOptions, type RecordedStep, type RecordedToolCall, type RenderAdmissionOptions, type RepairArm as RepairAnalystArm, type RepairArm$1 as RepairArm, type RepairArmAffordance, type RepairArmAnswer, type RepairArmAsymmetry, type RepairArmAsymmetryReport, type RepairArmBudgetRecord, type RepairArmCertification, type RepairArmDeclaration, type RepairArmEvidence, type RepairArmRejectedRow, type RepairArmRepairTurn, type RepairArmReply, type RepairArmRequest, RepairArmSymmetryError, type RepairArmUsage, type RepairContinuationOutcome, type RepairContinuationRequest, type RepairContinuationRunner, type RepairCredit, type RepairDeclined, type RepairExecResult, type RepairFinding, type RepairFunnelCounts, type RepairGrade, type RepairIntervention, type RepairRejection, type RepairRolloutEvidence, type RepairRowResult, type RepairSession, type RepairSessionFactory, type RepairSessionRequest, type RepairThreat, type RepairThreatId, type ReproductionEvidence, type ResponseParseFailure, type RunAdmissionOptions, type RunContinuationOptions, SCAFFOLD_INTERVENTION_BUDGET, SUBMIT_SENTINEL, SUITE_REWARD_UNIT, TB_REPAIR_ADMISSION_CRITERIA, type TaskOracleRegistry, type TaskOracleRegistryDocument, type TestOracle, type TestOracleContext, TestOracleError, type TestOracleOutcome, type TestRunEvidence, type TestSuiteFile, TestSuiteTamperedError, UncalibratedControlError, admissionArtifact, admitRow, admittedCount, admittedRowIds, askRepairArm, assertAnalystIndependent, assertArmSymmetry, assertChainReconciles, assertControlCalibrated, assertDenominatorIntact, assertDeterministicOracle, assertDspyRepairEngine, blindTrajectory, buildDenominatorChain, checkInterventionBudget, classifyActionPayload, continuationPolicyDigest, continuationSeed, controlCanRescue, countFunnel, createCompletionRepairArm, createDockerContinuationEnvironment, createDspyRepairArm, defineControlPolicy, definePinnedContinuationPolicy, degenerateStrategy, deltaRepair, dockerRunArgs, dspyRepairInstructions, gradeRepairRow, injectedTestOracle, isPreStratumReason, isRecordedTimeout, noOpInjectionStep, nodeProcessRunner, normalizeActionForComparison, oracleDeterminism, parseAction, parseAnalystResponse, parseTaskOracleRegistry, reachedT0, reachedT1, reachedT2, readRepairPayload, renderAdmissionReport, renderDeltaRepairReport, renderFormatErrorObservation, renderInstanceMessage, renderObservation, renderRepairTrajectory, renderTimeoutObservation, repairArmAsymmetries, repairArmPromptSha256, repairArmResponse, repairCredit, repairFinding, repairQuestionSha256, repairTaskDefinition, repairTaskPolicy, repairTrajectoryHeader, resolveAdmissionConfig, rolloutDigest, rolloutRecordedSteps, runAdmission, runContinuation, scanShellAction, stratumOf, submissionOf, taskOracleRegistry, testSuiteDigest, toRecordedSteps, totalCost, totalUsage };
2102
+ //# sourceMappingURL=index.d.ts.map