@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,3878 @@
|
|
|
1
|
+
import { c as ValidationError, n as CaptureIntegrityError } from "../errors-D-LKuDhb.js";
|
|
2
|
+
import { r as contentHash } from "../verdict-cache-BCcOh0kF.js";
|
|
3
|
+
import { E as pairedBootstrap } from "../statistics-ByxzSiOM.js";
|
|
4
|
+
import { d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage } from "../prime-protocol-BfSalTfR.js";
|
|
5
|
+
import { o as parseRecordedReturncode, r as deriveFailureSignature, t as wrapActionForExec } from "../exec-BLtYZdWo.js";
|
|
6
|
+
import { createHash } from "node:crypto";
|
|
7
|
+
import { spawn } from "node:child_process";
|
|
8
|
+
import { constants } from "node:os";
|
|
9
|
+
//#region src/trace-repair/mini-swe-scaffold.ts
|
|
10
|
+
/**
|
|
11
|
+
* The mini-swe-agent scaffold as the Terminal-Bench-2 trajectory corpus
|
|
12
|
+
* recorded it: one bash block per turn, one observation per command, and a
|
|
13
|
+
* sentinel command that ends the run.
|
|
14
|
+
*
|
|
15
|
+
* Every template here is byte-verified against
|
|
16
|
+
* `yoonholee/terminalbench-trajectories` (agent = `mini-swe-agent`, 6663 rows,
|
|
17
|
+
* one distinct system prompt across all of them). A continuation that renders
|
|
18
|
+
* different bytes puts the model in a different distribution than the prefix
|
|
19
|
+
* it inherits, so these strings are pinned, not configurable.
|
|
20
|
+
*/
|
|
21
|
+
/** Whole-line marker that ends a run. The first output line must equal it and the command must exit 0. */
|
|
22
|
+
const SUBMIT_SENTINEL = "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT";
|
|
23
|
+
/** Outputs at or above this length are elided head+tail instead of shown whole. */
|
|
24
|
+
const OUTPUT_ELISION_THRESHOLD = 1e4;
|
|
25
|
+
/** Characters kept from each end of an elided output. */
|
|
26
|
+
const OUTPUT_ELISION_WINDOW = 5e3;
|
|
27
|
+
const MINI_SWE_SYSTEM_MESSAGE = `You are a helpful assistant that can interact with a computer.
|
|
28
|
+
|
|
29
|
+
Your response must contain exactly ONE bash code block with ONE command (or commands connected with && or ||).
|
|
30
|
+
Include a THOUGHT section before your command where you explain your reasoning process.
|
|
31
|
+
Format your response as shown in <format_example>.
|
|
32
|
+
|
|
33
|
+
<format_example>
|
|
34
|
+
Your reasoning and analysis here. Explain why you want to perform the action.
|
|
35
|
+
|
|
36
|
+
\`\`\`bash
|
|
37
|
+
your_command_here
|
|
38
|
+
\`\`\`
|
|
39
|
+
</format_example>
|
|
40
|
+
|
|
41
|
+
Failure to follow these rules will cause your response to be rejected.
|
|
42
|
+
`;
|
|
43
|
+
/** The second message of every recorded run: task plus workflow rules. */
|
|
44
|
+
function renderInstanceMessage(input) {
|
|
45
|
+
return `Please solve this issue: ${input.task}
|
|
46
|
+
|
|
47
|
+
You can execute bash commands and edit files to implement the necessary changes.
|
|
48
|
+
|
|
49
|
+
## Recommended Workflow
|
|
50
|
+
|
|
51
|
+
This workflows should be done step-by-step so that you can iterate on your changes and any possible problems.
|
|
52
|
+
|
|
53
|
+
1. Analyze the codebase by finding and reading relevant files
|
|
54
|
+
2. Create a script to reproduce the issue
|
|
55
|
+
3. Edit the source code to resolve the issue
|
|
56
|
+
4. Verify your fix works by running your script again
|
|
57
|
+
5. Test edge cases to ensure your fix is robust
|
|
58
|
+
6. Submit your changes and finish your work by issuing the following command: \`echo ${SUBMIT_SENTINEL}\`.
|
|
59
|
+
Do not combine it with any other command. <important>After this command, you cannot continue working on this task.</important>
|
|
60
|
+
|
|
61
|
+
## Important Rules
|
|
62
|
+
|
|
63
|
+
1. Every response must contain exactly one action
|
|
64
|
+
2. The action must be enclosed in triple backticks
|
|
65
|
+
3. Directory or environment variable changes are not persistent. Every action is executed in a new subshell.
|
|
66
|
+
However, you can prefix any action with \`MY_ENV_VAR=MY_VALUE cd /path/to/working/dir && ...\` or write/load environment variables from files
|
|
67
|
+
|
|
68
|
+
<system_information>
|
|
69
|
+
${input.systemInformation}
|
|
70
|
+
</system_information>
|
|
71
|
+
|
|
72
|
+
## Formatting your response
|
|
73
|
+
|
|
74
|
+
Here is an example of a correct response:
|
|
75
|
+
|
|
76
|
+
<example_response>
|
|
77
|
+
THOUGHT: I need to understand the structure of the repository first. Let me check what files are in the current directory to get a better understanding of the codebase.
|
|
78
|
+
|
|
79
|
+
\`\`\`bash
|
|
80
|
+
ls -la
|
|
81
|
+
\`\`\`
|
|
82
|
+
</example_response>
|
|
83
|
+
|
|
84
|
+
## Useful command examples
|
|
85
|
+
|
|
86
|
+
### Create a new file:
|
|
87
|
+
|
|
88
|
+
\`\`\`bash
|
|
89
|
+
cat <<'EOF' > newfile.py
|
|
90
|
+
import numpy as np
|
|
91
|
+
hello = "world"
|
|
92
|
+
print(hello)
|
|
93
|
+
EOF
|
|
94
|
+
\`\`\`
|
|
95
|
+
|
|
96
|
+
### Edit files with sed:\`\`\`bash
|
|
97
|
+
# Replace all occurrences
|
|
98
|
+
sed -i 's/old_string/new_string/g' filename.py
|
|
99
|
+
|
|
100
|
+
# Replace only first occurrence
|
|
101
|
+
sed -i 's/old_string/new_string/' filename.py
|
|
102
|
+
|
|
103
|
+
# Replace first occurrence on line 1
|
|
104
|
+
sed -i '1s/old_string/new_string/' filename.py
|
|
105
|
+
|
|
106
|
+
# Replace all occurrences in lines 1-10
|
|
107
|
+
sed -i '1,10s/old_string/new_string/g' filename.py
|
|
108
|
+
\`\`\`
|
|
109
|
+
|
|
110
|
+
### View file content:
|
|
111
|
+
|
|
112
|
+
\`\`\`bash
|
|
113
|
+
# View specific lines with numbers
|
|
114
|
+
nl -ba filename.py | sed -n '10,20p'
|
|
115
|
+
\`\`\`
|
|
116
|
+
|
|
117
|
+
### Any other command you want to run
|
|
118
|
+
|
|
119
|
+
\`\`\`bash
|
|
120
|
+
anything
|
|
121
|
+
\`\`\`
|
|
122
|
+
`;
|
|
123
|
+
}
|
|
124
|
+
const BASH_BLOCK = /```bash\n(.*?)\n```/gs;
|
|
125
|
+
/**
|
|
126
|
+
* Exactly one fenced bash block is an action; zero or many is a format error.
|
|
127
|
+
* The scaffold trims the command, so a block padded with blank lines executes
|
|
128
|
+
* the same command as an unpadded one.
|
|
129
|
+
*/
|
|
130
|
+
function parseAction(assistantMessage) {
|
|
131
|
+
const blocks = [...assistantMessage.matchAll(BASH_BLOCK)].map((match) => match[1] ?? "");
|
|
132
|
+
if (blocks.length !== 1) return {
|
|
133
|
+
kind: "format-error",
|
|
134
|
+
actionCount: blocks.length
|
|
135
|
+
};
|
|
136
|
+
return {
|
|
137
|
+
kind: "action",
|
|
138
|
+
command: (blocks[0] ?? "").trim()
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* The observation the agent reads after a command. Short outputs are shown
|
|
143
|
+
* whole; long ones keep the first and last `OUTPUT_ELISION_WINDOW` characters
|
|
144
|
+
* with the dropped count between them.
|
|
145
|
+
*/
|
|
146
|
+
function renderObservation(output) {
|
|
147
|
+
const head = output.exceptionInfo ? `<exception>${output.exceptionInfo}</exception>\n` : "";
|
|
148
|
+
const returncode = `<returncode>${output.returncode}</returncode>\n`;
|
|
149
|
+
if (output.output.length < 1e4) return `${head}${returncode}<output>\n${output.output}</output>`;
|
|
150
|
+
const elided = output.output.length - OUTPUT_ELISION_THRESHOLD;
|
|
151
|
+
return `${head}${returncode}<warning>\nThe output of your last command was too long.
|
|
152
|
+
Please try a different command that produces less output.
|
|
153
|
+
If you're looking at a file you can try use head, tail or sed to view a smaller number of lines selectively.
|
|
154
|
+
If you're using grep or find and it produced too much output, you can use a more selective search pattern.
|
|
155
|
+
If you really need to see something from the full command's output, you can redirect output to a file and then search in that file.
|
|
156
|
+
</warning><output_head>\n${output.output.slice(0, OUTPUT_ELISION_WINDOW)}\n</output_head>\n<elided_chars>\n${elided} characters elided\n</elided_chars>\n<output_tail>\n${output.output.slice(-5e3)}\n</output_tail>`;
|
|
157
|
+
}
|
|
158
|
+
/** The observation after the environment killed a command for exceeding its timeout. */
|
|
159
|
+
function renderTimeoutObservation(command, partialOutput) {
|
|
160
|
+
return `The last command <command>${command}</command> timed out and has been killed.\nThe output of the command was:\n <output>\n${partialOutput}\n</output>\nPlease try another command and make sure to avoid those requiring interactive input.`;
|
|
161
|
+
}
|
|
162
|
+
/** Substring `renderTimeoutObservation` always writes, whatever the command was. */
|
|
163
|
+
const TIMEOUT_OBSERVATION_MARKER = "timed out and has been killed";
|
|
164
|
+
/**
|
|
165
|
+
* True when the recording shows the environment killed this step at its
|
|
166
|
+
* wall-clock bound.
|
|
167
|
+
*
|
|
168
|
+
* Such a step carries no returncode, so no replay can confirm or contradict
|
|
169
|
+
* it. Callers use this to bound the replay of that step cheaply rather than to
|
|
170
|
+
* decide agreement.
|
|
171
|
+
*/
|
|
172
|
+
function isRecordedTimeout(observation) {
|
|
173
|
+
return observation?.includes(TIMEOUT_OBSERVATION_MARKER) === true;
|
|
174
|
+
}
|
|
175
|
+
/** The observation after a turn that did not contain exactly one bash block. */
|
|
176
|
+
function renderFormatErrorObservation(actionCount) {
|
|
177
|
+
return `Please always provide EXACTLY ONE action in triple backticks, found ${actionCount} actions.\nIf you want to end the task, please issue the following command: \`echo ${SUBMIT_SENTINEL}\`\nwithout any other command.
|
|
178
|
+
Else, please format your response exactly as follows:
|
|
179
|
+
|
|
180
|
+
<response_example>
|
|
181
|
+
Here are some thoughts about why you want to perform the action.
|
|
182
|
+
|
|
183
|
+
\`\`\`bash
|
|
184
|
+
<action>
|
|
185
|
+
\`\`\`
|
|
186
|
+
</response_example>
|
|
187
|
+
|
|
188
|
+
Note: In rare cases, if you need to reference a similar format in your command, you might have
|
|
189
|
+
to proceed in two steps, first writing TRIPLEBACKTICKSBASH, then replacing them with \`\`\`bash.`;
|
|
190
|
+
}
|
|
191
|
+
/**
|
|
192
|
+
* The submission text when this output ends the run, `null` otherwise.
|
|
193
|
+
* A non-zero exit does not submit even when the sentinel is echoed, so an
|
|
194
|
+
* agent cannot end the run through a command that failed.
|
|
195
|
+
*/
|
|
196
|
+
function submissionOf(output) {
|
|
197
|
+
if (output.returncode !== 0) return null;
|
|
198
|
+
const lines = output.output.replace(/^\s+/, "").split(/(?<=\n)/);
|
|
199
|
+
const first = lines[0];
|
|
200
|
+
if (first === void 0 || first.trim() !== "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT") return null;
|
|
201
|
+
return lines.slice(1).join("");
|
|
202
|
+
}
|
|
203
|
+
//#endregion
|
|
204
|
+
//#region src/trace-repair/action-budget.ts
|
|
205
|
+
/**
|
|
206
|
+
* The action budget an intervention must fit inside.
|
|
207
|
+
*
|
|
208
|
+
* The analyst answers with one action, applied at step k, drawn from the same
|
|
209
|
+
* action space the scaffold had: one shell command or one edit, at most 4 KB.
|
|
210
|
+
* An answer that buys a bigger action than the scaffold could take is not a
|
|
211
|
+
* counterfactual about the recorded run, so the budget is enforced before any
|
|
212
|
+
* container is opened and a violation never reaches the reproduction gate.
|
|
213
|
+
*
|
|
214
|
+
* "One action" is decided by counting top-level statements, not lines. A
|
|
215
|
+
* command list joined by `&&`, `||` or a pipe is one statement, because that
|
|
216
|
+
* is one thing the shell runs and one thing the scaffold could have typed.
|
|
217
|
+
* Two statements separated by a newline or `;` are two actions and are
|
|
218
|
+
* rejected. Heredoc bodies, comments and compound blocks (`if`, `for`,
|
|
219
|
+
* `while`, `until`, `case`, `{ … }`) are inside a statement, never separators.
|
|
220
|
+
*/
|
|
221
|
+
/** The scaffold's own per-action budget, pre-registered for the campaign. */
|
|
222
|
+
const SCAFFOLD_INTERVENTION_BUDGET = Object.freeze({
|
|
223
|
+
maxBytes: 4096,
|
|
224
|
+
maxStatements: 1,
|
|
225
|
+
maxHeredocs: 1
|
|
226
|
+
});
|
|
227
|
+
/**
|
|
228
|
+
* Actions whose only effect is to consume a turn. They are rejected before a
|
|
229
|
+
* container opens: an intervention that changes nothing is measurably
|
|
230
|
+
* identical to the no-op control, and paying rollouts to rediscover that
|
|
231
|
+
* wastes the corpus.
|
|
232
|
+
*/
|
|
233
|
+
const NO_OP_ACTIONS = Object.freeze([
|
|
234
|
+
":",
|
|
235
|
+
"true",
|
|
236
|
+
"/bin/true",
|
|
237
|
+
"exit",
|
|
238
|
+
"exit 0"
|
|
239
|
+
]);
|
|
240
|
+
const BLOCK_OPENERS = /* @__PURE__ */ new Set([
|
|
241
|
+
"if",
|
|
242
|
+
"for",
|
|
243
|
+
"while",
|
|
244
|
+
"until",
|
|
245
|
+
"case",
|
|
246
|
+
"select",
|
|
247
|
+
"do",
|
|
248
|
+
"then"
|
|
249
|
+
]);
|
|
250
|
+
const BLOCK_CLOSERS = /* @__PURE__ */ new Set([
|
|
251
|
+
"fi",
|
|
252
|
+
"done",
|
|
253
|
+
"esac"
|
|
254
|
+
]);
|
|
255
|
+
/**
|
|
256
|
+
* Split a shell script into top-level statements and count its heredocs.
|
|
257
|
+
*
|
|
258
|
+
* The scan tracks quoting, escapes, command substitution, brace and paren
|
|
259
|
+
* grouping, comments, compound-block keywords, and heredoc bodies. Anything
|
|
260
|
+
* it cannot resolve stays inside the current statement, so an unparseable
|
|
261
|
+
* action reads as one oversized statement and is rejected on bytes rather
|
|
262
|
+
* than silently accepted as one clean action.
|
|
263
|
+
*/
|
|
264
|
+
function scanShellAction(script) {
|
|
265
|
+
const state = {
|
|
266
|
+
statements: [],
|
|
267
|
+
current: "",
|
|
268
|
+
heredocs: 0
|
|
269
|
+
};
|
|
270
|
+
let index = 0;
|
|
271
|
+
let parenDepth = 0;
|
|
272
|
+
let braceDepth = 0;
|
|
273
|
+
let blockDepth = 0;
|
|
274
|
+
let inSingle = false;
|
|
275
|
+
let inDouble = false;
|
|
276
|
+
let inBacktick = false;
|
|
277
|
+
let pendingHeredocs = [];
|
|
278
|
+
let word = "";
|
|
279
|
+
let atStatementStart = true;
|
|
280
|
+
const flushWord = () => {
|
|
281
|
+
if (word.length === 0) return;
|
|
282
|
+
if (BLOCK_OPENERS.has(word)) blockDepth += 1;
|
|
283
|
+
else if (BLOCK_CLOSERS.has(word)) blockDepth = Math.max(0, blockDepth - 1);
|
|
284
|
+
word = "";
|
|
285
|
+
};
|
|
286
|
+
const endStatement = () => {
|
|
287
|
+
flushWord();
|
|
288
|
+
const text = state.current.trim();
|
|
289
|
+
if (text.length > 0 && !isCommentOnly(text)) state.statements.push(text);
|
|
290
|
+
state.current = "";
|
|
291
|
+
atStatementStart = true;
|
|
292
|
+
};
|
|
293
|
+
while (index < script.length) {
|
|
294
|
+
const char = script[index];
|
|
295
|
+
if (inSingle) {
|
|
296
|
+
state.current += char;
|
|
297
|
+
if (char === "'") inSingle = false;
|
|
298
|
+
index += 1;
|
|
299
|
+
continue;
|
|
300
|
+
}
|
|
301
|
+
if (char === "\\" && index + 1 < script.length) {
|
|
302
|
+
const next = script[index + 1];
|
|
303
|
+
if (next === "\n" && !inDouble) {
|
|
304
|
+
state.current += char + next;
|
|
305
|
+
index += 2;
|
|
306
|
+
continue;
|
|
307
|
+
}
|
|
308
|
+
state.current += char + next;
|
|
309
|
+
index += 2;
|
|
310
|
+
continue;
|
|
311
|
+
}
|
|
312
|
+
if (inDouble) {
|
|
313
|
+
state.current += char;
|
|
314
|
+
if (char === "\"") inDouble = false;
|
|
315
|
+
index += 1;
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
318
|
+
if (char === "'") {
|
|
319
|
+
flushWord();
|
|
320
|
+
inSingle = true;
|
|
321
|
+
state.current += char;
|
|
322
|
+
index += 1;
|
|
323
|
+
atStatementStart = false;
|
|
324
|
+
continue;
|
|
325
|
+
}
|
|
326
|
+
if (char === "\"") {
|
|
327
|
+
flushWord();
|
|
328
|
+
inDouble = true;
|
|
329
|
+
state.current += char;
|
|
330
|
+
index += 1;
|
|
331
|
+
atStatementStart = false;
|
|
332
|
+
continue;
|
|
333
|
+
}
|
|
334
|
+
if (char === "`") {
|
|
335
|
+
inBacktick = !inBacktick;
|
|
336
|
+
state.current += char;
|
|
337
|
+
index += 1;
|
|
338
|
+
atStatementStart = false;
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
341
|
+
if (char === "#" && (atStatementStart || /\s/.test(script[index - 1] ?? " "))) {
|
|
342
|
+
const end = script.indexOf("\n", index);
|
|
343
|
+
const stop = end === -1 ? script.length : end;
|
|
344
|
+
state.current += script.slice(index, stop);
|
|
345
|
+
index = stop;
|
|
346
|
+
continue;
|
|
347
|
+
}
|
|
348
|
+
if (char === "<" && script[index + 1] === "<" && script[index + 2] !== "<" && script[index - 1] !== "<") {
|
|
349
|
+
const mark = readHeredocMark(script, index);
|
|
350
|
+
if (mark) {
|
|
351
|
+
pendingHeredocs.push(mark.mark);
|
|
352
|
+
state.heredocs += 1;
|
|
353
|
+
state.current += script.slice(index, mark.nextIndex);
|
|
354
|
+
index = mark.nextIndex;
|
|
355
|
+
atStatementStart = false;
|
|
356
|
+
continue;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
if (char === "(") {
|
|
360
|
+
flushWord();
|
|
361
|
+
parenDepth += 1;
|
|
362
|
+
state.current += char;
|
|
363
|
+
index += 1;
|
|
364
|
+
atStatementStart = true;
|
|
365
|
+
continue;
|
|
366
|
+
}
|
|
367
|
+
if (char === ")") {
|
|
368
|
+
flushWord();
|
|
369
|
+
parenDepth = Math.max(0, parenDepth - 1);
|
|
370
|
+
state.current += char;
|
|
371
|
+
index += 1;
|
|
372
|
+
atStatementStart = false;
|
|
373
|
+
continue;
|
|
374
|
+
}
|
|
375
|
+
if (char === "{") {
|
|
376
|
+
flushWord();
|
|
377
|
+
braceDepth += 1;
|
|
378
|
+
state.current += char;
|
|
379
|
+
index += 1;
|
|
380
|
+
atStatementStart = true;
|
|
381
|
+
continue;
|
|
382
|
+
}
|
|
383
|
+
if (char === "}") {
|
|
384
|
+
flushWord();
|
|
385
|
+
braceDepth = Math.max(0, braceDepth - 1);
|
|
386
|
+
state.current += char;
|
|
387
|
+
index += 1;
|
|
388
|
+
atStatementStart = false;
|
|
389
|
+
continue;
|
|
390
|
+
}
|
|
391
|
+
if (char === "\n") {
|
|
392
|
+
flushWord();
|
|
393
|
+
if (pendingHeredocs.length > 0) {
|
|
394
|
+
const consumed = consumeHeredocBodies(script, index + 1, pendingHeredocs);
|
|
395
|
+
state.current += script.slice(index, consumed);
|
|
396
|
+
pendingHeredocs = [];
|
|
397
|
+
index = consumed;
|
|
398
|
+
continue;
|
|
399
|
+
}
|
|
400
|
+
if (parenDepth > 0 || braceDepth > 0 || blockDepth > 0 || inBacktick || endsWithContinuation(state.current)) {
|
|
401
|
+
state.current += char;
|
|
402
|
+
index += 1;
|
|
403
|
+
atStatementStart = true;
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
endStatement();
|
|
407
|
+
index += 1;
|
|
408
|
+
continue;
|
|
409
|
+
}
|
|
410
|
+
if (char === ";") {
|
|
411
|
+
flushWord();
|
|
412
|
+
if (parenDepth > 0 || braceDepth > 0 || blockDepth > 0 || inBacktick) {
|
|
413
|
+
state.current += char;
|
|
414
|
+
index += 1;
|
|
415
|
+
atStatementStart = true;
|
|
416
|
+
continue;
|
|
417
|
+
}
|
|
418
|
+
endStatement();
|
|
419
|
+
index += 1;
|
|
420
|
+
continue;
|
|
421
|
+
}
|
|
422
|
+
if (/\s/.test(char)) {
|
|
423
|
+
flushWord();
|
|
424
|
+
state.current += char;
|
|
425
|
+
index += 1;
|
|
426
|
+
continue;
|
|
427
|
+
}
|
|
428
|
+
if (char === "&" || char === "|") {
|
|
429
|
+
flushWord();
|
|
430
|
+
state.current += char;
|
|
431
|
+
index += 1;
|
|
432
|
+
atStatementStart = true;
|
|
433
|
+
continue;
|
|
434
|
+
}
|
|
435
|
+
word += char;
|
|
436
|
+
state.current += char;
|
|
437
|
+
atStatementStart = false;
|
|
438
|
+
index += 1;
|
|
439
|
+
}
|
|
440
|
+
endStatement();
|
|
441
|
+
return {
|
|
442
|
+
statements: state.statements,
|
|
443
|
+
heredocs: state.heredocs
|
|
444
|
+
};
|
|
445
|
+
}
|
|
446
|
+
function isCommentOnly(text) {
|
|
447
|
+
return text.split("\n").every((line) => line.trim().length === 0 || line.trim().startsWith("#"));
|
|
448
|
+
}
|
|
449
|
+
function endsWithContinuation(current) {
|
|
450
|
+
const trimmed = current.trimEnd();
|
|
451
|
+
return /(&&|\|\||\||&|\\)$/.test(trimmed);
|
|
452
|
+
}
|
|
453
|
+
function readHeredocMark(script, index) {
|
|
454
|
+
let cursor = index + 2;
|
|
455
|
+
let stripTabs = false;
|
|
456
|
+
if (script[cursor] === "-") {
|
|
457
|
+
stripTabs = true;
|
|
458
|
+
cursor += 1;
|
|
459
|
+
}
|
|
460
|
+
while (script[cursor] === " " || script[cursor] === " ") cursor += 1;
|
|
461
|
+
const quote = script[cursor];
|
|
462
|
+
if (quote === "'" || quote === "\"") {
|
|
463
|
+
const close = script.indexOf(quote, cursor + 1);
|
|
464
|
+
if (close === -1) return null;
|
|
465
|
+
return {
|
|
466
|
+
mark: {
|
|
467
|
+
delimiter: script.slice(cursor + 1, close),
|
|
468
|
+
stripTabs,
|
|
469
|
+
quoted: true
|
|
470
|
+
},
|
|
471
|
+
nextIndex: close + 1
|
|
472
|
+
};
|
|
473
|
+
}
|
|
474
|
+
const match = /^[A-Za-z_][A-Za-z0-9_]*/.exec(script.slice(cursor));
|
|
475
|
+
if (!match) return null;
|
|
476
|
+
return {
|
|
477
|
+
mark: {
|
|
478
|
+
delimiter: match[0],
|
|
479
|
+
stripTabs,
|
|
480
|
+
quoted: false
|
|
481
|
+
},
|
|
482
|
+
nextIndex: cursor + match[0].length
|
|
483
|
+
};
|
|
484
|
+
}
|
|
485
|
+
/** Consume every pending heredoc body; returns the offset just past the last
|
|
486
|
+
* terminator, or the end of the script when a terminator never arrives. */
|
|
487
|
+
function consumeHeredocBodies(script, from, marks) {
|
|
488
|
+
let cursor = from;
|
|
489
|
+
for (const mark of marks) {
|
|
490
|
+
let closed = false;
|
|
491
|
+
while (cursor < script.length) {
|
|
492
|
+
const lineEnd = script.indexOf("\n", cursor);
|
|
493
|
+
const stop = lineEnd === -1 ? script.length : lineEnd;
|
|
494
|
+
const line = script.slice(cursor, stop);
|
|
495
|
+
const candidate = mark.stripTabs ? line.replace(/^\t+/, "") : line;
|
|
496
|
+
cursor = lineEnd === -1 ? script.length : lineEnd + 1;
|
|
497
|
+
if (candidate === mark.delimiter) {
|
|
498
|
+
closed = true;
|
|
499
|
+
break;
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
if (!closed) return script.length;
|
|
503
|
+
}
|
|
504
|
+
return cursor;
|
|
505
|
+
}
|
|
506
|
+
/** Payload shape of an action: `edit` when it authors file content through a
|
|
507
|
+
* heredoc, `shell` otherwise. */
|
|
508
|
+
function classifyActionPayload(action) {
|
|
509
|
+
return scanShellAction(action).heredocs > 0 ? "edit" : "shell";
|
|
510
|
+
}
|
|
511
|
+
/**
|
|
512
|
+
* Measure an action against the budget.
|
|
513
|
+
*
|
|
514
|
+
* The budget bounds what the scaffold can execute: one top-level statement,
|
|
515
|
+
* one authored file, a byte cap, and neither a no-op nor a submit. Every
|
|
516
|
+
* rejection here is one of those.
|
|
517
|
+
*
|
|
518
|
+
* `declaredKind` is what the analyst called its own action. It is recorded
|
|
519
|
+
* beside the measured payload and never rejected on, because the scaffold runs
|
|
520
|
+
* the action identically either way — so rejecting the label scores an arm on
|
|
521
|
+
* how it described a repair rather than on the repair. A reader who wants the
|
|
522
|
+
* mismatch counts `declared` against `payload`.
|
|
523
|
+
*/
|
|
524
|
+
function checkInterventionBudget(action, declaredKind, budget = SCAFFOLD_INTERVENTION_BUDGET) {
|
|
525
|
+
assertBudget(budget);
|
|
526
|
+
const scan = scanShellAction(action);
|
|
527
|
+
const payload = scan.heredocs > 0 ? "edit" : "shell";
|
|
528
|
+
const measurement = {
|
|
529
|
+
bytes: Buffer.byteLength(action, "utf8"),
|
|
530
|
+
statements: scan.statements.length,
|
|
531
|
+
heredocs: scan.heredocs,
|
|
532
|
+
payload,
|
|
533
|
+
declared: declaredKind
|
|
534
|
+
};
|
|
535
|
+
const reject = (violation, detail) => ({
|
|
536
|
+
admissible: false,
|
|
537
|
+
violation,
|
|
538
|
+
detail,
|
|
539
|
+
measurement
|
|
540
|
+
});
|
|
541
|
+
if (action.trim().length === 0) return reject("empty", "the intervention is empty");
|
|
542
|
+
if (measurement.bytes > budget.maxBytes) return reject("over-byte-cap", `${measurement.bytes} bytes exceeds the ${budget.maxBytes}-byte action budget`);
|
|
543
|
+
if (measurement.statements > budget.maxStatements) return reject("multiple-statements", `${measurement.statements} top-level statements exceeds the ${budget.maxStatements} the scaffold takes per action`);
|
|
544
|
+
if (measurement.heredocs > budget.maxHeredocs) return reject("multiple-heredocs", `${measurement.heredocs} heredocs exceeds the ${budget.maxHeredocs} one edit may author`);
|
|
545
|
+
if (action.includes("COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT")) return reject("submit-instead-of-repair", "the action submits the run instead of repairing it");
|
|
546
|
+
if (NO_OP_ACTIONS.includes(action.trim())) return reject("no-op-action", `"${action.trim()}" changes nothing`);
|
|
547
|
+
return {
|
|
548
|
+
admissible: true,
|
|
549
|
+
measurement
|
|
550
|
+
};
|
|
551
|
+
}
|
|
552
|
+
function assertBudget(budget) {
|
|
553
|
+
for (const field of [
|
|
554
|
+
"maxBytes",
|
|
555
|
+
"maxStatements",
|
|
556
|
+
"maxHeredocs"
|
|
557
|
+
]) {
|
|
558
|
+
const value = budget[field];
|
|
559
|
+
if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`intervention budget ${field} must be a positive integer, got ${value}`);
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
/**
|
|
563
|
+
* Whitespace-insensitive comparison used to reject an intervention that is
|
|
564
|
+
* the recorded action again. Trailing whitespace and blank lines are the only
|
|
565
|
+
* differences a re-proposal can carry without changing what runs.
|
|
566
|
+
*/
|
|
567
|
+
function normalizeActionForComparison(action) {
|
|
568
|
+
return action.split("\n").map((line) => line.trimEnd()).filter((line, index, lines) => line.length > 0 || index > 0 && index < lines.length - 1).join("\n").trim();
|
|
569
|
+
}
|
|
570
|
+
//#endregion
|
|
571
|
+
//#region src/trace-repair/admission-records.ts
|
|
572
|
+
/**
|
|
573
|
+
* The vocabulary of the TB-Repair admission pre-pass: what a corpus row is,
|
|
574
|
+
* which failure population it belongs to, why it can leave the funnel, and the
|
|
575
|
+
* denominator chain those exclusions add up to.
|
|
576
|
+
*
|
|
577
|
+
* Pure data and pure functions. The gate that spends containers on these types
|
|
578
|
+
* lives in `admission.ts`, and the artifact that publishes them lives in
|
|
579
|
+
* `admission-report.ts`.
|
|
580
|
+
*/
|
|
581
|
+
/** A campaign scored rows the pre-pass did not admit, or dropped rows it did. */
|
|
582
|
+
var AdmissionDenominatorError = class extends CaptureIntegrityError {};
|
|
583
|
+
/** A row handed to the pre-pass carried a field only an analyst could produce. */
|
|
584
|
+
var AdmissionIndependenceError = class extends CaptureIntegrityError {};
|
|
585
|
+
const ADMISSION_ROW_KEYS = [
|
|
586
|
+
"rowId",
|
|
587
|
+
"taskName",
|
|
588
|
+
"recordedModel",
|
|
589
|
+
"recordedCommands",
|
|
590
|
+
"finalReturncode"
|
|
591
|
+
];
|
|
592
|
+
/**
|
|
593
|
+
* Reject rows that carry anything beyond the recording.
|
|
594
|
+
*
|
|
595
|
+
* The check is a closed key list rather than a list of known analyst field
|
|
596
|
+
* names, because the failure to catch is "some new analyst output leaked into
|
|
597
|
+
* the gate", and only a closed shape catches the fields nobody thought of.
|
|
598
|
+
*/
|
|
599
|
+
function assertAnalystIndependent(rows) {
|
|
600
|
+
const allowed = new Set(ADMISSION_ROW_KEYS);
|
|
601
|
+
for (const row of rows) {
|
|
602
|
+
const extra = Object.keys(row).filter((key) => !allowed.has(key));
|
|
603
|
+
if (extra.length > 0) throw new AdmissionIndependenceError(`admission row ${String(row.rowId)} carries analyst-side fields: ${extra.sort().join(", ")}`);
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
const ADMISSION_STRATA = [
|
|
607
|
+
"clean-exit",
|
|
608
|
+
"command-error",
|
|
609
|
+
"signal-kill"
|
|
610
|
+
];
|
|
611
|
+
/** `null` when the recording holds no parseable final return code. */
|
|
612
|
+
function stratumOf(finalReturncode) {
|
|
613
|
+
if (finalReturncode === null || !Number.isInteger(finalReturncode)) return null;
|
|
614
|
+
if (finalReturncode === 0) return "clean-exit";
|
|
615
|
+
return finalReturncode < 0 ? "signal-kill" : "command-error";
|
|
616
|
+
}
|
|
617
|
+
const ADMISSION_EXCLUSION_ORDER = [
|
|
618
|
+
"no-recorded-commands",
|
|
619
|
+
"unparseable-final-returncode",
|
|
620
|
+
"stratum-not-admitted",
|
|
621
|
+
"task-oracle-uncertified",
|
|
622
|
+
"task-oracle-nondeterministic",
|
|
623
|
+
"prefix-replay-error",
|
|
624
|
+
"prefix-replay-empty",
|
|
625
|
+
"prefix-replay-truncated",
|
|
626
|
+
"prefix-divergence-above-threshold",
|
|
627
|
+
"end-state-oracle-error",
|
|
628
|
+
"end-state-tests-pass",
|
|
629
|
+
"no-fix-control-error",
|
|
630
|
+
"no-fix-control-rescued",
|
|
631
|
+
"no-op-control-error",
|
|
632
|
+
"no-op-control-rescued"
|
|
633
|
+
];
|
|
634
|
+
/** Reasons decided from the recording alone, before a row can be stratified. */
|
|
635
|
+
const PRE_STRATUM_REASONS = ["no-recorded-commands", "unparseable-final-returncode"];
|
|
636
|
+
function isPreStratumReason(reason) {
|
|
637
|
+
return PRE_STRATUM_REASONS.includes(reason);
|
|
638
|
+
}
|
|
639
|
+
/** One line of prose per reason, for the rendered chain. */
|
|
640
|
+
const ADMISSION_EXCLUSION_MEANING = Object.freeze({
|
|
641
|
+
"no-recorded-commands": "the recording holds no command to substitute",
|
|
642
|
+
"unparseable-final-returncode": "the last observation carries no return code",
|
|
643
|
+
"stratum-not-admitted": "the row belongs to a population this campaign excluded",
|
|
644
|
+
"task-oracle-uncertified": "the task grader has no determinism certification on file",
|
|
645
|
+
"task-oracle-nondeterministic": "the task grader returns different verdicts on byte-identical state",
|
|
646
|
+
"prefix-replay-error": "the replay boundary failed, so divergence is unmeasured",
|
|
647
|
+
"prefix-replay-empty": "the replay executed no recorded step",
|
|
648
|
+
"prefix-replay-truncated": "the replay stopped short of the recorded end state",
|
|
649
|
+
"prefix-divergence-above-threshold": "replaying the prefix did not reproduce the recording",
|
|
650
|
+
"end-state-oracle-error": "the task grader failed, so the end state is unjudged",
|
|
651
|
+
"end-state-tests-pass": "the task tests pass on the recorded end state, so nothing failed",
|
|
652
|
+
"no-fix-control-error": "a no-fix rollout failed to run, so the control is unmeasured",
|
|
653
|
+
"no-fix-control-rescued": "the continuation policy passes the task with no intervention",
|
|
654
|
+
"no-op-control-error": "a no-op rollout failed to run, so the control is unmeasured",
|
|
655
|
+
"no-op-control-rescued": "an inert action plus continuation passes the task"
|
|
656
|
+
});
|
|
657
|
+
/**
|
|
658
|
+
* Build the funnel every campaign report publishes.
|
|
659
|
+
*
|
|
660
|
+
* Each stage names the reason, the rows that reached it, the rows it removed,
|
|
661
|
+
* and the rows that survived, so `input = admitted + sum(excluded)` can be read
|
|
662
|
+
* off the table instead of trusted.
|
|
663
|
+
*/
|
|
664
|
+
function buildDenominatorChain(verdicts, admitStrata) {
|
|
665
|
+
const reasonTotals = emptyReasonTotals();
|
|
666
|
+
for (const verdict of verdicts) if (verdict.excludedBy !== null) reasonTotals[verdict.excludedBy] += 1;
|
|
667
|
+
const stratumReasons = ADMISSION_EXCLUSION_ORDER.filter((reason) => !isPreStratumReason(reason));
|
|
668
|
+
const artifact = {
|
|
669
|
+
version: 1,
|
|
670
|
+
overall: chainOf("all", verdicts, ADMISSION_EXCLUSION_ORDER),
|
|
671
|
+
byStratum: ADMISSION_STRATA.filter((stratum) => verdicts.some((verdict) => verdict.stratum === stratum)).map((stratum) => chainOf(stratum, verdicts.filter((verdict) => verdict.stratum === stratum), stratumReasons)),
|
|
672
|
+
reasonTotals,
|
|
673
|
+
unstratified: verdicts.filter((verdict) => verdict.stratum === null).length,
|
|
674
|
+
admitStrata: [...admitStrata]
|
|
675
|
+
};
|
|
676
|
+
assertChainReconciles(artifact);
|
|
677
|
+
return artifact;
|
|
678
|
+
}
|
|
679
|
+
function chainOf(scope, verdicts, reasons) {
|
|
680
|
+
let remaining = verdicts.length;
|
|
681
|
+
const stages = [];
|
|
682
|
+
for (const reason of reasons) {
|
|
683
|
+
const excluded = verdicts.filter((verdict) => verdict.excludedBy === reason).length;
|
|
684
|
+
const entering = remaining;
|
|
685
|
+
remaining -= excluded;
|
|
686
|
+
stages.push({
|
|
687
|
+
reason,
|
|
688
|
+
entering,
|
|
689
|
+
excluded,
|
|
690
|
+
remaining
|
|
691
|
+
});
|
|
692
|
+
}
|
|
693
|
+
return {
|
|
694
|
+
scope,
|
|
695
|
+
input: verdicts.length,
|
|
696
|
+
stages,
|
|
697
|
+
admitted: verdicts.filter((verdict) => verdict.admitted).length
|
|
698
|
+
};
|
|
699
|
+
}
|
|
700
|
+
function emptyReasonTotals() {
|
|
701
|
+
const totals = {};
|
|
702
|
+
for (const reason of ADMISSION_EXCLUSION_ORDER) totals[reason] = 0;
|
|
703
|
+
return totals;
|
|
704
|
+
}
|
|
705
|
+
/** A chain that does not add up is a broken denominator, so this throws. */
|
|
706
|
+
function assertChainReconciles(artifact) {
|
|
707
|
+
for (const chain of [artifact.overall, ...artifact.byStratum]) {
|
|
708
|
+
const excluded = chain.stages.reduce((total, stage) => total + stage.excluded, 0);
|
|
709
|
+
if (chain.input !== chain.admitted + excluded) throw new AdmissionDenominatorError(`denominator chain (${chain.scope}) does not reconcile: input ${chain.input} != admitted ${chain.admitted} + excluded ${excluded}`);
|
|
710
|
+
const last = chain.stages[chain.stages.length - 1];
|
|
711
|
+
if (last && last.remaining !== chain.admitted) throw new AdmissionDenominatorError(`denominator chain (${chain.scope}) ends at ${last.remaining} rows but reports ${chain.admitted} admitted`);
|
|
712
|
+
}
|
|
713
|
+
const stratumInputs = artifact.byStratum.reduce((total, chain) => total + chain.input, 0);
|
|
714
|
+
if (stratumInputs + artifact.unstratified !== artifact.overall.input) throw new AdmissionDenominatorError(`stratum inputs ${stratumInputs} plus ${artifact.unstratified} unstratified do not cover ${artifact.overall.input} input rows`);
|
|
715
|
+
}
|
|
716
|
+
//#endregion
|
|
717
|
+
//#region src/trace-repair/continuation-policy.ts
|
|
718
|
+
/** A rollout ran outside the pinned policy, so its evidence cannot be used. */
|
|
719
|
+
var ContinuationPolicyViolationError = class extends CaptureIntegrityError {};
|
|
720
|
+
/** Two arms did not run the same policy, so their difference is not the intervention. */
|
|
721
|
+
var ContinuationSymmetryError = class extends CaptureIntegrityError {};
|
|
722
|
+
/**
|
|
723
|
+
* Everything the policy fixes except the two a campaign must choose.
|
|
724
|
+
*
|
|
725
|
+
* `commandTimeoutSeconds` is 30 because the recorded runs used the scaffold's
|
|
726
|
+
* own 30-second environment timeout; a longer limit would let the continuation
|
|
727
|
+
* finish commands the recorded agent could not.
|
|
728
|
+
*/
|
|
729
|
+
const CONTINUATION_POLICY_DEFAULTS = {
|
|
730
|
+
id: "tb-repair-continuation-v1",
|
|
731
|
+
stepBudget: 20,
|
|
732
|
+
temperature: 0,
|
|
733
|
+
maxTokens: 4096,
|
|
734
|
+
commandTimeoutSeconds: 30,
|
|
735
|
+
maxConsecutiveFormatErrors: 3,
|
|
736
|
+
networkMode: "none",
|
|
737
|
+
scaffold: "mini-swe-agent"
|
|
738
|
+
};
|
|
739
|
+
function definePinnedContinuationPolicy(input) {
|
|
740
|
+
const policy = {
|
|
741
|
+
...CONTINUATION_POLICY_DEFAULTS,
|
|
742
|
+
...input
|
|
743
|
+
};
|
|
744
|
+
if (!policy.model.trim()) throw new ValidationError("continuation policy requires a model id");
|
|
745
|
+
if (!Number.isInteger(policy.seed)) throw new ValidationError(`continuation policy seed must be an integer, got ${policy.seed}`);
|
|
746
|
+
requirePositiveInteger$1(policy.stepBudget, "stepBudget");
|
|
747
|
+
requirePositiveInteger$1(policy.maxTokens, "maxTokens");
|
|
748
|
+
requirePositiveInteger$1(policy.commandTimeoutSeconds, "commandTimeoutSeconds");
|
|
749
|
+
requirePositiveInteger$1(policy.maxConsecutiveFormatErrors, "maxConsecutiveFormatErrors");
|
|
750
|
+
if (!Number.isFinite(policy.temperature) || policy.temperature < 0) throw new ValidationError(`continuation policy temperature must be a non-negative number, got ${policy.temperature}`);
|
|
751
|
+
return Object.freeze(policy);
|
|
752
|
+
}
|
|
753
|
+
function requirePositiveInteger$1(value, field) {
|
|
754
|
+
if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`continuation policy ${field} must be a positive integer, got ${value}`);
|
|
755
|
+
}
|
|
756
|
+
/**
|
|
757
|
+
* Hash over the policy and the scaffold text it renders. A changed template
|
|
758
|
+
* changes the digest, so rollouts recorded before and after an edit cannot be
|
|
759
|
+
* pooled by accident.
|
|
760
|
+
*/
|
|
761
|
+
function continuationPolicyDigest(policy) {
|
|
762
|
+
return contentHash({
|
|
763
|
+
policy: { ...policy },
|
|
764
|
+
systemMessage: MINI_SWE_SYSTEM_MESSAGE,
|
|
765
|
+
instanceMessage: renderInstanceMessage({
|
|
766
|
+
task: "<task>",
|
|
767
|
+
systemInformation: "<system>"
|
|
768
|
+
}),
|
|
769
|
+
formatErrorObservation: renderFormatErrorObservation(0),
|
|
770
|
+
timeoutObservation: renderTimeoutObservation("<command>", "<output>"),
|
|
771
|
+
observation: renderObservation({
|
|
772
|
+
returncode: 0,
|
|
773
|
+
output: ""
|
|
774
|
+
})
|
|
775
|
+
});
|
|
776
|
+
}
|
|
777
|
+
/**
|
|
778
|
+
* Per-rollout seed. It reads the policy seed, the row, and the rollout index —
|
|
779
|
+
* deliberately not the arm, so paired rollouts across arms draw identically.
|
|
780
|
+
*/
|
|
781
|
+
function continuationSeed(policySeed, rowId, rolloutIndex) {
|
|
782
|
+
const input = `${policySeed}:${rowId}:${rolloutIndex}`;
|
|
783
|
+
let hash = 2166136261;
|
|
784
|
+
for (let i = 0; i < input.length; i += 1) {
|
|
785
|
+
hash ^= input.charCodeAt(i);
|
|
786
|
+
hash = Math.imul(hash, 16777619) >>> 0;
|
|
787
|
+
}
|
|
788
|
+
return hash >>> 1;
|
|
789
|
+
}
|
|
790
|
+
/**
|
|
791
|
+
* Run the scaffold forward for `rollouts` independent continuations.
|
|
792
|
+
*
|
|
793
|
+
* Each rollout gets its own environment from the factory, because a rollout
|
|
794
|
+
* mutates the container it runs in and the next one must start from the same
|
|
795
|
+
* state, not from the previous rollout's leftovers.
|
|
796
|
+
*/
|
|
797
|
+
async function runContinuation(options) {
|
|
798
|
+
const { policy, arm, rowId, prefix, rollouts, model, environments } = options;
|
|
799
|
+
const clock = options.clock ?? Date.now;
|
|
800
|
+
requirePositiveInteger$1(rollouts, "rollouts");
|
|
801
|
+
assertPrefix(prefix);
|
|
802
|
+
const policyDigest = continuationPolicyDigest(policy);
|
|
803
|
+
const records = [];
|
|
804
|
+
for (let index = 0; index < rollouts; index += 1) records.push(await runOneRollout({
|
|
805
|
+
policy,
|
|
806
|
+
policyDigest,
|
|
807
|
+
arm,
|
|
808
|
+
rowId,
|
|
809
|
+
index,
|
|
810
|
+
prefix,
|
|
811
|
+
model,
|
|
812
|
+
environments,
|
|
813
|
+
clock
|
|
814
|
+
}));
|
|
815
|
+
return records;
|
|
816
|
+
}
|
|
817
|
+
function assertPrefix(prefix) {
|
|
818
|
+
if (prefix.length < 2) throw new ValidationError("continuation prefix needs the system and task messages");
|
|
819
|
+
if (prefix[0]?.role !== "system" || prefix[1]?.role !== "user") throw new ValidationError("continuation prefix must start with a system then a user message");
|
|
820
|
+
if (prefix[prefix.length - 1]?.role !== "user") throw new ValidationError("continuation prefix must end on a user message; an assistant turn with no observation means the replay left an action unanswered");
|
|
821
|
+
}
|
|
822
|
+
async function runOneRollout(input) {
|
|
823
|
+
const { policy, arm, rowId, index, clock } = input;
|
|
824
|
+
const seed = continuationSeed(policy.seed, rowId, index);
|
|
825
|
+
const startedMs = clock();
|
|
826
|
+
const environment = await input.environments.create({
|
|
827
|
+
rowId,
|
|
828
|
+
arm,
|
|
829
|
+
rolloutIndex: index
|
|
830
|
+
});
|
|
831
|
+
try {
|
|
832
|
+
const description = await environment.describe();
|
|
833
|
+
if (description.networkMode !== policy.networkMode) throw new ContinuationPolicyViolationError(`continuation requires network mode "${policy.networkMode}", container ${environment.containerRef} reports "${description.networkMode}"`);
|
|
834
|
+
const messages = [...input.prefix];
|
|
835
|
+
const steps = [];
|
|
836
|
+
let exitStatus = "step-budget-exhausted";
|
|
837
|
+
let submission = null;
|
|
838
|
+
let terminalError;
|
|
839
|
+
let consecutiveFormatErrors = 0;
|
|
840
|
+
for (let step = 1; step <= policy.stepBudget; step += 1) {
|
|
841
|
+
const callStartedMs = clock();
|
|
842
|
+
let response;
|
|
843
|
+
try {
|
|
844
|
+
response = await input.model({
|
|
845
|
+
model: policy.model,
|
|
846
|
+
messages: [...messages],
|
|
847
|
+
seed,
|
|
848
|
+
temperature: policy.temperature,
|
|
849
|
+
maxTokens: policy.maxTokens
|
|
850
|
+
});
|
|
851
|
+
} catch (error) {
|
|
852
|
+
exitStatus = "model-error";
|
|
853
|
+
terminalError = errorMessage(error);
|
|
854
|
+
break;
|
|
855
|
+
}
|
|
856
|
+
const call = {
|
|
857
|
+
servedModel: response.servedModel,
|
|
858
|
+
seed,
|
|
859
|
+
latencyMs: clock() - callStartedMs,
|
|
860
|
+
usage: response.usage,
|
|
861
|
+
costUsd: response.costUsd,
|
|
862
|
+
finishReason: response.finishReason ?? null,
|
|
863
|
+
contentChars: response.content.length
|
|
864
|
+
};
|
|
865
|
+
messages.push({
|
|
866
|
+
role: "assistant",
|
|
867
|
+
content: response.content
|
|
868
|
+
});
|
|
869
|
+
const parsed = parseAction(response.content);
|
|
870
|
+
if (parsed.kind === "format-error") {
|
|
871
|
+
consecutiveFormatErrors += 1;
|
|
872
|
+
const observation = renderFormatErrorObservation(parsed.actionCount);
|
|
873
|
+
messages.push({
|
|
874
|
+
role: "user",
|
|
875
|
+
content: observation
|
|
876
|
+
});
|
|
877
|
+
steps.push({
|
|
878
|
+
step,
|
|
879
|
+
assistantMessage: response.content,
|
|
880
|
+
action: null,
|
|
881
|
+
observation,
|
|
882
|
+
execution: null,
|
|
883
|
+
model: call
|
|
884
|
+
});
|
|
885
|
+
if (consecutiveFormatErrors >= policy.maxConsecutiveFormatErrors) {
|
|
886
|
+
exitStatus = "repeated-format-error";
|
|
887
|
+
break;
|
|
888
|
+
}
|
|
889
|
+
continue;
|
|
890
|
+
}
|
|
891
|
+
consecutiveFormatErrors = 0;
|
|
892
|
+
const execStartedMs = clock();
|
|
893
|
+
let execution;
|
|
894
|
+
try {
|
|
895
|
+
execution = await environment.exec(parsed.command, { timeoutSeconds: policy.commandTimeoutSeconds });
|
|
896
|
+
} catch (error) {
|
|
897
|
+
exitStatus = "environment-error";
|
|
898
|
+
terminalError = errorMessage(error);
|
|
899
|
+
steps.push({
|
|
900
|
+
step,
|
|
901
|
+
assistantMessage: response.content,
|
|
902
|
+
action: parsed.command,
|
|
903
|
+
observation: null,
|
|
904
|
+
execution: null,
|
|
905
|
+
model: call,
|
|
906
|
+
error: errorMessage(error)
|
|
907
|
+
});
|
|
908
|
+
break;
|
|
909
|
+
}
|
|
910
|
+
const execRecord = {
|
|
911
|
+
command: parsed.command,
|
|
912
|
+
returncode: execution.returncode,
|
|
913
|
+
timedOut: execution.timedOut,
|
|
914
|
+
outputChars: execution.output.length,
|
|
915
|
+
durationMs: clock() - execStartedMs
|
|
916
|
+
};
|
|
917
|
+
if (execution.timedOut) {
|
|
918
|
+
const observation = renderTimeoutObservation(parsed.command, execution.output);
|
|
919
|
+
messages.push({
|
|
920
|
+
role: "user",
|
|
921
|
+
content: observation
|
|
922
|
+
});
|
|
923
|
+
steps.push({
|
|
924
|
+
step,
|
|
925
|
+
assistantMessage: response.content,
|
|
926
|
+
action: parsed.command,
|
|
927
|
+
observation,
|
|
928
|
+
execution: execRecord,
|
|
929
|
+
model: call
|
|
930
|
+
});
|
|
931
|
+
continue;
|
|
932
|
+
}
|
|
933
|
+
const submitted = submissionOf(execution);
|
|
934
|
+
if (submitted !== null) {
|
|
935
|
+
exitStatus = "submitted";
|
|
936
|
+
submission = submitted;
|
|
937
|
+
steps.push({
|
|
938
|
+
step,
|
|
939
|
+
assistantMessage: response.content,
|
|
940
|
+
action: parsed.command,
|
|
941
|
+
observation: null,
|
|
942
|
+
execution: execRecord,
|
|
943
|
+
model: call
|
|
944
|
+
});
|
|
945
|
+
break;
|
|
946
|
+
}
|
|
947
|
+
const observation = renderObservation(execution);
|
|
948
|
+
messages.push({
|
|
949
|
+
role: "user",
|
|
950
|
+
content: observation
|
|
951
|
+
});
|
|
952
|
+
steps.push({
|
|
953
|
+
step,
|
|
954
|
+
assistantMessage: response.content,
|
|
955
|
+
action: parsed.command,
|
|
956
|
+
observation,
|
|
957
|
+
execution: execRecord,
|
|
958
|
+
model: call
|
|
959
|
+
});
|
|
960
|
+
}
|
|
961
|
+
const endedMs = clock();
|
|
962
|
+
const rollout = {
|
|
963
|
+
rolloutId: `${rowId}:${arm}:${index}`,
|
|
964
|
+
arm,
|
|
965
|
+
rowId,
|
|
966
|
+
index,
|
|
967
|
+
seed,
|
|
968
|
+
policyDigest: input.policyDigest,
|
|
969
|
+
environmentId: input.environments.id,
|
|
970
|
+
containerRef: environment.containerRef,
|
|
971
|
+
environment: description,
|
|
972
|
+
steps,
|
|
973
|
+
exitStatus,
|
|
974
|
+
submission,
|
|
975
|
+
usage: totalUsage(steps),
|
|
976
|
+
costProvenance: totalCost(steps),
|
|
977
|
+
wallMs: endedMs - startedMs,
|
|
978
|
+
startedAt: new Date(startedMs).toISOString(),
|
|
979
|
+
endedAt: new Date(endedMs).toISOString()
|
|
980
|
+
};
|
|
981
|
+
return terminalError === void 0 ? rollout : {
|
|
982
|
+
...rollout,
|
|
983
|
+
terminalError
|
|
984
|
+
};
|
|
985
|
+
} finally {
|
|
986
|
+
await environment.dispose();
|
|
987
|
+
}
|
|
988
|
+
}
|
|
989
|
+
function errorMessage(error) {
|
|
990
|
+
return error instanceof Error ? error.message : String(error);
|
|
991
|
+
}
|
|
992
|
+
/**
|
|
993
|
+
* Sum only what the provider reported. A call with no usage raises
|
|
994
|
+
* `callsWithUsage` short of `calls` and clears `captured`, so a partially
|
|
995
|
+
* reported rollout can never read as a fully measured one.
|
|
996
|
+
*/
|
|
997
|
+
function totalUsage(steps) {
|
|
998
|
+
let input = 0;
|
|
999
|
+
let output = 0;
|
|
1000
|
+
let reasoning = 0;
|
|
1001
|
+
let cached = 0;
|
|
1002
|
+
let cacheWrite = 0;
|
|
1003
|
+
let sawReasoning = false;
|
|
1004
|
+
let sawCached = false;
|
|
1005
|
+
let sawCacheWrite = false;
|
|
1006
|
+
let callsWithUsage = 0;
|
|
1007
|
+
for (const step of steps) {
|
|
1008
|
+
const usage = step.model.usage;
|
|
1009
|
+
if (!usage) continue;
|
|
1010
|
+
callsWithUsage += 1;
|
|
1011
|
+
input += usage.input;
|
|
1012
|
+
output += usage.output;
|
|
1013
|
+
if (usage.reasoning !== void 0) {
|
|
1014
|
+
reasoning += usage.reasoning;
|
|
1015
|
+
sawReasoning = true;
|
|
1016
|
+
}
|
|
1017
|
+
if (usage.cached !== void 0) {
|
|
1018
|
+
cached += usage.cached;
|
|
1019
|
+
sawCached = true;
|
|
1020
|
+
}
|
|
1021
|
+
if (usage.cacheWrite !== void 0) {
|
|
1022
|
+
cacheWrite += usage.cacheWrite;
|
|
1023
|
+
sawCacheWrite = true;
|
|
1024
|
+
}
|
|
1025
|
+
}
|
|
1026
|
+
const totals = {
|
|
1027
|
+
calls: steps.length,
|
|
1028
|
+
callsWithUsage,
|
|
1029
|
+
captured: steps.length > 0 && callsWithUsage === steps.length,
|
|
1030
|
+
input,
|
|
1031
|
+
output
|
|
1032
|
+
};
|
|
1033
|
+
if (sawReasoning) totals.reasoning = reasoning;
|
|
1034
|
+
if (sawCached) totals.cached = cached;
|
|
1035
|
+
if (sawCacheWrite) totals.cacheWrite = cacheWrite;
|
|
1036
|
+
return totals;
|
|
1037
|
+
}
|
|
1038
|
+
/**
|
|
1039
|
+
* One unpriced call makes the rollout's cost unknown. Summing the rest would
|
|
1040
|
+
* report a number smaller than what was spent.
|
|
1041
|
+
*/
|
|
1042
|
+
function totalCost(steps) {
|
|
1043
|
+
if (steps.length === 0) return {
|
|
1044
|
+
kind: "uncaptured",
|
|
1045
|
+
usd: null
|
|
1046
|
+
};
|
|
1047
|
+
let usd = 0;
|
|
1048
|
+
for (const step of steps) {
|
|
1049
|
+
if (step.model.costUsd === null) return {
|
|
1050
|
+
kind: "uncaptured",
|
|
1051
|
+
usd: null
|
|
1052
|
+
};
|
|
1053
|
+
usd += step.model.costUsd;
|
|
1054
|
+
}
|
|
1055
|
+
return {
|
|
1056
|
+
kind: "observed",
|
|
1057
|
+
usd
|
|
1058
|
+
};
|
|
1059
|
+
}
|
|
1060
|
+
/**
|
|
1061
|
+
* Prove the arms ran the same policy. Rollouts paired by row and index must
|
|
1062
|
+
* carry the same policy digest and the same seed; anything else means the
|
|
1063
|
+
* measured difference includes a policy change, not only the intervention.
|
|
1064
|
+
*/
|
|
1065
|
+
function assertArmSymmetry(rollouts) {
|
|
1066
|
+
const digests = new Set(rollouts.map((rollout) => rollout.policyDigest));
|
|
1067
|
+
if (digests.size > 1) throw new ContinuationSymmetryError(`arms ran different policies: ${[...digests].sort().join(", ")}`);
|
|
1068
|
+
const seeds = /* @__PURE__ */ new Map();
|
|
1069
|
+
for (const rollout of rollouts) {
|
|
1070
|
+
const key = `${rollout.rowId}:${rollout.index}`;
|
|
1071
|
+
const seen = seeds.get(key);
|
|
1072
|
+
if (!seen) {
|
|
1073
|
+
seeds.set(key, {
|
|
1074
|
+
seed: rollout.seed,
|
|
1075
|
+
arm: rollout.arm
|
|
1076
|
+
});
|
|
1077
|
+
continue;
|
|
1078
|
+
}
|
|
1079
|
+
if (seen.seed !== rollout.seed) throw new ContinuationSymmetryError(`paired rollouts ${key} drew different seeds: ${seen.arm}=${seen.seed}, ${rollout.arm}=${rollout.seed}`);
|
|
1080
|
+
}
|
|
1081
|
+
}
|
|
1082
|
+
//#endregion
|
|
1083
|
+
//#region src/trace-repair/control-policy.ts
|
|
1084
|
+
/**
|
|
1085
|
+
* The control policy admission screens under, declared rather than assumed.
|
|
1086
|
+
*
|
|
1087
|
+
* Admission conditions 3 and 4 ask whether a row is rescued by continuing from
|
|
1088
|
+
* the recorded end state with no intervention, and by continuing after an
|
|
1089
|
+
* action that changes nothing. Both are questions about a policy, and neither
|
|
1090
|
+
* is answerable without knowing what that policy is allowed to do.
|
|
1091
|
+
*
|
|
1092
|
+
* One number decides whether the question can be answered at all. Condition 2
|
|
1093
|
+
* has already graded the recorded end state and found it failing. A control
|
|
1094
|
+
* rollout that makes no model call executes no command, so the container it
|
|
1095
|
+
* grades holds those same bytes — the ones already graded as failing. Under
|
|
1096
|
+
* such a policy a control pass is not a rescue; it is the task's own grader
|
|
1097
|
+
* answering differently about identical state. The condition cannot fire for
|
|
1098
|
+
* the reason it exists, and every row walks through it.
|
|
1099
|
+
*
|
|
1100
|
+
* So the policy is a required, hashed parameter that lands on every admission
|
|
1101
|
+
* decision, and a configuration whose control cannot reach the outcome it
|
|
1102
|
+
* screens for is refused where it is configured rather than passed silently.
|
|
1103
|
+
*/
|
|
1104
|
+
/** The declared control cannot produce the outcome the criteria screen for. */
|
|
1105
|
+
var UncalibratedControlError = class extends CaptureIntegrityError {};
|
|
1106
|
+
const CONTROL_SCREENING_MODES = ["enforced", "declared-inert"];
|
|
1107
|
+
/**
|
|
1108
|
+
* A control rollout changes the graded state only by executing something, and
|
|
1109
|
+
* it executes only what a model call asks for. At a zero budget it grades the
|
|
1110
|
+
* bytes it was handed.
|
|
1111
|
+
*/
|
|
1112
|
+
function controlCanRescue(stepBudget) {
|
|
1113
|
+
return stepBudget > 0;
|
|
1114
|
+
}
|
|
1115
|
+
function defineControlPolicy(input) {
|
|
1116
|
+
if (!input.id.trim()) throw new ValidationError("control policy requires an id");
|
|
1117
|
+
if (!input.scaffold.trim()) throw new ValidationError("control policy requires a scaffold name");
|
|
1118
|
+
if (!Number.isInteger(input.stepBudget) || input.stepBudget < 0) throw new ValidationError(`control policy stepBudget must be a non-negative integer, got ${input.stepBudget}`);
|
|
1119
|
+
if (!Number.isInteger(input.commandTimeoutSeconds) || input.commandTimeoutSeconds <= 0) throw new ValidationError(`control policy commandTimeoutSeconds must be a positive integer, got ${input.commandTimeoutSeconds}`);
|
|
1120
|
+
if (input.stepBudget === 0 && input.model !== null) throw new ValidationError(`control policy ${input.id} declares model ${input.model} at a zero step budget; a policy that makes no model call must record model: null`);
|
|
1121
|
+
if (input.stepBudget > 0 && !input.model?.trim()) throw new ValidationError(`control policy ${input.id} allows ${input.stepBudget} model call(s) but names no model`);
|
|
1122
|
+
const declaration = {
|
|
1123
|
+
id: input.id,
|
|
1124
|
+
stepBudget: input.stepBudget,
|
|
1125
|
+
scaffold: input.scaffold,
|
|
1126
|
+
model: input.model,
|
|
1127
|
+
commandTimeoutSeconds: input.commandTimeoutSeconds
|
|
1128
|
+
};
|
|
1129
|
+
return Object.freeze({
|
|
1130
|
+
...declaration,
|
|
1131
|
+
digest: contentHash(declaration),
|
|
1132
|
+
canRescue: controlCanRescue(input.stepBudget)
|
|
1133
|
+
});
|
|
1134
|
+
}
|
|
1135
|
+
/**
|
|
1136
|
+
* Refuse a configuration whose control and screening mode contradict.
|
|
1137
|
+
*
|
|
1138
|
+
* Both directions are faults, and both are silent without this. A screening
|
|
1139
|
+
* control that cannot act passes every row through a condition it can never
|
|
1140
|
+
* fire. A control declared inert that can in fact act hides a real screen
|
|
1141
|
+
* behind a label that says nothing was screened.
|
|
1142
|
+
*/
|
|
1143
|
+
function assertControlCalibrated(policy, screening) {
|
|
1144
|
+
if (!CONTROL_SCREENING_MODES.includes(screening)) throw new ValidationError(`unknown control screening mode: ${screening}`);
|
|
1145
|
+
const canRescue = controlCanRescue(policy.stepBudget);
|
|
1146
|
+
if (screening === "enforced" && !canRescue) throw new UncalibratedControlError(`control policy ${policy.id} (digest ${policy.digest}) allows ${policy.stepBudget} model call(s) per rollout, so a control rollout executes no command and grades the same bytes the end-state check already graded as failing. Under it conditions 3 and 4 can only fire on a grader that disagrees with itself, so they screen nothing. Give the control a step budget of at least 1, or set controlScreening to 'declared-inert' and read a control pass as the oracle flip it is.`);
|
|
1147
|
+
if (screening === "declared-inert" && canRescue) throw new UncalibratedControlError(`control policy ${policy.id} (digest ${policy.digest}) allows ${policy.stepBudget} model call(s) per rollout, so it can rescue a row, but the criteria declare it inert. Set controlScreening to 'enforced' so a control pass is recorded as a rescue.`);
|
|
1148
|
+
}
|
|
1149
|
+
//#endregion
|
|
1150
|
+
//#region src/trace-repair/admission.ts
|
|
1151
|
+
const ADMISSION_CONFIG_DEFAULTS = Object.freeze({
|
|
1152
|
+
maxPrefixDivergence: .1,
|
|
1153
|
+
controlRollouts: 3,
|
|
1154
|
+
admitStrata: Object.freeze(["clean-exit", "command-error"]),
|
|
1155
|
+
inertAction: "true",
|
|
1156
|
+
controlScreening: "enforced",
|
|
1157
|
+
concurrency: 1
|
|
1158
|
+
});
|
|
1159
|
+
function resolveAdmissionConfig(input = {}) {
|
|
1160
|
+
const config = {
|
|
1161
|
+
...ADMISSION_CONFIG_DEFAULTS,
|
|
1162
|
+
...input
|
|
1163
|
+
};
|
|
1164
|
+
if (!Number.isFinite(config.maxPrefixDivergence)) throw new ValidationError(`admission maxPrefixDivergence must be a number, got ${config.maxPrefixDivergence}`);
|
|
1165
|
+
if (config.maxPrefixDivergence < 0 || config.maxPrefixDivergence > 1) throw new ValidationError(`admission maxPrefixDivergence must be a share between 0 and 1, got ${config.maxPrefixDivergence}`);
|
|
1166
|
+
requirePositiveInteger(config.controlRollouts, "controlRollouts");
|
|
1167
|
+
requirePositiveInteger(config.concurrency, "concurrency");
|
|
1168
|
+
if (config.admitStrata.length === 0) throw new ValidationError("admission admitStrata must name at least one stratum");
|
|
1169
|
+
for (const stratum of config.admitStrata) if (!ADMISSION_STRATA.includes(stratum)) throw new ValidationError(`admission admitStrata holds an unknown stratum: ${stratum}`);
|
|
1170
|
+
if (config.inertAction.trim().length === 0) throw new ValidationError("admission inertAction must be a non-empty command");
|
|
1171
|
+
if (!CONTROL_SCREENING_MODES.includes(config.controlScreening)) throw new ValidationError(`admission controlScreening must be one of ${CONTROL_SCREENING_MODES.join(", ")}, got ${config.controlScreening}`);
|
|
1172
|
+
return Object.freeze({
|
|
1173
|
+
...config,
|
|
1174
|
+
admitStrata: Object.freeze([...config.admitStrata])
|
|
1175
|
+
});
|
|
1176
|
+
}
|
|
1177
|
+
function requirePositiveInteger(value, field) {
|
|
1178
|
+
if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`admission ${field} must be a positive integer, got ${value}`);
|
|
1179
|
+
}
|
|
1180
|
+
/**
|
|
1181
|
+
* The recorded command the no-op control replaces, drawn per rollout.
|
|
1182
|
+
*
|
|
1183
|
+
* The draw reads the policy seed, the row, and the rollout index, so it is
|
|
1184
|
+
* reproducible from the artifact and cannot depend on when the pre-pass ran.
|
|
1185
|
+
*/
|
|
1186
|
+
function noOpInjectionStep(policySeed, rowId, rolloutIndex, recordedCommands) {
|
|
1187
|
+
requirePositiveInteger(recordedCommands, "recordedCommands");
|
|
1188
|
+
const input = `no-op:${policySeed}:${rowId}:${rolloutIndex}`;
|
|
1189
|
+
let hash = 2166136261;
|
|
1190
|
+
for (let i = 0; i < input.length; i += 1) {
|
|
1191
|
+
hash ^= input.charCodeAt(i);
|
|
1192
|
+
hash = Math.imul(hash, 16777619) >>> 0;
|
|
1193
|
+
}
|
|
1194
|
+
return (hash >>> 1) % recordedCommands + 1;
|
|
1195
|
+
}
|
|
1196
|
+
/**
|
|
1197
|
+
* Run the pre-pass over every row and publish the denominator it produced.
|
|
1198
|
+
*
|
|
1199
|
+
* Nothing here reads an analyst output. The report is frozen, and a campaign
|
|
1200
|
+
* proves it measured this denominator by passing the report back to
|
|
1201
|
+
* `assertDenominatorIntact`.
|
|
1202
|
+
*/
|
|
1203
|
+
async function runAdmission(options) {
|
|
1204
|
+
const { rows, policy, replayer, oracle, controls, taskOracles } = options;
|
|
1205
|
+
const config = resolveAdmissionConfig(options.config);
|
|
1206
|
+
const clock = options.clock ?? Date.now;
|
|
1207
|
+
assertAnalystIndependent(rows);
|
|
1208
|
+
assertUniqueRowIds(rows);
|
|
1209
|
+
const policyDigest = continuationPolicyDigest(policy);
|
|
1210
|
+
assertControlCalibrated({
|
|
1211
|
+
id: policy.id,
|
|
1212
|
+
digest: policyDigest,
|
|
1213
|
+
stepBudget: policy.stepBudget
|
|
1214
|
+
}, config.controlScreening);
|
|
1215
|
+
const verdicts = await mapOrdered(rows, config.concurrency, (row) => admitRow$1({
|
|
1216
|
+
row,
|
|
1217
|
+
policy,
|
|
1218
|
+
policyDigest,
|
|
1219
|
+
replayer,
|
|
1220
|
+
oracle,
|
|
1221
|
+
controls,
|
|
1222
|
+
taskOracles,
|
|
1223
|
+
config
|
|
1224
|
+
}));
|
|
1225
|
+
const strata = groupAdmittedByStratum(verdicts);
|
|
1226
|
+
const provenance = {
|
|
1227
|
+
replayerId: replayer.id,
|
|
1228
|
+
oracleId: oracle.id,
|
|
1229
|
+
controlRunnerId: controls.id,
|
|
1230
|
+
policyId: policy.id,
|
|
1231
|
+
policyModel: policy.model,
|
|
1232
|
+
policySeed: policy.seed,
|
|
1233
|
+
policyDigest,
|
|
1234
|
+
policyStepBudget: policy.stepBudget,
|
|
1235
|
+
controlScreening: config.controlScreening,
|
|
1236
|
+
certifiedTasks: Object.freeze(Object.fromEntries([...taskOracles].map(([task, verdict]) => [task, verdict.flipRate])))
|
|
1237
|
+
};
|
|
1238
|
+
return Object.freeze({
|
|
1239
|
+
config,
|
|
1240
|
+
provenance,
|
|
1241
|
+
rows: Object.freeze(verdicts),
|
|
1242
|
+
strata,
|
|
1243
|
+
chain: buildDenominatorChain(verdicts, config.admitStrata),
|
|
1244
|
+
controlCost: summarizeControlCost(verdicts),
|
|
1245
|
+
digest: contentHash({
|
|
1246
|
+
config,
|
|
1247
|
+
provenance,
|
|
1248
|
+
admitted: ADMISSION_STRATA.map((stratum) => ({
|
|
1249
|
+
stratum,
|
|
1250
|
+
rowIds: strata[stratum]
|
|
1251
|
+
}))
|
|
1252
|
+
}),
|
|
1253
|
+
generatedAt: new Date(clock()).toISOString()
|
|
1254
|
+
});
|
|
1255
|
+
}
|
|
1256
|
+
function assertUniqueRowIds(rows) {
|
|
1257
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1258
|
+
for (const row of rows) {
|
|
1259
|
+
if (typeof row.rowId !== "string" || row.rowId.length === 0) throw new ValidationError("admission row requires a non-empty rowId");
|
|
1260
|
+
if (seen.has(row.rowId)) throw new ValidationError(`admission received rowId ${row.rowId} twice`);
|
|
1261
|
+
seen.add(row.rowId);
|
|
1262
|
+
}
|
|
1263
|
+
}
|
|
1264
|
+
async function admitRow$1(input) {
|
|
1265
|
+
const { row, config } = input;
|
|
1266
|
+
const checks = [];
|
|
1267
|
+
const rollouts = [];
|
|
1268
|
+
const summaries = [];
|
|
1269
|
+
const certification = input.taskOracles.get(row.taskName) ?? null;
|
|
1270
|
+
const base = {
|
|
1271
|
+
rowId: row.rowId,
|
|
1272
|
+
taskName: row.taskName,
|
|
1273
|
+
recordedModel: row.recordedModel,
|
|
1274
|
+
recordedCommands: row.recordedCommands,
|
|
1275
|
+
finalReturncode: row.finalReturncode,
|
|
1276
|
+
controlPolicyDigest: input.policyDigest,
|
|
1277
|
+
controlScreening: config.controlScreening,
|
|
1278
|
+
oracleFlipRate: certification === null ? null : certification.flipRate
|
|
1279
|
+
};
|
|
1280
|
+
const finish = (stratum, reason, errorDetail = null) => ({
|
|
1281
|
+
...base,
|
|
1282
|
+
stratum,
|
|
1283
|
+
admitted: reason === null,
|
|
1284
|
+
excludedBy: reason,
|
|
1285
|
+
errorDetail,
|
|
1286
|
+
checks,
|
|
1287
|
+
rollouts: summaries
|
|
1288
|
+
});
|
|
1289
|
+
if (!Number.isInteger(row.recordedCommands) || row.recordedCommands < 1) return finish(null, "no-recorded-commands");
|
|
1290
|
+
const stratum = stratumOf(row.finalReturncode);
|
|
1291
|
+
if (stratum === null) return finish(null, "unparseable-final-returncode");
|
|
1292
|
+
checks.push({
|
|
1293
|
+
check: "stratum",
|
|
1294
|
+
stratum
|
|
1295
|
+
});
|
|
1296
|
+
if (!config.admitStrata.includes(stratum)) return finish(stratum, "stratum-not-admitted");
|
|
1297
|
+
if (certification === null) return finish(stratum, "task-oracle-uncertified");
|
|
1298
|
+
checks.push({
|
|
1299
|
+
check: "task-oracle",
|
|
1300
|
+
stable: certification.stable,
|
|
1301
|
+
flipRate: certification.flipRate,
|
|
1302
|
+
replicates: certification.replicates
|
|
1303
|
+
});
|
|
1304
|
+
if (!certification.stable) return finish(stratum, "task-oracle-nondeterministic", certification.detail);
|
|
1305
|
+
const replay = await input.replayer.replay(row);
|
|
1306
|
+
if (!replay.succeeded) return finish(stratum, "prefix-replay-error", replay.error);
|
|
1307
|
+
const { prefixExecuted, prefixDivergences } = replay.value;
|
|
1308
|
+
if (!Number.isInteger(prefixExecuted) || prefixExecuted < 1) return finish(stratum, "prefix-replay-empty");
|
|
1309
|
+
if (prefixExecuted < row.recordedCommands) return finish(stratum, "prefix-replay-truncated");
|
|
1310
|
+
const divergenceRatio = prefixDivergences.length / prefixExecuted;
|
|
1311
|
+
checks.push({
|
|
1312
|
+
check: "prefix-replay",
|
|
1313
|
+
prefixExecuted,
|
|
1314
|
+
divergences: prefixDivergences.length,
|
|
1315
|
+
divergenceRatio
|
|
1316
|
+
});
|
|
1317
|
+
if (divergenceRatio > config.maxPrefixDivergence) return finish(stratum, "prefix-divergence-above-threshold");
|
|
1318
|
+
const endState = await input.oracle.grade(row);
|
|
1319
|
+
if (!endState.succeeded) return finish(stratum, "end-state-oracle-error", endState.error);
|
|
1320
|
+
checks.push({
|
|
1321
|
+
check: "end-state-tests",
|
|
1322
|
+
passed: endState.value.passed,
|
|
1323
|
+
reward: endState.value.reward
|
|
1324
|
+
});
|
|
1325
|
+
if (endState.value.passed) return finish(stratum, "end-state-tests-pass");
|
|
1326
|
+
for (const arm of CONTROL_ARMS) {
|
|
1327
|
+
const control = await runControlArm({
|
|
1328
|
+
...input,
|
|
1329
|
+
arm
|
|
1330
|
+
});
|
|
1331
|
+
checks.push(control.record);
|
|
1332
|
+
rollouts.push(...control.rollouts);
|
|
1333
|
+
summaries.push(...control.summaries);
|
|
1334
|
+
if (control.outcome === "error") return finish(stratum, CONTROL_EXCLUSIONS[arm].error, control.error);
|
|
1335
|
+
if (control.outcome === "rescued") return finish(stratum, CONTROL_EXCLUSIONS[arm].rescued);
|
|
1336
|
+
}
|
|
1337
|
+
assertArmSymmetry(rollouts);
|
|
1338
|
+
return finish(stratum, null);
|
|
1339
|
+
}
|
|
1340
|
+
const CONTROL_ARMS = ["no-fix-control", "no-op-control"];
|
|
1341
|
+
const CONTROL_EXCLUSIONS = Object.freeze({
|
|
1342
|
+
"no-fix-control": {
|
|
1343
|
+
error: "no-fix-control-error",
|
|
1344
|
+
rescued: "no-fix-control-rescued"
|
|
1345
|
+
},
|
|
1346
|
+
"no-op-control": {
|
|
1347
|
+
error: "no-op-control-error",
|
|
1348
|
+
rescued: "no-op-control-rescued"
|
|
1349
|
+
}
|
|
1350
|
+
});
|
|
1351
|
+
/**
|
|
1352
|
+
* Run one control arm until it is decided.
|
|
1353
|
+
*
|
|
1354
|
+
* A rollout that passes decides the arm at once, so the remaining rollouts go
|
|
1355
|
+
* unpaid. A boundary failure decides it too, and as an error: counting an
|
|
1356
|
+
* unmeasured rollout as a failure would admit a row nobody verified.
|
|
1357
|
+
*/
|
|
1358
|
+
async function runControlArm(input) {
|
|
1359
|
+
const { row, arm, config, policy } = input;
|
|
1360
|
+
const rollouts = [];
|
|
1361
|
+
const summaries = [];
|
|
1362
|
+
const injections = [];
|
|
1363
|
+
let passes = 0;
|
|
1364
|
+
let error = null;
|
|
1365
|
+
for (let index = 0; index < config.controlRollouts; index += 1) {
|
|
1366
|
+
const injection = arm === "no-op-control" ? {
|
|
1367
|
+
step: noOpInjectionStep(policy.seed, row.rowId, index, row.recordedCommands),
|
|
1368
|
+
action: config.inertAction
|
|
1369
|
+
} : null;
|
|
1370
|
+
if (injection) injections.push(injection);
|
|
1371
|
+
const result = await input.controls.run({
|
|
1372
|
+
row,
|
|
1373
|
+
arm,
|
|
1374
|
+
rolloutIndex: index,
|
|
1375
|
+
injection
|
|
1376
|
+
});
|
|
1377
|
+
if (!result.succeeded) {
|
|
1378
|
+
error = result.error;
|
|
1379
|
+
break;
|
|
1380
|
+
}
|
|
1381
|
+
const { rollout, tests } = result.value;
|
|
1382
|
+
assertRolloutMatchesRequest(rollout, {
|
|
1383
|
+
row,
|
|
1384
|
+
arm,
|
|
1385
|
+
index
|
|
1386
|
+
});
|
|
1387
|
+
rollouts.push(rollout);
|
|
1388
|
+
summaries.push({
|
|
1389
|
+
arm,
|
|
1390
|
+
index: rollout.index,
|
|
1391
|
+
seed: rollout.seed,
|
|
1392
|
+
policyDigest: rollout.policyDigest,
|
|
1393
|
+
exitStatus: rollout.exitStatus,
|
|
1394
|
+
testsPassed: tests.passed,
|
|
1395
|
+
costUsd: rollout.costProvenance.usd
|
|
1396
|
+
});
|
|
1397
|
+
if (tests.passed) {
|
|
1398
|
+
passes += 1;
|
|
1399
|
+
break;
|
|
1400
|
+
}
|
|
1401
|
+
}
|
|
1402
|
+
const record = {
|
|
1403
|
+
check: "control",
|
|
1404
|
+
arm,
|
|
1405
|
+
rolloutsRun: rollouts.length,
|
|
1406
|
+
passes,
|
|
1407
|
+
injections
|
|
1408
|
+
};
|
|
1409
|
+
if (error !== null) return {
|
|
1410
|
+
outcome: "error",
|
|
1411
|
+
error,
|
|
1412
|
+
record,
|
|
1413
|
+
rollouts,
|
|
1414
|
+
summaries
|
|
1415
|
+
};
|
|
1416
|
+
if (passes > 0) return {
|
|
1417
|
+
outcome: "rescued",
|
|
1418
|
+
error: null,
|
|
1419
|
+
record,
|
|
1420
|
+
rollouts,
|
|
1421
|
+
summaries
|
|
1422
|
+
};
|
|
1423
|
+
return {
|
|
1424
|
+
outcome: "failed-every-rollout",
|
|
1425
|
+
error: null,
|
|
1426
|
+
record,
|
|
1427
|
+
rollouts,
|
|
1428
|
+
summaries
|
|
1429
|
+
};
|
|
1430
|
+
}
|
|
1431
|
+
function assertRolloutMatchesRequest(rollout, request) {
|
|
1432
|
+
if (rollout.arm !== request.arm) throw new AdmissionDenominatorError(`control runner returned a ${rollout.arm} rollout for the ${request.arm} arm`);
|
|
1433
|
+
if (rollout.rowId !== request.row.rowId) throw new AdmissionDenominatorError(`control runner returned a rollout for row ${rollout.rowId} while deciding ${request.row.rowId}`);
|
|
1434
|
+
if (rollout.index !== request.index) throw new AdmissionDenominatorError(`control runner returned rollout index ${rollout.index} for requested index ${request.index}`);
|
|
1435
|
+
}
|
|
1436
|
+
/**
|
|
1437
|
+
* Model cost of every control rollout that ran.
|
|
1438
|
+
*
|
|
1439
|
+
* One unpriced rollout makes the total uncaptured, because summing the priced
|
|
1440
|
+
* ones reports less than the pre-pass spent. Zero rollouts is an observed zero:
|
|
1441
|
+
* nothing ran, so nothing is missing.
|
|
1442
|
+
*/
|
|
1443
|
+
function summarizeControlCost(verdicts) {
|
|
1444
|
+
let usd = 0;
|
|
1445
|
+
for (const verdict of verdicts) for (const rollout of verdict.rollouts) {
|
|
1446
|
+
if (rollout.costUsd === null) return {
|
|
1447
|
+
kind: "uncaptured",
|
|
1448
|
+
usd: null
|
|
1449
|
+
};
|
|
1450
|
+
usd += rollout.costUsd;
|
|
1451
|
+
}
|
|
1452
|
+
return {
|
|
1453
|
+
kind: "observed",
|
|
1454
|
+
usd
|
|
1455
|
+
};
|
|
1456
|
+
}
|
|
1457
|
+
function groupAdmittedByStratum(verdicts) {
|
|
1458
|
+
const grouped = {
|
|
1459
|
+
"clean-exit": [],
|
|
1460
|
+
"command-error": [],
|
|
1461
|
+
"signal-kill": []
|
|
1462
|
+
};
|
|
1463
|
+
for (const verdict of verdicts) {
|
|
1464
|
+
if (!verdict.admitted || verdict.stratum === null) continue;
|
|
1465
|
+
grouped[verdict.stratum].push(verdict.rowId);
|
|
1466
|
+
}
|
|
1467
|
+
return Object.freeze({
|
|
1468
|
+
"clean-exit": Object.freeze(grouped["clean-exit"]),
|
|
1469
|
+
"command-error": Object.freeze(grouped["command-error"]),
|
|
1470
|
+
"signal-kill": Object.freeze(grouped["signal-kill"])
|
|
1471
|
+
});
|
|
1472
|
+
}
|
|
1473
|
+
/** Bounded worker pool that keeps results in input order. */
|
|
1474
|
+
async function mapOrdered(items, concurrency, fn) {
|
|
1475
|
+
const results = new Array(items.length);
|
|
1476
|
+
let next = 0;
|
|
1477
|
+
const workerCount = Math.max(1, Math.min(concurrency, items.length));
|
|
1478
|
+
const workers = Array.from({ length: workerCount }, async () => {
|
|
1479
|
+
for (;;) {
|
|
1480
|
+
const index = next;
|
|
1481
|
+
next += 1;
|
|
1482
|
+
if (index >= items.length) return;
|
|
1483
|
+
const item = items[index];
|
|
1484
|
+
if (item === void 0) return;
|
|
1485
|
+
results[index] = await fn(item);
|
|
1486
|
+
}
|
|
1487
|
+
});
|
|
1488
|
+
await Promise.all(workers);
|
|
1489
|
+
return results;
|
|
1490
|
+
}
|
|
1491
|
+
/** Admitted row ids in one stratum. No call returns them pooled. */
|
|
1492
|
+
function admittedRowIds(report, stratum) {
|
|
1493
|
+
return report.strata[stratum];
|
|
1494
|
+
}
|
|
1495
|
+
function admittedCount(report) {
|
|
1496
|
+
return ADMISSION_STRATA.reduce((total, stratum) => total + report.strata[stratum].length, 0);
|
|
1497
|
+
}
|
|
1498
|
+
/**
|
|
1499
|
+
* Prove a campaign measured the denominator admission published.
|
|
1500
|
+
*
|
|
1501
|
+
* Three ways a denominator moves after the fact, each rejected here: scoring a
|
|
1502
|
+
* row that was never admitted, scoring a row that was never sampled, and
|
|
1503
|
+
* dropping a sampled row instead of scoring it. The third is the one an analyst
|
|
1504
|
+
* can cause on its own — declining a row it cannot solve — so a sampled row
|
|
1505
|
+
* with no outcome is an error, not a smaller `n`.
|
|
1506
|
+
*/
|
|
1507
|
+
function assertDenominatorIntact(input) {
|
|
1508
|
+
const { report, strata, sampled, scored } = input;
|
|
1509
|
+
for (const stratum of strata) if (!ADMISSION_STRATA.includes(stratum)) throw new ValidationError(`unknown stratum in denominator check: ${stratum}`);
|
|
1510
|
+
assertNoDuplicates(sampled, "sampled");
|
|
1511
|
+
assertNoDuplicates(scored, "scored");
|
|
1512
|
+
const eligible = new Set(strata.flatMap((stratum) => [...report.strata[stratum]]));
|
|
1513
|
+
const notAdmitted = sampled.filter((rowId) => !eligible.has(rowId));
|
|
1514
|
+
if (notAdmitted.length > 0) throw new AdmissionDenominatorError(`${notAdmitted.length} sampled row(s) are not admitted in strata ${strata.join(", ")}: ${preview(notAdmitted)}`);
|
|
1515
|
+
const sampledSet = new Set(sampled);
|
|
1516
|
+
const unsampled = scored.filter((rowId) => !sampledSet.has(rowId));
|
|
1517
|
+
if (unsampled.length > 0) throw new AdmissionDenominatorError(`${unsampled.length} scored row(s) were never sampled: ${preview(unsampled)}`);
|
|
1518
|
+
const scoredSet = new Set(scored);
|
|
1519
|
+
const dropped = sampled.filter((rowId) => !scoredSet.has(rowId));
|
|
1520
|
+
if (dropped.length > 0) throw new AdmissionDenominatorError(`denominator shrank by ${dropped.length} row(s): sampled but never scored: ${preview(dropped)}`);
|
|
1521
|
+
}
|
|
1522
|
+
/** A row counted twice raises `n` without measuring anything twice. */
|
|
1523
|
+
function assertNoDuplicates(rowIds, field) {
|
|
1524
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1525
|
+
const duplicates = rowIds.filter((rowId) => {
|
|
1526
|
+
if (seen.has(rowId)) return true;
|
|
1527
|
+
seen.add(rowId);
|
|
1528
|
+
return false;
|
|
1529
|
+
});
|
|
1530
|
+
if (duplicates.length > 0) throw new AdmissionDenominatorError(`${field} lists ${duplicates.length} duplicate row id(s): ${preview([...new Set(duplicates)])}`);
|
|
1531
|
+
}
|
|
1532
|
+
function preview(rowIds) {
|
|
1533
|
+
const shown = rowIds.slice(0, 5).join(", ");
|
|
1534
|
+
return rowIds.length > 5 ? `${shown}, +${rowIds.length - 5} more` : shown;
|
|
1535
|
+
}
|
|
1536
|
+
//#endregion
|
|
1537
|
+
//#region src/trace-repair/admission-contract.ts
|
|
1538
|
+
/**
|
|
1539
|
+
* Admission: the executed checks a row must pass before an analyst is allowed
|
|
1540
|
+
* to see it, and the brand that proves it did.
|
|
1541
|
+
*
|
|
1542
|
+
* Every check is analyst-independent. None of them reads a finding, a label,
|
|
1543
|
+
* or a k, and all of them are anchored at the recorded end state — the point
|
|
1544
|
+
* the agent actually stopped at — so the same evidence admits a row no matter
|
|
1545
|
+
* which step an analyst later blames:
|
|
1546
|
+
*
|
|
1547
|
+
* oracle-determinism the task's own suite returns one verdict on identical
|
|
1548
|
+
* bytes, so its answer is about the state
|
|
1549
|
+
* prefix-fidelity the recorded trajectory replays with at most 10 % of
|
|
1550
|
+
* its steps diverging from their recorded returncode
|
|
1551
|
+
* end-state-fails the held-out suite fails on the recorded end state
|
|
1552
|
+
* no-fix-control 3 of 3 continuations from the end state fail
|
|
1553
|
+
* no-op-control 3 of 3 continuations from the end state, after an
|
|
1554
|
+
* action that changes nothing, fail
|
|
1555
|
+
*
|
|
1556
|
+
* The two controls are what make `Delta-repair` a difference rather than a
|
|
1557
|
+
* rate, and what stops a row where the agent was one free step from success
|
|
1558
|
+
* from counting as a repair the analyst caused. They only do that under a
|
|
1559
|
+
* control that can act: `control-policy.ts` holds the declaration, and the
|
|
1560
|
+
* criteria name which reading of a control pass applies.
|
|
1561
|
+
*
|
|
1562
|
+
* The determinism check sits in front of all of it. Every downstream check
|
|
1563
|
+
* reads the same suite, so a suite whose verdict is not a function of the state
|
|
1564
|
+
* makes each of them a coin flip rather than a measurement.
|
|
1565
|
+
*
|
|
1566
|
+
* This module owns the CONTRACT, not the execution. `runAdmission` in
|
|
1567
|
+
* `./admission` executes the checks against real containers and hands the
|
|
1568
|
+
* measured evidence to `admitRow`, which decides and brands. Splitting it that
|
|
1569
|
+
* way keeps the decision auditable from the recorded numbers alone: a reviewer
|
|
1570
|
+
* can re-derive every admission from the evidence file without re-running a
|
|
1571
|
+
* container.
|
|
1572
|
+
*/
|
|
1573
|
+
const TB_REPAIR_ADMISSION_CRITERIA = Object.freeze({
|
|
1574
|
+
maxPrefixDivergenceRatio: .1,
|
|
1575
|
+
controlRollouts: 3,
|
|
1576
|
+
controlScreening: "enforced"
|
|
1577
|
+
});
|
|
1578
|
+
/**
|
|
1579
|
+
* Decide admission from executed evidence.
|
|
1580
|
+
*
|
|
1581
|
+
* Pure: it opens no container and calls no model. Every threshold it applies
|
|
1582
|
+
* is in `criteria`, and every number it reads is in `evidence`, so an
|
|
1583
|
+
* admission decision is reproducible from the recorded evidence alone.
|
|
1584
|
+
*
|
|
1585
|
+
* It throws, rather than rejecting, when the criteria and the declared control
|
|
1586
|
+
* contradict. A contradiction there is a property of the configuration and not
|
|
1587
|
+
* of the row, so it must stop the run instead of producing one verdict per row
|
|
1588
|
+
* that reads as if a check had been applied.
|
|
1589
|
+
*/
|
|
1590
|
+
function admitRow(evidence, criteria = TB_REPAIR_ADMISSION_CRITERIA) {
|
|
1591
|
+
assertCriteria(criteria);
|
|
1592
|
+
assertControlCalibrated(evidence.controlPolicy, criteria.controlScreening);
|
|
1593
|
+
if (evidence.oracleDeterminism.taskName !== evidence.taskName) throw new ValidationError(`row ${evidence.rowId} is from task ${evidence.taskName} but carries the oracle certification for ${evidence.oracleDeterminism.taskName}`);
|
|
1594
|
+
const screening = {
|
|
1595
|
+
controlPolicy: evidence.controlPolicy,
|
|
1596
|
+
controlScreening: criteria.controlScreening,
|
|
1597
|
+
controlPolicyDigest: evidence.controlPolicy.digest,
|
|
1598
|
+
taskName: evidence.oracleDeterminism.taskName,
|
|
1599
|
+
oracleStable: evidence.oracleDeterminism.stable,
|
|
1600
|
+
oracleFlipRate: evidence.oracleDeterminism.flipRate
|
|
1601
|
+
};
|
|
1602
|
+
const reject = (rejection, detail) => ({
|
|
1603
|
+
admitted: false,
|
|
1604
|
+
rowId: evidence.rowId,
|
|
1605
|
+
screening,
|
|
1606
|
+
rejection,
|
|
1607
|
+
detail
|
|
1608
|
+
});
|
|
1609
|
+
const oracle = evidence.oracleDeterminism;
|
|
1610
|
+
if (!oracle.stable) return reject("task-oracle-nondeterministic", `task ${oracle.taskName} graded byte-identical state inconsistently: flip rate ${(oracle.flipRate * 100).toFixed(1)} % over ${oracle.replicates} replicates (${oracle.detail}). Every check below reads that suite, so none of them measures state.`);
|
|
1611
|
+
if (evidence.steps.length === 0) return reject("empty-trajectory", "the row records no steps");
|
|
1612
|
+
const { stepsReplayed, divergences } = evidence.prefixFidelity;
|
|
1613
|
+
if (stepsReplayed <= 0) return reject("empty-trajectory", "no recorded step was replayed");
|
|
1614
|
+
const prefixDivergenceRatio = divergences / stepsReplayed;
|
|
1615
|
+
if (prefixDivergenceRatio > criteria.maxPrefixDivergenceRatio) return reject("prefix-divergence-too-high", `${divergences}/${stepsReplayed} replayed steps diverged (${(prefixDivergenceRatio * 100).toFixed(1)} %), above the ${(criteria.maxPrefixDivergenceRatio * 100).toFixed(1)} % ceiling`);
|
|
1616
|
+
if (evidence.endStatePassed) return reject("end-state-already-passes", "the held-out suite passes on the recorded end state, so the row records no failure to repair");
|
|
1617
|
+
for (const [name, arm, rescued] of [[
|
|
1618
|
+
"no-fix control",
|
|
1619
|
+
evidence.noFixControl,
|
|
1620
|
+
"no-fix-control-passed"
|
|
1621
|
+
], [
|
|
1622
|
+
"no-op control",
|
|
1623
|
+
evidence.noOpControl,
|
|
1624
|
+
"no-op-control-passed"
|
|
1625
|
+
]]) {
|
|
1626
|
+
if (arm.rollouts !== criteria.controlRollouts) return reject("control-rollouts-short", `${name} ran ${arm.rollouts} rollouts, the criteria pre-register ${criteria.controlRollouts}`);
|
|
1627
|
+
if (arm.policyDigest !== evidence.controlPolicy.digest) return reject("control-policy-mismatch", `${name} reported policy ${arm.policyDigest} but the criteria screen under ${evidence.controlPolicy.id} (${evidence.controlPolicy.digest}); the arm did not run the declared control`);
|
|
1628
|
+
if (arm.passes !== 0) return criteria.controlScreening === "enforced" ? reject(rescued, `${name} passed ${arm.passes}/${arm.rollouts}; the row is repairable by continuing alone`) : reject("control-passed-on-identical-state", `${name} passed ${arm.passes}/${arm.rollouts} under ${evidence.controlPolicy.id}, which executes no command, so it graded the same bytes the end-state check graded as failing. The task's certification reports flip rate ${(oracle.flipRate * 100).toFixed(1)} %; this row is a further flip, not a rescue.`);
|
|
1629
|
+
}
|
|
1630
|
+
return {
|
|
1631
|
+
admitted: true,
|
|
1632
|
+
screening,
|
|
1633
|
+
row: {
|
|
1634
|
+
rowId: evidence.rowId,
|
|
1635
|
+
taskName: evidence.taskName,
|
|
1636
|
+
image: evidence.image,
|
|
1637
|
+
cwd: evidence.cwd,
|
|
1638
|
+
taskStatement: evidence.taskStatement,
|
|
1639
|
+
steps: evidence.steps,
|
|
1640
|
+
criteria,
|
|
1641
|
+
prefixDivergenceRatio,
|
|
1642
|
+
suiteDigest: evidence.suiteDigest,
|
|
1643
|
+
controlPolicy: evidence.controlPolicy,
|
|
1644
|
+
controlScreening: criteria.controlScreening,
|
|
1645
|
+
policyDigest: evidence.controlPolicy.digest,
|
|
1646
|
+
controlRate: evidence.noFixControl.passes / evidence.noFixControl.rollouts,
|
|
1647
|
+
controlRollouts: evidence.noFixControl.rollouts
|
|
1648
|
+
}
|
|
1649
|
+
};
|
|
1650
|
+
}
|
|
1651
|
+
function assertCriteria(criteria) {
|
|
1652
|
+
const { maxPrefixDivergenceRatio, controlRollouts } = criteria;
|
|
1653
|
+
if (!(maxPrefixDivergenceRatio >= 0 && maxPrefixDivergenceRatio <= 1)) throw new ValidationError(`admission maxPrefixDivergenceRatio must be within [0,1], got ${maxPrefixDivergenceRatio}`);
|
|
1654
|
+
if (!Number.isInteger(controlRollouts) || controlRollouts <= 0) throw new ValidationError(`admission controlRollouts must be a positive integer, got ${controlRollouts}`);
|
|
1655
|
+
}
|
|
1656
|
+
//#endregion
|
|
1657
|
+
//#region src/trace-repair/admission-report.ts
|
|
1658
|
+
/** Plain JSON. `JSON.stringify` of this object is the machine-readable artifact. */
|
|
1659
|
+
function admissionArtifact(report) {
|
|
1660
|
+
assertChainReconciles(report.chain);
|
|
1661
|
+
return {
|
|
1662
|
+
version: 1,
|
|
1663
|
+
kind: "tb-repair-admission",
|
|
1664
|
+
generatedAt: report.generatedAt,
|
|
1665
|
+
digest: report.digest,
|
|
1666
|
+
config: report.config,
|
|
1667
|
+
provenance: report.provenance,
|
|
1668
|
+
chain: report.chain,
|
|
1669
|
+
admitted: {
|
|
1670
|
+
"clean-exit": report.strata["clean-exit"],
|
|
1671
|
+
"command-error": report.strata["command-error"],
|
|
1672
|
+
"signal-kill": report.strata["signal-kill"]
|
|
1673
|
+
},
|
|
1674
|
+
controlCost: {
|
|
1675
|
+
kind: report.controlCost.kind,
|
|
1676
|
+
usd: report.controlCost.usd
|
|
1677
|
+
},
|
|
1678
|
+
rows: report.rows
|
|
1679
|
+
};
|
|
1680
|
+
}
|
|
1681
|
+
/** Markdown for the campaign report. Every number here is also in the artifact. */
|
|
1682
|
+
function renderAdmissionReport(artifact, options = {}) {
|
|
1683
|
+
const lines = [];
|
|
1684
|
+
lines.push("# TB-Repair admission", "");
|
|
1685
|
+
lines.push(`${artifact.chain.overall.admitted} of ${artifact.chain.overall.input} rows admitted. Digest \`${artifact.digest}\`.`, "");
|
|
1686
|
+
lines.push(...provenanceSection(artifact));
|
|
1687
|
+
lines.push(...chainSection(artifact.chain.overall, "Denominator chain"));
|
|
1688
|
+
for (const chain of artifact.chain.byStratum) lines.push(...chainSection(chain, `Denominator chain — ${chain.scope}`));
|
|
1689
|
+
lines.push(...stratumSection(artifact));
|
|
1690
|
+
lines.push(...reasonSection(artifact.chain.reasonTotals));
|
|
1691
|
+
const rowLimit = options.rowLimit ?? 0;
|
|
1692
|
+
if (rowLimit > 0) lines.push(...rowSection(artifact.rows, rowLimit));
|
|
1693
|
+
return `${lines.join("\n").trimEnd()}\n`;
|
|
1694
|
+
}
|
|
1695
|
+
function provenanceSection(artifact) {
|
|
1696
|
+
const { provenance, config, controlCost } = artifact;
|
|
1697
|
+
const cost = controlCost.usd === null ? "uncaptured" : `$${controlCost.usd.toFixed(4)}`;
|
|
1698
|
+
return [
|
|
1699
|
+
"## Provenance",
|
|
1700
|
+
"",
|
|
1701
|
+
"| field | value |",
|
|
1702
|
+
"| --- | --- |",
|
|
1703
|
+
`| generated | ${artifact.generatedAt} |`,
|
|
1704
|
+
`| prefix replayer | \`${provenance.replayerId}\` |`,
|
|
1705
|
+
`| end-state oracle | \`${provenance.oracleId}\` |`,
|
|
1706
|
+
`| control runner | \`${provenance.controlRunnerId}\` |`,
|
|
1707
|
+
`| continuation policy | \`${provenance.policyId}\` |`,
|
|
1708
|
+
`| policy model | \`${provenance.policyModel}\` |`,
|
|
1709
|
+
`| policy seed | ${provenance.policySeed} |`,
|
|
1710
|
+
`| policy digest | \`${provenance.policyDigest}\` |`,
|
|
1711
|
+
`| policy step budget | ${provenance.policyStepBudget} model call(s) per control rollout |`,
|
|
1712
|
+
`| control screening | \`${provenance.controlScreening}\` |`,
|
|
1713
|
+
`| certified task oracles | ${formatCertifiedTasks(provenance.certifiedTasks)} |`,
|
|
1714
|
+
`| max prefix divergence | ${formatShare(config.maxPrefixDivergence)} |`,
|
|
1715
|
+
`| control rollouts per arm | ${config.controlRollouts} |`,
|
|
1716
|
+
`| inert action | \`${config.inertAction}\` |`,
|
|
1717
|
+
`| strata admitted | ${config.admitStrata.join(", ")} |`,
|
|
1718
|
+
`| control model cost | ${cost} (${controlCost.kind}) |`,
|
|
1719
|
+
""
|
|
1720
|
+
];
|
|
1721
|
+
}
|
|
1722
|
+
function formatCertifiedTasks(certified) {
|
|
1723
|
+
const entries = Object.entries(certified).sort(([a], [b]) => a.localeCompare(b));
|
|
1724
|
+
if (entries.length === 0) return "none";
|
|
1725
|
+
return entries.map(([task, flipRate]) => `${task} (flip ${formatShare(flipRate)})`).join(", ");
|
|
1726
|
+
}
|
|
1727
|
+
function chainSection(chain, heading) {
|
|
1728
|
+
const lines = [
|
|
1729
|
+
`## ${heading}`,
|
|
1730
|
+
"",
|
|
1731
|
+
"| stage | exclusion reason | entering | excluded | remaining |",
|
|
1732
|
+
"| --- | --- | --- | --- | --- |"
|
|
1733
|
+
];
|
|
1734
|
+
chain.stages.forEach((stage, index) => {
|
|
1735
|
+
lines.push(`| ${index + 1} | \`${stage.reason}\` | ${stage.entering} | ${stage.excluded} | ${stage.remaining} |`);
|
|
1736
|
+
});
|
|
1737
|
+
const excluded = chain.input - chain.admitted;
|
|
1738
|
+
lines.push("", `Input ${chain.input} = admitted ${chain.admitted} + excluded ${excluded}. Admission rate ${formatShare(rate(chain.admitted, chain.input))}.`, "");
|
|
1739
|
+
return lines;
|
|
1740
|
+
}
|
|
1741
|
+
function stratumSection(artifact) {
|
|
1742
|
+
const lines = [
|
|
1743
|
+
"## Strata",
|
|
1744
|
+
"",
|
|
1745
|
+
"| stratum | admitted rows | in this campaign |",
|
|
1746
|
+
"| --- | --- | --- |"
|
|
1747
|
+
];
|
|
1748
|
+
for (const stratum of ADMISSION_STRATA) {
|
|
1749
|
+
const admitted = artifact.admitted[stratum].length;
|
|
1750
|
+
const eligible = artifact.chain.admitStrata.includes(stratum) ? "yes" : "no";
|
|
1751
|
+
lines.push(`| ${stratum} | ${admitted} | ${eligible} |`);
|
|
1752
|
+
}
|
|
1753
|
+
lines.push("", "Sample within a stratum. A command-level repair cannot address a signal kill, so pooling that population with the others averages an addressable class with an unaddressable one.", "");
|
|
1754
|
+
return lines;
|
|
1755
|
+
}
|
|
1756
|
+
function reasonSection(totals) {
|
|
1757
|
+
const lines = [
|
|
1758
|
+
"## Exclusions",
|
|
1759
|
+
"",
|
|
1760
|
+
"| reason | rows | what it means |",
|
|
1761
|
+
"| --- | --- | --- |"
|
|
1762
|
+
];
|
|
1763
|
+
for (const reason of ADMISSION_EXCLUSION_ORDER) lines.push(`| \`${reason}\` | ${totals[reason]} | ${ADMISSION_EXCLUSION_MEANING[reason]} |`);
|
|
1764
|
+
lines.push("");
|
|
1765
|
+
return lines;
|
|
1766
|
+
}
|
|
1767
|
+
function rowSection(rows, limit) {
|
|
1768
|
+
const lines = [
|
|
1769
|
+
"## Rows",
|
|
1770
|
+
"",
|
|
1771
|
+
"| row | task | stratum | final rc | admitted | excluded by |",
|
|
1772
|
+
"| --- | --- | --- | --- | --- | --- |"
|
|
1773
|
+
];
|
|
1774
|
+
for (const row of rows.slice(0, limit)) lines.push(`| \`${row.rowId}\` | ${row.taskName} | ${row.stratum ?? "—"} | ${row.finalReturncode ?? "—"} | ${row.admitted ? "yes" : "no"} | ${row.excludedBy === null ? "—" : `\`${row.excludedBy}\``} |`);
|
|
1775
|
+
if (rows.length > limit) lines.push("", `${rows.length - limit} further rows in the artifact.`);
|
|
1776
|
+
lines.push("");
|
|
1777
|
+
return lines;
|
|
1778
|
+
}
|
|
1779
|
+
function rate(part, total) {
|
|
1780
|
+
return total === 0 ? 0 : part / total;
|
|
1781
|
+
}
|
|
1782
|
+
function formatShare(share) {
|
|
1783
|
+
return `${(share * 100).toFixed(2)}%`;
|
|
1784
|
+
}
|
|
1785
|
+
//#endregion
|
|
1786
|
+
//#region src/trace-repair/blinding.ts
|
|
1787
|
+
/**
|
|
1788
|
+
* What the analyst is shown.
|
|
1789
|
+
*
|
|
1790
|
+
* A blinded trajectory prefix: the task the agent was given and the actions
|
|
1791
|
+
* and observations it produced, and nothing about how the run was judged. No
|
|
1792
|
+
* held-out suite, no suite digest, no control rates, no end-state result, no
|
|
1793
|
+
* label.
|
|
1794
|
+
*
|
|
1795
|
+
* The blinding is by construction rather than by redaction. The returned
|
|
1796
|
+
* object is assembled field by field from an admitted row, so a field added to
|
|
1797
|
+
* `AdmittedRow` later cannot leak into an analyst prompt by default — it has
|
|
1798
|
+
* to be copied here on purpose.
|
|
1799
|
+
*
|
|
1800
|
+
* Requiring an `AdmittedRow` is the other half of the guarantee: a row cannot
|
|
1801
|
+
* be shown to an analyst before the four admission checks have executed and
|
|
1802
|
+
* passed.
|
|
1803
|
+
*/
|
|
1804
|
+
function blindTrajectory(row, options = {}) {
|
|
1805
|
+
const through = options.throughStep ?? row.steps.length;
|
|
1806
|
+
if (!Number.isInteger(through) || through < 1 || through > row.steps.length) throw new ValidationError(`blindTrajectory throughStep must be within [1, ${row.steps.length}], got ${through}`);
|
|
1807
|
+
const steps = row.steps.slice(0, through).map((step) => ({
|
|
1808
|
+
step_id: step.step_id,
|
|
1809
|
+
action: step.action,
|
|
1810
|
+
observation: step.observation
|
|
1811
|
+
}));
|
|
1812
|
+
return {
|
|
1813
|
+
rowId: row.rowId,
|
|
1814
|
+
taskStatement: row.taskStatement,
|
|
1815
|
+
steps,
|
|
1816
|
+
recordedSteps: row.steps.length,
|
|
1817
|
+
maxK: through
|
|
1818
|
+
};
|
|
1819
|
+
}
|
|
1820
|
+
/**
|
|
1821
|
+
* Fields an analyst prompt must never carry. Exported so a consumer that
|
|
1822
|
+
* builds its own prompt can assert against the same list this module honours.
|
|
1823
|
+
*/
|
|
1824
|
+
const BLINDED_FIELDS = Object.freeze([
|
|
1825
|
+
"criteria",
|
|
1826
|
+
"controlPolicy",
|
|
1827
|
+
"controlRate",
|
|
1828
|
+
"controlRollouts",
|
|
1829
|
+
"controlScreening",
|
|
1830
|
+
"suiteDigest",
|
|
1831
|
+
"policyDigest",
|
|
1832
|
+
"prefixDivergenceRatio",
|
|
1833
|
+
"image",
|
|
1834
|
+
"cwd"
|
|
1835
|
+
]);
|
|
1836
|
+
//#endregion
|
|
1837
|
+
//#region src/trace-repair/repair-prompt.ts
|
|
1838
|
+
/**
|
|
1839
|
+
* The one question every repair arm answers.
|
|
1840
|
+
*
|
|
1841
|
+
* Arms differ in what EXECUTES the question — a single chat completion, an
|
|
1842
|
+
* agent harness, a DSPy program — and in nothing else. One prompt module is
|
|
1843
|
+
* what makes a comparison between them an ablation of the execution path
|
|
1844
|
+
* rather than a comparison of two prompts.
|
|
1845
|
+
*
|
|
1846
|
+
* The execution environment is stated, not hinted. The recorded scaffold runs
|
|
1847
|
+
* every action in a fresh `/bin/sh` and takes exactly one top-level statement,
|
|
1848
|
+
* so an arm that is not told that writes bash-only syntax and dies at
|
|
1849
|
+
* execution for a reason that has nothing to do with whether it localized the
|
|
1850
|
+
* failure. Stating the shell leaks no label: the held-out suite, the control
|
|
1851
|
+
* rates and the end-state verdict all stay out, and the prompt is built from a
|
|
1852
|
+
* `BlindedTrajectoryPrefix`, which cannot carry them.
|
|
1853
|
+
*
|
|
1854
|
+
* The action caps in the prose are read from the budget the grader enforces,
|
|
1855
|
+
* so the instruction and the rejection can never drift apart.
|
|
1856
|
+
*/
|
|
1857
|
+
const REPAIR_QUESTION = "This coding agent ran and did not finish the task. Name the ONE recorded step whose action you would replace, and give the single shell action to run instead of it.";
|
|
1858
|
+
/**
|
|
1859
|
+
* The task definition. Parameterised by the budget so the stated caps are the
|
|
1860
|
+
* enforced caps.
|
|
1861
|
+
*/
|
|
1862
|
+
function repairTaskPolicy(budget = SCAFFOLD_INTERVENTION_BUDGET) {
|
|
1863
|
+
return Object.freeze([
|
|
1864
|
+
"You are given a task statement and the complete recorded trajectory of an agent that attempted it and failed.",
|
|
1865
|
+
"Your answer is executed. The action you return is run inside the real task container, at the point in the trajectory you name, and the task's held-out test suite is then run against the resulting state. You are scored on whether the suite passes, never on how your explanation reads.",
|
|
1866
|
+
"",
|
|
1867
|
+
"HOW YOUR ACTION IS EXECUTED — read this before writing it:",
|
|
1868
|
+
"- The action runs in a FRESH `/bin/sh`, which is dash on this image, not bash. C-style `for ((i=0;i<n;i++))`, `[[ ]]`, arrays, and `local` are bash-only and will fail with a syntax error.",
|
|
1869
|
+
`- The scaffold accepts exactly ${budget.maxStatements === 1 ? "ONE top-level statement" : `${budget.maxStatements} top-level statements`}. Wrap multiple commands in a single \`{ ...; }\` block or chain them with \`&&\`.`,
|
|
1870
|
+
`- The action is at most ${budget.maxBytes} bytes, and must not be a no-op such as \`true\`, \`:\` or an empty string.`,
|
|
1871
|
+
"- It runs from the task working directory, as root, with the network available.",
|
|
1872
|
+
"- Prefer writing a complete correct file with a heredoc over patching in place; you cannot see the result of your own action, so it must work on the first attempt.",
|
|
1873
|
+
"Name the step whose action you would replace as `k`, using the step_id shown in the trajectory. Choosing the last step means \"instead of the last thing the agent did, do this\". You are not repairing the agent's reasoning; you are producing the state the suite requires.",
|
|
1874
|
+
"Return no finding only when no single replacement action could plausibly make the suite pass."
|
|
1875
|
+
]);
|
|
1876
|
+
}
|
|
1877
|
+
/** The reply grammar, stated to an arm that answers in one JSON object. */
|
|
1878
|
+
const REPAIR_CONTRACT_LINES = Object.freeze([
|
|
1879
|
+
"OUTPUT CONTRACT — reply with ONE fenced JSON object and nothing else:",
|
|
1880
|
+
"```json",
|
|
1881
|
+
"{\"answer\": \"<one sentence on what went wrong>\", \"findings\": [{\"k\": <int>, \"failure_claim\": \"<what went wrong at step k>\", \"intervention\": {\"kind\": \"shell\", \"action\": \"<the single shell action to run instead>\"}}]}",
|
|
1882
|
+
"```",
|
|
1883
|
+
"Rules:",
|
|
1884
|
+
"- `findings` carries AT MOST ONE object. Never more.",
|
|
1885
|
+
"- `k` is an integer step_id present in the trajectory below.",
|
|
1886
|
+
"- `failure_claim` is a non-empty string. It is recorded and never scored, so spend your effort on the action.",
|
|
1887
|
+
"- `intervention.kind` is the string \"shell\", or \"edit\" when the action authors a file with a heredoc.",
|
|
1888
|
+
"- `intervention.action` is the exact shell text to execute. It is not a description and not a diff.",
|
|
1889
|
+
"- Return `\"findings\": []` only to say that no single action could repair this run."
|
|
1890
|
+
]);
|
|
1891
|
+
/** The same grammar restated for the bounded repair turn, which never carries
|
|
1892
|
+
* the trajectory again. */
|
|
1893
|
+
const REPAIR_REPAIR_CONTRACT_LINES = Object.freeze([
|
|
1894
|
+
"Your previous reply did not satisfy the output contract. Resend the answer, and nothing else, as ONE fenced JSON object:",
|
|
1895
|
+
"```json",
|
|
1896
|
+
"{\"answer\": \"<one sentence>\", \"findings\": [{\"k\": <int>, \"failure_claim\": \"<string>\", \"intervention\": {\"kind\": \"shell\", \"action\": \"<shell text>\"}}]}",
|
|
1897
|
+
"```",
|
|
1898
|
+
"`findings` carries at most one object. Do not restate the trajectory."
|
|
1899
|
+
]);
|
|
1900
|
+
/** The trajectory as an arm sees it: actions, observations, nothing about grading. */
|
|
1901
|
+
function renderRepairTrajectory(prefix) {
|
|
1902
|
+
return prefix.steps.map((step) => [
|
|
1903
|
+
`--- step_id ${step.step_id} ---`,
|
|
1904
|
+
"ACTION:",
|
|
1905
|
+
step.action,
|
|
1906
|
+
"OBSERVATION:",
|
|
1907
|
+
step.observation === null ? "(no observation recorded)" : step.observation
|
|
1908
|
+
].join("\n")).join("\n\n");
|
|
1909
|
+
}
|
|
1910
|
+
function repairTrajectoryHeader(prefix) {
|
|
1911
|
+
return `RECORDED TRAJECTORY (${prefix.steps.length} steps; valid k is any step_id below, the last is ${prefix.maxK}):`;
|
|
1912
|
+
}
|
|
1913
|
+
/** The task definition an arm receives: the policy plus the statement the
|
|
1914
|
+
* recorded agent was given. */
|
|
1915
|
+
function repairTaskDefinition(prefix, budget = SCAFFOLD_INTERVENTION_BUDGET) {
|
|
1916
|
+
return [
|
|
1917
|
+
...repairTaskPolicy(budget),
|
|
1918
|
+
"",
|
|
1919
|
+
"TASK STATEMENT GIVEN TO THE AGENT:",
|
|
1920
|
+
prefix.taskStatement
|
|
1921
|
+
].join("\n");
|
|
1922
|
+
}
|
|
1923
|
+
/**
|
|
1924
|
+
* Digest of the question and task policy every arm shares.
|
|
1925
|
+
*
|
|
1926
|
+
* This is the part of the prompt that is equal by construction across arms:
|
|
1927
|
+
* the question, the execution rules, and the budget the caps are read from.
|
|
1928
|
+
* An arm's own contract text — its output grammar, its typed-signature
|
|
1929
|
+
* instructions — is deliberately outside it, because arms differ there.
|
|
1930
|
+
*/
|
|
1931
|
+
function repairQuestionSha256(budget = SCAFFOLD_INTERVENTION_BUDGET) {
|
|
1932
|
+
return createHash("sha256").update(JSON.stringify({
|
|
1933
|
+
kind: "tb-repair-analyst-question",
|
|
1934
|
+
question: REPAIR_QUESTION,
|
|
1935
|
+
taskPolicy: repairTaskPolicy(budget),
|
|
1936
|
+
budget
|
|
1937
|
+
})).digest("hex");
|
|
1938
|
+
}
|
|
1939
|
+
/**
|
|
1940
|
+
* Digest of the composed question one arm answered.
|
|
1941
|
+
*
|
|
1942
|
+
* Covers the shared question and the arm's own declared contract text, so two
|
|
1943
|
+
* arms that ask materially different composed questions — a JSON grammar
|
|
1944
|
+
* versus a typed SUBMIT signature — stamp different digests, and two arms
|
|
1945
|
+
* that ask the identical composed question share one.
|
|
1946
|
+
*/
|
|
1947
|
+
function repairArmPromptSha256(budget, contract) {
|
|
1948
|
+
return createHash("sha256").update(JSON.stringify({
|
|
1949
|
+
kind: "tb-repair-analyst-prompt",
|
|
1950
|
+
question: REPAIR_QUESTION,
|
|
1951
|
+
taskPolicy: repairTaskPolicy(budget),
|
|
1952
|
+
budget,
|
|
1953
|
+
contract
|
|
1954
|
+
})).digest("hex");
|
|
1955
|
+
}
|
|
1956
|
+
//#endregion
|
|
1957
|
+
//#region src/trace-repair/analyst-arm.ts
|
|
1958
|
+
/**
|
|
1959
|
+
* What an analyst arm is, and what every arm owes the comparison.
|
|
1960
|
+
*
|
|
1961
|
+
* An arm is one way of EXECUTING the repair question: a single chat
|
|
1962
|
+
* completion, an agent harness, a DSPy program. The question, the reply
|
|
1963
|
+
* grammar, the action budget and the bounded repair turn belong to the
|
|
1964
|
+
* comparison, not to the arm — so they live here and every arm inherits them.
|
|
1965
|
+
*
|
|
1966
|
+
* Three rules this module makes structural rather than customary:
|
|
1967
|
+
*
|
|
1968
|
+
* one contract an arm returns `RepairFinding` or an honest decline,
|
|
1969
|
+
* and nothing else parses. An arm that produced neither
|
|
1970
|
+
* fails loud with a typed reason; nothing is defaulted.
|
|
1971
|
+
* one budget `askRepairArm` measures every action against the
|
|
1972
|
+
* scaffold budget and records the measurement. The grader
|
|
1973
|
+
* remains the single authority that rejects, so a
|
|
1974
|
+
* violation is reported here and refused there — never
|
|
1975
|
+
* silently trimmed to fit.
|
|
1976
|
+
* one repair turn arms declare how many bounded retries a malformed reply
|
|
1977
|
+
* earns. `repairArmAsymmetries` refuses a set whose arms
|
|
1978
|
+
* disagree, because a second attempt is a second sample
|
|
1979
|
+
* the other arms never got.
|
|
1980
|
+
*
|
|
1981
|
+
* What arms are ALLOWED to differ in is declared, not hidden: `affordances`
|
|
1982
|
+
* names what an arm can do that another cannot — read the trajectory through
|
|
1983
|
+
* a code interpreter, take internal turns — and `repairArmAsymmetries` renders
|
|
1984
|
+
* those differences so a reader sees them beside the result instead of having
|
|
1985
|
+
* to infer them from two runners' source.
|
|
1986
|
+
*/
|
|
1987
|
+
/**
|
|
1988
|
+
* Ask one arm about one admitted row.
|
|
1989
|
+
*
|
|
1990
|
+
* Blinding, prompt identity and budget measurement happen here, once, for
|
|
1991
|
+
* every arm. An arm that wants a different question does not get one.
|
|
1992
|
+
*/
|
|
1993
|
+
async function askRepairArm(options) {
|
|
1994
|
+
const { arm, row } = options;
|
|
1995
|
+
assertDeclaration(arm.declaration);
|
|
1996
|
+
const now = options.now ?? Date.now;
|
|
1997
|
+
const startedMs = now();
|
|
1998
|
+
const prefix = blindTrajectory(row, options.throughStep === void 0 ? {} : { throughStep: options.throughStep });
|
|
1999
|
+
const reply = await arm.ask({
|
|
2000
|
+
prefix,
|
|
2001
|
+
...options.signal ? { signal: options.signal } : {}
|
|
2002
|
+
});
|
|
2003
|
+
assertReplyWithinDeclaration(arm.declaration, reply);
|
|
2004
|
+
const budget = reply.status === "finding" ? checkInterventionBudget(reply.intervention.action, reply.intervention.kind, arm.declaration.budget) : null;
|
|
2005
|
+
return {
|
|
2006
|
+
rowId: row.rowId,
|
|
2007
|
+
armId: arm.declaration.id,
|
|
2008
|
+
promptSha256: repairArmPromptSha256(arm.declaration.budget, arm.declaration.promptContract),
|
|
2009
|
+
reply,
|
|
2010
|
+
budget,
|
|
2011
|
+
wallMs: now() - startedMs
|
|
2012
|
+
};
|
|
2013
|
+
}
|
|
2014
|
+
/**
|
|
2015
|
+
* The answer in the grammar the grader consumes.
|
|
2016
|
+
*
|
|
2017
|
+
* Null for a failure: a run that could not answer has no answer to grade, and
|
|
2018
|
+
* turning it into a decline would credit an arm with an honest null it never
|
|
2019
|
+
* produced.
|
|
2020
|
+
*/
|
|
2021
|
+
function repairArmResponse(answer) {
|
|
2022
|
+
const { reply } = answer;
|
|
2023
|
+
if (reply.status === "declined") return { kind: "no-decisive-failure" };
|
|
2024
|
+
if (reply.status === "failed") return null;
|
|
2025
|
+
return {
|
|
2026
|
+
kind: "finding",
|
|
2027
|
+
k: reply.k,
|
|
2028
|
+
failureClaim: reply.failureClaim,
|
|
2029
|
+
intervention: reply.intervention
|
|
2030
|
+
};
|
|
2031
|
+
}
|
|
2032
|
+
/**
|
|
2033
|
+
* Refuse a set of arms that cannot be compared, and describe what still
|
|
2034
|
+
* differs between the ones that can.
|
|
2035
|
+
*
|
|
2036
|
+
* Two properties are hard: the arms measure actions against the same budget,
|
|
2037
|
+
* and a malformed reply earns the same number of retries everywhere. Both are
|
|
2038
|
+
* things that would move a score without moving the thing being measured.
|
|
2039
|
+
*
|
|
2040
|
+
* Certification is the third: a set where some arms run certified text and
|
|
2041
|
+
* others do not is refused outright. An optimisation applies to every arm or
|
|
2042
|
+
* to none — a mixed set measures the optimisation, then reports it as the
|
|
2043
|
+
* harness.
|
|
2044
|
+
*/
|
|
2045
|
+
function repairArmAsymmetries(arms, options = {}) {
|
|
2046
|
+
if (arms.length === 0) throw new ValidationError("a repair-arm comparison needs at least one arm");
|
|
2047
|
+
const declarations = arms.map((arm) => arm.declaration);
|
|
2048
|
+
for (const declaration of declarations) assertDeclaration(declaration);
|
|
2049
|
+
const { ids, repairTurns } = assertEqualDeclarativeTerms("repair arm", declarations.map((declaration) => ({
|
|
2050
|
+
id: declaration.id,
|
|
2051
|
+
repairTurns: declaration.repairTurns
|
|
2052
|
+
})));
|
|
2053
|
+
const budget = options.budget ?? declarations[0].budget;
|
|
2054
|
+
const mismatched = declarations.find((declaration) => !sameBudget(declaration.budget, budget));
|
|
2055
|
+
if (mismatched) throw new ValidationError(`arm '${mismatched.id}' measures actions against ${JSON.stringify(mismatched.budget)}, the comparison runs ${JSON.stringify(budget)}; one arm buying a bigger action than another is a difference in what was asked, not in what answered`);
|
|
2056
|
+
const certified = declarations.filter((declaration) => declaration.certification.kind === "certified");
|
|
2057
|
+
if (certified.length > 0 && certified.length !== declarations.length) throw new ValidationError(`${certified.length}/${declarations.length} arms run certified text (${certified.map((declaration) => declaration.id).join(", ")}); an optimisation applies to every arm or to none`);
|
|
2058
|
+
return {
|
|
2059
|
+
armIds: ids,
|
|
2060
|
+
sharedAffordances: ALL_AFFORDANCES.filter((affordance) => declarations.every((declaration) => declaration.affordances.includes(affordance))),
|
|
2061
|
+
asymmetries: declarations.map((declaration) => ({
|
|
2062
|
+
armId: declaration.id,
|
|
2063
|
+
extraAffordances: ALL_AFFORDANCES.filter((affordance) => declaration.affordances.includes(affordance) && declarations.some((other) => !other.affordances.includes(affordance))),
|
|
2064
|
+
missingAffordances: ALL_AFFORDANCES.filter((affordance) => !declaration.affordances.includes(affordance) && declarations.some((other) => other.affordances.includes(affordance))),
|
|
2065
|
+
certification: declaration.certification,
|
|
2066
|
+
promptSha256: repairArmPromptSha256(declaration.budget, declaration.promptContract)
|
|
2067
|
+
})),
|
|
2068
|
+
noArmCertified: certified.length === 0,
|
|
2069
|
+
budget,
|
|
2070
|
+
repairTurns,
|
|
2071
|
+
questionSha256: repairQuestionSha256(budget)
|
|
2072
|
+
};
|
|
2073
|
+
}
|
|
2074
|
+
const ALL_AFFORDANCES = Object.freeze([
|
|
2075
|
+
"inline-trajectory",
|
|
2076
|
+
"trajectory-tools",
|
|
2077
|
+
"code-interpreter",
|
|
2078
|
+
"agent-loop"
|
|
2079
|
+
]);
|
|
2080
|
+
function sameBudget(a, b) {
|
|
2081
|
+
return a.maxBytes === b.maxBytes && a.maxStatements === b.maxStatements && a.maxHeredocs === b.maxHeredocs;
|
|
2082
|
+
}
|
|
2083
|
+
function assertDeclaration(declaration) {
|
|
2084
|
+
if (typeof declaration.id !== "string" || declaration.id.trim() !== declaration.id || !declaration.id) throw new ValidationError("a repair arm id must be a trimmed non-empty string");
|
|
2085
|
+
if (typeof declaration.execution !== "string" || !declaration.execution.trim()) throw new ValidationError(`arm '${declaration.id}' must state what executes its answer`);
|
|
2086
|
+
if (!Number.isInteger(declaration.repairTurns) || declaration.repairTurns < 0) throw new ValidationError(`arm '${declaration.id}' repairTurns must be a non-negative integer, got ${declaration.repairTurns}`);
|
|
2087
|
+
if (declaration.certification.kind === "none" && (typeof declaration.certification.reason !== "string" || !declaration.certification.reason.trim())) throw new ValidationError(`arm '${declaration.id}' carries no certification and must say why in one sentence`);
|
|
2088
|
+
if (!Array.isArray(declaration.promptContract) || declaration.promptContract.length === 0 || declaration.promptContract.some((line) => typeof line !== "string")) throw new ValidationError(`arm '${declaration.id}' must declare its contract text as a non-empty array of strings; the per-arm prompt digest is computed from it`);
|
|
2089
|
+
for (const affordance of declaration.affordances) if (!ALL_AFFORDANCES.includes(affordance)) throw new ValidationError(`arm '${declaration.id}' declares unknown affordance '${affordance}'`);
|
|
2090
|
+
}
|
|
2091
|
+
/** An arm that reports more repair turns than it declared is not the arm the
|
|
2092
|
+
* comparison admitted, so the answer never reaches a grader. */
|
|
2093
|
+
function assertReplyWithinDeclaration(declaration, reply) {
|
|
2094
|
+
if (reply.repair.attempted && declaration.repairTurns === 0) throw new ValidationError(`arm '${declaration.id}' declares no bounded repair turn but took one`);
|
|
2095
|
+
if (reply.status === "failed" && !reply.failure.trim()) throw new ValidationError(`arm '${declaration.id}' failed without stating why`);
|
|
2096
|
+
if (reply.status === "finding" && !reply.intervention.action) throw new ValidationError(`arm '${declaration.id}' returned a finding with an empty intervention`);
|
|
2097
|
+
}
|
|
2098
|
+
//#endregion
|
|
2099
|
+
//#region src/trace-repair/analyst-response.ts
|
|
2100
|
+
/**
|
|
2101
|
+
* What an analyst is allowed to say about a blinded trajectory prefix.
|
|
2102
|
+
*
|
|
2103
|
+
* Exactly one of two answers:
|
|
2104
|
+
*
|
|
2105
|
+
* a finding {k, failureClaim, intervention}
|
|
2106
|
+
* the literal string no-decisive-failure
|
|
2107
|
+
*
|
|
2108
|
+
* Nothing else parses. A finding names one step, states what went wrong
|
|
2109
|
+
* there, and supplies one action to run instead. Declining is a real answer
|
|
2110
|
+
* with its own cell in the funnel, not a parse failure — an admitted row has
|
|
2111
|
+
* a failure by construction, but it need not have a single-action repair, and
|
|
2112
|
+
* an analyst that says so is answering the question it was asked.
|
|
2113
|
+
*/
|
|
2114
|
+
/** The literal an analyst returns when no single step carries the failure. */
|
|
2115
|
+
const NO_DECISIVE_FAILURE = "no-decisive-failure";
|
|
2116
|
+
/**
|
|
2117
|
+
* Read an analyst reply.
|
|
2118
|
+
*
|
|
2119
|
+
* Accepts the bare literal, or a JSON object carrying one finding. The reply
|
|
2120
|
+
* is untrusted text, so every field is checked and nothing is defaulted: a
|
|
2121
|
+
* missing `k` is a parse failure, never step 1.
|
|
2122
|
+
*/
|
|
2123
|
+
function parseAnalystResponse(reply) {
|
|
2124
|
+
const trimmed = reply.trim();
|
|
2125
|
+
if (trimmed.length === 0) return {
|
|
2126
|
+
succeeded: false,
|
|
2127
|
+
failure: "unreadable",
|
|
2128
|
+
detail: "the reply is empty"
|
|
2129
|
+
};
|
|
2130
|
+
if (trimmed === "no-decisive-failure") return {
|
|
2131
|
+
succeeded: true,
|
|
2132
|
+
value: { kind: "no-decisive-failure" }
|
|
2133
|
+
};
|
|
2134
|
+
const json = extractJsonObject(trimmed);
|
|
2135
|
+
if (!json) return {
|
|
2136
|
+
succeeded: false,
|
|
2137
|
+
failure: "unreadable",
|
|
2138
|
+
detail: `expected the literal "${NO_DECISIVE_FAILURE}" or one JSON object`
|
|
2139
|
+
};
|
|
2140
|
+
if (json.count > 1) return {
|
|
2141
|
+
succeeded: false,
|
|
2142
|
+
failure: "not-a-single-answer",
|
|
2143
|
+
detail: `the reply carries ${json.count} findings; exactly one is allowed`
|
|
2144
|
+
};
|
|
2145
|
+
const body = json.value;
|
|
2146
|
+
if (body.no_decisive_failure === true || body.finding === "no-decisive-failure") return {
|
|
2147
|
+
succeeded: true,
|
|
2148
|
+
value: { kind: "no-decisive-failure" }
|
|
2149
|
+
};
|
|
2150
|
+
const k = body.k;
|
|
2151
|
+
if (typeof k !== "number" || !Number.isInteger(k)) return {
|
|
2152
|
+
succeeded: false,
|
|
2153
|
+
failure: "missing-k",
|
|
2154
|
+
detail: `k must be an integer, got ${describe$2(k)}`
|
|
2155
|
+
};
|
|
2156
|
+
const failureClaim = body.failure_claim ?? body.failureClaim;
|
|
2157
|
+
if (typeof failureClaim !== "string" || failureClaim.trim().length === 0) return {
|
|
2158
|
+
succeeded: false,
|
|
2159
|
+
failure: "missing-failure-claim",
|
|
2160
|
+
detail: "failure_claim must be a non-empty string"
|
|
2161
|
+
};
|
|
2162
|
+
const raw = body.intervention;
|
|
2163
|
+
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return {
|
|
2164
|
+
succeeded: false,
|
|
2165
|
+
failure: "missing-intervention",
|
|
2166
|
+
detail: "intervention must be an object with kind and action"
|
|
2167
|
+
};
|
|
2168
|
+
const intervention = raw;
|
|
2169
|
+
const action = intervention.action;
|
|
2170
|
+
if (typeof action !== "string") return {
|
|
2171
|
+
succeeded: false,
|
|
2172
|
+
failure: "missing-intervention",
|
|
2173
|
+
detail: "intervention.action must be a string"
|
|
2174
|
+
};
|
|
2175
|
+
const kind = intervention.kind;
|
|
2176
|
+
if (kind !== "shell" && kind !== "edit") return {
|
|
2177
|
+
succeeded: false,
|
|
2178
|
+
failure: "unknown-intervention-kind",
|
|
2179
|
+
detail: `intervention.kind must be "shell" or "edit", got ${describe$2(kind)}`
|
|
2180
|
+
};
|
|
2181
|
+
return {
|
|
2182
|
+
succeeded: true,
|
|
2183
|
+
value: {
|
|
2184
|
+
kind: "finding",
|
|
2185
|
+
k,
|
|
2186
|
+
failureClaim: failureClaim.trim(),
|
|
2187
|
+
intervention: {
|
|
2188
|
+
kind,
|
|
2189
|
+
action
|
|
2190
|
+
}
|
|
2191
|
+
}
|
|
2192
|
+
};
|
|
2193
|
+
}
|
|
2194
|
+
/** Build a finding directly, for callers that already hold typed fields. */
|
|
2195
|
+
function repairFinding(input) {
|
|
2196
|
+
if (!Number.isInteger(input.k) || input.k < 1) throw new ValidationError(`repair finding k must be a positive integer, got ${input.k}`);
|
|
2197
|
+
if (input.failureClaim.trim().length === 0) throw new ValidationError("repair finding requires a non-empty failure claim");
|
|
2198
|
+
return {
|
|
2199
|
+
kind: "finding",
|
|
2200
|
+
k: input.k,
|
|
2201
|
+
failureClaim: input.failureClaim.trim(),
|
|
2202
|
+
intervention: input.intervention
|
|
2203
|
+
};
|
|
2204
|
+
}
|
|
2205
|
+
function describe$2(value) {
|
|
2206
|
+
if (value === null) return "null";
|
|
2207
|
+
if (value === void 0) return "undefined";
|
|
2208
|
+
return typeof value === "string" ? JSON.stringify(value) : String(value);
|
|
2209
|
+
}
|
|
2210
|
+
/**
|
|
2211
|
+
* Pull the single JSON object out of a reply, tolerating a fenced block and
|
|
2212
|
+
* surrounding prose. `count` reports how many top-level objects were found so
|
|
2213
|
+
* a reply hedging with several findings is rejected rather than silently
|
|
2214
|
+
* reduced to the first one.
|
|
2215
|
+
*/
|
|
2216
|
+
function extractJsonObject(reply) {
|
|
2217
|
+
const fenced = [...reply.matchAll(/```(?:json)?\s*\n([\s\S]*?)```/g)].map((m) => m[1].trim());
|
|
2218
|
+
const candidates = fenced.length > 0 ? fenced : [reply];
|
|
2219
|
+
const parsed = [];
|
|
2220
|
+
for (const candidate of candidates) for (const slice of topLevelObjectSlices(candidate)) try {
|
|
2221
|
+
const value = JSON.parse(slice);
|
|
2222
|
+
if (Array.isArray(value)) {
|
|
2223
|
+
for (const item of value) if (isPlainObject(item)) parsed.push(item);
|
|
2224
|
+
} else if (isPlainObject(value)) parsed.push(value);
|
|
2225
|
+
} catch {}
|
|
2226
|
+
if (parsed.length === 0) return null;
|
|
2227
|
+
return {
|
|
2228
|
+
value: parsed[0],
|
|
2229
|
+
count: parsed.length
|
|
2230
|
+
};
|
|
2231
|
+
}
|
|
2232
|
+
function isPlainObject(value) {
|
|
2233
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2234
|
+
}
|
|
2235
|
+
/** Balanced `{…}` and `[…]` slices at the top level of a candidate string. */
|
|
2236
|
+
function topLevelObjectSlices(text) {
|
|
2237
|
+
const slices = [];
|
|
2238
|
+
let depth = 0;
|
|
2239
|
+
let start = -1;
|
|
2240
|
+
let inString = false;
|
|
2241
|
+
let escaped = false;
|
|
2242
|
+
for (let i = 0; i < text.length; i += 1) {
|
|
2243
|
+
const char = text[i];
|
|
2244
|
+
if (inString) {
|
|
2245
|
+
if (escaped) escaped = false;
|
|
2246
|
+
else if (char === "\\") escaped = true;
|
|
2247
|
+
else if (char === "\"") inString = false;
|
|
2248
|
+
continue;
|
|
2249
|
+
}
|
|
2250
|
+
if (char === "\"") {
|
|
2251
|
+
inString = true;
|
|
2252
|
+
continue;
|
|
2253
|
+
}
|
|
2254
|
+
if (char === "{" || char === "[") {
|
|
2255
|
+
if (depth === 0) start = i;
|
|
2256
|
+
depth += 1;
|
|
2257
|
+
continue;
|
|
2258
|
+
}
|
|
2259
|
+
if (char === "}" || char === "]") {
|
|
2260
|
+
depth -= 1;
|
|
2261
|
+
if (depth === 0 && start >= 0) {
|
|
2262
|
+
slices.push(text.slice(start, i + 1));
|
|
2263
|
+
start = -1;
|
|
2264
|
+
}
|
|
2265
|
+
if (depth < 0) depth = 0;
|
|
2266
|
+
}
|
|
2267
|
+
}
|
|
2268
|
+
return slices;
|
|
2269
|
+
}
|
|
2270
|
+
//#endregion
|
|
2271
|
+
//#region src/trace-repair/arm-completion.ts
|
|
2272
|
+
function createCompletionRepairArm(options) {
|
|
2273
|
+
if (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0) throw new ValidationError(`repair arm '${options.id}' timeoutMs must be a positive safe integer, got ${options.timeoutMs}`);
|
|
2274
|
+
const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
|
|
2275
|
+
return {
|
|
2276
|
+
declaration: {
|
|
2277
|
+
id: options.id,
|
|
2278
|
+
execution: options.execution,
|
|
2279
|
+
certification: options.certification,
|
|
2280
|
+
budget,
|
|
2281
|
+
repairTurns: 1,
|
|
2282
|
+
affordances: options.affordances,
|
|
2283
|
+
promptContract: [...REPAIR_CONTRACT_LINES, ...REPAIR_REPAIR_CONTRACT_LINES]
|
|
2284
|
+
},
|
|
2285
|
+
async ask(request) {
|
|
2286
|
+
const prompt = buildPrimePrompt({
|
|
2287
|
+
question: REPAIR_QUESTION,
|
|
2288
|
+
taskDefinition: repairTaskDefinition(request.prefix, budget),
|
|
2289
|
+
contractLines: REPAIR_CONTRACT_LINES,
|
|
2290
|
+
trajectoryHeader: repairTrajectoryHeader(request.prefix),
|
|
2291
|
+
renderedTrajectory: renderRepairTrajectory(request.prefix)
|
|
2292
|
+
});
|
|
2293
|
+
const outcome = await runPrimeExchange({
|
|
2294
|
+
contract: replyContract(request.prefix),
|
|
2295
|
+
prompt,
|
|
2296
|
+
transport: options.transport,
|
|
2297
|
+
url: options.url,
|
|
2298
|
+
model: options.model,
|
|
2299
|
+
timeoutMs: options.timeoutMs,
|
|
2300
|
+
repair: true,
|
|
2301
|
+
...request.signal ? { signal: request.signal } : {}
|
|
2302
|
+
});
|
|
2303
|
+
const usage = armUsage(outcome.usage, options.pricing);
|
|
2304
|
+
if (!outcome.ok) return {
|
|
2305
|
+
status: "failed",
|
|
2306
|
+
failure: `${outcome.failure.kind}: ${outcome.failure.message}`,
|
|
2307
|
+
answer: null,
|
|
2308
|
+
rejectedRows: [],
|
|
2309
|
+
repair: outcome.repair,
|
|
2310
|
+
usage
|
|
2311
|
+
};
|
|
2312
|
+
const row = outcome.rows[0];
|
|
2313
|
+
if (row === void 0) return {
|
|
2314
|
+
status: "declined",
|
|
2315
|
+
answer: outcome.answer,
|
|
2316
|
+
reportedRows: outcome.reportedRows,
|
|
2317
|
+
rejectedRows: outcome.rejected,
|
|
2318
|
+
repair: outcome.repair,
|
|
2319
|
+
usage
|
|
2320
|
+
};
|
|
2321
|
+
return {
|
|
2322
|
+
status: "finding",
|
|
2323
|
+
k: row.k,
|
|
2324
|
+
failureClaim: row.failureClaim,
|
|
2325
|
+
intervention: {
|
|
2326
|
+
kind: row.kind,
|
|
2327
|
+
action: row.action
|
|
2328
|
+
},
|
|
2329
|
+
answer: outcome.answer,
|
|
2330
|
+
reportedRows: outcome.reportedRows,
|
|
2331
|
+
rejectedRows: outcome.rejected,
|
|
2332
|
+
repair: outcome.repair,
|
|
2333
|
+
usage
|
|
2334
|
+
};
|
|
2335
|
+
}
|
|
2336
|
+
};
|
|
2337
|
+
}
|
|
2338
|
+
/**
|
|
2339
|
+
* Decode one reply row.
|
|
2340
|
+
*
|
|
2341
|
+
* Every field is checked and nothing is defaulted: a `k` outside the recorded
|
|
2342
|
+
* step ids is a rejected row rather than a clamped one, because an arm that
|
|
2343
|
+
* names a step the recording does not hold has not localized anything.
|
|
2344
|
+
*/
|
|
2345
|
+
function replyContract(prefix) {
|
|
2346
|
+
const validSteps = prefix.steps.map((step) => step.step_id);
|
|
2347
|
+
return {
|
|
2348
|
+
rowsField: "findings",
|
|
2349
|
+
contractLines: REPAIR_CONTRACT_LINES,
|
|
2350
|
+
repairContractLines: REPAIR_REPAIR_CONTRACT_LINES,
|
|
2351
|
+
maxRows: 1,
|
|
2352
|
+
decodeRow(row, index) {
|
|
2353
|
+
if (typeof row !== "object" || row === null || Array.isArray(row)) return {
|
|
2354
|
+
ok: false,
|
|
2355
|
+
reason: `row ${index} is not an object`
|
|
2356
|
+
};
|
|
2357
|
+
const record = row;
|
|
2358
|
+
const k = record.k;
|
|
2359
|
+
if (!Number.isInteger(k) || !validSteps.includes(k)) return {
|
|
2360
|
+
ok: false,
|
|
2361
|
+
reason: `k must be a recorded step_id in [${validSteps[0] ?? 1}, ${prefix.maxK}], got ${describe$1(k)}`
|
|
2362
|
+
};
|
|
2363
|
+
const claim = record.failure_claim ?? record.failureClaim;
|
|
2364
|
+
if (typeof claim !== "string" || claim.trim().length === 0) return {
|
|
2365
|
+
ok: false,
|
|
2366
|
+
reason: "failure_claim must be a non-empty string"
|
|
2367
|
+
};
|
|
2368
|
+
const raw = record.intervention;
|
|
2369
|
+
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return {
|
|
2370
|
+
ok: false,
|
|
2371
|
+
reason: "intervention must be an object with kind and action"
|
|
2372
|
+
};
|
|
2373
|
+
const intervention = raw;
|
|
2374
|
+
const action = intervention.action;
|
|
2375
|
+
if (typeof action !== "string" || action.length === 0) return {
|
|
2376
|
+
ok: false,
|
|
2377
|
+
reason: "intervention.action must be a non-empty string"
|
|
2378
|
+
};
|
|
2379
|
+
const kind = intervention.kind;
|
|
2380
|
+
if (kind !== "shell" && kind !== "edit") return {
|
|
2381
|
+
ok: false,
|
|
2382
|
+
reason: `intervention.kind must be "shell" or "edit", got ${describe$1(kind)}`
|
|
2383
|
+
};
|
|
2384
|
+
return {
|
|
2385
|
+
ok: true,
|
|
2386
|
+
row: {
|
|
2387
|
+
k,
|
|
2388
|
+
failureClaim: claim.trim(),
|
|
2389
|
+
kind,
|
|
2390
|
+
action
|
|
2391
|
+
}
|
|
2392
|
+
};
|
|
2393
|
+
}
|
|
2394
|
+
};
|
|
2395
|
+
}
|
|
2396
|
+
function armUsage(usage, pricing) {
|
|
2397
|
+
const receipt = analystUsageReceiptFromPrimeUsage(usage, pricing);
|
|
2398
|
+
return {
|
|
2399
|
+
calls: usage.calls,
|
|
2400
|
+
inputTokens: usage.inputTokens,
|
|
2401
|
+
outputTokens: usage.outputTokens,
|
|
2402
|
+
costUsd: receipt.cost.usd ?? receipt.knownCostUsd ?? null
|
|
2403
|
+
};
|
|
2404
|
+
}
|
|
2405
|
+
function describe$1(value) {
|
|
2406
|
+
if (value === null) return "null";
|
|
2407
|
+
if (value === void 0) return "undefined";
|
|
2408
|
+
return typeof value === "string" ? JSON.stringify(value) : String(value);
|
|
2409
|
+
}
|
|
2410
|
+
//#endregion
|
|
2411
|
+
//#region src/trace-repair/arm-dspy.ts
|
|
2412
|
+
/**
|
|
2413
|
+
* Token the bridge reads to select the typed repair signature.
|
|
2414
|
+
*
|
|
2415
|
+
* The wire protocol carries instructions as opaque text, so the task is named
|
|
2416
|
+
* inside them rather than by adding a field every unrelated caller would have
|
|
2417
|
+
* to set. The same mechanism selects the CodeTraceBench typed signature.
|
|
2418
|
+
*/
|
|
2419
|
+
const DSPY_REPAIR_TASK_TOKEN = "tb-repair-typed-";
|
|
2420
|
+
/** Signature the bridge reports for a repair analysis. */
|
|
2421
|
+
const DSPY_REPAIR_SIGNATURE = "tb-repair-typed-v1";
|
|
2422
|
+
/** The name the trajectory is bound to inside the program's environment. */
|
|
2423
|
+
const DSPY_REPAIR_TRAJECTORY_INPUT = "trajectory";
|
|
2424
|
+
/**
|
|
2425
|
+
* The repair contract restated for the typed signature.
|
|
2426
|
+
*
|
|
2427
|
+
* Transport differs from the chat-completion arms — typed SUBMIT instead of a
|
|
2428
|
+
* fenced JSON object — and the QUESTION does not. Both read from
|
|
2429
|
+
* `repairTaskPolicy`, so the execution rules an arm is held to cannot drift
|
|
2430
|
+
* between arms.
|
|
2431
|
+
*/
|
|
2432
|
+
function dspyRepairInstructions(budget = SCAFFOLD_INTERVENTION_BUDGET) {
|
|
2433
|
+
return [
|
|
2434
|
+
`TASK ${DSPY_REPAIR_TASK_TOKEN}${DSPY_REPAIR_SIGNATURE}`,
|
|
2435
|
+
"",
|
|
2436
|
+
...repairTaskPolicy(budget),
|
|
2437
|
+
"",
|
|
2438
|
+
`The recorded trajectory is already loaded into your environment as \`${DSPY_REPAIR_TRAJECTORY_INPUT}\`,`,
|
|
2439
|
+
"a list of {step_id, action, observation} objects in recorded order. Read it with code",
|
|
2440
|
+
"rather than re-fetching it, and do not claim to have inspected anything you did not read.",
|
|
2441
|
+
"",
|
|
2442
|
+
"SUBMIT at most ONE repair, shaped as:",
|
|
2443
|
+
"- k: the step_id whose action you replace. It must be a step_id present in the trajectory.",
|
|
2444
|
+
"- failure_claim: what went wrong at step k. Recorded, never scored.",
|
|
2445
|
+
"- intervention_kind: \"edit\" when the action authors a file with a heredoc, \"shell\" otherwise.",
|
|
2446
|
+
`- action: the EXACT shell text to run instead, at most ${budget.maxBytes} bytes and exactly`,
|
|
2447
|
+
` ${budget.maxStatements === 1 ? "one top-level statement" : `${budget.maxStatements} top-level statements`}. It is executed verbatim: not a description, not a diff, not a plan.`,
|
|
2448
|
+
"SUBMIT an empty list only when no single replacement action could make the suite pass."
|
|
2449
|
+
].join("\n");
|
|
2450
|
+
}
|
|
2451
|
+
function createDspyRepairArm(options) {
|
|
2452
|
+
assertDspyRepairEngine(options.engine);
|
|
2453
|
+
const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
|
|
2454
|
+
const instructions = dspyRepairInstructions(budget);
|
|
2455
|
+
return {
|
|
2456
|
+
declaration: {
|
|
2457
|
+
id: options.id ?? "dspy-rlm",
|
|
2458
|
+
execution: `DSPy RLM program (${options.engine.id} v${options.engine.version}) with a typed repair signature and a code environment`,
|
|
2459
|
+
certification: {
|
|
2460
|
+
kind: "none",
|
|
2461
|
+
reason: "the GEPA-certified analyst text was earned on the CodeTraceBench incorrect-step contract, which asks for blocks of wrong steps rather than one executable repair; this arm runs instruction text authored for the repair contract and certified by nothing"
|
|
2462
|
+
},
|
|
2463
|
+
budget,
|
|
2464
|
+
repairTurns: 1,
|
|
2465
|
+
affordances: [
|
|
2466
|
+
"inline-trajectory",
|
|
2467
|
+
"code-interpreter",
|
|
2468
|
+
"agent-loop"
|
|
2469
|
+
],
|
|
2470
|
+
promptContract: [instructions]
|
|
2471
|
+
},
|
|
2472
|
+
async ask(request) {
|
|
2473
|
+
const steps = request.prefix.steps.map((step) => ({
|
|
2474
|
+
step_id: step.step_id,
|
|
2475
|
+
action: step.action,
|
|
2476
|
+
observation: step.observation
|
|
2477
|
+
}));
|
|
2478
|
+
let result;
|
|
2479
|
+
try {
|
|
2480
|
+
result = await options.engine.analyze({
|
|
2481
|
+
analystId: options.analystId,
|
|
2482
|
+
question: [
|
|
2483
|
+
REPAIR_QUESTION,
|
|
2484
|
+
"",
|
|
2485
|
+
`The trajectory has ${steps.length} recorded steps, step_ids ${steps.map((step) => step.step_id).join(", ")}.`
|
|
2486
|
+
].join("\n"),
|
|
2487
|
+
instructions,
|
|
2488
|
+
tools: [],
|
|
2489
|
+
taskInputs: {
|
|
2490
|
+
[DSPY_REPAIR_TRAJECTORY_INPUT]: steps,
|
|
2491
|
+
taskStatement: request.prefix.taskStatement
|
|
2492
|
+
},
|
|
2493
|
+
limits: options.limits,
|
|
2494
|
+
costLedger: options.costLedger,
|
|
2495
|
+
costPhase: options.costPhase,
|
|
2496
|
+
...options.costTags ? { costTags: options.costTags } : {},
|
|
2497
|
+
...request.signal ? { signal: request.signal } : {},
|
|
2498
|
+
...options.log ? { log: options.log } : {}
|
|
2499
|
+
});
|
|
2500
|
+
} catch (error) {
|
|
2501
|
+
return {
|
|
2502
|
+
status: "failed",
|
|
2503
|
+
failure: `engine threw: ${error instanceof Error ? error.message : String(error)}`,
|
|
2504
|
+
answer: null,
|
|
2505
|
+
rejectedRows: [],
|
|
2506
|
+
repair: {
|
|
2507
|
+
attempted: false,
|
|
2508
|
+
succeeded: null
|
|
2509
|
+
},
|
|
2510
|
+
usage: engineUsage(null)
|
|
2511
|
+
};
|
|
2512
|
+
}
|
|
2513
|
+
const payload = readRepairPayload(result.runtime, { validSteps: steps.map((step) => step.step_id) });
|
|
2514
|
+
const usage = engineUsage(result.modelCalls);
|
|
2515
|
+
if (!payload.ok) return {
|
|
2516
|
+
status: "failed",
|
|
2517
|
+
failure: payload.reason,
|
|
2518
|
+
answer: result.answer,
|
|
2519
|
+
rejectedRows: [],
|
|
2520
|
+
repair: {
|
|
2521
|
+
attempted: false,
|
|
2522
|
+
succeeded: null
|
|
2523
|
+
},
|
|
2524
|
+
usage
|
|
2525
|
+
};
|
|
2526
|
+
const { rows, dropped, repair, reported, failure } = payload.value;
|
|
2527
|
+
const repairTurn = {
|
|
2528
|
+
attempted: repair !== null,
|
|
2529
|
+
succeeded: repair === null ? null : failure === null
|
|
2530
|
+
};
|
|
2531
|
+
if (failure !== null) return {
|
|
2532
|
+
status: "failed",
|
|
2533
|
+
failure,
|
|
2534
|
+
answer: result.answer,
|
|
2535
|
+
rejectedRows: dropped,
|
|
2536
|
+
repair: repairTurn,
|
|
2537
|
+
usage
|
|
2538
|
+
};
|
|
2539
|
+
const row = rows[0];
|
|
2540
|
+
if (row === void 0) return {
|
|
2541
|
+
status: "declined",
|
|
2542
|
+
answer: result.answer,
|
|
2543
|
+
reportedRows: reported,
|
|
2544
|
+
rejectedRows: dropped,
|
|
2545
|
+
repair: repairTurn,
|
|
2546
|
+
usage
|
|
2547
|
+
};
|
|
2548
|
+
return {
|
|
2549
|
+
status: "finding",
|
|
2550
|
+
k: row.k,
|
|
2551
|
+
failureClaim: row.failureClaim,
|
|
2552
|
+
intervention: {
|
|
2553
|
+
kind: row.kind,
|
|
2554
|
+
action: row.action
|
|
2555
|
+
},
|
|
2556
|
+
answer: result.answer,
|
|
2557
|
+
reportedRows: reported,
|
|
2558
|
+
rejectedRows: dropped,
|
|
2559
|
+
repair: repairTurn,
|
|
2560
|
+
usage
|
|
2561
|
+
};
|
|
2562
|
+
}
|
|
2563
|
+
};
|
|
2564
|
+
}
|
|
2565
|
+
/**
|
|
2566
|
+
* Read the bridge's typed repair block.
|
|
2567
|
+
*
|
|
2568
|
+
* Structure is checked and fails loud: a missing block, a wrong signature, or
|
|
2569
|
+
* a row that does not carry an integer k and a non-empty action is a wiring
|
|
2570
|
+
* fault with a reason — never a default, and never an empty list that would
|
|
2571
|
+
* grade as a decline the program did not make. A structurally sound row whose
|
|
2572
|
+
* k is not a recorded step id is the model's mistake, and it is dropped with
|
|
2573
|
+
* its reason instead.
|
|
2574
|
+
*/
|
|
2575
|
+
function readRepairPayload(runtime, options) {
|
|
2576
|
+
const block = runtime.repair;
|
|
2577
|
+
if (!isRecord(block)) return {
|
|
2578
|
+
ok: false,
|
|
2579
|
+
reason: "the engine returned no runtime.repair block; the bridge did not run the typed repair signature"
|
|
2580
|
+
};
|
|
2581
|
+
if (block.signature !== "tb-repair-typed-v1") return {
|
|
2582
|
+
ok: false,
|
|
2583
|
+
reason: `runtime.repair.signature is ${describe(block.signature)}, expected "${DSPY_REPAIR_SIGNATURE}"`
|
|
2584
|
+
};
|
|
2585
|
+
if (!Number.isSafeInteger(block.reported) || block.reported < 0) return {
|
|
2586
|
+
ok: false,
|
|
2587
|
+
reason: `runtime.repair.reported must be a non-negative integer, got ${describe(block.reported)}`
|
|
2588
|
+
};
|
|
2589
|
+
if (!Array.isArray(block.rows)) return {
|
|
2590
|
+
ok: false,
|
|
2591
|
+
reason: "runtime.repair.rows must be an array"
|
|
2592
|
+
};
|
|
2593
|
+
const repair = block.repair;
|
|
2594
|
+
if (repair !== null && typeof repair !== "string") return {
|
|
2595
|
+
ok: false,
|
|
2596
|
+
reason: `runtime.repair.repair must be a string or null, got ${describe(repair)}`
|
|
2597
|
+
};
|
|
2598
|
+
const rawFailure = block.failure;
|
|
2599
|
+
if (rawFailure !== void 0 && rawFailure !== null && typeof rawFailure !== "string") return {
|
|
2600
|
+
ok: false,
|
|
2601
|
+
reason: `runtime.repair.failure must be a string or null, got ${describe(rawFailure)}`
|
|
2602
|
+
};
|
|
2603
|
+
const failure = typeof rawFailure === "string" ? rawFailure : null;
|
|
2604
|
+
const dropped = [];
|
|
2605
|
+
if (block.dropped !== void 0) {
|
|
2606
|
+
if (!Array.isArray(block.dropped)) return {
|
|
2607
|
+
ok: false,
|
|
2608
|
+
reason: "runtime.repair.dropped must be an array"
|
|
2609
|
+
};
|
|
2610
|
+
for (const [index, raw] of block.dropped.entries()) {
|
|
2611
|
+
if (!isRecord(raw) || typeof raw.reason !== "string") return {
|
|
2612
|
+
ok: false,
|
|
2613
|
+
reason: `runtime.repair.dropped[${index}] must carry a reason string`
|
|
2614
|
+
};
|
|
2615
|
+
dropped.push({
|
|
2616
|
+
index: Number.isSafeInteger(raw.index) ? raw.index : index,
|
|
2617
|
+
reason: raw.reason
|
|
2618
|
+
});
|
|
2619
|
+
}
|
|
2620
|
+
}
|
|
2621
|
+
const rows = [];
|
|
2622
|
+
for (const [index, raw] of block.rows.entries()) {
|
|
2623
|
+
const decoded = decodeRepairRow(raw);
|
|
2624
|
+
if (!decoded.ok) return {
|
|
2625
|
+
ok: false,
|
|
2626
|
+
reason: `runtime.repair.rows[${index}]: ${decoded.reason}`
|
|
2627
|
+
};
|
|
2628
|
+
if (!options.validSteps.includes(decoded.row.k)) {
|
|
2629
|
+
dropped.push({
|
|
2630
|
+
index,
|
|
2631
|
+
reason: `k must be a recorded step_id in [${options.validSteps[0] ?? 1}, ${options.validSteps[options.validSteps.length - 1] ?? 1}], got ${decoded.row.k}`
|
|
2632
|
+
});
|
|
2633
|
+
continue;
|
|
2634
|
+
}
|
|
2635
|
+
if (rows.length >= 1) {
|
|
2636
|
+
dropped.push({
|
|
2637
|
+
index,
|
|
2638
|
+
reason: "exceeds the one-repair cap"
|
|
2639
|
+
});
|
|
2640
|
+
continue;
|
|
2641
|
+
}
|
|
2642
|
+
rows.push(decoded.row);
|
|
2643
|
+
}
|
|
2644
|
+
return {
|
|
2645
|
+
ok: true,
|
|
2646
|
+
value: {
|
|
2647
|
+
signature: DSPY_REPAIR_SIGNATURE,
|
|
2648
|
+
reported: block.reported,
|
|
2649
|
+
rows,
|
|
2650
|
+
dropped,
|
|
2651
|
+
repair: repair ?? null,
|
|
2652
|
+
failure
|
|
2653
|
+
}
|
|
2654
|
+
};
|
|
2655
|
+
}
|
|
2656
|
+
function decodeRepairRow(raw) {
|
|
2657
|
+
if (!isRecord(raw)) return {
|
|
2658
|
+
ok: false,
|
|
2659
|
+
reason: "not an object"
|
|
2660
|
+
};
|
|
2661
|
+
const k = raw.k;
|
|
2662
|
+
if (!Number.isInteger(k) || k < 1) return {
|
|
2663
|
+
ok: false,
|
|
2664
|
+
reason: `k must be a positive integer, got ${describe(k)}`
|
|
2665
|
+
};
|
|
2666
|
+
const claim = raw.failure_claim;
|
|
2667
|
+
if (typeof claim !== "string" || claim.trim().length === 0) return {
|
|
2668
|
+
ok: false,
|
|
2669
|
+
reason: "failure_claim must be a non-empty string"
|
|
2670
|
+
};
|
|
2671
|
+
const intervention = raw.intervention;
|
|
2672
|
+
if (!isRecord(intervention)) return {
|
|
2673
|
+
ok: false,
|
|
2674
|
+
reason: "intervention must be an object with kind and action"
|
|
2675
|
+
};
|
|
2676
|
+
const kind = intervention.kind;
|
|
2677
|
+
if (kind !== "shell" && kind !== "edit") return {
|
|
2678
|
+
ok: false,
|
|
2679
|
+
reason: `intervention.kind must be "shell" or "edit", got ${describe(kind)}`
|
|
2680
|
+
};
|
|
2681
|
+
const action = intervention.action;
|
|
2682
|
+
if (typeof action !== "string" || action.length === 0) return {
|
|
2683
|
+
ok: false,
|
|
2684
|
+
reason: "intervention.action must be a non-empty string"
|
|
2685
|
+
};
|
|
2686
|
+
return {
|
|
2687
|
+
ok: true,
|
|
2688
|
+
row: {
|
|
2689
|
+
k,
|
|
2690
|
+
failureClaim: claim.trim(),
|
|
2691
|
+
kind,
|
|
2692
|
+
action
|
|
2693
|
+
}
|
|
2694
|
+
};
|
|
2695
|
+
}
|
|
2696
|
+
/**
|
|
2697
|
+
* The engine reports model calls; token counts and cost live in the caller's
|
|
2698
|
+
* cost ledger, which meters the proxy the engine ran through. Reporting them
|
|
2699
|
+
* here as zero would state a measurement this arm did not make.
|
|
2700
|
+
*/
|
|
2701
|
+
function engineUsage(modelCalls) {
|
|
2702
|
+
return {
|
|
2703
|
+
calls: modelCalls,
|
|
2704
|
+
inputTokens: null,
|
|
2705
|
+
outputTokens: null,
|
|
2706
|
+
costUsd: null
|
|
2707
|
+
};
|
|
2708
|
+
}
|
|
2709
|
+
function isRecord(value) {
|
|
2710
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2711
|
+
}
|
|
2712
|
+
function describe(value) {
|
|
2713
|
+
if (value === null) return "null";
|
|
2714
|
+
if (value === void 0) return "undefined";
|
|
2715
|
+
return typeof value === "string" ? JSON.stringify(value) : String(value);
|
|
2716
|
+
}
|
|
2717
|
+
/** Thrown by callers that build a repair arm with a non-DSPy engine. */
|
|
2718
|
+
function assertDspyRepairEngine(engine) {
|
|
2719
|
+
if (engine.id !== "dspy-rlm") throw new ValidationError(`the DSPy repair arm needs the dspy-rlm engine, got '${engine.id}'; another engine does not run the typed repair signature and would answer a different contract`);
|
|
2720
|
+
}
|
|
2721
|
+
//#endregion
|
|
2722
|
+
//#region src/trace-repair/continuation-records.ts
|
|
2723
|
+
/**
|
|
2724
|
+
* Project a full message list into corpus steps: the system and task messages
|
|
2725
|
+
* become their own steps, and every assistant turn carries the command it
|
|
2726
|
+
* requested plus the observation that answered it. A trailing assistant turn
|
|
2727
|
+
* with no answer keeps `obs: null`, which is how the corpus records a run that
|
|
2728
|
+
* ended on its last command.
|
|
2729
|
+
*/
|
|
2730
|
+
function toRecordedSteps(messages) {
|
|
2731
|
+
const steps = [];
|
|
2732
|
+
for (let i = 0; i < messages.length; i += 1) {
|
|
2733
|
+
const message = messages[i];
|
|
2734
|
+
if (!message) continue;
|
|
2735
|
+
if (message.role === "assistant") {
|
|
2736
|
+
const parsed = parseAction(message.content);
|
|
2737
|
+
const next = messages[i + 1];
|
|
2738
|
+
steps.push({
|
|
2739
|
+
src: "agent",
|
|
2740
|
+
msg: message.content,
|
|
2741
|
+
tools: parsed.kind === "action" ? [{
|
|
2742
|
+
fn: "bash_command",
|
|
2743
|
+
cmd: parsed.command
|
|
2744
|
+
}] : null,
|
|
2745
|
+
obs: next && next.role === "user" ? next.content : null
|
|
2746
|
+
});
|
|
2747
|
+
continue;
|
|
2748
|
+
}
|
|
2749
|
+
if (message.role === "system") {
|
|
2750
|
+
steps.push({
|
|
2751
|
+
src: "system",
|
|
2752
|
+
msg: message.content,
|
|
2753
|
+
tools: null,
|
|
2754
|
+
obs: null
|
|
2755
|
+
});
|
|
2756
|
+
continue;
|
|
2757
|
+
}
|
|
2758
|
+
const previous = messages[i - 1];
|
|
2759
|
+
if (!previous || previous.role === "system") steps.push({
|
|
2760
|
+
src: "user",
|
|
2761
|
+
msg: message.content,
|
|
2762
|
+
tools: null,
|
|
2763
|
+
obs: null
|
|
2764
|
+
});
|
|
2765
|
+
}
|
|
2766
|
+
return steps;
|
|
2767
|
+
}
|
|
2768
|
+
/** Corpus steps for the continuation alone, excluding the prefix it inherited. */
|
|
2769
|
+
function rolloutRecordedSteps(rollout) {
|
|
2770
|
+
return rollout.steps.map((step) => ({
|
|
2771
|
+
src: "agent",
|
|
2772
|
+
msg: step.assistantMessage,
|
|
2773
|
+
tools: step.action === null ? null : [{
|
|
2774
|
+
fn: "bash_command",
|
|
2775
|
+
cmd: step.action
|
|
2776
|
+
}],
|
|
2777
|
+
obs: step.observation
|
|
2778
|
+
}));
|
|
2779
|
+
}
|
|
2780
|
+
/**
|
|
2781
|
+
* Hash over everything the policy determines: the actions taken, the
|
|
2782
|
+
* observations they produced, the seeds, the exit, and the usage.
|
|
2783
|
+
*
|
|
2784
|
+
* Wall-clock fields are excluded because they vary between identical runs;
|
|
2785
|
+
* two rollouts with the same digest did the same work, whatever they cost in
|
|
2786
|
+
* seconds. Use it to assert determinism, never to assert equal latency.
|
|
2787
|
+
*/
|
|
2788
|
+
function rolloutDigest(rollout) {
|
|
2789
|
+
return contentHash({
|
|
2790
|
+
arm: rollout.arm,
|
|
2791
|
+
rowId: rollout.rowId,
|
|
2792
|
+
index: rollout.index,
|
|
2793
|
+
seed: rollout.seed,
|
|
2794
|
+
policyDigest: rollout.policyDigest,
|
|
2795
|
+
exitStatus: rollout.exitStatus,
|
|
2796
|
+
submission: rollout.submission,
|
|
2797
|
+
usage: rollout.usage,
|
|
2798
|
+
costUsd: rollout.costProvenance.usd,
|
|
2799
|
+
costKind: rollout.costProvenance.kind,
|
|
2800
|
+
steps: rollout.steps.map((step) => ({
|
|
2801
|
+
step: step.step,
|
|
2802
|
+
assistantMessage: step.assistantMessage,
|
|
2803
|
+
action: step.action,
|
|
2804
|
+
observation: step.observation,
|
|
2805
|
+
returncode: step.execution?.returncode ?? null,
|
|
2806
|
+
timedOut: step.execution?.timedOut ?? null,
|
|
2807
|
+
seed: step.model.seed,
|
|
2808
|
+
servedModel: step.model.servedModel,
|
|
2809
|
+
usage: step.model.usage,
|
|
2810
|
+
costUsd: step.model.costUsd
|
|
2811
|
+
}))
|
|
2812
|
+
});
|
|
2813
|
+
}
|
|
2814
|
+
//#endregion
|
|
2815
|
+
//#region src/trace-repair/degenerate-strategies.ts
|
|
2816
|
+
const DEGENERATE_STRATEGIES = [
|
|
2817
|
+
{
|
|
2818
|
+
id: "point-at-any-nonzero-exit-step",
|
|
2819
|
+
strategy: "Name the first step with a nonzero returncode, claim it failed, and let the reproduction gate confirm it.",
|
|
2820
|
+
defeat: "Reproduction is a gate, not a tier that pays. The credit vector has no term for it, so a reproduced step with no working intervention scores exactly what an unreproduced one scores: nothing.",
|
|
2821
|
+
defeatKind: "gate",
|
|
2822
|
+
enforcedIn: "funnel.ts (repairCredit has no reproduction term)"
|
|
2823
|
+
},
|
|
2824
|
+
{
|
|
2825
|
+
id: "propose-the-recorded-command-again",
|
|
2826
|
+
strategy: "Return the action the agent already ran at step k, so the arm reproduces the recorded state and looks like a faithful replay.",
|
|
2827
|
+
defeat: "The intervention is compared against the recorded action at k after whitespace normalisation and rejected before a container opens.",
|
|
2828
|
+
defeatKind: "gate",
|
|
2829
|
+
enforcedIn: "grade.ts (recorded-action-reproposed rejection)"
|
|
2830
|
+
},
|
|
2831
|
+
{
|
|
2832
|
+
id: "propose-a-no-op",
|
|
2833
|
+
strategy: "Return an action that changes nothing, so the arm inherits whatever the trajectory would have done anyway.",
|
|
2834
|
+
defeat: "Literal no-ops are rejected at the budget check. A semantic no-op cannot be detected syntactically, so it runs and measures at the no-op control floor: its paired delta is zero.",
|
|
2835
|
+
defeatKind: "measurement",
|
|
2836
|
+
enforcedIn: "action-budget.ts (NO_OP_ACTIONS) and delta-repair.ts (paired delta)"
|
|
2837
|
+
},
|
|
2838
|
+
{
|
|
2839
|
+
id: "submit-instead-of-repair",
|
|
2840
|
+
strategy: "Return the submit sentinel so the run ends immediately and the arm terminates cleanly.",
|
|
2841
|
+
defeat: "An action carrying the submit sentinel is rejected at the budget check. Ending the run is what the recorded agent already did and the tests already failed on it.",
|
|
2842
|
+
defeatKind: "gate",
|
|
2843
|
+
enforcedIn: "action-budget.ts (SUBMIT_SENTINEL rejection)"
|
|
2844
|
+
},
|
|
2845
|
+
{
|
|
2846
|
+
id: "touch-the-test-suite",
|
|
2847
|
+
strategy: "Write a passing suite, or a reward file, at the path the grader will read, so the oracle grades the trajectory's own artifact.",
|
|
2848
|
+
defeat: "The oracle purges the suite root and uploads the held-out suite from outside the session at grade time, then verifies the bytes it reads back. A planted suite is overwritten; a session that refuses the overwrite raises a tamper error instead of returning a pass.",
|
|
2849
|
+
defeatKind: "gate",
|
|
2850
|
+
enforcedIn: "test-oracle.ts (purge, upload, read-back digest)"
|
|
2851
|
+
},
|
|
2852
|
+
{
|
|
2853
|
+
id: "buy-a-bigger-action",
|
|
2854
|
+
strategy: "Return a multi-command script or a whole-file rewrite that does far more than one scaffold turn could.",
|
|
2855
|
+
defeat: "The budget counts top-level statements, heredocs and bytes. More than one action, more than one authored file, or more than 4 KB is rejected before a container opens.",
|
|
2856
|
+
defeatKind: "gate",
|
|
2857
|
+
enforcedIn: "action-budget.ts (checkInterventionBudget)"
|
|
2858
|
+
},
|
|
2859
|
+
{
|
|
2860
|
+
id: "decline-every-hard-row",
|
|
2861
|
+
strategy: "Answer no-decisive-failure on everything except the rows that are obviously repairable, so the reported rate is computed on an easy subset.",
|
|
2862
|
+
defeat: "A declined row keeps its cell in the funnel and stays in the denominator with a paired delta of zero, because its intervention arm is definitionally its control arm. Declining cannot raise the headline; it can only dilute it.",
|
|
2863
|
+
defeatKind: "measurement",
|
|
2864
|
+
enforcedIn: "grade.ts (declined outcome) and delta-repair.ts (full admitted denominator)"
|
|
2865
|
+
},
|
|
2866
|
+
{
|
|
2867
|
+
id: "repair-somewhere-other-than-k",
|
|
2868
|
+
strategy: "Name a plausible-looking k, then submit an intervention that fixes the task from any state, so the answer scores without localising anything.",
|
|
2869
|
+
defeat: "The intervention is executed at the k the analyst named, on the state produced by replaying steps 1..k-1. There is no separate localisation credit to win and no label the grader reads, so a wrong k is only penalised through the repair failing to work there.",
|
|
2870
|
+
defeatKind: "measurement",
|
|
2871
|
+
enforcedIn: "grade.ts (the intervention is applied at the named k only)"
|
|
2872
|
+
}
|
|
2873
|
+
];
|
|
2874
|
+
function degenerateStrategy(id) {
|
|
2875
|
+
const found = DEGENERATE_STRATEGIES.find((entry) => entry.id === id);
|
|
2876
|
+
if (!found) throw new Error(`unknown degenerate strategy: ${id}`);
|
|
2877
|
+
return found;
|
|
2878
|
+
}
|
|
2879
|
+
//#endregion
|
|
2880
|
+
//#region src/trace-repair/funnel.ts
|
|
2881
|
+
const CREDIT_TERMS = [
|
|
2882
|
+
"executes",
|
|
2883
|
+
"localFlip",
|
|
2884
|
+
"repairRate"
|
|
2885
|
+
];
|
|
2886
|
+
const ZERO_CREDIT = Object.freeze({
|
|
2887
|
+
executes: 0,
|
|
2888
|
+
localFlip: 0,
|
|
2889
|
+
repairRate: 0
|
|
2890
|
+
});
|
|
2891
|
+
/** What an answer earned. Every outcome but `measured` earns nothing. */
|
|
2892
|
+
function repairCredit(grade) {
|
|
2893
|
+
if (grade.outcome !== "measured") return ZERO_CREDIT;
|
|
2894
|
+
return {
|
|
2895
|
+
executes: 1,
|
|
2896
|
+
localFlip: grade.localFlip.passed ? 1 : 0,
|
|
2897
|
+
repairRate: grade.repair.rollouts === 0 ? 0 : grade.repair.passes / grade.repair.rollouts
|
|
2898
|
+
};
|
|
2899
|
+
}
|
|
2900
|
+
/** True once the answer is a well-formed, budget-admissible answer. A decline
|
|
2901
|
+
* is well formed, so it parses. */
|
|
2902
|
+
function reachedT0(grade) {
|
|
2903
|
+
return grade.outcome !== "rejected";
|
|
2904
|
+
}
|
|
2905
|
+
/** True once the recorded state at k came back. A decline never reaches the
|
|
2906
|
+
* gate, because it names no k. */
|
|
2907
|
+
function reachedT1(grade) {
|
|
2908
|
+
return grade.outcome === "did-not-execute" || grade.outcome === "measured";
|
|
2909
|
+
}
|
|
2910
|
+
/** True once the intervention ran at k and exited cleanly. */
|
|
2911
|
+
function reachedT2(grade) {
|
|
2912
|
+
return grade.outcome === "measured";
|
|
2913
|
+
}
|
|
2914
|
+
function countFunnel(grades) {
|
|
2915
|
+
let rejected = 0;
|
|
2916
|
+
let declined = 0;
|
|
2917
|
+
let t1 = 0;
|
|
2918
|
+
let t2 = 0;
|
|
2919
|
+
let t3 = 0;
|
|
2920
|
+
let t4Any = 0;
|
|
2921
|
+
let t4All = 0;
|
|
2922
|
+
for (const grade of grades) {
|
|
2923
|
+
if (grade.outcome === "rejected") rejected += 1;
|
|
2924
|
+
if (grade.outcome === "declined") declined += 1;
|
|
2925
|
+
if (reachedT1(grade)) t1 += 1;
|
|
2926
|
+
if (reachedT2(grade)) t2 += 1;
|
|
2927
|
+
if (grade.outcome === "measured") {
|
|
2928
|
+
if (grade.localFlip.passed) t3 += 1;
|
|
2929
|
+
if (grade.repair.passes > 0) t4Any += 1;
|
|
2930
|
+
if (grade.repair.rollouts > 0 && grade.repair.passes === grade.repair.rollouts) t4All += 1;
|
|
2931
|
+
}
|
|
2932
|
+
}
|
|
2933
|
+
return {
|
|
2934
|
+
rows: grades.length,
|
|
2935
|
+
rejected,
|
|
2936
|
+
declined,
|
|
2937
|
+
t0Parsed: grades.length - rejected,
|
|
2938
|
+
t1Reproduced: t1,
|
|
2939
|
+
t2Executed: t2,
|
|
2940
|
+
t3LocalFlip: t3,
|
|
2941
|
+
t4RepairFlipAny: t4Any,
|
|
2942
|
+
t4RepairFlipAll: t4All
|
|
2943
|
+
};
|
|
2944
|
+
}
|
|
2945
|
+
//#endregion
|
|
2946
|
+
//#region src/trace-repair/delta-repair.ts
|
|
2947
|
+
/**
|
|
2948
|
+
* The headline metric.
|
|
2949
|
+
*
|
|
2950
|
+
* Delta-repair = P(tests pass | intervention) − P(tests pass | no-fix control)
|
|
2951
|
+
*
|
|
2952
|
+
* Paired per row, then bootstrapped over rows. Paired because the rows differ
|
|
2953
|
+
* enormously from each other and not at all between arms: the same trajectory,
|
|
2954
|
+
* the same image, the same held-out suite, the same continuation policy, and
|
|
2955
|
+
* one difference — the state the continuation starts from.
|
|
2956
|
+
*
|
|
2957
|
+
* Every admitted row stays in the denominator, including the ones an analyst
|
|
2958
|
+
* declined and the ones whose answer was rejected. Those rows contribute a
|
|
2959
|
+
* paired difference of exactly zero, because with no intervention to run their
|
|
2960
|
+
* arm IS their control arm. Declining is therefore free of error and free of
|
|
2961
|
+
* reward, which is what makes it an honest answer rather than a way to pick an
|
|
2962
|
+
* easy subset.
|
|
2963
|
+
*
|
|
2964
|
+
* The report never collapses to this one number. The funnel counts, the rates
|
|
2965
|
+
* on measured rows alone, the per-row table and the threats travel with it,
|
|
2966
|
+
* because a difference of means computed on a corpus that was admitted on the
|
|
2967
|
+
* control failing is conditional on that admission and the reader has to see
|
|
2968
|
+
* it to price it.
|
|
2969
|
+
*/
|
|
2970
|
+
function deltaRepair(rowResults, options = {}) {
|
|
2971
|
+
if (rowResults.length === 0) throw new ValidationError("deltaRepair needs at least one graded row");
|
|
2972
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2973
|
+
for (const row of rowResults) {
|
|
2974
|
+
if (seen.has(row.rowId)) throw new ValidationError(`deltaRepair received row ${row.rowId} twice`);
|
|
2975
|
+
seen.add(row.rowId);
|
|
2976
|
+
}
|
|
2977
|
+
const funnel = countFunnel(rowResults.map((row) => row.grade));
|
|
2978
|
+
const control = rowResults.map((row) => row.controlRate);
|
|
2979
|
+
const intervention = rowResults.map((row) => row.interventionRate);
|
|
2980
|
+
const deltaAll = interval(control, intervention, options);
|
|
2981
|
+
const measured = rowResults.filter((row) => row.grade.outcome === "measured");
|
|
2982
|
+
const measuredOnly = measured.length === 0 ? emptyInterval(options) : interval(measured.map((row) => row.controlRate), measured.map((row) => row.interventionRate), options);
|
|
2983
|
+
return {
|
|
2984
|
+
rows: rowResults.length,
|
|
2985
|
+
funnel,
|
|
2986
|
+
interventionRate: mean(intervention),
|
|
2987
|
+
controlRate: mean(control),
|
|
2988
|
+
deltaRepair: deltaAll,
|
|
2989
|
+
measuredOnly,
|
|
2990
|
+
measuredRows: measured.length,
|
|
2991
|
+
rowResults,
|
|
2992
|
+
threats: collectThreats(rowResults, funnel, deltaAll)
|
|
2993
|
+
};
|
|
2994
|
+
}
|
|
2995
|
+
function interval(control, intervention, options) {
|
|
2996
|
+
const result = pairedBootstrap([...control], [...intervention], {
|
|
2997
|
+
statistic: "mean",
|
|
2998
|
+
seed: options.seed,
|
|
2999
|
+
resamples: options.resamples,
|
|
3000
|
+
confidence: options.confidence
|
|
3001
|
+
});
|
|
3002
|
+
return {
|
|
3003
|
+
n: result.n,
|
|
3004
|
+
mean: result.mean,
|
|
3005
|
+
median: result.median,
|
|
3006
|
+
low: result.low,
|
|
3007
|
+
high: result.high,
|
|
3008
|
+
confidence: result.confidence,
|
|
3009
|
+
resamples: result.resamples,
|
|
3010
|
+
gateEligible: result.gateEligible
|
|
3011
|
+
};
|
|
3012
|
+
}
|
|
3013
|
+
function emptyInterval(options) {
|
|
3014
|
+
return {
|
|
3015
|
+
n: 0,
|
|
3016
|
+
mean: 0,
|
|
3017
|
+
median: 0,
|
|
3018
|
+
low: 0,
|
|
3019
|
+
high: 0,
|
|
3020
|
+
confidence: options.confidence ?? .95,
|
|
3021
|
+
resamples: options.resamples ?? 2e3,
|
|
3022
|
+
gateEligible: false
|
|
3023
|
+
};
|
|
3024
|
+
}
|
|
3025
|
+
function collectThreats(rowResults, funnel, delta) {
|
|
3026
|
+
const threats = [{
|
|
3027
|
+
id: "control-position-asymmetry",
|
|
3028
|
+
statement: "The no-fix control continues from the recorded end state, while the intervention arm continues from step k and has to redo the work the recording did after k inside the same step budget. The arms are matched on policy and budget, not on position.",
|
|
3029
|
+
direction: "understates"
|
|
3030
|
+
}];
|
|
3031
|
+
const inert = rowResults.filter((row) => row.controlScreening === "declared-inert");
|
|
3032
|
+
if (inert.length > 0) threats.push({
|
|
3033
|
+
id: "control-cannot-rescue",
|
|
3034
|
+
statement: `${inert.length}/${rowResults.length} rows were screened under a control that makes no model call, so its rollouts graded the same bytes the end-state check graded as failing. A control rate of zero on those rows is a restatement of the end-state check, not a measurement of what continuing alone can repair.`,
|
|
3035
|
+
direction: "unknown"
|
|
3036
|
+
});
|
|
3037
|
+
if (inert.length === 0 && rowResults.every((row) => row.controlRate === 0)) threats.push({
|
|
3038
|
+
id: "admission-conditions-on-control-failure",
|
|
3039
|
+
statement: "Every row was admitted on its no-fix control failing every rollout, so the control rate is zero everywhere and Delta-repair equals the intervention rate. The estimate is conditional on that admission and does not describe rows the control can already repair.",
|
|
3040
|
+
direction: "unknown"
|
|
3041
|
+
});
|
|
3042
|
+
if (!delta.gateEligible) threats.push({
|
|
3043
|
+
id: "bootstrap-below-min-n",
|
|
3044
|
+
statement: `The interval covers ${delta.n} paired rows, below the 20 where a percentile bootstrap holds its nominal error rate. Read it as descriptive spread; a promotion must not turn on it.`,
|
|
3045
|
+
direction: "unknown"
|
|
3046
|
+
});
|
|
3047
|
+
const divergent = rowResults.filter((row) => measuredPrefixDivergences(row.grade) > 0).length;
|
|
3048
|
+
if (divergent > 0) threats.push({
|
|
3049
|
+
id: "prefix-divergence-present",
|
|
3050
|
+
statement: `${divergent}/${rowResults.length} rows replayed at least one prefix step to a different exit code than the recording. The state the intervention landed on is close to the recorded one, not identical to it.`,
|
|
3051
|
+
direction: "unknown"
|
|
3052
|
+
});
|
|
3053
|
+
const withFailures = rowResults.filter((row) => row.grade.outcome === "measured" && row.grade.repair.interventionFailures > 0).length;
|
|
3054
|
+
if (withFailures > 0) threats.push({
|
|
3055
|
+
id: "intervention-failures-present",
|
|
3056
|
+
statement: `${withFailures} rows had at least one rollout where the intervention failed to run after it had already run cleanly in the local-flip session. Those rollouts count as non-passes.`,
|
|
3057
|
+
direction: "understates"
|
|
3058
|
+
});
|
|
3059
|
+
if (delta.low === delta.high) threats.push({
|
|
3060
|
+
id: "zero-variance-interval",
|
|
3061
|
+
statement: "Every resample produced the same statistic, so the interval has zero width. That is an absence of variation in the data, not certainty about the effect.",
|
|
3062
|
+
direction: "unknown"
|
|
3063
|
+
});
|
|
3064
|
+
if (funnel.declined + funnel.rejected > funnel.rows / 2) threats.push({
|
|
3065
|
+
id: "declines-carry-the-denominator",
|
|
3066
|
+
statement: `${funnel.declined} declined and ${funnel.rejected} rejected of ${funnel.rows} rows contribute a paired delta of zero. The headline is dominated by rows where no intervention ran.`,
|
|
3067
|
+
direction: "understates"
|
|
3068
|
+
});
|
|
3069
|
+
return threats;
|
|
3070
|
+
}
|
|
3071
|
+
function measuredPrefixDivergences(grade) {
|
|
3072
|
+
if (grade.outcome === "measured" || grade.outcome === "did-not-execute") return grade.execution.prefix.divergences;
|
|
3073
|
+
if (grade.outcome === "not-reproduced" && grade.reproduction.basis !== "no-recorded-observation") return grade.reproduction.prefix.divergences;
|
|
3074
|
+
return 0;
|
|
3075
|
+
}
|
|
3076
|
+
function mean(values) {
|
|
3077
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
3078
|
+
}
|
|
3079
|
+
/** Markdown report: provenance, the funnel, every per-row column, the
|
|
3080
|
+
* distribution, and the threats. */
|
|
3081
|
+
function renderDeltaRepairReport(report) {
|
|
3082
|
+
const lines = [];
|
|
3083
|
+
const pct = (value) => `${(value * 100).toFixed(1)} %`;
|
|
3084
|
+
lines.push("# Delta-repair");
|
|
3085
|
+
lines.push("");
|
|
3086
|
+
lines.push(`**${signed(report.deltaRepair.mean)}** paired mean over ${report.rows} admitted rows (${(report.deltaRepair.confidence * 100).toFixed(0)} % CI ${signed(report.deltaRepair.low)} … ${signed(report.deltaRepair.high)}, ${report.deltaRepair.resamples} resamples, gate-eligible: ${report.deltaRepair.gateEligible}).`);
|
|
3087
|
+
lines.push("");
|
|
3088
|
+
lines.push(`P(tests pass | intervention) = ${pct(report.interventionRate)}; P(tests pass | no-fix control) = ${pct(report.controlRate)}.`);
|
|
3089
|
+
lines.push("");
|
|
3090
|
+
lines.push("## Funnel");
|
|
3091
|
+
lines.push("");
|
|
3092
|
+
lines.push("| cell | rows | share |");
|
|
3093
|
+
lines.push("| --- | --- | --- |");
|
|
3094
|
+
const f = report.funnel;
|
|
3095
|
+
const share = (value) => pct(f.rows === 0 ? 0 : value / f.rows);
|
|
3096
|
+
for (const [label, value] of [
|
|
3097
|
+
["admitted rows", f.rows],
|
|
3098
|
+
["t0 parsed", f.t0Parsed],
|
|
3099
|
+
["t1 reproduced (gate, pays nothing)", f.t1Reproduced],
|
|
3100
|
+
["t2 intervention executes", f.t2Executed],
|
|
3101
|
+
["t3 local flip", f.t3LocalFlip],
|
|
3102
|
+
["t4 repair flip (any rollout)", f.t4RepairFlipAny],
|
|
3103
|
+
["t4 repair flip (every rollout)", f.t4RepairFlipAll],
|
|
3104
|
+
["no-decisive-failure", f.declined],
|
|
3105
|
+
["rejected", f.rejected]
|
|
3106
|
+
]) lines.push(`| ${label} | ${value} | ${share(value)} |`);
|
|
3107
|
+
lines.push("");
|
|
3108
|
+
lines.push("## Rows");
|
|
3109
|
+
lines.push("");
|
|
3110
|
+
lines.push("| row | outcome | k | reproduction basis | reproduced | intervention exit | local flip | repair passes | rollouts | intervention failures | prefix steps | prefix divergences | P(int) | P(ctl) | delta | wall ms |");
|
|
3111
|
+
lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
|
|
3112
|
+
for (const row of report.rowResults) lines.push(`| ${rowCells(row).join(" | ")} |`);
|
|
3113
|
+
lines.push("");
|
|
3114
|
+
lines.push("## Measured rows only");
|
|
3115
|
+
lines.push("");
|
|
3116
|
+
lines.push(`${report.measuredRows} rows reached t2. Paired mean ${signed(report.measuredOnly.mean)} (CI ${signed(report.measuredOnly.low)} … ${signed(report.measuredOnly.high)}). Conditional on the analyst answering, so it is not the headline.`);
|
|
3117
|
+
lines.push("");
|
|
3118
|
+
lines.push("## Threats");
|
|
3119
|
+
lines.push("");
|
|
3120
|
+
for (const threat of report.threats) lines.push(`- **${threat.id}** (${threat.direction} the effect) — ${threat.statement}`);
|
|
3121
|
+
lines.push("");
|
|
3122
|
+
return lines.join("\n");
|
|
3123
|
+
}
|
|
3124
|
+
function rowCells(row) {
|
|
3125
|
+
const grade = row.grade;
|
|
3126
|
+
const reproduction = grade.outcome === "not-reproduced" || grade.outcome === "did-not-execute" || grade.outcome === "measured" ? grade.reproduction : null;
|
|
3127
|
+
const execution = grade.outcome === "did-not-execute" || grade.outcome === "measured" ? grade.execution : null;
|
|
3128
|
+
const repair = grade.outcome === "measured" ? grade.repair : null;
|
|
3129
|
+
const prefix = execution?.prefix ?? (reproduction && reproduction.basis !== "no-recorded-observation" ? reproduction.prefix : null);
|
|
3130
|
+
const rejection = grade.outcome === "rejected" ? `rejected: ${grade.rejection.reason}` : grade.outcome;
|
|
3131
|
+
return [
|
|
3132
|
+
row.rowId,
|
|
3133
|
+
rejection,
|
|
3134
|
+
grade.outcome === "rejected" || grade.outcome === "declined" ? "—" : String(grade.k),
|
|
3135
|
+
reproduction ? reproduction.basis : "—",
|
|
3136
|
+
reproduction ? String(reproduction.reproduced) : "—",
|
|
3137
|
+
execution ? String(execution.exitCode) : "—",
|
|
3138
|
+
grade.outcome === "measured" ? String(grade.localFlip.passed) : "—",
|
|
3139
|
+
repair ? String(repair.passes) : "—",
|
|
3140
|
+
repair ? String(repair.rollouts) : "0",
|
|
3141
|
+
repair ? String(repair.interventionFailures) : "—",
|
|
3142
|
+
prefix ? String(prefix.stepsReplayed) : "—",
|
|
3143
|
+
prefix ? String(prefix.divergences) : "—",
|
|
3144
|
+
row.interventionRate.toFixed(3),
|
|
3145
|
+
row.controlRate.toFixed(3),
|
|
3146
|
+
signed(row.delta),
|
|
3147
|
+
String(row.wallMs)
|
|
3148
|
+
];
|
|
3149
|
+
}
|
|
3150
|
+
function signed(value) {
|
|
3151
|
+
const points = value * 100;
|
|
3152
|
+
return `${points >= 0 ? "+" : ""}${points.toFixed(1)} pp`;
|
|
3153
|
+
}
|
|
3154
|
+
//#endregion
|
|
3155
|
+
//#region src/trace-repair/docker-environment.ts
|
|
3156
|
+
/**
|
|
3157
|
+
* Docker-backed continuation environment.
|
|
3158
|
+
*
|
|
3159
|
+
* The runner refuses any container whose network mode is not `none`, so this
|
|
3160
|
+
* builds containers that way and reports the mode it reads back from the
|
|
3161
|
+
* daemon rather than the mode it asked for. A container created elsewhere and
|
|
3162
|
+
* attached here is described the same way, so an environment that quietly kept
|
|
3163
|
+
* its network is rejected instead of producing evidence.
|
|
3164
|
+
*
|
|
3165
|
+
* Commands are bounded twice. `timeout` inside the container is the primary
|
|
3166
|
+
* bound, because killing the host-side `docker exec` client leaves the process
|
|
3167
|
+
* it started running in the container, where it would keep writing files under
|
|
3168
|
+
* later steps. The host-side kill is a backstop for a daemon that stops
|
|
3169
|
+
* answering. The container must therefore provide `timeout`; `describe()`
|
|
3170
|
+
* checks for it and refuses the container when it is absent.
|
|
3171
|
+
*/
|
|
3172
|
+
/**
|
|
3173
|
+
* Spawns a process, merges stdout and stderr the way the scaffold reads them,
|
|
3174
|
+
* and kills the whole process group on timeout so a killed command leaves no
|
|
3175
|
+
* children running in the container's namespace.
|
|
3176
|
+
*/
|
|
3177
|
+
const nodeProcessRunner = (request) => new Promise((resolve, reject) => {
|
|
3178
|
+
const [command, ...args] = request.argv;
|
|
3179
|
+
if (!command) {
|
|
3180
|
+
reject(new ValidationError("process runner requires a command"));
|
|
3181
|
+
return;
|
|
3182
|
+
}
|
|
3183
|
+
const child = spawn(command, args, {
|
|
3184
|
+
detached: true,
|
|
3185
|
+
stdio: [
|
|
3186
|
+
"ignore",
|
|
3187
|
+
"pipe",
|
|
3188
|
+
"pipe"
|
|
3189
|
+
]
|
|
3190
|
+
});
|
|
3191
|
+
const chunks = [];
|
|
3192
|
+
let timedOut = false;
|
|
3193
|
+
child.stdout.setEncoding("utf8");
|
|
3194
|
+
child.stderr.setEncoding("utf8");
|
|
3195
|
+
child.stdout.on("data", (chunk) => chunks.push(chunk));
|
|
3196
|
+
child.stderr.on("data", (chunk) => chunks.push(chunk));
|
|
3197
|
+
const timer = request.timeoutSeconds === void 0 ? void 0 : setTimeout(() => {
|
|
3198
|
+
timedOut = true;
|
|
3199
|
+
if (child.pid === void 0) return;
|
|
3200
|
+
try {
|
|
3201
|
+
process.kill(-child.pid, "SIGKILL");
|
|
3202
|
+
} catch (error) {
|
|
3203
|
+
if (error.code !== "ESRCH") reject(error);
|
|
3204
|
+
}
|
|
3205
|
+
}, request.timeoutSeconds * 1e3);
|
|
3206
|
+
child.on("error", (error) => {
|
|
3207
|
+
if (timer) clearTimeout(timer);
|
|
3208
|
+
reject(error);
|
|
3209
|
+
});
|
|
3210
|
+
child.on("close", (code, signal) => {
|
|
3211
|
+
if (timer) clearTimeout(timer);
|
|
3212
|
+
const output = chunks.join("");
|
|
3213
|
+
if (code !== null) {
|
|
3214
|
+
resolve({
|
|
3215
|
+
output,
|
|
3216
|
+
exitCode: code,
|
|
3217
|
+
timedOut
|
|
3218
|
+
});
|
|
3219
|
+
return;
|
|
3220
|
+
}
|
|
3221
|
+
const signalNumber = signal === null ? void 0 : constants.signals[signal];
|
|
3222
|
+
if (signalNumber === void 0) {
|
|
3223
|
+
reject(new ValidationError(`process closed with no exit code and no known signal: ${signal}`));
|
|
3224
|
+
return;
|
|
3225
|
+
}
|
|
3226
|
+
resolve({
|
|
3227
|
+
output,
|
|
3228
|
+
exitCode: -signalNumber,
|
|
3229
|
+
timedOut
|
|
3230
|
+
});
|
|
3231
|
+
});
|
|
3232
|
+
});
|
|
3233
|
+
const DEFAULT_INTERPRETER = ["bash", "-lc"];
|
|
3234
|
+
const DEFAULT_EXECUTABLE = "docker";
|
|
3235
|
+
/** Seconds `timeout` waits after SIGTERM before sending SIGKILL. */
|
|
3236
|
+
const KILL_GRACE_SECONDS = 5;
|
|
3237
|
+
/** Extra seconds the host-side backstop waits after the in-container bound. */
|
|
3238
|
+
const BACKSTOP_SECONDS = 10;
|
|
3239
|
+
/** `timeout` reports this when it stopped the command. */
|
|
3240
|
+
const TIMEOUT_EXIT_CODE = 124;
|
|
3241
|
+
/**
|
|
3242
|
+
* Arguments that create a container the policy accepts. `--network none` is
|
|
3243
|
+
* not optional: a continuation with network access could install what the
|
|
3244
|
+
* recorded run could not, and the arms would no longer differ only by the
|
|
3245
|
+
* intervention.
|
|
3246
|
+
*/
|
|
3247
|
+
function dockerRunArgs(input) {
|
|
3248
|
+
return [
|
|
3249
|
+
input.executable ?? DEFAULT_EXECUTABLE,
|
|
3250
|
+
"run",
|
|
3251
|
+
"-d",
|
|
3252
|
+
"--name",
|
|
3253
|
+
input.name,
|
|
3254
|
+
"--network",
|
|
3255
|
+
"none",
|
|
3256
|
+
"-w",
|
|
3257
|
+
input.cwd,
|
|
3258
|
+
"--rm",
|
|
3259
|
+
input.image,
|
|
3260
|
+
"sleep",
|
|
3261
|
+
input.containerLifetime ?? "2h"
|
|
3262
|
+
];
|
|
3263
|
+
}
|
|
3264
|
+
function createDockerContinuationEnvironment(options) {
|
|
3265
|
+
if (!options.containerRef.trim()) throw new ValidationError("docker continuation environment requires a container reference");
|
|
3266
|
+
const executable = options.executable ?? DEFAULT_EXECUTABLE;
|
|
3267
|
+
const interpreter = options.interpreter ?? DEFAULT_INTERPRETER;
|
|
3268
|
+
const envEntries = Object.entries(options.env ?? {});
|
|
3269
|
+
return {
|
|
3270
|
+
containerRef: options.containerRef,
|
|
3271
|
+
async describe() {
|
|
3272
|
+
const inspect = await options.runProcess({ argv: [
|
|
3273
|
+
executable,
|
|
3274
|
+
"inspect",
|
|
3275
|
+
"--format",
|
|
3276
|
+
"{{.HostConfig.NetworkMode}} {{.Config.Image}}",
|
|
3277
|
+
options.containerRef
|
|
3278
|
+
] });
|
|
3279
|
+
if (inspect.exitCode !== 0) throw new ValidationError(`docker inspect failed for ${options.containerRef} (exit ${inspect.exitCode}): ${inspect.output.trim()}`);
|
|
3280
|
+
const [networkMode, image] = inspect.output.trim().split(" ");
|
|
3281
|
+
if (!networkMode) throw new ValidationError(`docker inspect returned no network mode for ${options.containerRef}`);
|
|
3282
|
+
if ((await options.runProcess({ argv: [
|
|
3283
|
+
executable,
|
|
3284
|
+
"exec",
|
|
3285
|
+
options.containerRef,
|
|
3286
|
+
...interpreter,
|
|
3287
|
+
"command -v timeout"
|
|
3288
|
+
] })).exitCode !== 0) throw new ValidationError(`container ${options.containerRef} provides no \`timeout\`, so a long command cannot be bounded inside it`);
|
|
3289
|
+
return image ? {
|
|
3290
|
+
networkMode,
|
|
3291
|
+
image
|
|
3292
|
+
} : { networkMode };
|
|
3293
|
+
},
|
|
3294
|
+
async exec(command, execOptions) {
|
|
3295
|
+
const argv = [
|
|
3296
|
+
executable,
|
|
3297
|
+
"exec",
|
|
3298
|
+
"-w",
|
|
3299
|
+
options.cwd
|
|
3300
|
+
];
|
|
3301
|
+
for (const [key, value] of envEntries) argv.push("-e", `${key}=${value}`);
|
|
3302
|
+
argv.push(options.containerRef, "timeout", `--kill-after=${KILL_GRACE_SECONDS}s`, `${execOptions.timeoutSeconds}s`, ...interpreter, command);
|
|
3303
|
+
const result = await options.runProcess({
|
|
3304
|
+
argv,
|
|
3305
|
+
timeoutSeconds: execOptions.timeoutSeconds + KILL_GRACE_SECONDS + BACKSTOP_SECONDS
|
|
3306
|
+
});
|
|
3307
|
+
const timedOut = result.timedOut || result.exitCode === TIMEOUT_EXIT_CODE;
|
|
3308
|
+
return {
|
|
3309
|
+
output: result.output,
|
|
3310
|
+
returncode: result.exitCode,
|
|
3311
|
+
timedOut
|
|
3312
|
+
};
|
|
3313
|
+
},
|
|
3314
|
+
async dispose() {
|
|
3315
|
+
if (!options.removeOnDispose) return;
|
|
3316
|
+
await options.runProcess({ argv: [
|
|
3317
|
+
executable,
|
|
3318
|
+
"rm",
|
|
3319
|
+
"-f",
|
|
3320
|
+
options.containerRef
|
|
3321
|
+
] });
|
|
3322
|
+
}
|
|
3323
|
+
};
|
|
3324
|
+
}
|
|
3325
|
+
//#endregion
|
|
3326
|
+
//#region src/trace-repair/grade.ts
|
|
3327
|
+
/**
|
|
3328
|
+
* Grade one analyst answer about one admitted row.
|
|
3329
|
+
*
|
|
3330
|
+
* The whole funnel runs here, in the order that spends the least before it
|
|
3331
|
+
* knows: parse and budget checks open no container, the reproduction gate
|
|
3332
|
+
* opens one, the local flip opens one more, and only then does the repair arm
|
|
3333
|
+
* pay for its rollouts.
|
|
3334
|
+
*
|
|
3335
|
+
* Two properties a reviewer should be able to check by reading this file:
|
|
3336
|
+
*
|
|
3337
|
+
* The intervention is applied at the k the analyst named, on the state
|
|
3338
|
+
* produced by replaying steps 1..k-1, and nowhere else. There is no search
|
|
3339
|
+
* over nearby steps, no credit for being close, and no label in scope — the
|
|
3340
|
+
* grader never receives one. A wrong k therefore scores zero the only honest
|
|
3341
|
+
* way it can: the repair has to work where the analyst said the failure was.
|
|
3342
|
+
*
|
|
3343
|
+
* The row must arrive branded by `admitRow`. That is the type-level form of
|
|
3344
|
+
* "admission runs before any analyst sees a row": there is no signature here
|
|
3345
|
+
* that accepts an unadmitted row.
|
|
3346
|
+
*/
|
|
3347
|
+
/** Rollouts of the repair arm disagreed with the controls about the policy. */
|
|
3348
|
+
var RepairArmSymmetryError = class extends CaptureIntegrityError {};
|
|
3349
|
+
const DEFAULT_STEP_TIMEOUT_MS = 3e5;
|
|
3350
|
+
const OUTPUT_EXCERPT_CHARS = 4e3;
|
|
3351
|
+
async function gradeRepairRow(options) {
|
|
3352
|
+
const startedMs = Date.now();
|
|
3353
|
+
const grade = await produceGrade(options);
|
|
3354
|
+
return finish(options.row, grade, startedMs);
|
|
3355
|
+
}
|
|
3356
|
+
function finish(row, grade, startedMs) {
|
|
3357
|
+
const credit = repairCredit(grade);
|
|
3358
|
+
const interventionRate = grade.outcome === "measured" ? credit.repairRate : row.controlRate;
|
|
3359
|
+
return {
|
|
3360
|
+
rowId: row.rowId,
|
|
3361
|
+
grade,
|
|
3362
|
+
credit,
|
|
3363
|
+
interventionRate,
|
|
3364
|
+
controlRate: row.controlRate,
|
|
3365
|
+
controlRollouts: row.controlRollouts,
|
|
3366
|
+
controlScreening: row.controlScreening,
|
|
3367
|
+
controlPolicyDigest: row.policyDigest,
|
|
3368
|
+
repairRollouts: grade.outcome === "measured" ? grade.repair.rollouts : 0,
|
|
3369
|
+
delta: interventionRate - row.controlRate,
|
|
3370
|
+
wallMs: Date.now() - startedMs
|
|
3371
|
+
};
|
|
3372
|
+
}
|
|
3373
|
+
async function produceGrade(options) {
|
|
3374
|
+
const { row, response } = options;
|
|
3375
|
+
if (response.kind === "no-decisive-failure") return { outcome: "declined" };
|
|
3376
|
+
const target = targetStep(row, response);
|
|
3377
|
+
if (!target.ok) return target.grade;
|
|
3378
|
+
const budget = options.budget ?? SCAFFOLD_INTERVENTION_BUDGET;
|
|
3379
|
+
const check = checkInterventionBudget(response.intervention.action, response.intervention.kind, budget);
|
|
3380
|
+
if (!check.admissible) return {
|
|
3381
|
+
outcome: "rejected",
|
|
3382
|
+
rejection: {
|
|
3383
|
+
source: "budget",
|
|
3384
|
+
reason: check.violation,
|
|
3385
|
+
detail: check.detail,
|
|
3386
|
+
measurement: check.measurement
|
|
3387
|
+
}
|
|
3388
|
+
};
|
|
3389
|
+
if (normalizeActionForComparison(response.intervention.action) === normalizeActionForComparison(target.step.action)) return {
|
|
3390
|
+
outcome: "rejected",
|
|
3391
|
+
rejection: {
|
|
3392
|
+
source: "target",
|
|
3393
|
+
reason: "recorded-action-reproposed",
|
|
3394
|
+
detail: `the intervention is the action already recorded at step ${response.k}`
|
|
3395
|
+
}
|
|
3396
|
+
};
|
|
3397
|
+
const reproduction = await runReproductionGate(options, response, target.step);
|
|
3398
|
+
if (!reproduction.reproduced) return {
|
|
3399
|
+
outcome: "not-reproduced",
|
|
3400
|
+
k: response.k,
|
|
3401
|
+
reproduction
|
|
3402
|
+
};
|
|
3403
|
+
const local = await runLocalFlip(options, response);
|
|
3404
|
+
if (local.kind === "did-not-execute") return {
|
|
3405
|
+
outcome: "did-not-execute",
|
|
3406
|
+
k: response.k,
|
|
3407
|
+
reproduction,
|
|
3408
|
+
execution: local.execution
|
|
3409
|
+
};
|
|
3410
|
+
const repair = await runRepairArm(options, response);
|
|
3411
|
+
return {
|
|
3412
|
+
outcome: "measured",
|
|
3413
|
+
k: response.k,
|
|
3414
|
+
reproduction,
|
|
3415
|
+
execution: local.execution,
|
|
3416
|
+
localFlip: local.tests,
|
|
3417
|
+
repair
|
|
3418
|
+
};
|
|
3419
|
+
}
|
|
3420
|
+
function targetStep(row, finding) {
|
|
3421
|
+
if (finding.k < 1 || finding.k > row.steps.length) return {
|
|
3422
|
+
ok: false,
|
|
3423
|
+
grade: {
|
|
3424
|
+
outcome: "rejected",
|
|
3425
|
+
rejection: {
|
|
3426
|
+
source: "target",
|
|
3427
|
+
reason: "k-out-of-range",
|
|
3428
|
+
detail: `k=${finding.k} is outside [1, ${row.steps.length}]`
|
|
3429
|
+
}
|
|
3430
|
+
}
|
|
3431
|
+
};
|
|
3432
|
+
const step = row.steps[finding.k - 1];
|
|
3433
|
+
if (step.step_id !== finding.k) throw new ValidationError(`trace-repair: row ${row.rowId} steps[${finding.k - 1}].step_id=${step.step_id} != ${finding.k}; an admitted row must carry 1-based contiguous step ids`);
|
|
3434
|
+
return {
|
|
3435
|
+
ok: true,
|
|
3436
|
+
step
|
|
3437
|
+
};
|
|
3438
|
+
}
|
|
3439
|
+
async function runReproductionGate(options, finding, target) {
|
|
3440
|
+
const { row } = options;
|
|
3441
|
+
if (target.observation === null) {
|
|
3442
|
+
options.onProgress?.(`row ${row.rowId}: step ${finding.k} recorded no observation; the reproduction gate passes vacuously`);
|
|
3443
|
+
return {
|
|
3444
|
+
basis: "no-recorded-observation",
|
|
3445
|
+
reproduced: true
|
|
3446
|
+
};
|
|
3447
|
+
}
|
|
3448
|
+
const recordedReturncode = parseRecordedReturncode(target.observation);
|
|
3449
|
+
if (recordedReturncode === null) throw new ValidationError(`trace-repair: row ${row.rowId} step ${finding.k} carries an observation with no <returncode>; the corpus row is malformed and must not have been admitted`);
|
|
3450
|
+
const signature = deriveFailureSignature(target.observation);
|
|
3451
|
+
const session = await options.sessions.open({
|
|
3452
|
+
rowId: row.rowId,
|
|
3453
|
+
image: row.image,
|
|
3454
|
+
arm: "reproduce",
|
|
3455
|
+
rolloutIndex: 0
|
|
3456
|
+
});
|
|
3457
|
+
try {
|
|
3458
|
+
const prefix = await replayPrefix(session, options, finding.k);
|
|
3459
|
+
const result = await session.exec(wrapActionForExec(target.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
|
|
3460
|
+
const output = `${result.stdout}\n${result.stderr}`;
|
|
3461
|
+
const signatureObserved = signature === null ? null : output.includes(signature);
|
|
3462
|
+
const reproduced = !result.timedOut && result.exitCode === recordedReturncode && signatureObserved !== false;
|
|
3463
|
+
options.onProgress?.(`row ${row.rowId}: reproduction at step ${finding.k} exit=${result.exitCode} (recorded ${recordedReturncode}) reproduced=${reproduced}`);
|
|
3464
|
+
return {
|
|
3465
|
+
basis: signature === null ? "returncode-only" : "returncode+output-substring",
|
|
3466
|
+
reproduced,
|
|
3467
|
+
recordedReturncode,
|
|
3468
|
+
observedExitCode: result.exitCode,
|
|
3469
|
+
signature,
|
|
3470
|
+
signatureObserved,
|
|
3471
|
+
prefix
|
|
3472
|
+
};
|
|
3473
|
+
} finally {
|
|
3474
|
+
await session.close();
|
|
3475
|
+
}
|
|
3476
|
+
}
|
|
3477
|
+
async function runLocalFlip(options, finding) {
|
|
3478
|
+
const { row } = options;
|
|
3479
|
+
const session = await options.sessions.open({
|
|
3480
|
+
rowId: row.rowId,
|
|
3481
|
+
image: row.image,
|
|
3482
|
+
arm: "local-flip",
|
|
3483
|
+
rolloutIndex: 0
|
|
3484
|
+
});
|
|
3485
|
+
try {
|
|
3486
|
+
const prefix = await replayPrefix(session, options, finding.k);
|
|
3487
|
+
const startedMs = Date.now();
|
|
3488
|
+
const result = await session.exec(wrapActionForExec(finding.intervention.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
|
|
3489
|
+
const output = `${result.stdout}\n${result.stderr}`.trim();
|
|
3490
|
+
const execution = {
|
|
3491
|
+
command: finding.intervention.action,
|
|
3492
|
+
exitCode: result.exitCode,
|
|
3493
|
+
timedOut: result.timedOut,
|
|
3494
|
+
wallMs: Date.now() - startedMs,
|
|
3495
|
+
output: excerpt(output),
|
|
3496
|
+
prefix
|
|
3497
|
+
};
|
|
3498
|
+
if (result.timedOut || result.exitCode !== 0) {
|
|
3499
|
+
options.onProgress?.(`row ${row.rowId}: the intervention did not run at step ${finding.k} (exit=${result.exitCode} timedOut=${result.timedOut})`);
|
|
3500
|
+
return {
|
|
3501
|
+
kind: "did-not-execute",
|
|
3502
|
+
execution
|
|
3503
|
+
};
|
|
3504
|
+
}
|
|
3505
|
+
const tests = await gradeTests(options, session, "local-flip", 0);
|
|
3506
|
+
options.onProgress?.(`row ${row.rowId}: local flip=${tests.passed} after the intervention at step ${finding.k}`);
|
|
3507
|
+
return {
|
|
3508
|
+
kind: "measured",
|
|
3509
|
+
execution,
|
|
3510
|
+
tests
|
|
3511
|
+
};
|
|
3512
|
+
} finally {
|
|
3513
|
+
await session.close();
|
|
3514
|
+
}
|
|
3515
|
+
}
|
|
3516
|
+
async function runRepairArm(options, finding) {
|
|
3517
|
+
const { row } = options;
|
|
3518
|
+
const rollouts = options.repairRollouts ?? row.controlRollouts;
|
|
3519
|
+
if (!Number.isInteger(rollouts) || rollouts <= 0) throw new ValidationError(`repairRollouts must be a positive integer, got ${rollouts}`);
|
|
3520
|
+
const evidence = [];
|
|
3521
|
+
let passes = 0;
|
|
3522
|
+
let interventionFailures = 0;
|
|
3523
|
+
let policyDigest = null;
|
|
3524
|
+
for (let index = 0; index < rollouts; index += 1) {
|
|
3525
|
+
const session = await options.sessions.open({
|
|
3526
|
+
rowId: row.rowId,
|
|
3527
|
+
image: row.image,
|
|
3528
|
+
arm: "intervention",
|
|
3529
|
+
rolloutIndex: index
|
|
3530
|
+
});
|
|
3531
|
+
try {
|
|
3532
|
+
await replayPrefix(session, options, finding.k);
|
|
3533
|
+
const result = await session.exec(wrapActionForExec(finding.intervention.action, row.cwd), options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS);
|
|
3534
|
+
if (result.timedOut || result.exitCode !== 0) {
|
|
3535
|
+
interventionFailures += 1;
|
|
3536
|
+
evidence.push({
|
|
3537
|
+
rolloutIndex: index,
|
|
3538
|
+
status: "intervention-failed",
|
|
3539
|
+
interventionExitCode: result.exitCode,
|
|
3540
|
+
timedOut: result.timedOut
|
|
3541
|
+
});
|
|
3542
|
+
continue;
|
|
3543
|
+
}
|
|
3544
|
+
const continuation = await options.continuation({
|
|
3545
|
+
rowId: row.rowId,
|
|
3546
|
+
arm: "intervention",
|
|
3547
|
+
rolloutIndex: index,
|
|
3548
|
+
session,
|
|
3549
|
+
steps: row.steps,
|
|
3550
|
+
k: finding.k,
|
|
3551
|
+
injected: {
|
|
3552
|
+
action: finding.intervention.action,
|
|
3553
|
+
returncode: result.exitCode,
|
|
3554
|
+
output: `${result.stdout}\n${result.stderr}`.trim(),
|
|
3555
|
+
timedOut: result.timedOut
|
|
3556
|
+
},
|
|
3557
|
+
taskStatement: row.taskStatement
|
|
3558
|
+
});
|
|
3559
|
+
assertPolicySymmetry(row, continuation, index);
|
|
3560
|
+
policyDigest = continuation.policyDigest;
|
|
3561
|
+
const tests = await gradeTests(options, session, "intervention", index);
|
|
3562
|
+
if (tests.passed) passes += 1;
|
|
3563
|
+
evidence.push({
|
|
3564
|
+
rolloutIndex: index,
|
|
3565
|
+
status: "completed",
|
|
3566
|
+
interventionExitCode: result.exitCode,
|
|
3567
|
+
continuation,
|
|
3568
|
+
tests
|
|
3569
|
+
});
|
|
3570
|
+
options.onProgress?.(`row ${row.rowId}: repair rollout ${index} tests=${tests.passed} after ${continuation.steps} continuation steps (${continuation.exitStatus})`);
|
|
3571
|
+
} finally {
|
|
3572
|
+
await session.close();
|
|
3573
|
+
}
|
|
3574
|
+
}
|
|
3575
|
+
return {
|
|
3576
|
+
rollouts,
|
|
3577
|
+
passes,
|
|
3578
|
+
interventionFailures,
|
|
3579
|
+
policyDigest,
|
|
3580
|
+
rolloutEvidence: evidence
|
|
3581
|
+
};
|
|
3582
|
+
}
|
|
3583
|
+
function assertPolicySymmetry(row, continuation, index) {
|
|
3584
|
+
if (continuation.policyDigest !== row.policyDigest) throw new RepairArmSymmetryError(`row ${row.rowId} repair rollout ${index} ran policy ${continuation.policyDigest} but its controls ran ${row.policyDigest}; the difference between the arms would not be the intervention`);
|
|
3585
|
+
}
|
|
3586
|
+
async function gradeTests(options, session, arm, rolloutIndex) {
|
|
3587
|
+
const outcome = await options.oracle.grade(session, {
|
|
3588
|
+
rowId: options.row.rowId,
|
|
3589
|
+
arm,
|
|
3590
|
+
rolloutIndex
|
|
3591
|
+
});
|
|
3592
|
+
if (outcome.suiteDigest !== options.row.suiteDigest) throw new RepairArmSymmetryError(`row ${options.row.rowId} was admitted against suite ${options.row.suiteDigest} but the ${arm} arm graded against ${outcome.suiteDigest}; the arms did not answer the same question`);
|
|
3593
|
+
return {
|
|
3594
|
+
passed: outcome.passed,
|
|
3595
|
+
exitCode: outcome.exitCode,
|
|
3596
|
+
timedOut: outcome.timedOut,
|
|
3597
|
+
suiteDigest: outcome.suiteDigest
|
|
3598
|
+
};
|
|
3599
|
+
}
|
|
3600
|
+
/** Replay recorded steps 1..k-1 to rebuild the state the intervention lands
|
|
3601
|
+
* on. Divergence is counted and reported, never repaired. */
|
|
3602
|
+
async function replayPrefix(session, options, k) {
|
|
3603
|
+
const { row } = options;
|
|
3604
|
+
const timeoutMs = options.stepTimeoutMs ?? DEFAULT_STEP_TIMEOUT_MS;
|
|
3605
|
+
const recordedTimeoutMs = options.recordedTimeoutStepMs ?? timeoutMs;
|
|
3606
|
+
const startedMs = Date.now();
|
|
3607
|
+
let divergences = 0;
|
|
3608
|
+
let stepsReplayed = 0;
|
|
3609
|
+
for (const step of row.steps.slice(0, k - 1)) {
|
|
3610
|
+
const bound = isRecordedTimeout(step.observation) ? recordedTimeoutMs : timeoutMs;
|
|
3611
|
+
const result = await session.exec(wrapActionForExec(step.action, row.cwd), bound);
|
|
3612
|
+
stepsReplayed += 1;
|
|
3613
|
+
const recorded = parseRecordedReturncode(step.observation);
|
|
3614
|
+
if (recorded !== null && recorded !== result.exitCode) divergences += 1;
|
|
3615
|
+
}
|
|
3616
|
+
return {
|
|
3617
|
+
stepsReplayed,
|
|
3618
|
+
divergences,
|
|
3619
|
+
wallMs: Date.now() - startedMs
|
|
3620
|
+
};
|
|
3621
|
+
}
|
|
3622
|
+
function excerpt(text) {
|
|
3623
|
+
if (text.length <= OUTPUT_EXCERPT_CHARS) return text;
|
|
3624
|
+
const half = Math.floor(OUTPUT_EXCERPT_CHARS / 2);
|
|
3625
|
+
return `${text.slice(0, half)}\n… [${text.length - OUTPUT_EXCERPT_CHARS} chars elided] …\n${text.slice(-2e3)}`;
|
|
3626
|
+
}
|
|
3627
|
+
//#endregion
|
|
3628
|
+
//#region src/trace-repair/oracle-determinism.ts
|
|
3629
|
+
/**
|
|
3630
|
+
* Whether a task's own grader is a function of the container state.
|
|
3631
|
+
*
|
|
3632
|
+
* A repair benchmark uses the task's held-out suite as ground truth: the suite
|
|
3633
|
+
* says the state before the intervention fails and the state after it passes,
|
|
3634
|
+
* and the difference is the measurement. That reading needs the suite to answer
|
|
3635
|
+
* the same way twice on the same bytes. A suite that asserts on wall clock does
|
|
3636
|
+
* not: the same container passes or fails by chance, so a control arm can
|
|
3637
|
+
* "rescue" a row nothing touched, and an intervention arm can lose a repair it
|
|
3638
|
+
* made.
|
|
3639
|
+
*
|
|
3640
|
+
* The check is direct. Grade the same container N times with nothing written
|
|
3641
|
+
* between the runs, pool the replicates by the state they graded, and read the
|
|
3642
|
+
* minority share.
|
|
3643
|
+
*
|
|
3644
|
+
* It reads that share per ASSERTION, not per suite, wherever the suite reports
|
|
3645
|
+
* assertions. A pass/fail reward is a conjunction over many assertions, so a
|
|
3646
|
+
* suite whose timing assertions each flip independently can still return the
|
|
3647
|
+
* same reward on every replicate of a state that sits far from the threshold —
|
|
3648
|
+
* and then return a coin flip on the states a campaign actually grades, which
|
|
3649
|
+
* sit near it. Per-assertion counting sees the flip at the anchor; reward
|
|
3650
|
+
* counting does not. A state whose replicates carried no assertion report falls
|
|
3651
|
+
* back to the reward and records that it did, so a coarser measurement is
|
|
3652
|
+
* visible rather than assumed equivalent.
|
|
3653
|
+
*
|
|
3654
|
+
* Replicates are also run under machine contention, because unanimity on an
|
|
3655
|
+
* idle box proves nothing about a threshold the load never approached.
|
|
3656
|
+
* Replicates on one state pool across both loads, so a verdict that moves when
|
|
3657
|
+
* the machine gets busy reads as the flip it is.
|
|
3658
|
+
*/
|
|
3659
|
+
/** A task graded a repair by something other than the state under test. */
|
|
3660
|
+
var NondeterministicOracleError = class extends CaptureIntegrityError {};
|
|
3661
|
+
/** The unit a whole-suite verdict is counted under when nothing finer exists. */
|
|
3662
|
+
const SUITE_REWARD_UNIT = "suite-reward";
|
|
3663
|
+
/** Replicates one group needs before a flip is measurable at all. */
|
|
3664
|
+
const MIN_ORACLE_REPLICATES = 2;
|
|
3665
|
+
/**
|
|
3666
|
+
* Reduce measured replicates to a verdict.
|
|
3667
|
+
*
|
|
3668
|
+
* Pure: it opens no container. The tool that runs the replicates hands the
|
|
3669
|
+
* counts here, so the rule that decides stability is one rule and a reviewer
|
|
3670
|
+
* can re-derive any verdict from the recorded replicates.
|
|
3671
|
+
*/
|
|
3672
|
+
function oracleDeterminism(evidence) {
|
|
3673
|
+
if (evidence.groups.length === 0) throw new ValidationError(`oracle determinism for ${evidence.taskName} received no replicate group`);
|
|
3674
|
+
for (const group of evidence.groups) if (group.replicates.length < 2) throw new ValidationError(`oracle determinism for ${evidence.taskName} received ${group.replicates.length} ${group.load} replicate(s) on the ${group.state} state; 2 are needed before a flip can be observed`);
|
|
3675
|
+
const byState = [];
|
|
3676
|
+
for (const state of ["unsolved", "solved"]) {
|
|
3677
|
+
const groups = evidence.groups.filter((group) => group.state === state);
|
|
3678
|
+
if (groups.length === 0) continue;
|
|
3679
|
+
byState.push(stateVerdict(state, groups));
|
|
3680
|
+
}
|
|
3681
|
+
const flipRate = byState.reduce((worst, state) => Math.max(worst, state.flipRate), 0);
|
|
3682
|
+
const replicates = byState.reduce((total, state) => total + state.replicates, 0);
|
|
3683
|
+
return {
|
|
3684
|
+
taskName: evidence.taskName,
|
|
3685
|
+
image: evidence.image,
|
|
3686
|
+
suiteDigest: evidence.suiteDigest,
|
|
3687
|
+
stable: flipRate === 0,
|
|
3688
|
+
replicates,
|
|
3689
|
+
flipRate,
|
|
3690
|
+
byState,
|
|
3691
|
+
measuredAt: evidence.measuredAt,
|
|
3692
|
+
detail: byState.map(describeState).join("; ")
|
|
3693
|
+
};
|
|
3694
|
+
}
|
|
3695
|
+
function stateVerdict(state, groups) {
|
|
3696
|
+
const replicates = groups.flatMap((group) => [...group.replicates]);
|
|
3697
|
+
const passes = replicates.filter((replicate) => replicate.passed).length;
|
|
3698
|
+
const rates = groups.map((group) => group.replicates.filter((r) => r.passed).length / group.replicates.length);
|
|
3699
|
+
const perAssertion = replicates.every((replicate) => replicate.assertions !== null);
|
|
3700
|
+
const flipped = [];
|
|
3701
|
+
const assertionSetUnstable = [];
|
|
3702
|
+
if (perAssertion) {
|
|
3703
|
+
const outcomes = /* @__PURE__ */ new Map();
|
|
3704
|
+
for (const replicate of replicates) for (const assertion of replicate.assertions ?? []) {
|
|
3705
|
+
const seen = outcomes.get(assertion.id);
|
|
3706
|
+
if (seen) seen.push(assertion.passed);
|
|
3707
|
+
else outcomes.set(assertion.id, [assertion.passed]);
|
|
3708
|
+
}
|
|
3709
|
+
for (const [id, results] of [...outcomes].sort(([a], [b]) => a.localeCompare(b))) {
|
|
3710
|
+
if (results.length !== replicates.length) assertionSetUnstable.push(id);
|
|
3711
|
+
const unitPasses = results.filter(Boolean).length;
|
|
3712
|
+
const unitFails = results.length - unitPasses;
|
|
3713
|
+
const minority = Math.min(unitPasses, unitFails);
|
|
3714
|
+
const absent = replicates.length - results.length;
|
|
3715
|
+
if (minority + absent > 0) flipped.push({
|
|
3716
|
+
unit: id,
|
|
3717
|
+
passes: unitPasses,
|
|
3718
|
+
fails: unitFails + absent,
|
|
3719
|
+
flipRate: (minority + absent) / replicates.length
|
|
3720
|
+
});
|
|
3721
|
+
}
|
|
3722
|
+
} else {
|
|
3723
|
+
const minority = Math.min(passes, replicates.length - passes);
|
|
3724
|
+
if (minority > 0) flipped.push({
|
|
3725
|
+
unit: SUITE_REWARD_UNIT,
|
|
3726
|
+
passes,
|
|
3727
|
+
fails: replicates.length - passes,
|
|
3728
|
+
flipRate: minority / replicates.length
|
|
3729
|
+
});
|
|
3730
|
+
}
|
|
3731
|
+
return {
|
|
3732
|
+
state,
|
|
3733
|
+
replicates: replicates.length,
|
|
3734
|
+
passes,
|
|
3735
|
+
fails: replicates.length - passes,
|
|
3736
|
+
granularity: perAssertion ? "per-assertion" : "reward",
|
|
3737
|
+
flipRate: flipped.reduce((worst, unit) => Math.max(worst, unit.flipRate), 0),
|
|
3738
|
+
flipped,
|
|
3739
|
+
assertionSetUnstable,
|
|
3740
|
+
loadSensitive: new Set(rates).size > 1,
|
|
3741
|
+
rewardsObserved: [...new Set(replicates.map((r) => r.reward ?? "NO_REWARD_FILE"))].sort()
|
|
3742
|
+
};
|
|
3743
|
+
}
|
|
3744
|
+
function describeState(state) {
|
|
3745
|
+
const base = `${state.state}: ${state.passes}/${state.replicates} suite pass, ${state.granularity} counting`;
|
|
3746
|
+
if (state.flipped.length === 0) return `${base}, no flip`;
|
|
3747
|
+
const worst = state.flipped.reduce((a, b) => b.flipRate > a.flipRate ? b : a);
|
|
3748
|
+
return `${base}, ${state.flipped.length} unit(s) flipped, worst ${worst.unit} ${worst.passes}/${worst.passes + worst.fails} pass` + (state.loadSensitive ? " (load-sensitive)" : "");
|
|
3749
|
+
}
|
|
3750
|
+
function parseTaskOracleRegistry(document) {
|
|
3751
|
+
if (typeof document !== "object" || document === null) throw new ValidationError("task oracle registry must be an object");
|
|
3752
|
+
const { version, measurements } = document;
|
|
3753
|
+
if (version !== 1) throw new ValidationError(`task oracle registry version must be 1, got ${String(version)}`);
|
|
3754
|
+
if (!Array.isArray(measurements)) throw new ValidationError("task oracle registry needs a measurements array");
|
|
3755
|
+
return taskOracleRegistry(measurements.map((evidence) => oracleDeterminism(evidence)));
|
|
3756
|
+
}
|
|
3757
|
+
function taskOracleRegistry(verdicts) {
|
|
3758
|
+
const registry = /* @__PURE__ */ new Map();
|
|
3759
|
+
for (const verdict of verdicts) {
|
|
3760
|
+
if (registry.has(verdict.taskName)) throw new ValidationError(`task oracle registry received ${verdict.taskName} twice; a task has one certification`);
|
|
3761
|
+
registry.set(verdict.taskName, verdict);
|
|
3762
|
+
}
|
|
3763
|
+
return registry;
|
|
3764
|
+
}
|
|
3765
|
+
/** Throws unless the task graded the same bytes the same way every replicate. */
|
|
3766
|
+
function assertDeterministicOracle(verdict) {
|
|
3767
|
+
if (verdict.stable) return;
|
|
3768
|
+
throw new NondeterministicOracleError(`task ${verdict.taskName} graded byte-identical state inconsistently: flip rate ${(verdict.flipRate * 100).toFixed(1)} % over ${verdict.replicates} replicates (${verdict.detail}). Its verdict is not a function of the state, so it cannot carry ground truth for an intervention study.`);
|
|
3769
|
+
}
|
|
3770
|
+
//#endregion
|
|
3771
|
+
//#region src/trace-repair/test-oracle.ts
|
|
3772
|
+
/**
|
|
3773
|
+
* The held-out suite, injected from outside the box at grade time.
|
|
3774
|
+
*
|
|
3775
|
+
* A repair is only measurable if the thing that decides pass or fail is out
|
|
3776
|
+
* of the trajectory's reach. Terminal-Bench gets that by uploading the suite
|
|
3777
|
+
* into the container at grade time, after the agent has stopped; a suite the
|
|
3778
|
+
* agent planted is overwritten before it is ever read. `injectedTestOracle`
|
|
3779
|
+
* reproduces that property and then proves it per call:
|
|
3780
|
+
*
|
|
3781
|
+
* 1. purge the suite root, so a planted extra file cannot survive
|
|
3782
|
+
* 2. upload every suite file from outside the session
|
|
3783
|
+
* 3. read the bytes back from inside and hash them
|
|
3784
|
+
* 4. refuse to grade when the read-back digest is not the uploaded digest
|
|
3785
|
+
*
|
|
3786
|
+
* Step 4 is why the property is asserted rather than assumed. A container
|
|
3787
|
+
* that silently drops the upload, or a filesystem trick that serves different
|
|
3788
|
+
* bytes to the reader, raises `TestSuiteTamperedError` instead of returning a
|
|
3789
|
+
* result. An oracle that cannot prove what it graded reports nothing.
|
|
3790
|
+
*/
|
|
3791
|
+
/** The suite the oracle read back is not the suite it uploaded. */
|
|
3792
|
+
var TestSuiteTamperedError = class extends CaptureIntegrityError {};
|
|
3793
|
+
/** The oracle could not place or run the suite, so it graded nothing. */
|
|
3794
|
+
var TestOracleError = class extends CaptureIntegrityError {};
|
|
3795
|
+
const DEFAULT_UPLOAD_TIMEOUT_MS = 6e4;
|
|
3796
|
+
const DEFAULT_COMMAND_TIMEOUT_MS = 9e5;
|
|
3797
|
+
/**
|
|
3798
|
+
* Content digest of the suite: sha256 over each path and its bytes, in path
|
|
3799
|
+
* order. Two suites with the same digest are the same suite.
|
|
3800
|
+
*/
|
|
3801
|
+
function testSuiteDigest(files) {
|
|
3802
|
+
const hash = createHash("sha256");
|
|
3803
|
+
for (const file of [...files].sort((a, b) => a.path.localeCompare(b.path))) {
|
|
3804
|
+
hash.update(file.path);
|
|
3805
|
+
hash.update("\0");
|
|
3806
|
+
hash.update(Buffer.from(file.contents, "utf8"));
|
|
3807
|
+
hash.update("\0");
|
|
3808
|
+
}
|
|
3809
|
+
return hash.digest("hex");
|
|
3810
|
+
}
|
|
3811
|
+
function injectedTestOracle(options) {
|
|
3812
|
+
assertOracleOptions(options);
|
|
3813
|
+
const uploadTimeoutMs = options.uploadTimeoutMs ?? DEFAULT_UPLOAD_TIMEOUT_MS;
|
|
3814
|
+
const commandTimeoutMs = options.commandTimeoutMs ?? DEFAULT_COMMAND_TIMEOUT_MS;
|
|
3815
|
+
const expectedDigest = testSuiteDigest(options.files);
|
|
3816
|
+
return { async grade(session, context) {
|
|
3817
|
+
const where = `${context.rowId}/${context.arm}#${context.rolloutIndex}`;
|
|
3818
|
+
for (const directory of options.purge ?? []) await mustSucceed(session, `rm -rf ${shellQuote(directory)}`, uploadTimeoutMs, `purge ${directory} in ${where}`);
|
|
3819
|
+
for (const file of options.files) {
|
|
3820
|
+
const directory = parentDirectory(file.path);
|
|
3821
|
+
if (directory) await mustSucceed(session, `mkdir -p ${shellQuote(directory)}`, uploadTimeoutMs, `mkdir ${directory} in ${where}`);
|
|
3822
|
+
await mustSucceed(session, `printf %s ${shellQuote(Buffer.from(file.contents, "utf8").toString("base64"))} | base64 -d > ${shellQuote(file.path)}`, uploadTimeoutMs, `upload ${file.path} in ${where}`);
|
|
3823
|
+
if (file.mode) await mustSucceed(session, `chmod ${file.mode} ${shellQuote(file.path)}`, uploadTimeoutMs, `chmod ${file.path} in ${where}`);
|
|
3824
|
+
}
|
|
3825
|
+
const readBack = [];
|
|
3826
|
+
for (const file of options.files) {
|
|
3827
|
+
const result = await mustSucceed(session, `base64 < ${shellQuote(file.path)} | tr -d '\\n'`, uploadTimeoutMs, `read back ${file.path} in ${where}`);
|
|
3828
|
+
readBack.push({
|
|
3829
|
+
path: file.path,
|
|
3830
|
+
contents: Buffer.from(result.stdout.trim(), "base64").toString("utf8")
|
|
3831
|
+
});
|
|
3832
|
+
}
|
|
3833
|
+
const observedDigest = testSuiteDigest(readBack);
|
|
3834
|
+
if (observedDigest !== expectedDigest) throw new TestSuiteTamperedError(`test suite in ${where} does not match the suite uploaded from outside (expected ${expectedDigest}, read back ${observedDigest}); the graded result is discarded`);
|
|
3835
|
+
const run = await session.exec(options.command, commandTimeoutMs);
|
|
3836
|
+
return {
|
|
3837
|
+
passed: run.exitCode === 0 && !run.timedOut,
|
|
3838
|
+
exitCode: run.exitCode,
|
|
3839
|
+
output: `${run.stdout}\n${run.stderr}`.trim(),
|
|
3840
|
+
suiteDigest: observedDigest,
|
|
3841
|
+
timedOut: run.timedOut
|
|
3842
|
+
};
|
|
3843
|
+
} };
|
|
3844
|
+
}
|
|
3845
|
+
/**
|
|
3846
|
+
* Run a setup command that must succeed. A failed upload is an oracle
|
|
3847
|
+
* failure, never a failed test: reporting it as a failing suite would turn a
|
|
3848
|
+
* broken container into evidence against the intervention.
|
|
3849
|
+
*/
|
|
3850
|
+
async function mustSucceed(session, command, timeoutMs, what) {
|
|
3851
|
+
const result = await session.exec(command, timeoutMs);
|
|
3852
|
+
if (result.timedOut) throw new TestOracleError(`test oracle timed out while it tried to ${what}`);
|
|
3853
|
+
if (result.exitCode !== 0) throw new TestOracleError(`test oracle failed to ${what}: exit ${result.exitCode}\n${result.stderr.trim()}`);
|
|
3854
|
+
return result;
|
|
3855
|
+
}
|
|
3856
|
+
function assertOracleOptions(options) {
|
|
3857
|
+
if (options.files.length === 0) throw new ValidationError("injectedTestOracle requires at least one suite file");
|
|
3858
|
+
const seen = /* @__PURE__ */ new Set();
|
|
3859
|
+
for (const file of options.files) {
|
|
3860
|
+
if (!file.path.startsWith("/")) throw new ValidationError(`suite file path must be absolute, got "${file.path}"`);
|
|
3861
|
+
if (seen.has(file.path)) throw new ValidationError(`suite file ${file.path} is listed twice`);
|
|
3862
|
+
seen.add(file.path);
|
|
3863
|
+
}
|
|
3864
|
+
if (options.command.trim().length === 0) throw new ValidationError("injectedTestOracle requires a suite command");
|
|
3865
|
+
for (const directory of options.purge ?? []) if (!directory.startsWith("/") || directory.trim() === "/") throw new ValidationError(`purge path must be an absolute directory below the root, got "${directory}"`);
|
|
3866
|
+
}
|
|
3867
|
+
function parentDirectory(path) {
|
|
3868
|
+
const index = path.lastIndexOf("/");
|
|
3869
|
+
if (index <= 0) return null;
|
|
3870
|
+
return path.slice(0, index);
|
|
3871
|
+
}
|
|
3872
|
+
function shellQuote(value) {
|
|
3873
|
+
return `'${value.replaceAll("'", `'\\''`)}'`;
|
|
3874
|
+
}
|
|
3875
|
+
//#endregion
|
|
3876
|
+
export { ADMISSION_CONFIG_DEFAULTS, ADMISSION_EXCLUSION_MEANING, ADMISSION_EXCLUSION_ORDER, ADMISSION_ROW_KEYS, ADMISSION_STRATA, AdmissionDenominatorError, AdmissionIndependenceError, BLINDED_FIELDS, CONTINUATION_POLICY_DEFAULTS, CONTROL_SCREENING_MODES, CREDIT_TERMS, ContinuationPolicyViolationError, ContinuationSymmetryError, DEGENERATE_STRATEGIES, DSPY_REPAIR_SIGNATURE, DSPY_REPAIR_TASK_TOKEN, DSPY_REPAIR_TRAJECTORY_INPUT, MINI_SWE_SYSTEM_MESSAGE, MIN_ORACLE_REPLICATES, NO_DECISIVE_FAILURE, NO_OP_ACTIONS, NondeterministicOracleError, OUTPUT_ELISION_THRESHOLD, OUTPUT_ELISION_WINDOW, REPAIR_CONTRACT_LINES, REPAIR_QUESTION, REPAIR_REPAIR_CONTRACT_LINES, RepairArmSymmetryError, SCAFFOLD_INTERVENTION_BUDGET, SUBMIT_SENTINEL, SUITE_REWARD_UNIT, TB_REPAIR_ADMISSION_CRITERIA, TestOracleError, TestSuiteTamperedError, UncalibratedControlError, admissionArtifact, admitRow, admittedCount, admittedRowIds, askRepairArm, assertAnalystIndependent, assertArmSymmetry, assertChainReconciles, assertControlCalibrated, assertDenominatorIntact, assertDeterministicOracle, assertDspyRepairEngine, blindTrajectory, buildDenominatorChain, checkInterventionBudget, classifyActionPayload, continuationPolicyDigest, continuationSeed, controlCanRescue, countFunnel, createCompletionRepairArm, createDockerContinuationEnvironment, createDspyRepairArm, defineControlPolicy, definePinnedContinuationPolicy, degenerateStrategy, deltaRepair, dockerRunArgs, dspyRepairInstructions, gradeRepairRow, injectedTestOracle, isPreStratumReason, isRecordedTimeout, noOpInjectionStep, nodeProcessRunner, normalizeActionForComparison, oracleDeterminism, parseAction, parseAnalystResponse, parseTaskOracleRegistry, reachedT0, reachedT1, reachedT2, readRepairPayload, renderAdmissionReport, renderDeltaRepairReport, renderFormatErrorObservation, renderInstanceMessage, renderObservation, renderRepairTrajectory, renderTimeoutObservation, repairArmAsymmetries, repairArmPromptSha256, repairArmResponse, repairCredit, repairFinding, repairQuestionSha256, repairTaskDefinition, repairTaskPolicy, repairTrajectoryHeader, resolveAdmissionConfig, rolloutDigest, rolloutRecordedSteps, runAdmission, runContinuation, scanShellAction, stratumOf, submissionOf, taskOracleRegistry, testSuiteDigest, toRecordedSteps, totalCost, totalUsage };
|
|
3877
|
+
|
|
3878
|
+
//# sourceMappingURL=index.js.map
|